新功能:个性化推荐算法

This commit is contained in:
吕新雨
2026-02-02 16:47:37 +08:00
parent 936094211b
commit 6dc4e2b943
119 changed files with 7427 additions and 357 deletions

View File

@@ -0,0 +1,22 @@
"""
个性化推荐Rerank & Freqcap 子模块(重排 / 去重 / 频控)
说明V1
- 本模块在 Soft Scoring 后执行,消费候选的 `final_score`,输出可下发的排序结果。
- 仅做 Dedup / Freqcap / Feed MMR不做 Soft Scoring 与 Hard Filter。
"""
from .defaults import get_default_config
from .rerank import rerank_and_freqcap
from .types import RerankConfig, RerankMeta, RerankResult, ScoredCandidate, Scene
__all__ = [
"RerankConfig",
"RerankMeta",
"RerankResult",
"ScoredCandidate",
"Scene",
"get_default_config",
"rerank_and_freqcap",
]

View File

@@ -0,0 +1,41 @@
from __future__ import annotations
from app.features.personalized_reco.rerank_freqcap.types import RerankConfig, Scene
_DEFAULTS: dict[Scene, RerankConfig] = {
# FeedMMR λ=0.7;冷却参数不强制使用
"feed": RerankConfig(
mmr_lambda=0.7,
top_n_for_mmr=200,
cooldown_sentence_days=0,
cooldown_author_days=0,
cooldown_template_days=0,
),
# Push工程默认来自算法规则的建议参数
"push": RerankConfig(
mmr_lambda=0.7,
top_n_for_mmr=200,
cooldown_sentence_days=14,
cooldown_author_days=7,
cooldown_template_days=7,
),
# Widget工程默认
"widget": RerankConfig(
mmr_lambda=0.7,
top_n_for_mmr=200,
cooldown_sentence_days=7,
cooldown_author_days=7,
cooldown_template_days=7,
),
}
def get_default_config(scene: Scene) -> RerankConfig:
"""
获取指定场景的默认参数(返回副本,避免被意外修改)。
"""
base = _DEFAULTS[scene]
return RerankConfig.model_validate(base.model_dump())

View File

@@ -0,0 +1,208 @@
from __future__ import annotations
from typing import Any, Iterable, Optional
from app.features.personalized_reco.rerank_freqcap.defaults import get_default_config
from app.features.personalized_reco.rerank_freqcap.types import RerankConfig, RerankMeta, RerankResult, ScoredCandidate, Scene
from app.features.personalized_reco.rerank_freqcap.utils import as_finite_float, build_tags, clamp, jaccard, normalize_int_id_set
def _sort_by_score_desc(cands: list[ScoredCandidate]) -> list[ScoredCandidate]:
return sorted(cands, key=lambda x: as_finite_float(x.final_score, default=float("-inf")), reverse=True)
def _dedup_by_seen_ids(
cands: list[ScoredCandidate],
*,
seen_ids: set[int],
) -> tuple[list[ScoredCandidate], int]:
kept: list[ScoredCandidate] = []
removed = 0
for c in cands:
if int(c.content_id) in seen_ids:
removed += 1
continue
kept.append(c)
return kept, removed
def _apply_author_template_freqcap(
cands: list[ScoredCandidate],
*,
recent_author_ids: Optional[Iterable[str]],
recent_template_ids: Optional[Iterable[str]],
) -> tuple[list[ScoredCandidate], dict[str, int], list[str]]:
"""
V1 策略:
- 若 recent_*_ids 未提供None不执行该维度过滤但在 meta 记录缺失
- 若提供,则执行硬过滤
"""
filtered_counts: dict[str, int] = {"author": 0, "template": 0}
missing: list[str] = []
author_set: set[str] | None
if recent_author_ids is None:
author_set = None
missing.append("author")
else:
author_set = set([a for a in recent_author_ids if a is not None and str(a).strip() != ""])
template_set: set[str] | None
if recent_template_ids is None:
template_set = None
missing.append("template")
else:
template_set = set([t for t in recent_template_ids if t is not None and str(t).strip() != ""])
out: list[ScoredCandidate] = []
for c in cands:
if author_set is not None and c.author_id and c.author_id in author_set:
filtered_counts["author"] += 1
continue
if template_set is not None and c.template_id and c.template_id in template_set:
filtered_counts["template"] += 1
continue
out.append(c)
# 只返回真正生效的维度计数(避免 meta 噪音)
effective_counts: dict[str, int] = {}
if author_set is not None:
effective_counts["author"] = int(filtered_counts["author"])
if template_set is not None:
effective_counts["template"] = int(filtered_counts["template"])
missing_sorted = sorted(set(missing))
return out, effective_counts, missing_sorted
def _sim(a: ScoredCandidate, b: ScoredCandidate, *, tags_a: set[str], tags_b: set[str]) -> float:
# 离散特征版V1 推荐),对齐 plan.md
if int(a.content_id) == int(b.content_id):
return 1.0
sim = 0.0
if a.template_id and b.template_id and a.template_id == b.template_id:
sim += 0.6
if a.author_id and b.author_id and a.author_id == b.author_id:
sim += 0.3
sim += 0.1 * jaccard(tags_a, tags_b)
return clamp(sim, 0.0, 1.0)
def _mmr_rerank(
*,
candidates: list[ScoredCandidate],
k: int,
lam: float,
) -> list[ScoredCandidate]:
if k <= 0:
return []
if not candidates:
return []
lam_f = clamp(as_finite_float(lam, default=0.7), 0.0, 1.0)
# 预计算 tags避免重复构造
tags_map: dict[int, set[str]] = {}
for c in candidates:
tags_map[int(c.content_id)] = build_tags(c)
remaining = _sort_by_score_desc(list(candidates))
selected: list[ScoredCandidate] = []
# Top1最高分
selected.append(remaining.pop(0))
while remaining and len(selected) < k:
best_idx = 0
best_val = float("-inf")
for idx, c in enumerate(remaining):
rel = as_finite_float(c.final_score, default=float("-inf"))
tags_c = tags_map.get(int(c.content_id), set())
max_sim = 0.0
for s in selected:
tags_s = tags_map.get(int(s.content_id), set())
max_sim = max(max_sim, _sim(c, s, tags_a=tags_c, tags_b=tags_s))
val = lam_f * float(rel) - (1.0 - lam_f) * float(max_sim)
if val > best_val:
best_val = val
best_idx = idx
selected.append(remaining.pop(best_idx))
return selected
def rerank_and_freqcap(
*,
scene: Scene,
scored_candidates: list[ScoredCandidate],
already_recommended_ids: list[Any],
touched_or_viewed_ids: list[Any],
k: int,
config: Optional[RerankConfig] = None,
recent_author_ids: Optional[list[str]] = None,
recent_template_ids: Optional[list[str]] = None,
) -> RerankResult:
"""
主入口:对 scored_candidates 做去重/频控/重排,输出最终可下发序列。
V1 约定:
- 冷却窗口“按天”由调用方保证输入集合已经裁剪到窗口内,本模块以“集合代表窗口内历史”为准
- Feed 默认只做 dedup + MMRPush/Widget 做 dedup + freqcap + TopK
"""
cfg = config or get_default_config(scene)
# seen_ids = already_recommended_ids touched_or_viewed_ids
seen_ids = normalize_int_id_set(list(already_recommended_ids) + list(touched_or_viewed_ids))
# 先按分数降序,保证 Top1 与 TopK 一致
base_sorted = _sort_by_score_desc(list(scored_candidates))
after_dedup, removed_sentence = _dedup_by_seen_ids(base_sorted, seen_ids=seen_ids)
candidate_pool_size_after_dedup = len(after_dedup)
missing_history_fields: list[str] = []
freqcap_counts: dict[str, int] = {"sentence": int(removed_sentence)}
after_freqcap = after_dedup
# Push/Widget作者/模板冷却(增强项)
if scene in {"push", "widget"}:
after_freqcap, dim_counts, missing = _apply_author_template_freqcap(
after_freqcap,
recent_author_ids=recent_author_ids,
recent_template_ids=recent_template_ids,
)
missing_history_fields = missing
freqcap_counts.update(dim_counts)
else:
# Feed不强制作者/模板冷却V1 可选,这里默认跳过)
missing_history_fields = []
candidate_pool_size_after_freqcap = len(after_freqcap)
ranked: list[ScoredCandidate]
if scene == "feed":
# MMR 前截断,避免性能问题
top_n = int(cfg.top_n_for_mmr) if int(cfg.top_n_for_mmr) > 0 else len(after_freqcap)
mmr_pool = after_freqcap[:top_n]
ranked = _mmr_rerank(candidates=mmr_pool, k=int(k), lam=cfg.mmr_lambda)
else:
ranked = after_freqcap[: max(0, int(k))]
meta = RerankMeta(
candidate_pool_size_after_dedup=int(candidate_pool_size_after_dedup),
candidate_pool_size_after_freqcap=int(candidate_pool_size_after_freqcap),
missing_history_fields=missing_history_fields,
freqcap_filtered_counts=freqcap_counts,
)
return RerankResult(ranked_items=ranked, meta=meta)

View File

@@ -0,0 +1,61 @@
from __future__ import annotations
from typing import Any, Literal, Optional
from pydantic import BaseModel, Field
from app.features.personalized_reco.content_repository.types import ContentProfileDTO
Scene = Literal["feed", "push", "widget"]
class ScoredCandidate(BaseModel):
"""
Soft Scoring 后的候选项(本模块消费的最小字段集合)。
说明:
- `content_profile` 用于 Feed 的标签/相似度计算;缺失时需降级为仅使用 author/template 等字段
"""
content_id: int
final_score: float
author_id: Optional[str] = None
template_id: Optional[str] = None
content_profile: Optional[ContentProfileDTO] = None
# 允许透传额外字段(例如 text、breakdown 等),便于上层直接下发
extra: dict[str, Any] = Field(default_factory=dict)
class RerankConfig(BaseModel):
"""
重排/频控配置(可调参)。
"""
# FeedMMR
mmr_lambda: float = 0.7
top_n_for_mmr: int = 200
# Push/Widget冷却窗口V1 主要用于配置与可观测;真正按天需要带时间戳的历史)
cooldown_sentence_days: int = 14
cooldown_author_days: int = 7
cooldown_template_days: int = 7
class RerankMeta(BaseModel):
candidate_pool_size_after_dedup: int
candidate_pool_size_after_freqcap: int
# 例如未提供 recent_author_ids/recent_template_ids 时记录 ["author","template"]
missing_history_fields: list[str] = Field(default_factory=list)
# 可选但建议:按维度统计被过滤数量
freqcap_filtered_counts: dict[str, int] = Field(default_factory=dict)
class RerankResult(BaseModel):
ranked_items: list[ScoredCandidate] = Field(default_factory=list)
meta: RerankMeta

View File

@@ -0,0 +1,107 @@
from __future__ import annotations
import logging
from typing import Any, Iterable
logger = logging.getLogger(__name__)
def clamp(value: float, min_value: float, max_value: float) -> float:
if value != value: # NaN
return min_value
return max(min_value, min(max_value, value))
def as_finite_float(value: Any, *, default: float) -> float:
try:
f = float(value)
except Exception:
return float(default)
if f != f:
return float(default)
if f == float("inf") or f == float("-inf"):
return float(default)
return f
def normalize_int_id_set(values: Iterable[Any]) -> set[int]:
"""
将历史 ID 列表归一化为 int 集合(支持 str/int 混用)。
说明:
- 无法转换的值会被忽略,并记录 debug 日志(不影响主流程)
"""
out: set[int] = set()
for v in values:
try:
if isinstance(v, bool):
# 避免 True/False 被当作 1/0
raise ValueError("bool 不是合法 id")
out.add(int(v))
except Exception:
logger.debug("历史 id 无法转为 int已忽略%r", v)
return out
def jaccard(a: set[str], b: set[str]) -> float:
if not a and not b:
return 0.0
inter = len(a & b)
union = len(a | b)
return float(inter) / float(union) if union > 0 else 0.0
def argmax_key(d: dict[str, Any] | None) -> str | None:
"""
从 suitability 字典中取最大值 keyV1 用作代表标签)。
- 空字典/None -> None
- 值非法 -> 按 default=0 处理
"""
if not d:
return None
best_k: str | None = None
best_v = float("-inf")
for k, v in d.items():
fv = as_finite_float(v, default=0.0)
if fv > best_v:
best_v = fv
best_k = k
return best_k
def build_tags(candidate: Any) -> set[str]:
"""
构造离散标签集合V1 写死):
- stage:<stage>
- need:<argmax_key>
- context:<argmax_key>
说明:
- candidate 可能是 ScoredCandidate 或具备 content_profile 的对象
- 字段缺失时自动降级(只返回可得标签)
"""
tags: set[str] = set()
cp = getattr(candidate, "content_profile", None)
if cp is None:
return tags
stage = getattr(cp, "stage", None)
if stage:
tags.add(f"stage:{stage}")
need = getattr(cp, "need_suitability", None)
need_k = argmax_key(need)
if need_k:
tags.add(f"need:{need_k}")
ctx = getattr(cp, "context_suitability", None)
ctx_k = argmax_key(ctx)
if ctx_k:
tags.add(f"context:{ctx_k}")
return tags