新功能:个性化推荐算法
This commit is contained in:
@@ -0,0 +1,22 @@
|
||||
"""
|
||||
个性化推荐|Rerank & Freqcap 子模块(重排 / 去重 / 频控)
|
||||
|
||||
说明(V1):
|
||||
- 本模块在 Soft Scoring 后执行,消费候选的 `final_score`,输出可下发的排序结果。
|
||||
- 仅做 Dedup / Freqcap / Feed MMR,不做 Soft Scoring 与 Hard Filter。
|
||||
"""
|
||||
|
||||
from .defaults import get_default_config
|
||||
from .rerank import rerank_and_freqcap
|
||||
from .types import RerankConfig, RerankMeta, RerankResult, ScoredCandidate, Scene
|
||||
|
||||
__all__ = [
|
||||
"RerankConfig",
|
||||
"RerankMeta",
|
||||
"RerankResult",
|
||||
"ScoredCandidate",
|
||||
"Scene",
|
||||
"get_default_config",
|
||||
"rerank_and_freqcap",
|
||||
]
|
||||
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,41 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from app.features.personalized_reco.rerank_freqcap.types import RerankConfig, Scene
|
||||
|
||||
|
||||
_DEFAULTS: dict[Scene, RerankConfig] = {
|
||||
# Feed:MMR λ=0.7;冷却参数不强制使用
|
||||
"feed": RerankConfig(
|
||||
mmr_lambda=0.7,
|
||||
top_n_for_mmr=200,
|
||||
cooldown_sentence_days=0,
|
||||
cooldown_author_days=0,
|
||||
cooldown_template_days=0,
|
||||
),
|
||||
# Push:工程默认(来自算法规则的建议参数)
|
||||
"push": RerankConfig(
|
||||
mmr_lambda=0.7,
|
||||
top_n_for_mmr=200,
|
||||
cooldown_sentence_days=14,
|
||||
cooldown_author_days=7,
|
||||
cooldown_template_days=7,
|
||||
),
|
||||
# Widget:工程默认
|
||||
"widget": RerankConfig(
|
||||
mmr_lambda=0.7,
|
||||
top_n_for_mmr=200,
|
||||
cooldown_sentence_days=7,
|
||||
cooldown_author_days=7,
|
||||
cooldown_template_days=7,
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def get_default_config(scene: Scene) -> RerankConfig:
|
||||
"""
|
||||
获取指定场景的默认参数(返回副本,避免被意外修改)。
|
||||
"""
|
||||
|
||||
base = _DEFAULTS[scene]
|
||||
return RerankConfig.model_validate(base.model_dump())
|
||||
|
||||
208
server/app/features/personalized_reco/rerank_freqcap/rerank.py
Normal file
208
server/app/features/personalized_reco/rerank_freqcap/rerank.py
Normal file
@@ -0,0 +1,208 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Iterable, Optional
|
||||
|
||||
from app.features.personalized_reco.rerank_freqcap.defaults import get_default_config
|
||||
from app.features.personalized_reco.rerank_freqcap.types import RerankConfig, RerankMeta, RerankResult, ScoredCandidate, Scene
|
||||
from app.features.personalized_reco.rerank_freqcap.utils import as_finite_float, build_tags, clamp, jaccard, normalize_int_id_set
|
||||
|
||||
|
||||
def _sort_by_score_desc(cands: list[ScoredCandidate]) -> list[ScoredCandidate]:
|
||||
return sorted(cands, key=lambda x: as_finite_float(x.final_score, default=float("-inf")), reverse=True)
|
||||
|
||||
|
||||
def _dedup_by_seen_ids(
|
||||
cands: list[ScoredCandidate],
|
||||
*,
|
||||
seen_ids: set[int],
|
||||
) -> tuple[list[ScoredCandidate], int]:
|
||||
kept: list[ScoredCandidate] = []
|
||||
removed = 0
|
||||
for c in cands:
|
||||
if int(c.content_id) in seen_ids:
|
||||
removed += 1
|
||||
continue
|
||||
kept.append(c)
|
||||
return kept, removed
|
||||
|
||||
|
||||
def _apply_author_template_freqcap(
|
||||
cands: list[ScoredCandidate],
|
||||
*,
|
||||
recent_author_ids: Optional[Iterable[str]],
|
||||
recent_template_ids: Optional[Iterable[str]],
|
||||
) -> tuple[list[ScoredCandidate], dict[str, int], list[str]]:
|
||||
"""
|
||||
V1 策略:
|
||||
- 若 recent_*_ids 未提供(None),不执行该维度过滤,但在 meta 记录缺失
|
||||
- 若提供,则执行硬过滤
|
||||
"""
|
||||
|
||||
filtered_counts: dict[str, int] = {"author": 0, "template": 0}
|
||||
missing: list[str] = []
|
||||
|
||||
author_set: set[str] | None
|
||||
if recent_author_ids is None:
|
||||
author_set = None
|
||||
missing.append("author")
|
||||
else:
|
||||
author_set = set([a for a in recent_author_ids if a is not None and str(a).strip() != ""])
|
||||
|
||||
template_set: set[str] | None
|
||||
if recent_template_ids is None:
|
||||
template_set = None
|
||||
missing.append("template")
|
||||
else:
|
||||
template_set = set([t for t in recent_template_ids if t is not None and str(t).strip() != ""])
|
||||
|
||||
out: list[ScoredCandidate] = []
|
||||
for c in cands:
|
||||
if author_set is not None and c.author_id and c.author_id in author_set:
|
||||
filtered_counts["author"] += 1
|
||||
continue
|
||||
if template_set is not None and c.template_id and c.template_id in template_set:
|
||||
filtered_counts["template"] += 1
|
||||
continue
|
||||
out.append(c)
|
||||
|
||||
# 只返回真正生效的维度计数(避免 meta 噪音)
|
||||
effective_counts: dict[str, int] = {}
|
||||
if author_set is not None:
|
||||
effective_counts["author"] = int(filtered_counts["author"])
|
||||
if template_set is not None:
|
||||
effective_counts["template"] = int(filtered_counts["template"])
|
||||
|
||||
missing_sorted = sorted(set(missing))
|
||||
return out, effective_counts, missing_sorted
|
||||
|
||||
|
||||
def _sim(a: ScoredCandidate, b: ScoredCandidate, *, tags_a: set[str], tags_b: set[str]) -> float:
|
||||
# 离散特征版(V1 推荐),对齐 plan.md
|
||||
if int(a.content_id) == int(b.content_id):
|
||||
return 1.0
|
||||
|
||||
sim = 0.0
|
||||
if a.template_id and b.template_id and a.template_id == b.template_id:
|
||||
sim += 0.6
|
||||
if a.author_id and b.author_id and a.author_id == b.author_id:
|
||||
sim += 0.3
|
||||
|
||||
sim += 0.1 * jaccard(tags_a, tags_b)
|
||||
return clamp(sim, 0.0, 1.0)
|
||||
|
||||
|
||||
def _mmr_rerank(
|
||||
*,
|
||||
candidates: list[ScoredCandidate],
|
||||
k: int,
|
||||
lam: float,
|
||||
) -> list[ScoredCandidate]:
|
||||
if k <= 0:
|
||||
return []
|
||||
|
||||
if not candidates:
|
||||
return []
|
||||
|
||||
lam_f = clamp(as_finite_float(lam, default=0.7), 0.0, 1.0)
|
||||
|
||||
# 预计算 tags,避免重复构造
|
||||
tags_map: dict[int, set[str]] = {}
|
||||
for c in candidates:
|
||||
tags_map[int(c.content_id)] = build_tags(c)
|
||||
|
||||
remaining = _sort_by_score_desc(list(candidates))
|
||||
selected: list[ScoredCandidate] = []
|
||||
|
||||
# Top1:最高分
|
||||
selected.append(remaining.pop(0))
|
||||
|
||||
while remaining and len(selected) < k:
|
||||
best_idx = 0
|
||||
best_val = float("-inf")
|
||||
|
||||
for idx, c in enumerate(remaining):
|
||||
rel = as_finite_float(c.final_score, default=float("-inf"))
|
||||
|
||||
tags_c = tags_map.get(int(c.content_id), set())
|
||||
max_sim = 0.0
|
||||
for s in selected:
|
||||
tags_s = tags_map.get(int(s.content_id), set())
|
||||
max_sim = max(max_sim, _sim(c, s, tags_a=tags_c, tags_b=tags_s))
|
||||
|
||||
val = lam_f * float(rel) - (1.0 - lam_f) * float(max_sim)
|
||||
if val > best_val:
|
||||
best_val = val
|
||||
best_idx = idx
|
||||
|
||||
selected.append(remaining.pop(best_idx))
|
||||
|
||||
return selected
|
||||
|
||||
|
||||
def rerank_and_freqcap(
|
||||
*,
|
||||
scene: Scene,
|
||||
scored_candidates: list[ScoredCandidate],
|
||||
already_recommended_ids: list[Any],
|
||||
touched_or_viewed_ids: list[Any],
|
||||
k: int,
|
||||
config: Optional[RerankConfig] = None,
|
||||
recent_author_ids: Optional[list[str]] = None,
|
||||
recent_template_ids: Optional[list[str]] = None,
|
||||
) -> RerankResult:
|
||||
"""
|
||||
主入口:对 scored_candidates 做去重/频控/重排,输出最终可下发序列。
|
||||
|
||||
V1 约定:
|
||||
- 冷却窗口“按天”由调用方保证输入集合已经裁剪到窗口内,本模块以“集合代表窗口内历史”为准
|
||||
- Feed 默认只做 dedup + MMR;Push/Widget 做 dedup + freqcap + TopK
|
||||
"""
|
||||
|
||||
cfg = config or get_default_config(scene)
|
||||
|
||||
# seen_ids = already_recommended_ids ∪ touched_or_viewed_ids
|
||||
seen_ids = normalize_int_id_set(list(already_recommended_ids) + list(touched_or_viewed_ids))
|
||||
|
||||
# 先按分数降序,保证 Top1 与 TopK 一致
|
||||
base_sorted = _sort_by_score_desc(list(scored_candidates))
|
||||
|
||||
after_dedup, removed_sentence = _dedup_by_seen_ids(base_sorted, seen_ids=seen_ids)
|
||||
candidate_pool_size_after_dedup = len(after_dedup)
|
||||
|
||||
missing_history_fields: list[str] = []
|
||||
freqcap_counts: dict[str, int] = {"sentence": int(removed_sentence)}
|
||||
|
||||
after_freqcap = after_dedup
|
||||
|
||||
# Push/Widget:作者/模板冷却(增强项)
|
||||
if scene in {"push", "widget"}:
|
||||
after_freqcap, dim_counts, missing = _apply_author_template_freqcap(
|
||||
after_freqcap,
|
||||
recent_author_ids=recent_author_ids,
|
||||
recent_template_ids=recent_template_ids,
|
||||
)
|
||||
missing_history_fields = missing
|
||||
freqcap_counts.update(dim_counts)
|
||||
else:
|
||||
# Feed:不强制作者/模板冷却(V1 可选,这里默认跳过)
|
||||
missing_history_fields = []
|
||||
|
||||
candidate_pool_size_after_freqcap = len(after_freqcap)
|
||||
|
||||
ranked: list[ScoredCandidate]
|
||||
if scene == "feed":
|
||||
# MMR 前截断,避免性能问题
|
||||
top_n = int(cfg.top_n_for_mmr) if int(cfg.top_n_for_mmr) > 0 else len(after_freqcap)
|
||||
mmr_pool = after_freqcap[:top_n]
|
||||
ranked = _mmr_rerank(candidates=mmr_pool, k=int(k), lam=cfg.mmr_lambda)
|
||||
else:
|
||||
ranked = after_freqcap[: max(0, int(k))]
|
||||
|
||||
meta = RerankMeta(
|
||||
candidate_pool_size_after_dedup=int(candidate_pool_size_after_dedup),
|
||||
candidate_pool_size_after_freqcap=int(candidate_pool_size_after_freqcap),
|
||||
missing_history_fields=missing_history_fields,
|
||||
freqcap_filtered_counts=freqcap_counts,
|
||||
)
|
||||
return RerankResult(ranked_items=ranked, meta=meta)
|
||||
|
||||
@@ -0,0 +1,61 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Literal, Optional
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
from app.features.personalized_reco.content_repository.types import ContentProfileDTO
|
||||
|
||||
Scene = Literal["feed", "push", "widget"]
|
||||
|
||||
|
||||
class ScoredCandidate(BaseModel):
|
||||
"""
|
||||
Soft Scoring 后的候选项(本模块消费的最小字段集合)。
|
||||
|
||||
说明:
|
||||
- `content_profile` 用于 Feed 的标签/相似度计算;缺失时需降级为仅使用 author/template 等字段
|
||||
"""
|
||||
|
||||
content_id: int
|
||||
final_score: float
|
||||
|
||||
author_id: Optional[str] = None
|
||||
template_id: Optional[str] = None
|
||||
|
||||
content_profile: Optional[ContentProfileDTO] = None
|
||||
|
||||
# 允许透传额外字段(例如 text、breakdown 等),便于上层直接下发
|
||||
extra: dict[str, Any] = Field(default_factory=dict)
|
||||
|
||||
|
||||
class RerankConfig(BaseModel):
|
||||
"""
|
||||
重排/频控配置(可调参)。
|
||||
"""
|
||||
|
||||
# Feed:MMR
|
||||
mmr_lambda: float = 0.7
|
||||
top_n_for_mmr: int = 200
|
||||
|
||||
# Push/Widget:冷却窗口(V1 主要用于配置与可观测;真正按天需要带时间戳的历史)
|
||||
cooldown_sentence_days: int = 14
|
||||
cooldown_author_days: int = 7
|
||||
cooldown_template_days: int = 7
|
||||
|
||||
|
||||
class RerankMeta(BaseModel):
|
||||
candidate_pool_size_after_dedup: int
|
||||
candidate_pool_size_after_freqcap: int
|
||||
|
||||
# 例如未提供 recent_author_ids/recent_template_ids 时记录 ["author","template"]
|
||||
missing_history_fields: list[str] = Field(default_factory=list)
|
||||
|
||||
# 可选但建议:按维度统计被过滤数量
|
||||
freqcap_filtered_counts: dict[str, int] = Field(default_factory=dict)
|
||||
|
||||
|
||||
class RerankResult(BaseModel):
|
||||
ranked_items: list[ScoredCandidate] = Field(default_factory=list)
|
||||
meta: RerankMeta
|
||||
|
||||
107
server/app/features/personalized_reco/rerank_freqcap/utils.py
Normal file
107
server/app/features/personalized_reco/rerank_freqcap/utils.py
Normal file
@@ -0,0 +1,107 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import Any, Iterable
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def clamp(value: float, min_value: float, max_value: float) -> float:
|
||||
if value != value: # NaN
|
||||
return min_value
|
||||
return max(min_value, min(max_value, value))
|
||||
|
||||
|
||||
def as_finite_float(value: Any, *, default: float) -> float:
|
||||
try:
|
||||
f = float(value)
|
||||
except Exception:
|
||||
return float(default)
|
||||
if f != f:
|
||||
return float(default)
|
||||
if f == float("inf") or f == float("-inf"):
|
||||
return float(default)
|
||||
return f
|
||||
|
||||
|
||||
def normalize_int_id_set(values: Iterable[Any]) -> set[int]:
|
||||
"""
|
||||
将历史 ID 列表归一化为 int 集合(支持 str/int 混用)。
|
||||
|
||||
说明:
|
||||
- 无法转换的值会被忽略,并记录 debug 日志(不影响主流程)
|
||||
"""
|
||||
|
||||
out: set[int] = set()
|
||||
for v in values:
|
||||
try:
|
||||
if isinstance(v, bool):
|
||||
# 避免 True/False 被当作 1/0
|
||||
raise ValueError("bool 不是合法 id")
|
||||
out.add(int(v))
|
||||
except Exception:
|
||||
logger.debug("历史 id 无法转为 int,已忽略:%r", v)
|
||||
return out
|
||||
|
||||
|
||||
def jaccard(a: set[str], b: set[str]) -> float:
|
||||
if not a and not b:
|
||||
return 0.0
|
||||
inter = len(a & b)
|
||||
union = len(a | b)
|
||||
return float(inter) / float(union) if union > 0 else 0.0
|
||||
|
||||
|
||||
def argmax_key(d: dict[str, Any] | None) -> str | None:
|
||||
"""
|
||||
从 suitability 字典中取最大值 key(V1 用作代表标签)。
|
||||
- 空字典/None -> None
|
||||
- 值非法 -> 按 default=0 处理
|
||||
"""
|
||||
|
||||
if not d:
|
||||
return None
|
||||
best_k: str | None = None
|
||||
best_v = float("-inf")
|
||||
for k, v in d.items():
|
||||
fv = as_finite_float(v, default=0.0)
|
||||
if fv > best_v:
|
||||
best_v = fv
|
||||
best_k = k
|
||||
return best_k
|
||||
|
||||
|
||||
def build_tags(candidate: Any) -> set[str]:
|
||||
"""
|
||||
构造离散标签集合(V1 写死):
|
||||
- stage:<stage>
|
||||
- need:<argmax_key>
|
||||
- context:<argmax_key>
|
||||
|
||||
说明:
|
||||
- candidate 可能是 ScoredCandidate 或具备 content_profile 的对象
|
||||
- 字段缺失时自动降级(只返回可得标签)
|
||||
"""
|
||||
|
||||
tags: set[str] = set()
|
||||
|
||||
cp = getattr(candidate, "content_profile", None)
|
||||
if cp is None:
|
||||
return tags
|
||||
|
||||
stage = getattr(cp, "stage", None)
|
||||
if stage:
|
||||
tags.add(f"stage:{stage}")
|
||||
|
||||
need = getattr(cp, "need_suitability", None)
|
||||
need_k = argmax_key(need)
|
||||
if need_k:
|
||||
tags.add(f"need:{need_k}")
|
||||
|
||||
ctx = getattr(cp, "context_suitability", None)
|
||||
ctx_k = argmax_key(ctx)
|
||||
if ctx_k:
|
||||
tags.add(f"context:{ctx_k}")
|
||||
|
||||
return tags
|
||||
|
||||
Reference in New Issue
Block a user