keyword cleanup: v2 engine, alias rule layer, LLM semantic suggestions

- build_review_bundle.py: 新增 _compute_percentile/_compute_growth,
  候选池从固定阈值改为百分位排名 + 增速因子 (v2 policy)
- term_cleanup_policy.json: 升级 v2 schema
- generate_term_cleanup_suggestions.py: 新增 _prepare_alias_suggestions,
  规则层输出 alias (大小写/单复数/分词变体)
- generate_term_cleanup_semantic_suggestions.py: 新增 LLM 语义建议脚本
  (DeepSeek API, 产出 semantic alias/stopword/promote)
- SKILL.md: 更新为 5 Phase 工作流程
- 首轮清洗 apply: interest 54, aliases 17组, stopwords 17个
- docs/design/keyword-cleanup-flow-overview.md: 流程文档
- plans/: 引擎设计方案
This commit is contained in:
root
2026-05-14 17:17:49 +08:00
parent 4399c9ca90
commit 590d050218
13 changed files with 1592 additions and 57 deletions
+128 -3
View File
@@ -175,6 +175,121 @@ def _prepare_watch_suggestions(bundle: dict[str, Any], reserved_terms: set[str])
return suggestions
def _prepare_alias_suggestions(
bundle: dict[str, Any],
all_terms: list[dict[str, Any]] | None = None,
) -> list[dict[str, Any]]:
"""
Generate alias suggestions using surface-form rules (no LLM).
Rules:
1. casefold match — same normalized form, different original casing
2. trailing-s singularization — singular/plural variants
3. whitespace/hyphen normalization — word boundary variants
Scans all_terms (full term_stats) if provided; otherwise falls back
to top_global_terms from the bundle.
"""
current_config = _require_dict(bundle.get("current_config"), "bundle.current_config")
interest_keywords = _require_list(
current_config.get("interest_keywords"), "bundle.current_config.interest_keywords"
)
top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms")
source_terms = all_terms if all_terms is not None else top_global_terms
interest_set = {_term_key(t) for t in interest_keywords if isinstance(t, str)}
interest_originals: set[str] = {t for t in interest_keywords if isinstance(t, str)}
# Build full casefold → [original forms] map
cf_map: dict[str, list[str]] = {}
for item in source_terms:
term = None
if isinstance(item, dict):
term = item.get("term")
elif isinstance(item, str):
term = item
if not isinstance(term, str) or not term.strip():
continue
key = _term_key(term)
if key not in cf_map:
cf_map[key] = []
if term not in cf_map[key]:
cf_map[key].append(term)
suggestions: list[dict[str, Any]] = []
seen_pairs: set[tuple[str, str]] = set()
def _add(from_term: str, to_term: str, reason: str) -> None:
pair = (_term_key(from_term), _term_key(to_term))
if pair in seen_pairs:
return
seen_pairs.add(pair)
suggestions.append({"from": from_term, "to": to_term, "reason": reason})
# Build a set of all term keys from source for quick lookup
source_keys = set(cf_map.keys())
# Rule 1: casefold match — same normalized form, different casing
for key, variants in cf_map.items():
if len(variants) < 2:
continue
canonical = None
alt_forms = []
for v in variants:
if v in interest_originals:
canonical = v
else:
alt_forms.append(v)
if canonical and alt_forms:
for alt in alt_forms:
_add(alt, canonical, "Case variant")
elif len(variants) >= 2 and not canonical:
# None is canonical — suggest the highest-frequency form
ranked = sorted(variants, key=lambda t: -(
next(
(it.get("total_count", 0) for it in top_global_terms if it.get("term") == t),
0,
)
))
for alt in ranked[1:]:
_add(alt, ranked[0], "Case variant (auto-ranked)")
# Rule 2: singular/plural — trailing-s normalization
# Check all source terms (not just interest keys) for bidirectional matching
for key in source_keys:
if key in interest_set:
continue
if key.endswith("s") and len(key) > 2:
singular_key = key.rstrip("s")
if singular_key in interest_set and singular_key != key:
# Find canonical interest keyword
canon = next((t for t in interest_keywords if _term_key(t) == singular_key), None)
from_form = cf_map[key][0]
if canon:
_add(from_form, canon, "Plural variant")
# singular form → interest has plural
plural_key = key + "s"
if plural_key in interest_set and plural_key != key:
canon = next((t for t in interest_keywords if _term_key(t) == plural_key), None)
from_form = cf_map[key][0]
if canon:
_add(from_form, canon, "Singular variant")
# Rule 3: whitespace/hyphen normalization
for key in source_keys:
if key in interest_set:
continue
normalized = key.replace("-", "").replace("_", "").replace(" ", "")
if normalized in interest_set and normalized != key:
canon = next((t for t in interest_keywords if _term_key(t) == normalized), None)
from_form = cf_map[key][0]
if canon:
_add(from_form, canon, "Whitespace/punctuation variant")
suggestions.sort(key=lambda x: (x["from"].casefold(), x["to"].casefold()))
return suggestions
def _render_table(items: list[dict[str, Any]]) -> str:
if not items:
return "_None in this pass._\n"
@@ -361,9 +476,19 @@ def main() -> None:
markdown_output=args.markdown_output,
)
# Load full term_stats for alias scanning (bundle only has top N)
stats_path = REPO_ROOT / "data" / "term_index" / "term_stats.json"
all_stats_terms: list[str] = []
if stats_path.exists():
stats_payload = _load_json(stats_path)
raw_terms = stats_payload.get("terms") if isinstance(stats_payload, dict) else []
if isinstance(raw_terms, list):
all_stats_terms = [str(t["term"]) for t in raw_terms if isinstance(t, dict) and isinstance(t.get("term"), str)]
interest_items = _prepare_interest_suggestions(bundle)
reserved_terms = {_term_key(str(item.get("term") or "")) for item in interest_items}
watch_items = _prepare_watch_suggestions(bundle, reserved_terms=reserved_terms)
alias_items = _prepare_alias_suggestions(bundle, all_terms=all_stats_terms)
suggestions = {
"date": suggestion_date,
@@ -373,10 +498,10 @@ def main() -> None:
"summary": {
"interest_keyword_suggestions": len(interest_items),
"watch_terms": len(watch_items),
"alias_suggestions": 0,
"alias_suggestions": len(alias_items),
"stopword_suggestions": 0,
},
"alias_suggestions": [],
"alias_suggestions": alias_items,
"stopword_suggestions": [],
"interest_keyword_suggestions": interest_items,
"watch_terms": watch_items,
@@ -400,7 +525,7 @@ def main() -> None:
"markdown_output": str(markdown_output_path) if args.emit_markdown else None,
"interest_keyword_suggestions": len(interest_items),
"watch_terms": len(watch_items),
"alias_suggestions": 0,
"alias_suggestions": len(alias_items),
"stopword_suggestions": 0,
"emit_markdown": args.emit_markdown,
}