keyword cleanup: v2 engine, alias rule layer, LLM semantic suggestions
- build_review_bundle.py: 新增 _compute_percentile/_compute_growth, 候选池从固定阈值改为百分位排名 + 增速因子 (v2 policy) - term_cleanup_policy.json: 升级 v2 schema - generate_term_cleanup_suggestions.py: 新增 _prepare_alias_suggestions, 规则层输出 alias (大小写/单复数/分词变体) - generate_term_cleanup_semantic_suggestions.py: 新增 LLM 语义建议脚本 (DeepSeek API, 产出 semantic alias/stopword/promote) - SKILL.md: 更新为 5 Phase 工作流程 - 首轮清洗 apply: interest 54, aliases 17组, stopwords 17个 - docs/design/keyword-cleanup-flow-overview.md: 流程文档 - plans/: 引擎设计方案
This commit is contained in:
@@ -175,6 +175,121 @@ def _prepare_watch_suggestions(bundle: dict[str, Any], reserved_terms: set[str])
|
||||
return suggestions
|
||||
|
||||
|
||||
def _prepare_alias_suggestions(
|
||||
bundle: dict[str, Any],
|
||||
all_terms: list[dict[str, Any]] | None = None,
|
||||
) -> list[dict[str, Any]]:
|
||||
"""
|
||||
Generate alias suggestions using surface-form rules (no LLM).
|
||||
|
||||
Rules:
|
||||
1. casefold match — same normalized form, different original casing
|
||||
2. trailing-s singularization — singular/plural variants
|
||||
3. whitespace/hyphen normalization — word boundary variants
|
||||
|
||||
Scans all_terms (full term_stats) if provided; otherwise falls back
|
||||
to top_global_terms from the bundle.
|
||||
"""
|
||||
current_config = _require_dict(bundle.get("current_config"), "bundle.current_config")
|
||||
interest_keywords = _require_list(
|
||||
current_config.get("interest_keywords"), "bundle.current_config.interest_keywords"
|
||||
)
|
||||
top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms")
|
||||
source_terms = all_terms if all_terms is not None else top_global_terms
|
||||
|
||||
interest_set = {_term_key(t) for t in interest_keywords if isinstance(t, str)}
|
||||
interest_originals: set[str] = {t for t in interest_keywords if isinstance(t, str)}
|
||||
|
||||
# Build full casefold → [original forms] map
|
||||
cf_map: dict[str, list[str]] = {}
|
||||
for item in source_terms:
|
||||
term = None
|
||||
if isinstance(item, dict):
|
||||
term = item.get("term")
|
||||
elif isinstance(item, str):
|
||||
term = item
|
||||
if not isinstance(term, str) or not term.strip():
|
||||
continue
|
||||
key = _term_key(term)
|
||||
if key not in cf_map:
|
||||
cf_map[key] = []
|
||||
if term not in cf_map[key]:
|
||||
cf_map[key].append(term)
|
||||
|
||||
suggestions: list[dict[str, Any]] = []
|
||||
seen_pairs: set[tuple[str, str]] = set()
|
||||
|
||||
def _add(from_term: str, to_term: str, reason: str) -> None:
|
||||
pair = (_term_key(from_term), _term_key(to_term))
|
||||
if pair in seen_pairs:
|
||||
return
|
||||
seen_pairs.add(pair)
|
||||
suggestions.append({"from": from_term, "to": to_term, "reason": reason})
|
||||
|
||||
# Build a set of all term keys from source for quick lookup
|
||||
source_keys = set(cf_map.keys())
|
||||
|
||||
# Rule 1: casefold match — same normalized form, different casing
|
||||
for key, variants in cf_map.items():
|
||||
if len(variants) < 2:
|
||||
continue
|
||||
canonical = None
|
||||
alt_forms = []
|
||||
for v in variants:
|
||||
if v in interest_originals:
|
||||
canonical = v
|
||||
else:
|
||||
alt_forms.append(v)
|
||||
if canonical and alt_forms:
|
||||
for alt in alt_forms:
|
||||
_add(alt, canonical, "Case variant")
|
||||
elif len(variants) >= 2 and not canonical:
|
||||
# None is canonical — suggest the highest-frequency form
|
||||
ranked = sorted(variants, key=lambda t: -(
|
||||
next(
|
||||
(it.get("total_count", 0) for it in top_global_terms if it.get("term") == t),
|
||||
0,
|
||||
)
|
||||
))
|
||||
for alt in ranked[1:]:
|
||||
_add(alt, ranked[0], "Case variant (auto-ranked)")
|
||||
|
||||
# Rule 2: singular/plural — trailing-s normalization
|
||||
# Check all source terms (not just interest keys) for bidirectional matching
|
||||
for key in source_keys:
|
||||
if key in interest_set:
|
||||
continue
|
||||
if key.endswith("s") and len(key) > 2:
|
||||
singular_key = key.rstrip("s")
|
||||
if singular_key in interest_set and singular_key != key:
|
||||
# Find canonical interest keyword
|
||||
canon = next((t for t in interest_keywords if _term_key(t) == singular_key), None)
|
||||
from_form = cf_map[key][0]
|
||||
if canon:
|
||||
_add(from_form, canon, "Plural variant")
|
||||
# singular form → interest has plural
|
||||
plural_key = key + "s"
|
||||
if plural_key in interest_set and plural_key != key:
|
||||
canon = next((t for t in interest_keywords if _term_key(t) == plural_key), None)
|
||||
from_form = cf_map[key][0]
|
||||
if canon:
|
||||
_add(from_form, canon, "Singular variant")
|
||||
|
||||
# Rule 3: whitespace/hyphen normalization
|
||||
for key in source_keys:
|
||||
if key in interest_set:
|
||||
continue
|
||||
normalized = key.replace("-", "").replace("_", "").replace(" ", "")
|
||||
if normalized in interest_set and normalized != key:
|
||||
canon = next((t for t in interest_keywords if _term_key(t) == normalized), None)
|
||||
from_form = cf_map[key][0]
|
||||
if canon:
|
||||
_add(from_form, canon, "Whitespace/punctuation variant")
|
||||
|
||||
suggestions.sort(key=lambda x: (x["from"].casefold(), x["to"].casefold()))
|
||||
return suggestions
|
||||
|
||||
|
||||
def _render_table(items: list[dict[str, Any]]) -> str:
|
||||
if not items:
|
||||
return "_None in this pass._\n"
|
||||
@@ -361,9 +476,19 @@ def main() -> None:
|
||||
markdown_output=args.markdown_output,
|
||||
)
|
||||
|
||||
# Load full term_stats for alias scanning (bundle only has top N)
|
||||
stats_path = REPO_ROOT / "data" / "term_index" / "term_stats.json"
|
||||
all_stats_terms: list[str] = []
|
||||
if stats_path.exists():
|
||||
stats_payload = _load_json(stats_path)
|
||||
raw_terms = stats_payload.get("terms") if isinstance(stats_payload, dict) else []
|
||||
if isinstance(raw_terms, list):
|
||||
all_stats_terms = [str(t["term"]) for t in raw_terms if isinstance(t, dict) and isinstance(t.get("term"), str)]
|
||||
|
||||
interest_items = _prepare_interest_suggestions(bundle)
|
||||
reserved_terms = {_term_key(str(item.get("term") or "")) for item in interest_items}
|
||||
watch_items = _prepare_watch_suggestions(bundle, reserved_terms=reserved_terms)
|
||||
alias_items = _prepare_alias_suggestions(bundle, all_terms=all_stats_terms)
|
||||
|
||||
suggestions = {
|
||||
"date": suggestion_date,
|
||||
@@ -373,10 +498,10 @@ def main() -> None:
|
||||
"summary": {
|
||||
"interest_keyword_suggestions": len(interest_items),
|
||||
"watch_terms": len(watch_items),
|
||||
"alias_suggestions": 0,
|
||||
"alias_suggestions": len(alias_items),
|
||||
"stopword_suggestions": 0,
|
||||
},
|
||||
"alias_suggestions": [],
|
||||
"alias_suggestions": alias_items,
|
||||
"stopword_suggestions": [],
|
||||
"interest_keyword_suggestions": interest_items,
|
||||
"watch_terms": watch_items,
|
||||
@@ -400,7 +525,7 @@ def main() -> None:
|
||||
"markdown_output": str(markdown_output_path) if args.emit_markdown else None,
|
||||
"interest_keyword_suggestions": len(interest_items),
|
||||
"watch_terms": len(watch_items),
|
||||
"alias_suggestions": 0,
|
||||
"alias_suggestions": len(alias_items),
|
||||
"stopword_suggestions": 0,
|
||||
"emit_markdown": args.emit_markdown,
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user