keyword cleanup: v2 engine, alias rule layer, LLM semantic suggestions

- build_review_bundle.py: 新增 _compute_percentile/_compute_growth,
  候选池从固定阈值改为百分位排名 + 增速因子 (v2 policy)
- term_cleanup_policy.json: 升级 v2 schema
- generate_term_cleanup_suggestions.py: 新增 _prepare_alias_suggestions,
  规则层输出 alias (大小写/单复数/分词变体)
- generate_term_cleanup_semantic_suggestions.py: 新增 LLM 语义建议脚本
  (DeepSeek API, 产出 semantic alias/stopword/promote)
- SKILL.md: 更新为 5 Phase 工作流程
- 首轮清洗 apply: interest 54, aliases 17组, stopwords 17个
- docs/design/keyword-cleanup-flow-overview.md: 流程文档
- plans/: 引擎设计方案
This commit is contained in:
root
2026-05-14 17:17:49 +08:00
parent 4399c9ca90
commit 590d050218
13 changed files with 1592 additions and 57 deletions
@@ -9,16 +9,15 @@ from typing import Any
DEFAULT_POLICY: dict[str, Any] = {
"schema_version": "v1",
"schema_version": "v2",
"interest_keyword_review": {
"min_total_count": 3,
"min_days_seen": 2,
"percentile_min": 0.0,
"percentile_max": 0.05,
"growth_promotion": 0.5,
},
"watch_term_review": {
"min_total_count": 1,
"min_days_seen": 1,
"max_total_count": 2,
"max_days_seen": 2,
"percentile_min": 0.05,
"percentile_max": 0.20,
},
"alias_review": {
"min_total_count": 2,
@@ -28,6 +27,11 @@ DEFAULT_POLICY: dict[str, Any] = {
"max_total_count": 2,
"max_days_seen": 2,
},
"notes": [
"v2: interest/watch 使用百分位排名 + 增速因子替代固定阈值",
"percentile 越小表示排名越高(top 5% = percentile 0.05)",
"growth = recent_count / total_count,衡量近期活跃度",
],
}
@@ -126,6 +130,37 @@ def _within_watch_thresholds(item: dict[str, Any], thresholds: dict[str, Any]) -
)
def _compute_percentile(value: int, sorted_values: list[int]) -> float:
"""
Return the percentile rank of `value` in `sorted_values` (ascending).
0.0 = highest frequency (top rank), 1.0 = lowest frequency (bottom rank).
"""
if not sorted_values:
return 1.0
# bisect_left — count of values strictly less than `value`
lo, hi = 0, len(sorted_values)
while lo < hi:
mid = (lo + hi) // 2
if sorted_values[mid] < value:
lo = mid + 1
else:
hi = mid
rank = lo
# invert: smallest value → rank=0 → 1.0 (bottom)
# largest value → rank=len → 0.0 (top)
return 1.0 - (rank / len(sorted_values))
def _compute_growth(recent_count: int, total_count: int) -> float:
"""
Return growth factor: recent_count / total_count.
Only meaningful when total_count >= 3; returns 0.0 for small counts.
"""
if total_count < 3:
return 0.0
return recent_count / total_count
def main() -> None:
parser = argparse.ArgumentParser(
description="Build a compact review bundle for the keyword-cleanup-review skill."
@@ -235,6 +270,13 @@ def main() -> None:
alias_values = _casefold_set(list(aliases.values()))
watch_set = _casefold_set([str(item.get("term", "")) for item in watchlist])
# Build a sorted list of all total_counts for percentile computation
all_total_counts = sorted(
int(item.get("total_count") or 0)
for item in stats_terms
if isinstance(item, dict) and isinstance(item.get("term"), str)
)
top_global_terms = []
for item in stats_terms[: args.top]:
if not isinstance(item, dict):
@@ -256,42 +298,108 @@ def main() -> None:
"is_alias_target": folded in alias_values,
"in_watchlist": folded in watch_set,
"recent_count": recent_counter.get(term, 0),
"percentile": _compute_percentile(
int(item.get("total_count") or 0), all_total_counts
),
"growth": _compute_growth(
recent_counter.get(term, 0),
int(item.get("total_count") or 0),
),
}
)
# Keep more uncovered terms for percentile-based selection
uncovered_terms = [
item for item in top_global_terms if not item["in_interest_keywords"] and not item["is_stopword"]
][:20]
][:100]
policy_version = (policy.get("schema_version") if isinstance(policy, dict) else None) or "v1"
interest_thresholds = policy.get("interest_keyword_review") if isinstance(policy, dict) else {}
watch_thresholds = policy.get("watch_term_review") if isinstance(policy, dict) else {}
if policy_version == "v2" or "percentile_max" in interest_thresholds:
# v2: percentile + growth based selection
pct_min_interest = float(interest_thresholds.get("percentile_min", 0.0))
pct_max_interest = float(interest_thresholds.get("percentile_max", 0.05))
growth_promo = float(interest_thresholds.get("growth_promotion", 0.5))
pct_min_watch = float(watch_thresholds.get("percentile_min", 0.05))
pct_max_watch = float(watch_thresholds.get("percentile_max", 0.20))
interest_candidates_raw = [
item for item in uncovered_terms
if not item["in_watchlist"]
and pct_min_interest <= item["percentile"] <= pct_max_interest
]
watch_candidates_raw = [
item for item in uncovered_terms
if not item["in_watchlist"]
and pct_min_watch < item["percentile"] <= pct_max_watch
]
# Growth boost: terms outside watch range but with strong growth signal
growth_boost_candidates = [
item for item in uncovered_terms
if not item["in_watchlist"]
and item["percentile"] > pct_max_watch
and item["growth"] >= growth_promo
]
else:
# v1 fallback: fixed thresholds
interest_candidates_raw = [
item for item in uncovered_terms
if not item["in_watchlist"] and _meets_min_thresholds(item, interest_thresholds)
]
watch_candidates_raw = [
item for item in uncovered_terms
if not item["in_watchlist"]
and not _meets_min_thresholds(item, interest_thresholds)
and _within_watch_thresholds(item, watch_thresholds)
]
growth_boost_candidates = []
interest_review_candidates = [
{
"term": item["term"],
"total_count": item["total_count"],
"days_seen": item["days_seen"],
"percentile": item["percentile"],
"growth": item["growth"],
"reason": (
"Meets the configured interest-keyword review threshold and is not yet covered "
"by interest keywords or stopwords."
f"top {item['percentile']:.1%} by frequency,"
f"growth={item['growth']:.0%},"
"not yet covered by interest keywords or stopwords."
),
}
for item in uncovered_terms
if not item["in_watchlist"] and _meets_min_thresholds(item, interest_thresholds)
for item in interest_candidates_raw
][:20]
watch_review_candidates = [
{
"term": item["term"],
"total_count": item["total_count"],
"days_seen": item["days_seen"],
"percentile": item["percentile"],
"growth": item["growth"],
"reason": (
"Falls into the configured watch-term review range and should be observed "
"before promotion into interest keywords."
f"top {item['percentile']:.1%} by frequency,"
f"growth={item['growth']:.0%},"
"fell into watch-review range."
),
}
for item in uncovered_terms
if not item["in_watchlist"]
and not _meets_min_thresholds(item, interest_thresholds)
and _within_watch_thresholds(item, watch_thresholds)
for item in watch_candidates_raw
][:20]
growth_boost_review_items = [
{
"term": item["term"],
"total_count": item["total_count"],
"days_seen": item["days_seen"],
"percentile": item["percentile"],
"growth": item["growth"],
"reason": (
f"growth spike: {item['growth']:.0%} of occurrences in recent window "
f"(total={item['total_count']}, days={item['days_seen']})."
),
}
for item in growth_boost_candidates
][:5]
recent_hot_terms = sorted(
({"term": term, "recent_count": count} for term, count in recent_counter.items()),
key=lambda item: (-item["recent_count"], item["term"].casefold(), item["term"]),
@@ -330,6 +438,7 @@ def main() -> None:
"governance_hints": {
"interest_review_candidates": interest_review_candidates,
"watch_review_candidates": watch_review_candidates,
"growth_boost_review_items": growth_boost_review_items,
},
}
_save_json(args.output, bundle)