keyword cleanup: v2 engine, alias rule layer, LLM semantic suggestions
- build_review_bundle.py: 新增 _compute_percentile/_compute_growth, 候选池从固定阈值改为百分位排名 + 增速因子 (v2 policy) - term_cleanup_policy.json: 升级 v2 schema - generate_term_cleanup_suggestions.py: 新增 _prepare_alias_suggestions, 规则层输出 alias (大小写/单复数/分词变体) - generate_term_cleanup_semantic_suggestions.py: 新增 LLM 语义建议脚本 (DeepSeek API, 产出 semantic alias/stopword/promote) - SKILL.md: 更新为 5 Phase 工作流程 - 首轮清洗 apply: interest 54, aliases 17组, stopwords 17个 - docs/design/keyword-cleanup-flow-overview.md: 流程文档 - plans/: 引擎设计方案
This commit is contained in:
@@ -9,16 +9,15 @@ from typing import Any
|
||||
|
||||
|
||||
DEFAULT_POLICY: dict[str, Any] = {
|
||||
"schema_version": "v1",
|
||||
"schema_version": "v2",
|
||||
"interest_keyword_review": {
|
||||
"min_total_count": 3,
|
||||
"min_days_seen": 2,
|
||||
"percentile_min": 0.0,
|
||||
"percentile_max": 0.05,
|
||||
"growth_promotion": 0.5,
|
||||
},
|
||||
"watch_term_review": {
|
||||
"min_total_count": 1,
|
||||
"min_days_seen": 1,
|
||||
"max_total_count": 2,
|
||||
"max_days_seen": 2,
|
||||
"percentile_min": 0.05,
|
||||
"percentile_max": 0.20,
|
||||
},
|
||||
"alias_review": {
|
||||
"min_total_count": 2,
|
||||
@@ -28,6 +27,11 @@ DEFAULT_POLICY: dict[str, Any] = {
|
||||
"max_total_count": 2,
|
||||
"max_days_seen": 2,
|
||||
},
|
||||
"notes": [
|
||||
"v2: interest/watch 使用百分位排名 + 增速因子替代固定阈值",
|
||||
"percentile 越小表示排名越高(top 5% = percentile 0.05)",
|
||||
"growth = recent_count / total_count,衡量近期活跃度",
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
@@ -126,6 +130,37 @@ def _within_watch_thresholds(item: dict[str, Any], thresholds: dict[str, Any]) -
|
||||
)
|
||||
|
||||
|
||||
def _compute_percentile(value: int, sorted_values: list[int]) -> float:
|
||||
"""
|
||||
Return the percentile rank of `value` in `sorted_values` (ascending).
|
||||
0.0 = highest frequency (top rank), 1.0 = lowest frequency (bottom rank).
|
||||
"""
|
||||
if not sorted_values:
|
||||
return 1.0
|
||||
# bisect_left — count of values strictly less than `value`
|
||||
lo, hi = 0, len(sorted_values)
|
||||
while lo < hi:
|
||||
mid = (lo + hi) // 2
|
||||
if sorted_values[mid] < value:
|
||||
lo = mid + 1
|
||||
else:
|
||||
hi = mid
|
||||
rank = lo
|
||||
# invert: smallest value → rank=0 → 1.0 (bottom)
|
||||
# largest value → rank=len → 0.0 (top)
|
||||
return 1.0 - (rank / len(sorted_values))
|
||||
|
||||
|
||||
def _compute_growth(recent_count: int, total_count: int) -> float:
|
||||
"""
|
||||
Return growth factor: recent_count / total_count.
|
||||
Only meaningful when total_count >= 3; returns 0.0 for small counts.
|
||||
"""
|
||||
if total_count < 3:
|
||||
return 0.0
|
||||
return recent_count / total_count
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Build a compact review bundle for the keyword-cleanup-review skill."
|
||||
@@ -235,6 +270,13 @@ def main() -> None:
|
||||
alias_values = _casefold_set(list(aliases.values()))
|
||||
watch_set = _casefold_set([str(item.get("term", "")) for item in watchlist])
|
||||
|
||||
# Build a sorted list of all total_counts for percentile computation
|
||||
all_total_counts = sorted(
|
||||
int(item.get("total_count") or 0)
|
||||
for item in stats_terms
|
||||
if isinstance(item, dict) and isinstance(item.get("term"), str)
|
||||
)
|
||||
|
||||
top_global_terms = []
|
||||
for item in stats_terms[: args.top]:
|
||||
if not isinstance(item, dict):
|
||||
@@ -256,42 +298,108 @@ def main() -> None:
|
||||
"is_alias_target": folded in alias_values,
|
||||
"in_watchlist": folded in watch_set,
|
||||
"recent_count": recent_counter.get(term, 0),
|
||||
"percentile": _compute_percentile(
|
||||
int(item.get("total_count") or 0), all_total_counts
|
||||
),
|
||||
"growth": _compute_growth(
|
||||
recent_counter.get(term, 0),
|
||||
int(item.get("total_count") or 0),
|
||||
),
|
||||
}
|
||||
)
|
||||
|
||||
# Keep more uncovered terms for percentile-based selection
|
||||
uncovered_terms = [
|
||||
item for item in top_global_terms if not item["in_interest_keywords"] and not item["is_stopword"]
|
||||
][:20]
|
||||
][:100]
|
||||
|
||||
policy_version = (policy.get("schema_version") if isinstance(policy, dict) else None) or "v1"
|
||||
interest_thresholds = policy.get("interest_keyword_review") if isinstance(policy, dict) else {}
|
||||
watch_thresholds = policy.get("watch_term_review") if isinstance(policy, dict) else {}
|
||||
|
||||
if policy_version == "v2" or "percentile_max" in interest_thresholds:
|
||||
# v2: percentile + growth based selection
|
||||
pct_min_interest = float(interest_thresholds.get("percentile_min", 0.0))
|
||||
pct_max_interest = float(interest_thresholds.get("percentile_max", 0.05))
|
||||
growth_promo = float(interest_thresholds.get("growth_promotion", 0.5))
|
||||
pct_min_watch = float(watch_thresholds.get("percentile_min", 0.05))
|
||||
pct_max_watch = float(watch_thresholds.get("percentile_max", 0.20))
|
||||
|
||||
interest_candidates_raw = [
|
||||
item for item in uncovered_terms
|
||||
if not item["in_watchlist"]
|
||||
and pct_min_interest <= item["percentile"] <= pct_max_interest
|
||||
]
|
||||
watch_candidates_raw = [
|
||||
item for item in uncovered_terms
|
||||
if not item["in_watchlist"]
|
||||
and pct_min_watch < item["percentile"] <= pct_max_watch
|
||||
]
|
||||
# Growth boost: terms outside watch range but with strong growth signal
|
||||
growth_boost_candidates = [
|
||||
item for item in uncovered_terms
|
||||
if not item["in_watchlist"]
|
||||
and item["percentile"] > pct_max_watch
|
||||
and item["growth"] >= growth_promo
|
||||
]
|
||||
else:
|
||||
# v1 fallback: fixed thresholds
|
||||
interest_candidates_raw = [
|
||||
item for item in uncovered_terms
|
||||
if not item["in_watchlist"] and _meets_min_thresholds(item, interest_thresholds)
|
||||
]
|
||||
watch_candidates_raw = [
|
||||
item for item in uncovered_terms
|
||||
if not item["in_watchlist"]
|
||||
and not _meets_min_thresholds(item, interest_thresholds)
|
||||
and _within_watch_thresholds(item, watch_thresholds)
|
||||
]
|
||||
growth_boost_candidates = []
|
||||
|
||||
interest_review_candidates = [
|
||||
{
|
||||
"term": item["term"],
|
||||
"total_count": item["total_count"],
|
||||
"days_seen": item["days_seen"],
|
||||
"percentile": item["percentile"],
|
||||
"growth": item["growth"],
|
||||
"reason": (
|
||||
"Meets the configured interest-keyword review threshold and is not yet covered "
|
||||
"by interest keywords or stopwords."
|
||||
f"top {item['percentile']:.1%} by frequency,"
|
||||
f"growth={item['growth']:.0%},"
|
||||
"not yet covered by interest keywords or stopwords."
|
||||
),
|
||||
}
|
||||
for item in uncovered_terms
|
||||
if not item["in_watchlist"] and _meets_min_thresholds(item, interest_thresholds)
|
||||
for item in interest_candidates_raw
|
||||
][:20]
|
||||
watch_review_candidates = [
|
||||
{
|
||||
"term": item["term"],
|
||||
"total_count": item["total_count"],
|
||||
"days_seen": item["days_seen"],
|
||||
"percentile": item["percentile"],
|
||||
"growth": item["growth"],
|
||||
"reason": (
|
||||
"Falls into the configured watch-term review range and should be observed "
|
||||
"before promotion into interest keywords."
|
||||
f"top {item['percentile']:.1%} by frequency,"
|
||||
f"growth={item['growth']:.0%},"
|
||||
"fell into watch-review range."
|
||||
),
|
||||
}
|
||||
for item in uncovered_terms
|
||||
if not item["in_watchlist"]
|
||||
and not _meets_min_thresholds(item, interest_thresholds)
|
||||
and _within_watch_thresholds(item, watch_thresholds)
|
||||
for item in watch_candidates_raw
|
||||
][:20]
|
||||
growth_boost_review_items = [
|
||||
{
|
||||
"term": item["term"],
|
||||
"total_count": item["total_count"],
|
||||
"days_seen": item["days_seen"],
|
||||
"percentile": item["percentile"],
|
||||
"growth": item["growth"],
|
||||
"reason": (
|
||||
f"growth spike: {item['growth']:.0%} of occurrences in recent window "
|
||||
f"(total={item['total_count']}, days={item['days_seen']})."
|
||||
),
|
||||
}
|
||||
for item in growth_boost_candidates
|
||||
][:5]
|
||||
recent_hot_terms = sorted(
|
||||
({"term": term, "recent_count": count} for term, count in recent_counter.items()),
|
||||
key=lambda item: (-item["recent_count"], item["term"].casefold(), item["term"]),
|
||||
@@ -330,6 +438,7 @@ def main() -> None:
|
||||
"governance_hints": {
|
||||
"interest_review_candidates": interest_review_candidates,
|
||||
"watch_review_candidates": watch_review_candidates,
|
||||
"growth_boost_review_items": growth_boost_review_items,
|
||||
},
|
||||
}
|
||||
_save_json(args.output, bundle)
|
||||
|
||||
Reference in New Issue
Block a user