Files
root 590d050218 keyword cleanup: v2 engine, alias rule layer, LLM semantic suggestions
- build_review_bundle.py: 新增 _compute_percentile/_compute_growth,
  候选池从固定阈值改为百分位排名 + 增速因子 (v2 policy)
- term_cleanup_policy.json: 升级 v2 schema
- generate_term_cleanup_suggestions.py: 新增 _prepare_alias_suggestions,
  规则层输出 alias (大小写/单复数/分词变体)
- generate_term_cleanup_semantic_suggestions.py: 新增 LLM 语义建议脚本
  (DeepSeek API, 产出 semantic alias/stopword/promote)
- SKILL.md: 更新为 5 Phase 工作流程
- 首轮清洗 apply: interest 54, aliases 17组, stopwords 17个
- docs/design/keyword-cleanup-flow-overview.md: 流程文档
- plans/: 引擎设计方案
2026-05-14 17:17:49 +08:00

449 lines
16 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
from __future__ import annotations
import argparse
import json
from collections import defaultdict
from datetime import UTC, datetime
from pathlib import Path
from typing import Any
DEFAULT_POLICY: dict[str, Any] = {
"schema_version": "v2",
"interest_keyword_review": {
"percentile_min": 0.0,
"percentile_max": 0.05,
"growth_promotion": 0.5,
},
"watch_term_review": {
"percentile_min": 0.05,
"percentile_max": 0.20,
},
"alias_review": {
"min_total_count": 2,
"min_days_seen": 2,
},
"stopword_review": {
"max_total_count": 2,
"max_days_seen": 2,
},
"notes": [
"v2: interest/watch 使用百分位排名 + 增速因子替代固定阈值",
"percentile 越小表示排名越高(top 5% = percentile 0.05)",
"growth = recent_count / total_count,衡量近期活跃度",
],
}
def _load_json(path: Path) -> Any:
return json.loads(path.read_text(encoding="utf-8-sig"))
def _save_json(path: Path, payload: dict[str, Any]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
def _load_mapping(path: Path) -> dict[str, str]:
if not path.exists():
return {}
payload = _load_json(path)
return payload if isinstance(payload, dict) else {}
def _load_list(path: Path) -> list[str]:
if not path.exists():
return []
payload = _load_json(path)
if isinstance(payload, list):
return [item for item in payload if isinstance(item, str)]
return []
def _load_interest_keywords(path: Path) -> list[str]:
if not path.exists():
return []
payload = _load_json(path)
if not isinstance(payload, dict):
return []
values = payload.get("interest_keywords")
if not isinstance(values, list):
return []
return [item for item in values if isinstance(item, str)]
def _casefold_set(values: list[str]) -> set[str]:
return {value.strip().casefold() for value in values if value.strip()}
def _load_policy(path: Path) -> dict[str, Any]:
if not path.exists():
return DEFAULT_POLICY.copy()
payload = _load_json(path)
return payload if isinstance(payload, dict) else DEFAULT_POLICY.copy()
def _load_watchlist(path: Path) -> list[dict[str, Any]]:
if not path.exists():
return []
payload = _load_json(path)
if not isinstance(payload, dict):
return []
terms = payload.get("terms")
if not isinstance(terms, list):
return []
return [item for item in terms if isinstance(item, dict) and isinstance(item.get("term"), str)]
def _load_change_log(path: Path) -> list[dict[str, Any]]:
if not path.exists():
return []
payload = _load_json(path)
if not isinstance(payload, dict):
return []
entries = payload.get("entries")
if not isinstance(entries, list):
return []
return [item for item in entries if isinstance(item, dict)]
def _meets_min_thresholds(item: dict[str, Any], thresholds: dict[str, Any]) -> bool:
min_total_count = int(thresholds.get("min_total_count", 1))
min_days_seen = int(thresholds.get("min_days_seen", 1))
total_count = int(item.get("total_count") or 0)
days_seen = int(item.get("days_seen") or 0)
return total_count >= min_total_count and days_seen >= min_days_seen
def _within_watch_thresholds(item: dict[str, Any], thresholds: dict[str, Any]) -> bool:
min_total_count = int(thresholds.get("min_total_count", 1))
min_days_seen = int(thresholds.get("min_days_seen", 1))
max_total_count = int(thresholds.get("max_total_count", 999999))
max_days_seen = int(thresholds.get("max_days_seen", 999999))
total_count = int(item.get("total_count") or 0)
days_seen = int(item.get("days_seen") or 0)
return (
total_count >= min_total_count
and days_seen >= min_days_seen
and total_count <= max_total_count
and days_seen <= max_days_seen
)
def _compute_percentile(value: int, sorted_values: list[int]) -> float:
"""
Return the percentile rank of `value` in `sorted_values` (ascending).
0.0 = highest frequency (top rank), 1.0 = lowest frequency (bottom rank).
"""
if not sorted_values:
return 1.0
# bisect_left — count of values strictly less than `value`
lo, hi = 0, len(sorted_values)
while lo < hi:
mid = (lo + hi) // 2
if sorted_values[mid] < value:
lo = mid + 1
else:
hi = mid
rank = lo
# invert: smallest value → rank=0 → 1.0 (bottom)
# largest value → rank=len → 0.0 (top)
return 1.0 - (rank / len(sorted_values))
def _compute_growth(recent_count: int, total_count: int) -> float:
"""
Return growth factor: recent_count / total_count.
Only meaningful when total_count >= 3; returns 0.0 for small counts.
"""
if total_count < 3:
return 0.0
return recent_count / total_count
def main() -> None:
parser = argparse.ArgumentParser(
description="Build a compact review bundle for the keyword-cleanup-review skill."
)
repo_root = Path(__file__).resolve().parents[3]
parser.add_argument(
"--stats",
type=Path,
default=repo_root / "data" / "term_index" / "term_stats.json",
help="Global keyword stats file",
)
parser.add_argument(
"--daily-dir",
type=Path,
default=repo_root / "data" / "term_index" / "daily",
help="Directory containing daily keyword index JSON files",
)
parser.add_argument(
"--aliases",
type=Path,
default=repo_root / "configs" / "term_aliases.json",
help="Term aliases JSON file",
)
parser.add_argument(
"--stopwords",
type=Path,
default=repo_root / "configs" / "term_stopwords.json",
help="Term stopwords JSON file",
)
parser.add_argument(
"--context",
type=Path,
default=repo_root / "configs" / "filter_context.personal.json",
help="Personal filter context JSON file",
)
parser.add_argument(
"--policy",
type=Path,
default=repo_root / "configs" / "term_cleanup_policy.json",
help="Keyword cleanup policy JSON file",
)
parser.add_argument(
"--watchlist",
type=Path,
default=repo_root / "configs" / "term_watchlist.json",
help="Tracked watch terms JSON file",
)
parser.add_argument(
"--change-log",
type=Path,
default=repo_root / "configs" / "term_change_log.json",
help="Applied keyword cleanup change log JSON file",
)
parser.add_argument("--days", type=int, default=7, help="How many recent daily files to include")
parser.add_argument("--top", type=int, default=50, help="How many top global terms to include")
parser.add_argument(
"--output",
type=Path,
default=repo_root / "outputs" / "term_index" / "review" / "keyword-cleanup-bundle.json",
help="Output JSON file",
)
args = parser.parse_args()
stats_payload = _load_json(args.stats) if args.stats.exists() else {"terms": []}
stats_terms = stats_payload.get("terms") if isinstance(stats_payload, dict) else []
if not isinstance(stats_terms, list):
stats_terms = []
aliases = _load_mapping(args.aliases)
stopwords = _load_list(args.stopwords)
interest_keywords = _load_interest_keywords(args.context)
policy = _load_policy(args.policy)
watchlist = _load_watchlist(args.watchlist)
change_log = _load_change_log(args.change_log)
recent_daily_paths = sorted(args.daily_dir.glob("*.json"))[-args.days :] if args.daily_dir.exists() else []
recent_daily: list[dict[str, Any]] = []
recent_counter: defaultdict[str, int] = defaultdict(int)
for path in recent_daily_paths:
payload = _load_json(path)
if not isinstance(payload, dict):
continue
terms = payload.get("terms")
if not isinstance(terms, list):
terms = []
compact_terms = []
for term in terms:
if not isinstance(term, dict):
continue
normalized = term.get("normalized_term")
count = term.get("count")
if not isinstance(normalized, str) or not isinstance(count, int):
continue
compact_terms.append({"term": normalized, "count": count})
recent_counter[normalized] += count
recent_daily.append(
{
"date": payload.get("date"),
"candidate_count": payload.get("candidate_count"),
"top_terms": compact_terms[:20],
}
)
interest_set = _casefold_set(interest_keywords)
stopword_set = _casefold_set(stopwords)
alias_keys = _casefold_set(list(aliases.keys()))
alias_values = _casefold_set(list(aliases.values()))
watch_set = _casefold_set([str(item.get("term", "")) for item in watchlist])
# Build a sorted list of all total_counts for percentile computation
all_total_counts = sorted(
int(item.get("total_count") or 0)
for item in stats_terms
if isinstance(item, dict) and isinstance(item.get("term"), str)
)
top_global_terms = []
for item in stats_terms[: args.top]:
if not isinstance(item, dict):
continue
term = item.get("term")
if not isinstance(term, str):
continue
folded = term.strip().casefold()
top_global_terms.append(
{
"term": term,
"total_count": item.get("total_count"),
"days_seen": item.get("days_seen"),
"first_seen": item.get("first_seen"),
"last_seen": item.get("last_seen"),
"in_interest_keywords": folded in interest_set,
"is_stopword": folded in stopword_set,
"is_alias_source": folded in alias_keys,
"is_alias_target": folded in alias_values,
"in_watchlist": folded in watch_set,
"recent_count": recent_counter.get(term, 0),
"percentile": _compute_percentile(
int(item.get("total_count") or 0), all_total_counts
),
"growth": _compute_growth(
recent_counter.get(term, 0),
int(item.get("total_count") or 0),
),
}
)
# Keep more uncovered terms for percentile-based selection
uncovered_terms = [
item for item in top_global_terms if not item["in_interest_keywords"] and not item["is_stopword"]
][:100]
policy_version = (policy.get("schema_version") if isinstance(policy, dict) else None) or "v1"
interest_thresholds = policy.get("interest_keyword_review") if isinstance(policy, dict) else {}
watch_thresholds = policy.get("watch_term_review") if isinstance(policy, dict) else {}
if policy_version == "v2" or "percentile_max" in interest_thresholds:
# v2: percentile + growth based selection
pct_min_interest = float(interest_thresholds.get("percentile_min", 0.0))
pct_max_interest = float(interest_thresholds.get("percentile_max", 0.05))
growth_promo = float(interest_thresholds.get("growth_promotion", 0.5))
pct_min_watch = float(watch_thresholds.get("percentile_min", 0.05))
pct_max_watch = float(watch_thresholds.get("percentile_max", 0.20))
interest_candidates_raw = [
item for item in uncovered_terms
if not item["in_watchlist"]
and pct_min_interest <= item["percentile"] <= pct_max_interest
]
watch_candidates_raw = [
item for item in uncovered_terms
if not item["in_watchlist"]
and pct_min_watch < item["percentile"] <= pct_max_watch
]
# Growth boost: terms outside watch range but with strong growth signal
growth_boost_candidates = [
item for item in uncovered_terms
if not item["in_watchlist"]
and item["percentile"] > pct_max_watch
and item["growth"] >= growth_promo
]
else:
# v1 fallback: fixed thresholds
interest_candidates_raw = [
item for item in uncovered_terms
if not item["in_watchlist"] and _meets_min_thresholds(item, interest_thresholds)
]
watch_candidates_raw = [
item for item in uncovered_terms
if not item["in_watchlist"]
and not _meets_min_thresholds(item, interest_thresholds)
and _within_watch_thresholds(item, watch_thresholds)
]
growth_boost_candidates = []
interest_review_candidates = [
{
"term": item["term"],
"total_count": item["total_count"],
"days_seen": item["days_seen"],
"percentile": item["percentile"],
"growth": item["growth"],
"reason": (
f"top {item['percentile']:.1%} by frequency,"
f"growth={item['growth']:.0%},"
"not yet covered by interest keywords or stopwords."
),
}
for item in interest_candidates_raw
][:20]
watch_review_candidates = [
{
"term": item["term"],
"total_count": item["total_count"],
"days_seen": item["days_seen"],
"percentile": item["percentile"],
"growth": item["growth"],
"reason": (
f"top {item['percentile']:.1%} by frequency,"
f"growth={item['growth']:.0%},"
"fell into watch-review range."
),
}
for item in watch_candidates_raw
][:20]
growth_boost_review_items = [
{
"term": item["term"],
"total_count": item["total_count"],
"days_seen": item["days_seen"],
"percentile": item["percentile"],
"growth": item["growth"],
"reason": (
f"growth spike: {item['growth']:.0%} of occurrences in recent window "
f"(total={item['total_count']}, days={item['days_seen']})."
),
}
for item in growth_boost_candidates
][:5]
recent_hot_terms = sorted(
({"term": term, "recent_count": count} for term, count in recent_counter.items()),
key=lambda item: (-item["recent_count"], item["term"].casefold(), item["term"]),
)[:20]
bundle = {
"generated_at": datetime.now(tz=UTC).isoformat(),
"days": args.days,
"top": args.top,
"sources": {
"stats": str(args.stats),
"daily_dir": str(args.daily_dir),
"aliases": str(args.aliases),
"stopwords": str(args.stopwords),
"context": str(args.context),
"policy": str(args.policy),
"watchlist": str(args.watchlist),
"change_log": str(args.change_log),
},
"policy": policy,
"current_config": {
"alias_count": len(aliases),
"stopword_count": len(stopwords),
"interest_keyword_count": len(interest_keywords),
"watch_term_count": len(watchlist),
"aliases": aliases,
"stopwords": stopwords,
"interest_keywords": interest_keywords,
"watchlist": watchlist,
},
"change_log_tail": change_log[-10:],
"top_global_terms": top_global_terms,
"recent_daily": recent_daily,
"recent_hot_terms": recent_hot_terms,
"uncovered_terms": uncovered_terms,
"governance_hints": {
"interest_review_candidates": interest_review_candidates,
"watch_review_candidates": watch_review_candidates,
"growth_boost_review_items": growth_boost_review_items,
},
}
_save_json(args.output, bundle)
print(f"Saved keyword cleanup bundle to {args.output}")
if __name__ == "__main__":
main()