- build_review_bundle.py: 新增 _compute_percentile/_compute_growth, 候选池从固定阈值改为百分位排名 + 增速因子 (v2 policy) - term_cleanup_policy.json: 升级 v2 schema - generate_term_cleanup_suggestions.py: 新增 _prepare_alias_suggestions, 规则层输出 alias (大小写/单复数/分词变体) - generate_term_cleanup_semantic_suggestions.py: 新增 LLM 语义建议脚本 (DeepSeek API, 产出 semantic alias/stopword/promote) - SKILL.md: 更新为 5 Phase 工作流程 - 首轮清洗 apply: interest 54, aliases 17组, stopwords 17个 - docs/design/keyword-cleanup-flow-overview.md: 流程文档 - plans/: 引擎设计方案
449 lines
16 KiB
Python
449 lines
16 KiB
Python
from __future__ import annotations
|
||
|
||
import argparse
|
||
import json
|
||
from collections import defaultdict
|
||
from datetime import UTC, datetime
|
||
from pathlib import Path
|
||
from typing import Any
|
||
|
||
|
||
DEFAULT_POLICY: dict[str, Any] = {
|
||
"schema_version": "v2",
|
||
"interest_keyword_review": {
|
||
"percentile_min": 0.0,
|
||
"percentile_max": 0.05,
|
||
"growth_promotion": 0.5,
|
||
},
|
||
"watch_term_review": {
|
||
"percentile_min": 0.05,
|
||
"percentile_max": 0.20,
|
||
},
|
||
"alias_review": {
|
||
"min_total_count": 2,
|
||
"min_days_seen": 2,
|
||
},
|
||
"stopword_review": {
|
||
"max_total_count": 2,
|
||
"max_days_seen": 2,
|
||
},
|
||
"notes": [
|
||
"v2: interest/watch 使用百分位排名 + 增速因子替代固定阈值",
|
||
"percentile 越小表示排名越高(top 5% = percentile 0.05)",
|
||
"growth = recent_count / total_count,衡量近期活跃度",
|
||
],
|
||
}
|
||
|
||
|
||
def _load_json(path: Path) -> Any:
|
||
return json.loads(path.read_text(encoding="utf-8-sig"))
|
||
|
||
|
||
def _save_json(path: Path, payload: dict[str, Any]) -> None:
|
||
path.parent.mkdir(parents=True, exist_ok=True)
|
||
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
|
||
|
||
|
||
def _load_mapping(path: Path) -> dict[str, str]:
|
||
if not path.exists():
|
||
return {}
|
||
payload = _load_json(path)
|
||
return payload if isinstance(payload, dict) else {}
|
||
|
||
|
||
def _load_list(path: Path) -> list[str]:
|
||
if not path.exists():
|
||
return []
|
||
payload = _load_json(path)
|
||
if isinstance(payload, list):
|
||
return [item for item in payload if isinstance(item, str)]
|
||
return []
|
||
|
||
|
||
def _load_interest_keywords(path: Path) -> list[str]:
|
||
if not path.exists():
|
||
return []
|
||
payload = _load_json(path)
|
||
if not isinstance(payload, dict):
|
||
return []
|
||
values = payload.get("interest_keywords")
|
||
if not isinstance(values, list):
|
||
return []
|
||
return [item for item in values if isinstance(item, str)]
|
||
|
||
|
||
def _casefold_set(values: list[str]) -> set[str]:
|
||
return {value.strip().casefold() for value in values if value.strip()}
|
||
|
||
|
||
def _load_policy(path: Path) -> dict[str, Any]:
|
||
if not path.exists():
|
||
return DEFAULT_POLICY.copy()
|
||
payload = _load_json(path)
|
||
return payload if isinstance(payload, dict) else DEFAULT_POLICY.copy()
|
||
|
||
|
||
def _load_watchlist(path: Path) -> list[dict[str, Any]]:
|
||
if not path.exists():
|
||
return []
|
||
payload = _load_json(path)
|
||
if not isinstance(payload, dict):
|
||
return []
|
||
terms = payload.get("terms")
|
||
if not isinstance(terms, list):
|
||
return []
|
||
return [item for item in terms if isinstance(item, dict) and isinstance(item.get("term"), str)]
|
||
|
||
|
||
def _load_change_log(path: Path) -> list[dict[str, Any]]:
|
||
if not path.exists():
|
||
return []
|
||
payload = _load_json(path)
|
||
if not isinstance(payload, dict):
|
||
return []
|
||
entries = payload.get("entries")
|
||
if not isinstance(entries, list):
|
||
return []
|
||
return [item for item in entries if isinstance(item, dict)]
|
||
|
||
|
||
def _meets_min_thresholds(item: dict[str, Any], thresholds: dict[str, Any]) -> bool:
|
||
min_total_count = int(thresholds.get("min_total_count", 1))
|
||
min_days_seen = int(thresholds.get("min_days_seen", 1))
|
||
total_count = int(item.get("total_count") or 0)
|
||
days_seen = int(item.get("days_seen") or 0)
|
||
return total_count >= min_total_count and days_seen >= min_days_seen
|
||
|
||
|
||
def _within_watch_thresholds(item: dict[str, Any], thresholds: dict[str, Any]) -> bool:
|
||
min_total_count = int(thresholds.get("min_total_count", 1))
|
||
min_days_seen = int(thresholds.get("min_days_seen", 1))
|
||
max_total_count = int(thresholds.get("max_total_count", 999999))
|
||
max_days_seen = int(thresholds.get("max_days_seen", 999999))
|
||
total_count = int(item.get("total_count") or 0)
|
||
days_seen = int(item.get("days_seen") or 0)
|
||
return (
|
||
total_count >= min_total_count
|
||
and days_seen >= min_days_seen
|
||
and total_count <= max_total_count
|
||
and days_seen <= max_days_seen
|
||
)
|
||
|
||
|
||
def _compute_percentile(value: int, sorted_values: list[int]) -> float:
|
||
"""
|
||
Return the percentile rank of `value` in `sorted_values` (ascending).
|
||
0.0 = highest frequency (top rank), 1.0 = lowest frequency (bottom rank).
|
||
"""
|
||
if not sorted_values:
|
||
return 1.0
|
||
# bisect_left — count of values strictly less than `value`
|
||
lo, hi = 0, len(sorted_values)
|
||
while lo < hi:
|
||
mid = (lo + hi) // 2
|
||
if sorted_values[mid] < value:
|
||
lo = mid + 1
|
||
else:
|
||
hi = mid
|
||
rank = lo
|
||
# invert: smallest value → rank=0 → 1.0 (bottom)
|
||
# largest value → rank=len → 0.0 (top)
|
||
return 1.0 - (rank / len(sorted_values))
|
||
|
||
|
||
def _compute_growth(recent_count: int, total_count: int) -> float:
|
||
"""
|
||
Return growth factor: recent_count / total_count.
|
||
Only meaningful when total_count >= 3; returns 0.0 for small counts.
|
||
"""
|
||
if total_count < 3:
|
||
return 0.0
|
||
return recent_count / total_count
|
||
|
||
|
||
def main() -> None:
|
||
parser = argparse.ArgumentParser(
|
||
description="Build a compact review bundle for the keyword-cleanup-review skill."
|
||
)
|
||
repo_root = Path(__file__).resolve().parents[3]
|
||
parser.add_argument(
|
||
"--stats",
|
||
type=Path,
|
||
default=repo_root / "data" / "term_index" / "term_stats.json",
|
||
help="Global keyword stats file",
|
||
)
|
||
parser.add_argument(
|
||
"--daily-dir",
|
||
type=Path,
|
||
default=repo_root / "data" / "term_index" / "daily",
|
||
help="Directory containing daily keyword index JSON files",
|
||
)
|
||
parser.add_argument(
|
||
"--aliases",
|
||
type=Path,
|
||
default=repo_root / "configs" / "term_aliases.json",
|
||
help="Term aliases JSON file",
|
||
)
|
||
parser.add_argument(
|
||
"--stopwords",
|
||
type=Path,
|
||
default=repo_root / "configs" / "term_stopwords.json",
|
||
help="Term stopwords JSON file",
|
||
)
|
||
parser.add_argument(
|
||
"--context",
|
||
type=Path,
|
||
default=repo_root / "configs" / "filter_context.personal.json",
|
||
help="Personal filter context JSON file",
|
||
)
|
||
parser.add_argument(
|
||
"--policy",
|
||
type=Path,
|
||
default=repo_root / "configs" / "term_cleanup_policy.json",
|
||
help="Keyword cleanup policy JSON file",
|
||
)
|
||
parser.add_argument(
|
||
"--watchlist",
|
||
type=Path,
|
||
default=repo_root / "configs" / "term_watchlist.json",
|
||
help="Tracked watch terms JSON file",
|
||
)
|
||
parser.add_argument(
|
||
"--change-log",
|
||
type=Path,
|
||
default=repo_root / "configs" / "term_change_log.json",
|
||
help="Applied keyword cleanup change log JSON file",
|
||
)
|
||
parser.add_argument("--days", type=int, default=7, help="How many recent daily files to include")
|
||
parser.add_argument("--top", type=int, default=50, help="How many top global terms to include")
|
||
parser.add_argument(
|
||
"--output",
|
||
type=Path,
|
||
default=repo_root / "outputs" / "term_index" / "review" / "keyword-cleanup-bundle.json",
|
||
help="Output JSON file",
|
||
)
|
||
args = parser.parse_args()
|
||
|
||
stats_payload = _load_json(args.stats) if args.stats.exists() else {"terms": []}
|
||
stats_terms = stats_payload.get("terms") if isinstance(stats_payload, dict) else []
|
||
if not isinstance(stats_terms, list):
|
||
stats_terms = []
|
||
|
||
aliases = _load_mapping(args.aliases)
|
||
stopwords = _load_list(args.stopwords)
|
||
interest_keywords = _load_interest_keywords(args.context)
|
||
policy = _load_policy(args.policy)
|
||
watchlist = _load_watchlist(args.watchlist)
|
||
change_log = _load_change_log(args.change_log)
|
||
|
||
recent_daily_paths = sorted(args.daily_dir.glob("*.json"))[-args.days :] if args.daily_dir.exists() else []
|
||
recent_daily: list[dict[str, Any]] = []
|
||
recent_counter: defaultdict[str, int] = defaultdict(int)
|
||
for path in recent_daily_paths:
|
||
payload = _load_json(path)
|
||
if not isinstance(payload, dict):
|
||
continue
|
||
terms = payload.get("terms")
|
||
if not isinstance(terms, list):
|
||
terms = []
|
||
compact_terms = []
|
||
for term in terms:
|
||
if not isinstance(term, dict):
|
||
continue
|
||
normalized = term.get("normalized_term")
|
||
count = term.get("count")
|
||
if not isinstance(normalized, str) or not isinstance(count, int):
|
||
continue
|
||
compact_terms.append({"term": normalized, "count": count})
|
||
recent_counter[normalized] += count
|
||
recent_daily.append(
|
||
{
|
||
"date": payload.get("date"),
|
||
"candidate_count": payload.get("candidate_count"),
|
||
"top_terms": compact_terms[:20],
|
||
}
|
||
)
|
||
|
||
interest_set = _casefold_set(interest_keywords)
|
||
stopword_set = _casefold_set(stopwords)
|
||
alias_keys = _casefold_set(list(aliases.keys()))
|
||
alias_values = _casefold_set(list(aliases.values()))
|
||
watch_set = _casefold_set([str(item.get("term", "")) for item in watchlist])
|
||
|
||
# Build a sorted list of all total_counts for percentile computation
|
||
all_total_counts = sorted(
|
||
int(item.get("total_count") or 0)
|
||
for item in stats_terms
|
||
if isinstance(item, dict) and isinstance(item.get("term"), str)
|
||
)
|
||
|
||
top_global_terms = []
|
||
for item in stats_terms[: args.top]:
|
||
if not isinstance(item, dict):
|
||
continue
|
||
term = item.get("term")
|
||
if not isinstance(term, str):
|
||
continue
|
||
folded = term.strip().casefold()
|
||
top_global_terms.append(
|
||
{
|
||
"term": term,
|
||
"total_count": item.get("total_count"),
|
||
"days_seen": item.get("days_seen"),
|
||
"first_seen": item.get("first_seen"),
|
||
"last_seen": item.get("last_seen"),
|
||
"in_interest_keywords": folded in interest_set,
|
||
"is_stopword": folded in stopword_set,
|
||
"is_alias_source": folded in alias_keys,
|
||
"is_alias_target": folded in alias_values,
|
||
"in_watchlist": folded in watch_set,
|
||
"recent_count": recent_counter.get(term, 0),
|
||
"percentile": _compute_percentile(
|
||
int(item.get("total_count") or 0), all_total_counts
|
||
),
|
||
"growth": _compute_growth(
|
||
recent_counter.get(term, 0),
|
||
int(item.get("total_count") or 0),
|
||
),
|
||
}
|
||
)
|
||
|
||
# Keep more uncovered terms for percentile-based selection
|
||
uncovered_terms = [
|
||
item for item in top_global_terms if not item["in_interest_keywords"] and not item["is_stopword"]
|
||
][:100]
|
||
|
||
policy_version = (policy.get("schema_version") if isinstance(policy, dict) else None) or "v1"
|
||
interest_thresholds = policy.get("interest_keyword_review") if isinstance(policy, dict) else {}
|
||
watch_thresholds = policy.get("watch_term_review") if isinstance(policy, dict) else {}
|
||
|
||
if policy_version == "v2" or "percentile_max" in interest_thresholds:
|
||
# v2: percentile + growth based selection
|
||
pct_min_interest = float(interest_thresholds.get("percentile_min", 0.0))
|
||
pct_max_interest = float(interest_thresholds.get("percentile_max", 0.05))
|
||
growth_promo = float(interest_thresholds.get("growth_promotion", 0.5))
|
||
pct_min_watch = float(watch_thresholds.get("percentile_min", 0.05))
|
||
pct_max_watch = float(watch_thresholds.get("percentile_max", 0.20))
|
||
|
||
interest_candidates_raw = [
|
||
item for item in uncovered_terms
|
||
if not item["in_watchlist"]
|
||
and pct_min_interest <= item["percentile"] <= pct_max_interest
|
||
]
|
||
watch_candidates_raw = [
|
||
item for item in uncovered_terms
|
||
if not item["in_watchlist"]
|
||
and pct_min_watch < item["percentile"] <= pct_max_watch
|
||
]
|
||
# Growth boost: terms outside watch range but with strong growth signal
|
||
growth_boost_candidates = [
|
||
item for item in uncovered_terms
|
||
if not item["in_watchlist"]
|
||
and item["percentile"] > pct_max_watch
|
||
and item["growth"] >= growth_promo
|
||
]
|
||
else:
|
||
# v1 fallback: fixed thresholds
|
||
interest_candidates_raw = [
|
||
item for item in uncovered_terms
|
||
if not item["in_watchlist"] and _meets_min_thresholds(item, interest_thresholds)
|
||
]
|
||
watch_candidates_raw = [
|
||
item for item in uncovered_terms
|
||
if not item["in_watchlist"]
|
||
and not _meets_min_thresholds(item, interest_thresholds)
|
||
and _within_watch_thresholds(item, watch_thresholds)
|
||
]
|
||
growth_boost_candidates = []
|
||
|
||
interest_review_candidates = [
|
||
{
|
||
"term": item["term"],
|
||
"total_count": item["total_count"],
|
||
"days_seen": item["days_seen"],
|
||
"percentile": item["percentile"],
|
||
"growth": item["growth"],
|
||
"reason": (
|
||
f"top {item['percentile']:.1%} by frequency,"
|
||
f"growth={item['growth']:.0%},"
|
||
"not yet covered by interest keywords or stopwords."
|
||
),
|
||
}
|
||
for item in interest_candidates_raw
|
||
][:20]
|
||
watch_review_candidates = [
|
||
{
|
||
"term": item["term"],
|
||
"total_count": item["total_count"],
|
||
"days_seen": item["days_seen"],
|
||
"percentile": item["percentile"],
|
||
"growth": item["growth"],
|
||
"reason": (
|
||
f"top {item['percentile']:.1%} by frequency,"
|
||
f"growth={item['growth']:.0%},"
|
||
"fell into watch-review range."
|
||
),
|
||
}
|
||
for item in watch_candidates_raw
|
||
][:20]
|
||
growth_boost_review_items = [
|
||
{
|
||
"term": item["term"],
|
||
"total_count": item["total_count"],
|
||
"days_seen": item["days_seen"],
|
||
"percentile": item["percentile"],
|
||
"growth": item["growth"],
|
||
"reason": (
|
||
f"growth spike: {item['growth']:.0%} of occurrences in recent window "
|
||
f"(total={item['total_count']}, days={item['days_seen']})."
|
||
),
|
||
}
|
||
for item in growth_boost_candidates
|
||
][:5]
|
||
recent_hot_terms = sorted(
|
||
({"term": term, "recent_count": count} for term, count in recent_counter.items()),
|
||
key=lambda item: (-item["recent_count"], item["term"].casefold(), item["term"]),
|
||
)[:20]
|
||
|
||
bundle = {
|
||
"generated_at": datetime.now(tz=UTC).isoformat(),
|
||
"days": args.days,
|
||
"top": args.top,
|
||
"sources": {
|
||
"stats": str(args.stats),
|
||
"daily_dir": str(args.daily_dir),
|
||
"aliases": str(args.aliases),
|
||
"stopwords": str(args.stopwords),
|
||
"context": str(args.context),
|
||
"policy": str(args.policy),
|
||
"watchlist": str(args.watchlist),
|
||
"change_log": str(args.change_log),
|
||
},
|
||
"policy": policy,
|
||
"current_config": {
|
||
"alias_count": len(aliases),
|
||
"stopword_count": len(stopwords),
|
||
"interest_keyword_count": len(interest_keywords),
|
||
"watch_term_count": len(watchlist),
|
||
"aliases": aliases,
|
||
"stopwords": stopwords,
|
||
"interest_keywords": interest_keywords,
|
||
"watchlist": watchlist,
|
||
},
|
||
"change_log_tail": change_log[-10:],
|
||
"top_global_terms": top_global_terms,
|
||
"recent_daily": recent_daily,
|
||
"recent_hot_terms": recent_hot_terms,
|
||
"uncovered_terms": uncovered_terms,
|
||
"governance_hints": {
|
||
"interest_review_candidates": interest_review_candidates,
|
||
"watch_review_candidates": watch_review_candidates,
|
||
"growth_boost_review_items": growth_boost_review_items,
|
||
},
|
||
}
|
||
_save_json(args.output, bundle)
|
||
print(f"Saved keyword cleanup bundle to {args.output}")
|
||
|
||
|
||
if __name__ == "__main__":
|
||
main() |