from __future__ import annotations import argparse import json from collections import defaultdict from datetime import UTC, datetime from pathlib import Path from typing import Any DEFAULT_POLICY: dict[str, Any] = { "schema_version": "v2", "interest_keyword_review": { "percentile_min": 0.0, "percentile_max": 0.05, "growth_promotion": 0.5, }, "watch_term_review": { "percentile_min": 0.05, "percentile_max": 0.20, }, "alias_review": { "min_total_count": 2, "min_days_seen": 2, }, "stopword_review": { "max_total_count": 2, "max_days_seen": 2, }, "notes": [ "v2: interest/watch 使用百分位排名 + 增速因子替代固定阈值", "percentile 越小表示排名越高(top 5% = percentile 0.05)", "growth = recent_count / total_count,衡量近期活跃度", ], } def _load_json(path: Path) -> Any: return json.loads(path.read_text(encoding="utf-8-sig")) def _save_json(path: Path, payload: dict[str, Any]) -> None: path.parent.mkdir(parents=True, exist_ok=True) path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8") def _load_mapping(path: Path) -> dict[str, str]: if not path.exists(): return {} payload = _load_json(path) return payload if isinstance(payload, dict) else {} def _load_list(path: Path) -> list[str]: if not path.exists(): return [] payload = _load_json(path) if isinstance(payload, list): return [item for item in payload if isinstance(item, str)] return [] def _load_interest_keywords(path: Path) -> list[str]: if not path.exists(): return [] payload = _load_json(path) if not isinstance(payload, dict): return [] values = payload.get("interest_keywords") if not isinstance(values, list): return [] return [item for item in values if isinstance(item, str)] def _casefold_set(values: list[str]) -> set[str]: return {value.strip().casefold() for value in values if value.strip()} def _load_policy(path: Path) -> dict[str, Any]: if not path.exists(): return DEFAULT_POLICY.copy() payload = _load_json(path) return payload if isinstance(payload, dict) else DEFAULT_POLICY.copy() def _load_watchlist(path: Path) -> list[dict[str, Any]]: if not path.exists(): return [] payload = _load_json(path) if not isinstance(payload, dict): return [] terms = payload.get("terms") if not isinstance(terms, list): return [] return [item for item in terms if isinstance(item, dict) and isinstance(item.get("term"), str)] def _load_change_log(path: Path) -> list[dict[str, Any]]: if not path.exists(): return [] payload = _load_json(path) if not isinstance(payload, dict): return [] entries = payload.get("entries") if not isinstance(entries, list): return [] return [item for item in entries if isinstance(item, dict)] def _meets_min_thresholds(item: dict[str, Any], thresholds: dict[str, Any]) -> bool: min_total_count = int(thresholds.get("min_total_count", 1)) min_days_seen = int(thresholds.get("min_days_seen", 1)) total_count = int(item.get("total_count") or 0) days_seen = int(item.get("days_seen") or 0) return total_count >= min_total_count and days_seen >= min_days_seen def _within_watch_thresholds(item: dict[str, Any], thresholds: dict[str, Any]) -> bool: min_total_count = int(thresholds.get("min_total_count", 1)) min_days_seen = int(thresholds.get("min_days_seen", 1)) max_total_count = int(thresholds.get("max_total_count", 999999)) max_days_seen = int(thresholds.get("max_days_seen", 999999)) total_count = int(item.get("total_count") or 0) days_seen = int(item.get("days_seen") or 0) return ( total_count >= min_total_count and days_seen >= min_days_seen and total_count <= max_total_count and days_seen <= max_days_seen ) def _compute_percentile(value: int, sorted_values: list[int]) -> float: """ Return the percentile rank of `value` in `sorted_values` (ascending). 0.0 = highest frequency (top rank), 1.0 = lowest frequency (bottom rank). """ if not sorted_values: return 1.0 # bisect_left — count of values strictly less than `value` lo, hi = 0, len(sorted_values) while lo < hi: mid = (lo + hi) // 2 if sorted_values[mid] < value: lo = mid + 1 else: hi = mid rank = lo # invert: smallest value → rank=0 → 1.0 (bottom) # largest value → rank=len → 0.0 (top) return 1.0 - (rank / len(sorted_values)) def _compute_growth(recent_count: int, total_count: int) -> float: """ Return growth factor: recent_count / total_count. Only meaningful when total_count >= 3; returns 0.0 for small counts. """ if total_count < 3: return 0.0 return recent_count / total_count def main() -> None: parser = argparse.ArgumentParser( description="Build a compact review bundle for the keyword-cleanup-review skill." ) repo_root = Path(__file__).resolve().parents[3] parser.add_argument( "--stats", type=Path, default=repo_root / "data" / "term_index" / "term_stats.json", help="Global keyword stats file", ) parser.add_argument( "--daily-dir", type=Path, default=repo_root / "data" / "term_index" / "daily", help="Directory containing daily keyword index JSON files", ) parser.add_argument( "--aliases", type=Path, default=repo_root / "configs" / "term_aliases.json", help="Term aliases JSON file", ) parser.add_argument( "--stopwords", type=Path, default=repo_root / "configs" / "term_stopwords.json", help="Term stopwords JSON file", ) parser.add_argument( "--context", type=Path, default=repo_root / "configs" / "filter_context.personal.json", help="Personal filter context JSON file", ) parser.add_argument( "--policy", type=Path, default=repo_root / "configs" / "term_cleanup_policy.json", help="Keyword cleanup policy JSON file", ) parser.add_argument( "--watchlist", type=Path, default=repo_root / "configs" / "term_watchlist.json", help="Tracked watch terms JSON file", ) parser.add_argument( "--change-log", type=Path, default=repo_root / "configs" / "term_change_log.json", help="Applied keyword cleanup change log JSON file", ) parser.add_argument("--days", type=int, default=7, help="How many recent daily files to include") parser.add_argument("--top", type=int, default=50, help="How many top global terms to include") parser.add_argument( "--output", type=Path, default=repo_root / "outputs" / "term_index" / "review" / "keyword-cleanup-bundle.json", help="Output JSON file", ) args = parser.parse_args() stats_payload = _load_json(args.stats) if args.stats.exists() else {"terms": []} stats_terms = stats_payload.get("terms") if isinstance(stats_payload, dict) else [] if not isinstance(stats_terms, list): stats_terms = [] aliases = _load_mapping(args.aliases) stopwords = _load_list(args.stopwords) interest_keywords = _load_interest_keywords(args.context) policy = _load_policy(args.policy) watchlist = _load_watchlist(args.watchlist) change_log = _load_change_log(args.change_log) recent_daily_paths = sorted(args.daily_dir.glob("*.json"))[-args.days :] if args.daily_dir.exists() else [] recent_daily: list[dict[str, Any]] = [] recent_counter: defaultdict[str, int] = defaultdict(int) for path in recent_daily_paths: payload = _load_json(path) if not isinstance(payload, dict): continue terms = payload.get("terms") if not isinstance(terms, list): terms = [] compact_terms = [] for term in terms: if not isinstance(term, dict): continue normalized = term.get("normalized_term") count = term.get("count") if not isinstance(normalized, str) or not isinstance(count, int): continue compact_terms.append({"term": normalized, "count": count}) recent_counter[normalized] += count recent_daily.append( { "date": payload.get("date"), "candidate_count": payload.get("candidate_count"), "top_terms": compact_terms[:20], } ) interest_set = _casefold_set(interest_keywords) stopword_set = _casefold_set(stopwords) alias_keys = _casefold_set(list(aliases.keys())) alias_values = _casefold_set(list(aliases.values())) watch_set = _casefold_set([str(item.get("term", "")) for item in watchlist]) # Build a sorted list of all total_counts for percentile computation all_total_counts = sorted( int(item.get("total_count") or 0) for item in stats_terms if isinstance(item, dict) and isinstance(item.get("term"), str) ) top_global_terms = [] for item in stats_terms[: args.top]: if not isinstance(item, dict): continue term = item.get("term") if not isinstance(term, str): continue folded = term.strip().casefold() top_global_terms.append( { "term": term, "total_count": item.get("total_count"), "days_seen": item.get("days_seen"), "first_seen": item.get("first_seen"), "last_seen": item.get("last_seen"), "in_interest_keywords": folded in interest_set, "is_stopword": folded in stopword_set, "is_alias_source": folded in alias_keys, "is_alias_target": folded in alias_values, "in_watchlist": folded in watch_set, "recent_count": recent_counter.get(term, 0), "percentile": _compute_percentile( int(item.get("total_count") or 0), all_total_counts ), "growth": _compute_growth( recent_counter.get(term, 0), int(item.get("total_count") or 0), ), } ) # Keep more uncovered terms for percentile-based selection uncovered_terms = [ item for item in top_global_terms if not item["in_interest_keywords"] and not item["is_stopword"] ][:100] policy_version = (policy.get("schema_version") if isinstance(policy, dict) else None) or "v1" interest_thresholds = policy.get("interest_keyword_review") if isinstance(policy, dict) else {} watch_thresholds = policy.get("watch_term_review") if isinstance(policy, dict) else {} if policy_version == "v2" or "percentile_max" in interest_thresholds: # v2: percentile + growth based selection pct_min_interest = float(interest_thresholds.get("percentile_min", 0.0)) pct_max_interest = float(interest_thresholds.get("percentile_max", 0.05)) growth_promo = float(interest_thresholds.get("growth_promotion", 0.5)) pct_min_watch = float(watch_thresholds.get("percentile_min", 0.05)) pct_max_watch = float(watch_thresholds.get("percentile_max", 0.20)) interest_candidates_raw = [ item for item in uncovered_terms if not item["in_watchlist"] and pct_min_interest <= item["percentile"] <= pct_max_interest ] watch_candidates_raw = [ item for item in uncovered_terms if not item["in_watchlist"] and pct_min_watch < item["percentile"] <= pct_max_watch ] # Growth boost: terms outside watch range but with strong growth signal growth_boost_candidates = [ item for item in uncovered_terms if not item["in_watchlist"] and item["percentile"] > pct_max_watch and item["growth"] >= growth_promo ] else: # v1 fallback: fixed thresholds interest_candidates_raw = [ item for item in uncovered_terms if not item["in_watchlist"] and _meets_min_thresholds(item, interest_thresholds) ] watch_candidates_raw = [ item for item in uncovered_terms if not item["in_watchlist"] and not _meets_min_thresholds(item, interest_thresholds) and _within_watch_thresholds(item, watch_thresholds) ] growth_boost_candidates = [] interest_review_candidates = [ { "term": item["term"], "total_count": item["total_count"], "days_seen": item["days_seen"], "percentile": item["percentile"], "growth": item["growth"], "reason": ( f"top {item['percentile']:.1%} by frequency," f"growth={item['growth']:.0%}," "not yet covered by interest keywords or stopwords." ), } for item in interest_candidates_raw ][:20] watch_review_candidates = [ { "term": item["term"], "total_count": item["total_count"], "days_seen": item["days_seen"], "percentile": item["percentile"], "growth": item["growth"], "reason": ( f"top {item['percentile']:.1%} by frequency," f"growth={item['growth']:.0%}," "fell into watch-review range." ), } for item in watch_candidates_raw ][:20] growth_boost_review_items = [ { "term": item["term"], "total_count": item["total_count"], "days_seen": item["days_seen"], "percentile": item["percentile"], "growth": item["growth"], "reason": ( f"growth spike: {item['growth']:.0%} of occurrences in recent window " f"(total={item['total_count']}, days={item['days_seen']})." ), } for item in growth_boost_candidates ][:5] recent_hot_terms = sorted( ({"term": term, "recent_count": count} for term, count in recent_counter.items()), key=lambda item: (-item["recent_count"], item["term"].casefold(), item["term"]), )[:20] bundle = { "generated_at": datetime.now(tz=UTC).isoformat(), "days": args.days, "top": args.top, "sources": { "stats": str(args.stats), "daily_dir": str(args.daily_dir), "aliases": str(args.aliases), "stopwords": str(args.stopwords), "context": str(args.context), "policy": str(args.policy), "watchlist": str(args.watchlist), "change_log": str(args.change_log), }, "policy": policy, "current_config": { "alias_count": len(aliases), "stopword_count": len(stopwords), "interest_keyword_count": len(interest_keywords), "watch_term_count": len(watchlist), "aliases": aliases, "stopwords": stopwords, "interest_keywords": interest_keywords, "watchlist": watchlist, }, "change_log_tail": change_log[-10:], "top_global_terms": top_global_terms, "recent_daily": recent_daily, "recent_hot_terms": recent_hot_terms, "uncovered_terms": uncovered_terms, "governance_hints": { "interest_review_candidates": interest_review_candidates, "watch_review_candidates": watch_review_candidates, "growth_boost_review_items": growth_boost_review_items, }, } _save_json(args.output, bundle) print(f"Saved keyword cleanup bundle to {args.output}") if __name__ == "__main__": main()