from __future__ import annotations import argparse import json from datetime import datetime, timezone from pathlib import Path from typing import Any REPO_ROOT = Path(__file__).resolve().parents[1] DEFAULT_BUNDLE_PATH = REPO_ROOT / "outputs" / "term_index" / "review" / "keyword-cleanup-bundle.json" DEFAULT_OUTPUT_DIR = REPO_ROOT / "outputs" / "term_index" / "review" def _load_json(path: Path) -> Any: return json.loads(path.read_text(encoding="utf-8-sig")) def _save_json(path: Path, payload: dict[str, Any]) -> None: path.parent.mkdir(parents=True, exist_ok=True) path.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") def _save_text(path: Path, content: str) -> None: path.parent.mkdir(parents=True, exist_ok=True) path.write_text(content, encoding="utf-8") def _term_key(value: str) -> str: return value.strip().casefold() def _utc_today() -> str: return datetime.now(timezone.utc).date().isoformat() def _require_dict(payload: Any, name: str) -> dict[str, Any]: if not isinstance(payload, dict): raise RuntimeError(f"{name} must be a JSON object.") return payload def _require_list(payload: Any, name: str) -> list[Any]: if not isinstance(payload, list): raise RuntimeError(f"{name} must be a JSON array.") return payload def _bundle_date(bundle: dict[str, Any]) -> str: generated_at = bundle.get("generated_at") if isinstance(generated_at, str) and generated_at.strip(): normalized = generated_at.replace("Z", "+00:00") try: return datetime.fromisoformat(normalized).date().isoformat() except ValueError: pass return _utc_today() def _recent_count_map(top_global_terms: list[dict[str, Any]]) -> dict[str, int]: counts: dict[str, int] = {} for item in top_global_terms: term = item.get("term") recent_count = item.get("recent_count") if isinstance(term, str) and isinstance(recent_count, int): counts[term] = recent_count return counts def _covered_term_sets(bundle: dict[str, Any]) -> tuple[set[str], set[str], set[str]]: current_config = _require_dict(bundle.get("current_config"), "bundle.current_config") interest_keywords = _require_list(current_config.get("interest_keywords"), "bundle.current_config.interest_keywords") stopwords = _require_list(current_config.get("stopwords"), "bundle.current_config.stopwords") watchlist = _require_list(current_config.get("watchlist"), "bundle.current_config.watchlist") interest_set = {_term_key(item) for item in interest_keywords if isinstance(item, str) and item.strip()} stopword_set = {_term_key(item) for item in stopwords if isinstance(item, str) and item.strip()} watch_set = { _term_key(str(item.get("term", ""))) for item in watchlist if isinstance(item, dict) and isinstance(item.get("term"), str) and str(item.get("term", "")).strip() } return interest_set, stopword_set, watch_set def _sort_key(item: dict[str, Any]) -> tuple[int, int, int, str, str]: total_count = int(item.get("total_count") or 0) days_seen = int(item.get("days_seen") or 0) recent_count = int(item.get("recent_count") or 0) term = str(item.get("term") or "") return (-total_count, -days_seen, -recent_count, term.casefold(), term) def _prepare_interest_suggestions(bundle: dict[str, Any]) -> list[dict[str, Any]]: governance_hints = _require_dict(bundle.get("governance_hints"), "bundle.governance_hints") candidates = _require_list( governance_hints.get("interest_review_candidates"), "bundle.governance_hints.interest_review_candidates", ) top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms") recent_counts = _recent_count_map([item for item in top_global_terms if isinstance(item, dict)]) interest_set, stopword_set, watch_set = _covered_term_sets(bundle) suggestions: list[dict[str, Any]] = [] seen: set[str] = set() for item in candidates: if not isinstance(item, dict): continue term = item.get("term") if not isinstance(term, str) or not term.strip(): continue term_key = _term_key(term) if term_key in seen or term_key in interest_set or term_key in stopword_set: continue total_count = int(item.get("total_count") or 0) days_seen = int(item.get("days_seen") or 0) recent_count = recent_counts.get(term, 0) base_reason = str(item.get("reason") or "Meets the configured interest-keyword review threshold.") if term_key in watch_set: base_reason += " It is currently in watchlist and is ready for promotion." reason = f"{base_reason} Evidence: total_count={total_count}, days_seen={days_seen}, recent_count={recent_count}." suggestions.append( { "term": term, "reason": reason, "total_count": total_count, "days_seen": days_seen, "recent_count": recent_count, } ) seen.add(term_key) suggestions.sort(key=_sort_key) return suggestions def _prepare_watch_suggestions(bundle: dict[str, Any], reserved_terms: set[str]) -> list[dict[str, Any]]: governance_hints = _require_dict(bundle.get("governance_hints"), "bundle.governance_hints") candidates = _require_list( governance_hints.get("watch_review_candidates"), "bundle.governance_hints.watch_review_candidates", ) top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms") recent_counts = _recent_count_map([item for item in top_global_terms if isinstance(item, dict)]) interest_set, stopword_set, watch_set = _covered_term_sets(bundle) suggestions: list[dict[str, Any]] = [] seen: set[str] = set(reserved_terms) for item in candidates: if not isinstance(item, dict): continue term = item.get("term") if not isinstance(term, str) or not term.strip(): continue term_key = _term_key(term) if term_key in seen or term_key in interest_set or term_key in stopword_set or term_key in watch_set: continue total_count = int(item.get("total_count") or 0) days_seen = int(item.get("days_seen") or 0) recent_count = recent_counts.get(term, 0) base_reason = str(item.get("reason") or "Falls into the configured watch-term review range.") reason = f"{base_reason} Evidence: total_count={total_count}, days_seen={days_seen}, recent_count={recent_count}." suggestions.append( { "term": term, "reason": reason, "total_count": total_count, "days_seen": days_seen, "recent_count": recent_count, } ) seen.add(term_key) suggestions.sort(key=_sort_key) return suggestions def _prepare_alias_suggestions( bundle: dict[str, Any], all_terms: list[dict[str, Any]] | None = None, ) -> list[dict[str, Any]]: """ Generate alias suggestions using surface-form rules (no LLM). Rules: 1. casefold match — same normalized form, different original casing 2. trailing-s singularization — singular/plural variants 3. whitespace/hyphen normalization — word boundary variants Scans all_terms (full term_stats) if provided; otherwise falls back to top_global_terms from the bundle. """ current_config = _require_dict(bundle.get("current_config"), "bundle.current_config") interest_keywords = _require_list( current_config.get("interest_keywords"), "bundle.current_config.interest_keywords" ) top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms") source_terms = all_terms if all_terms is not None else top_global_terms interest_set = {_term_key(t) for t in interest_keywords if isinstance(t, str)} interest_originals: set[str] = {t for t in interest_keywords if isinstance(t, str)} # Build full casefold → [original forms] map cf_map: dict[str, list[str]] = {} for item in source_terms: term = None if isinstance(item, dict): term = item.get("term") elif isinstance(item, str): term = item if not isinstance(term, str) or not term.strip(): continue key = _term_key(term) if key not in cf_map: cf_map[key] = [] if term not in cf_map[key]: cf_map[key].append(term) suggestions: list[dict[str, Any]] = [] seen_pairs: set[tuple[str, str]] = set() def _add(from_term: str, to_term: str, reason: str) -> None: pair = (_term_key(from_term), _term_key(to_term)) if pair in seen_pairs: return seen_pairs.add(pair) suggestions.append({"from": from_term, "to": to_term, "reason": reason}) # Build a set of all term keys from source for quick lookup source_keys = set(cf_map.keys()) # Rule 1: casefold match — same normalized form, different casing for key, variants in cf_map.items(): if len(variants) < 2: continue canonical = None alt_forms = [] for v in variants: if v in interest_originals: canonical = v else: alt_forms.append(v) if canonical and alt_forms: for alt in alt_forms: _add(alt, canonical, "Case variant") elif len(variants) >= 2 and not canonical: # None is canonical — suggest the highest-frequency form ranked = sorted(variants, key=lambda t: -( next( (it.get("total_count", 0) for it in top_global_terms if it.get("term") == t), 0, ) )) for alt in ranked[1:]: _add(alt, ranked[0], "Case variant (auto-ranked)") # Rule 2: singular/plural — trailing-s normalization # Check all source terms (not just interest keys) for bidirectional matching for key in source_keys: if key in interest_set: continue if key.endswith("s") and len(key) > 2: singular_key = key.rstrip("s") if singular_key in interest_set and singular_key != key: # Find canonical interest keyword canon = next((t for t in interest_keywords if _term_key(t) == singular_key), None) from_form = cf_map[key][0] if canon: _add(from_form, canon, "Plural variant") # singular form → interest has plural plural_key = key + "s" if plural_key in interest_set and plural_key != key: canon = next((t for t in interest_keywords if _term_key(t) == plural_key), None) from_form = cf_map[key][0] if canon: _add(from_form, canon, "Singular variant") # Rule 3: whitespace/hyphen normalization for key in source_keys: if key in interest_set: continue normalized = key.replace("-", "").replace("_", "").replace(" ", "") if normalized in interest_set and normalized != key: canon = next((t for t in interest_keywords if _term_key(t) == normalized), None) from_form = cf_map[key][0] if canon: _add(from_form, canon, "Whitespace/punctuation variant") suggestions.sort(key=lambda x: (x["from"].casefold(), x["to"].casefold())) return suggestions def _render_table(items: list[dict[str, Any]]) -> str: if not items: return "_None in this pass._\n" lines = [ "| Term | Total | Days | Recent | Reason |", "| --- | ---: | ---: | ---: | --- |", ] for item in items: term = str(item.get("term") or "") total_count = int(item.get("total_count") or 0) days_seen = int(item.get("days_seen") or 0) recent_count = int(item.get("recent_count") or 0) reason = str(item.get("reason") or "").replace("|", "\\|") lines.append(f"| {term} | {total_count} | {days_seen} | {recent_count} | {reason} |") return "\n".join(lines) + "\n" def _render_simple_table(items: list[dict[str, Any]], first_column: str) -> str: if not items: return "_None in this pass._\n" lines = [ f"| {first_column} | Reason |", "| --- | --- |", ] for item in items: value = str(item.get(first_column.casefold()) or item.get(first_column) or "") reason = str(item.get("reason") or "").replace("|", "\\|") lines.append(f"| {value} | {reason} |") return "\n".join(lines) + "\n" def _render_markdown( *, suggestion_date: str, bundle_path: Path, json_output_path: Path, bundle: dict[str, Any], suggestions: dict[str, Any], ) -> str: policy = _require_dict(bundle.get("policy"), "bundle.policy") current_config = _require_dict(bundle.get("current_config"), "bundle.current_config") top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms") uncovered_terms = _require_list(bundle.get("uncovered_terms"), "bundle.uncovered_terms") top_preview = [item for item in top_global_terms if isinstance(item, dict)][:5] uncovered_preview = [item for item in uncovered_terms if isinstance(item, dict)][:5] interest_items = suggestions["interest_keyword_suggestions"] watch_items = suggestions["watch_terms"] alias_items = suggestions["alias_suggestions"] stopword_items = suggestions["stopword_suggestions"] lines = [ f"# Term Cleanup Suggestions - {suggestion_date}", "", "## Review Context", "", f"- Source bundle: `{bundle_path}`", f"- Suggestions JSON: `{json_output_path}`", f"- Bundle generated_at: `{bundle.get('generated_at', 'unknown')}`", f"- Based on days: `{suggestions['based_on_days']}`", f"- Policy schema version: `{policy.get('schema_version', 'unknown')}`", "- Scope: implement `interest_keyword_suggestions` and `watch_terms` main path first; keep alias/stopword conservative in this pass.", "", "## Current State", "", f"- Interest keywords: `{current_config.get('interest_keyword_count', 0)}`", f"- Watch terms: `{current_config.get('watch_term_count', 0)}`", f"- Stopwords: `{current_config.get('stopword_count', 0)}`", f"- Aliases: `{current_config.get('alias_count', 0)}`", f"- Top global terms considered: `{len(top_global_terms)}`", f"- Uncovered terms considered: `{len(uncovered_terms)}`", "", "### Top Terms Snapshot", "", ] if top_preview: for item in top_preview: lines.append( f"- `{item.get('term', '')}`: total_count={item.get('total_count', 0)}, days_seen={item.get('days_seen', 0)}, recent_count={item.get('recent_count', 0)}" ) else: lines.append("- No top terms available.") lines.extend([ "", "### Uncovered Terms Snapshot", "", ]) if uncovered_preview: for item in uncovered_preview: lines.append( f"- `{item.get('term', '')}`: total_count={item.get('total_count', 0)}, days_seen={item.get('days_seen', 0)}, recent_count={item.get('recent_count', 0)}" ) else: lines.append("- No uncovered terms available.") lines.extend([ "", "## Suggestion Summary", "", f"- `interest_keyword_suggestions`: `{len(interest_items)}`", f"- `watch_terms`: `{len(watch_items)}`", f"- `alias_suggestions`: `{len(alias_items)}`", f"- `stopword_suggestions`: `{len(stopword_items)}`", "", "## Interest Keyword Suggestions", "", _render_table(interest_items).rstrip(), "", "## Watch Terms", "", _render_table(watch_items).rstrip(), "", "## Alias Suggestions", "", "_Conservative by design in this minimal version; no automatic alias suggestions are emitted yet._" if not alias_items else _render_simple_table(alias_items, "from").rstrip(), "", "## Stopword Suggestions", "", "_Conservative by design in this minimal version; no automatic stopword suggestions are emitted yet._" if not stopword_items else _render_simple_table(stopword_items, "term").rstrip(), "", "## Apply", "", "Review the Markdown first, then selectively apply accepted suggestions with the JSON file.", "", "```bash", f"python scripts/apply_term_suggestions.py \\", f" --suggestions {json_output_path} \\", " --accept-interest \"Claude Code\" \\", " --accept-watch \"A2A\" \\", " --dry-run", "```", "", ]) return "\n".join(lines) def _build_output_paths( *, output_dir: Path, suggestion_date: str, json_output: Path | None, markdown_output: Path | None, ) -> tuple[Path, Path]: stem = f"term-cleanup-suggestions-{suggestion_date}" resolved_json = json_output or (output_dir / f"{stem}.json") resolved_markdown = markdown_output or (output_dir / f"{stem}.md") return resolved_json, resolved_markdown def main() -> None: parser = argparse.ArgumentParser(description="Generate term cleanup suggestions JSON and Markdown from review bundle.") parser.add_argument("--bundle", type=Path, default=DEFAULT_BUNDLE_PATH, help="Review bundle JSON file") parser.add_argument( "--output-dir", type=Path, default=DEFAULT_OUTPUT_DIR, help="Directory for generated suggestions outputs when explicit output paths are not provided", ) parser.add_argument("--date", type=str, default=None, help="Override suggestions date (YYYY-MM-DD)") parser.add_argument("--json-output", type=Path, default=None, help="Explicit suggestions JSON output path") parser.add_argument("--markdown-output", type=Path, default=None, help="Explicit suggestions Markdown output path") parser.add_argument( "--emit-markdown", action="store_true", help="Also write the human-readable Markdown review draft. JSON suggestions are always written.", ) args = parser.parse_args() if not args.bundle.exists(): raise RuntimeError(f"Bundle file not found: {args.bundle}") bundle = _require_dict(_load_json(args.bundle), "bundle") days = bundle.get("days") if not isinstance(days, int): raise RuntimeError("bundle.days must be an integer.") suggestion_date = args.date or _bundle_date(bundle) json_output_path, markdown_output_path = _build_output_paths( output_dir=args.output_dir, suggestion_date=suggestion_date, json_output=args.json_output, markdown_output=args.markdown_output, ) # Load full term_stats for alias scanning (bundle only has top N) stats_path = REPO_ROOT / "data" / "term_index" / "term_stats.json" all_stats_terms: list[str] = [] if stats_path.exists(): stats_payload = _load_json(stats_path) raw_terms = stats_payload.get("terms") if isinstance(stats_payload, dict) else [] if isinstance(raw_terms, list): all_stats_terms = [str(t["term"]) for t in raw_terms if isinstance(t, dict) and isinstance(t.get("term"), str)] interest_items = _prepare_interest_suggestions(bundle) reserved_terms = {_term_key(str(item.get("term") or "")) for item in interest_items} watch_items = _prepare_watch_suggestions(bundle, reserved_terms=reserved_terms) alias_items = _prepare_alias_suggestions(bundle, all_terms=all_stats_terms) suggestions = { "date": suggestion_date, "based_on_days": days, "source_bundle": str(args.bundle), "policy_schema_version": _require_dict(bundle.get("policy"), "bundle.policy").get("schema_version", "unknown"), "summary": { "interest_keyword_suggestions": len(interest_items), "watch_terms": len(watch_items), "alias_suggestions": len(alias_items), "stopword_suggestions": 0, }, "alias_suggestions": alias_items, "stopword_suggestions": [], "interest_keyword_suggestions": interest_items, "watch_terms": watch_items, } markdown = _render_markdown( suggestion_date=suggestion_date, bundle_path=args.bundle, json_output_path=json_output_path, bundle=bundle, suggestions=suggestions, ) _save_json(json_output_path, suggestions) if args.emit_markdown: _save_text(markdown_output_path, markdown) summary = { "bundle": str(args.bundle), "date": suggestion_date, "json_output": str(json_output_path), "markdown_output": str(markdown_output_path) if args.emit_markdown else None, "interest_keyword_suggestions": len(interest_items), "watch_terms": len(watch_items), "alias_suggestions": len(alias_items), "stopword_suggestions": 0, "emit_markdown": args.emit_markdown, } print(json.dumps(summary, ensure_ascii=False, indent=2)) if __name__ == "__main__": main()