from __future__ import annotations import argparse import json from datetime import datetime, timezone from pathlib import Path from typing import Any REPO_ROOT = Path(__file__).resolve().parents[1] DEFAULT_BUNDLE_PATH = REPO_ROOT / "outputs" / "term_index" / "review" / "keyword-cleanup-bundle.json" DEFAULT_OUTPUT_DIR = REPO_ROOT / "outputs" / "term_index" / "review" def _load_json(path: Path) -> Any: return json.loads(path.read_text(encoding="utf-8-sig")) def _save_json(path: Path, payload: dict[str, Any]) -> None: path.parent.mkdir(parents=True, exist_ok=True) path.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") def _save_text(path: Path, content: str) -> None: path.parent.mkdir(parents=True, exist_ok=True) path.write_text(content, encoding="utf-8") def _term_key(value: str) -> str: return value.strip().casefold() def _utc_today() -> str: return datetime.now(timezone.utc).date().isoformat() def _require_dict(payload: Any, name: str) -> dict[str, Any]: if not isinstance(payload, dict): raise RuntimeError(f"{name} must be a JSON object.") return payload def _require_list(payload: Any, name: str) -> list[Any]: if not isinstance(payload, list): raise RuntimeError(f"{name} must be a JSON array.") return payload def _bundle_date(bundle: dict[str, Any]) -> str: generated_at = bundle.get("generated_at") if isinstance(generated_at, str) and generated_at.strip(): normalized = generated_at.replace("Z", "+00:00") try: return datetime.fromisoformat(normalized).date().isoformat() except ValueError: pass return _utc_today() def _recent_count_map(top_global_terms: list[dict[str, Any]]) -> dict[str, int]: counts: dict[str, int] = {} for item in top_global_terms: term = item.get("term") recent_count = item.get("recent_count") if isinstance(term, str) and isinstance(recent_count, int): counts[term] = recent_count return counts def _covered_term_sets(bundle: dict[str, Any]) -> tuple[set[str], set[str], set[str]]: current_config = _require_dict(bundle.get("current_config"), "bundle.current_config") interest_keywords = _require_list(current_config.get("interest_keywords"), "bundle.current_config.interest_keywords") stopwords = _require_list(current_config.get("stopwords"), "bundle.current_config.stopwords") watchlist = _require_list(current_config.get("watchlist"), "bundle.current_config.watchlist") interest_set = {_term_key(item) for item in interest_keywords if isinstance(item, str) and item.strip()} stopword_set = {_term_key(item) for item in stopwords if isinstance(item, str) and item.strip()} watch_set = { _term_key(str(item.get("term", ""))) for item in watchlist if isinstance(item, dict) and isinstance(item.get("term"), str) and str(item.get("term", "")).strip() } return interest_set, stopword_set, watch_set def _sort_key(item: dict[str, Any]) -> tuple[int, int, int, str, str]: total_count = int(item.get("total_count") or 0) days_seen = int(item.get("days_seen") or 0) recent_count = int(item.get("recent_count") or 0) term = str(item.get("term") or "") return (-total_count, -days_seen, -recent_count, term.casefold(), term) def _prepare_interest_suggestions(bundle: dict[str, Any]) -> list[dict[str, Any]]: governance_hints = _require_dict(bundle.get("governance_hints"), "bundle.governance_hints") candidates = _require_list( governance_hints.get("interest_review_candidates"), "bundle.governance_hints.interest_review_candidates", ) top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms") recent_counts = _recent_count_map([item for item in top_global_terms if isinstance(item, dict)]) interest_set, stopword_set, watch_set = _covered_term_sets(bundle) suggestions: list[dict[str, Any]] = [] seen: set[str] = set() for item in candidates: if not isinstance(item, dict): continue term = item.get("term") if not isinstance(term, str) or not term.strip(): continue term_key = _term_key(term) if term_key in seen or term_key in interest_set or term_key in stopword_set: continue total_count = int(item.get("total_count") or 0) days_seen = int(item.get("days_seen") or 0) recent_count = recent_counts.get(term, 0) base_reason = str(item.get("reason") or "Meets the configured interest-keyword review threshold.") if term_key in watch_set: base_reason += " It is currently in watchlist and is ready for promotion." reason = f"{base_reason} Evidence: total_count={total_count}, days_seen={days_seen}, recent_count={recent_count}." suggestions.append( { "term": term, "reason": reason, "total_count": total_count, "days_seen": days_seen, "recent_count": recent_count, } ) seen.add(term_key) suggestions.sort(key=_sort_key) return suggestions def _prepare_watch_suggestions(bundle: dict[str, Any], reserved_terms: set[str]) -> list[dict[str, Any]]: governance_hints = _require_dict(bundle.get("governance_hints"), "bundle.governance_hints") candidates = _require_list( governance_hints.get("watch_review_candidates"), "bundle.governance_hints.watch_review_candidates", ) top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms") recent_counts = _recent_count_map([item for item in top_global_terms if isinstance(item, dict)]) interest_set, stopword_set, watch_set = _covered_term_sets(bundle) suggestions: list[dict[str, Any]] = [] seen: set[str] = set(reserved_terms) for item in candidates: if not isinstance(item, dict): continue term = item.get("term") if not isinstance(term, str) or not term.strip(): continue term_key = _term_key(term) if term_key in seen or term_key in interest_set or term_key in stopword_set or term_key in watch_set: continue total_count = int(item.get("total_count") or 0) days_seen = int(item.get("days_seen") or 0) recent_count = recent_counts.get(term, 0) base_reason = str(item.get("reason") or "Falls into the configured watch-term review range.") reason = f"{base_reason} Evidence: total_count={total_count}, days_seen={days_seen}, recent_count={recent_count}." suggestions.append( { "term": term, "reason": reason, "total_count": total_count, "days_seen": days_seen, "recent_count": recent_count, } ) seen.add(term_key) suggestions.sort(key=_sort_key) return suggestions def _render_table(items: list[dict[str, Any]]) -> str: if not items: return "_None in this pass._\n" lines = [ "| Term | Total | Days | Recent | Reason |", "| --- | ---: | ---: | ---: | --- |", ] for item in items: term = str(item.get("term") or "") total_count = int(item.get("total_count") or 0) days_seen = int(item.get("days_seen") or 0) recent_count = int(item.get("recent_count") or 0) reason = str(item.get("reason") or "").replace("|", "\\|") lines.append(f"| {term} | {total_count} | {days_seen} | {recent_count} | {reason} |") return "\n".join(lines) + "\n" def _render_simple_table(items: list[dict[str, Any]], first_column: str) -> str: if not items: return "_None in this pass._\n" lines = [ f"| {first_column} | Reason |", "| --- | --- |", ] for item in items: value = str(item.get(first_column.casefold()) or item.get(first_column) or "") reason = str(item.get("reason") or "").replace("|", "\\|") lines.append(f"| {value} | {reason} |") return "\n".join(lines) + "\n" def _render_markdown( *, suggestion_date: str, bundle_path: Path, json_output_path: Path, bundle: dict[str, Any], suggestions: dict[str, Any], ) -> str: policy = _require_dict(bundle.get("policy"), "bundle.policy") current_config = _require_dict(bundle.get("current_config"), "bundle.current_config") top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms") uncovered_terms = _require_list(bundle.get("uncovered_terms"), "bundle.uncovered_terms") top_preview = [item for item in top_global_terms if isinstance(item, dict)][:5] uncovered_preview = [item for item in uncovered_terms if isinstance(item, dict)][:5] interest_items = suggestions["interest_keyword_suggestions"] watch_items = suggestions["watch_terms"] alias_items = suggestions["alias_suggestions"] stopword_items = suggestions["stopword_suggestions"] lines = [ f"# Term Cleanup Suggestions - {suggestion_date}", "", "## Review Context", "", f"- Source bundle: `{bundle_path}`", f"- Suggestions JSON: `{json_output_path}`", f"- Bundle generated_at: `{bundle.get('generated_at', 'unknown')}`", f"- Based on days: `{suggestions['based_on_days']}`", f"- Policy schema version: `{policy.get('schema_version', 'unknown')}`", "- Scope: implement `interest_keyword_suggestions` and `watch_terms` main path first; keep alias/stopword conservative in this pass.", "", "## Current State", "", f"- Interest keywords: `{current_config.get('interest_keyword_count', 0)}`", f"- Watch terms: `{current_config.get('watch_term_count', 0)}`", f"- Stopwords: `{current_config.get('stopword_count', 0)}`", f"- Aliases: `{current_config.get('alias_count', 0)}`", f"- Top global terms considered: `{len(top_global_terms)}`", f"- Uncovered terms considered: `{len(uncovered_terms)}`", "", "### Top Terms Snapshot", "", ] if top_preview: for item in top_preview: lines.append( f"- `{item.get('term', '')}`: total_count={item.get('total_count', 0)}, days_seen={item.get('days_seen', 0)}, recent_count={item.get('recent_count', 0)}" ) else: lines.append("- No top terms available.") lines.extend([ "", "### Uncovered Terms Snapshot", "", ]) if uncovered_preview: for item in uncovered_preview: lines.append( f"- `{item.get('term', '')}`: total_count={item.get('total_count', 0)}, days_seen={item.get('days_seen', 0)}, recent_count={item.get('recent_count', 0)}" ) else: lines.append("- No uncovered terms available.") lines.extend([ "", "## Suggestion Summary", "", f"- `interest_keyword_suggestions`: `{len(interest_items)}`", f"- `watch_terms`: `{len(watch_items)}`", f"- `alias_suggestions`: `{len(alias_items)}`", f"- `stopword_suggestions`: `{len(stopword_items)}`", "", "## Interest Keyword Suggestions", "", _render_table(interest_items).rstrip(), "", "## Watch Terms", "", _render_table(watch_items).rstrip(), "", "## Alias Suggestions", "", "_Conservative by design in this minimal version; no automatic alias suggestions are emitted yet._" if not alias_items else _render_simple_table(alias_items, "from").rstrip(), "", "## Stopword Suggestions", "", "_Conservative by design in this minimal version; no automatic stopword suggestions are emitted yet._" if not stopword_items else _render_simple_table(stopword_items, "term").rstrip(), "", "## Apply", "", "Review the Markdown first, then selectively apply accepted suggestions with the JSON file.", "", "```bash", f"python scripts/apply_term_suggestions.py \\", f" --suggestions {json_output_path} \\", " --accept-interest \"Claude Code\" \\", " --accept-watch \"A2A\" \\", " --dry-run", "```", "", ]) return "\n".join(lines) def _build_output_paths( *, output_dir: Path, suggestion_date: str, json_output: Path | None, markdown_output: Path | None, ) -> tuple[Path, Path]: stem = f"term-cleanup-suggestions-{suggestion_date}" resolved_json = json_output or (output_dir / f"{stem}.json") resolved_markdown = markdown_output or (output_dir / f"{stem}.md") return resolved_json, resolved_markdown def main() -> None: parser = argparse.ArgumentParser(description="Generate term cleanup suggestions JSON and Markdown from review bundle.") parser.add_argument("--bundle", type=Path, default=DEFAULT_BUNDLE_PATH, help="Review bundle JSON file") parser.add_argument( "--output-dir", type=Path, default=DEFAULT_OUTPUT_DIR, help="Directory for generated suggestions outputs when explicit output paths are not provided", ) parser.add_argument("--date", type=str, default=None, help="Override suggestions date (YYYY-MM-DD)") parser.add_argument("--json-output", type=Path, default=None, help="Explicit suggestions JSON output path") parser.add_argument("--markdown-output", type=Path, default=None, help="Explicit suggestions Markdown output path") parser.add_argument( "--emit-markdown", action="store_true", help="Also write the human-readable Markdown review draft. JSON suggestions are always written.", ) args = parser.parse_args() if not args.bundle.exists(): raise RuntimeError(f"Bundle file not found: {args.bundle}") bundle = _require_dict(_load_json(args.bundle), "bundle") days = bundle.get("days") if not isinstance(days, int): raise RuntimeError("bundle.days must be an integer.") suggestion_date = args.date or _bundle_date(bundle) json_output_path, markdown_output_path = _build_output_paths( output_dir=args.output_dir, suggestion_date=suggestion_date, json_output=args.json_output, markdown_output=args.markdown_output, ) interest_items = _prepare_interest_suggestions(bundle) reserved_terms = {_term_key(str(item.get("term") or "")) for item in interest_items} watch_items = _prepare_watch_suggestions(bundle, reserved_terms=reserved_terms) suggestions = { "date": suggestion_date, "based_on_days": days, "source_bundle": str(args.bundle), "policy_schema_version": _require_dict(bundle.get("policy"), "bundle.policy").get("schema_version", "unknown"), "summary": { "interest_keyword_suggestions": len(interest_items), "watch_terms": len(watch_items), "alias_suggestions": 0, "stopword_suggestions": 0, }, "alias_suggestions": [], "stopword_suggestions": [], "interest_keyword_suggestions": interest_items, "watch_terms": watch_items, } markdown = _render_markdown( suggestion_date=suggestion_date, bundle_path=args.bundle, json_output_path=json_output_path, bundle=bundle, suggestions=suggestions, ) _save_json(json_output_path, suggestions) if args.emit_markdown: _save_text(markdown_output_path, markdown) summary = { "bundle": str(args.bundle), "date": suggestion_date, "json_output": str(json_output_path), "markdown_output": str(markdown_output_path) if args.emit_markdown else None, "interest_keyword_suggestions": len(interest_items), "watch_terms": len(watch_items), "alias_suggestions": 0, "stopword_suggestions": 0, "emit_markdown": args.emit_markdown, } print(json.dumps(summary, ensure_ascii=False, indent=2)) if __name__ == "__main__": main()