412 lines
16 KiB
Python
412 lines
16 KiB
Python
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
from datetime import datetime, timezone
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
|
|
REPO_ROOT = Path(__file__).resolve().parents[1]
|
|
DEFAULT_BUNDLE_PATH = REPO_ROOT / "outputs" / "term_index" / "review" / "keyword-cleanup-bundle.json"
|
|
DEFAULT_OUTPUT_DIR = REPO_ROOT / "outputs" / "term_index" / "review"
|
|
|
|
|
|
def _load_json(path: Path) -> Any:
|
|
return json.loads(path.read_text(encoding="utf-8-sig"))
|
|
|
|
|
|
def _save_json(path: Path, payload: dict[str, Any]) -> None:
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
|
|
|
|
|
def _save_text(path: Path, content: str) -> None:
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
path.write_text(content, encoding="utf-8")
|
|
|
|
|
|
def _term_key(value: str) -> str:
|
|
return value.strip().casefold()
|
|
|
|
|
|
def _utc_today() -> str:
|
|
return datetime.now(timezone.utc).date().isoformat()
|
|
|
|
|
|
def _require_dict(payload: Any, name: str) -> dict[str, Any]:
|
|
if not isinstance(payload, dict):
|
|
raise RuntimeError(f"{name} must be a JSON object.")
|
|
return payload
|
|
|
|
|
|
def _require_list(payload: Any, name: str) -> list[Any]:
|
|
if not isinstance(payload, list):
|
|
raise RuntimeError(f"{name} must be a JSON array.")
|
|
return payload
|
|
|
|
|
|
def _bundle_date(bundle: dict[str, Any]) -> str:
|
|
generated_at = bundle.get("generated_at")
|
|
if isinstance(generated_at, str) and generated_at.strip():
|
|
normalized = generated_at.replace("Z", "+00:00")
|
|
try:
|
|
return datetime.fromisoformat(normalized).date().isoformat()
|
|
except ValueError:
|
|
pass
|
|
return _utc_today()
|
|
|
|
|
|
def _recent_count_map(top_global_terms: list[dict[str, Any]]) -> dict[str, int]:
|
|
counts: dict[str, int] = {}
|
|
for item in top_global_terms:
|
|
term = item.get("term")
|
|
recent_count = item.get("recent_count")
|
|
if isinstance(term, str) and isinstance(recent_count, int):
|
|
counts[term] = recent_count
|
|
return counts
|
|
|
|
|
|
def _covered_term_sets(bundle: dict[str, Any]) -> tuple[set[str], set[str], set[str]]:
|
|
current_config = _require_dict(bundle.get("current_config"), "bundle.current_config")
|
|
interest_keywords = _require_list(current_config.get("interest_keywords"), "bundle.current_config.interest_keywords")
|
|
stopwords = _require_list(current_config.get("stopwords"), "bundle.current_config.stopwords")
|
|
watchlist = _require_list(current_config.get("watchlist"), "bundle.current_config.watchlist")
|
|
|
|
interest_set = {_term_key(item) for item in interest_keywords if isinstance(item, str) and item.strip()}
|
|
stopword_set = {_term_key(item) for item in stopwords if isinstance(item, str) and item.strip()}
|
|
watch_set = {
|
|
_term_key(str(item.get("term", "")))
|
|
for item in watchlist
|
|
if isinstance(item, dict) and isinstance(item.get("term"), str) and str(item.get("term", "")).strip()
|
|
}
|
|
return interest_set, stopword_set, watch_set
|
|
|
|
|
|
def _sort_key(item: dict[str, Any]) -> tuple[int, int, int, str, str]:
|
|
total_count = int(item.get("total_count") or 0)
|
|
days_seen = int(item.get("days_seen") or 0)
|
|
recent_count = int(item.get("recent_count") or 0)
|
|
term = str(item.get("term") or "")
|
|
return (-total_count, -days_seen, -recent_count, term.casefold(), term)
|
|
|
|
|
|
def _prepare_interest_suggestions(bundle: dict[str, Any]) -> list[dict[str, Any]]:
|
|
governance_hints = _require_dict(bundle.get("governance_hints"), "bundle.governance_hints")
|
|
candidates = _require_list(
|
|
governance_hints.get("interest_review_candidates"),
|
|
"bundle.governance_hints.interest_review_candidates",
|
|
)
|
|
top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms")
|
|
recent_counts = _recent_count_map([item for item in top_global_terms if isinstance(item, dict)])
|
|
interest_set, stopword_set, watch_set = _covered_term_sets(bundle)
|
|
|
|
suggestions: list[dict[str, Any]] = []
|
|
seen: set[str] = set()
|
|
for item in candidates:
|
|
if not isinstance(item, dict):
|
|
continue
|
|
term = item.get("term")
|
|
if not isinstance(term, str) or not term.strip():
|
|
continue
|
|
term_key = _term_key(term)
|
|
if term_key in seen or term_key in interest_set or term_key in stopword_set:
|
|
continue
|
|
total_count = int(item.get("total_count") or 0)
|
|
days_seen = int(item.get("days_seen") or 0)
|
|
recent_count = recent_counts.get(term, 0)
|
|
base_reason = str(item.get("reason") or "Meets the configured interest-keyword review threshold.")
|
|
if term_key in watch_set:
|
|
base_reason += " It is currently in watchlist and is ready for promotion."
|
|
reason = f"{base_reason} Evidence: total_count={total_count}, days_seen={days_seen}, recent_count={recent_count}."
|
|
suggestions.append(
|
|
{
|
|
"term": term,
|
|
"reason": reason,
|
|
"total_count": total_count,
|
|
"days_seen": days_seen,
|
|
"recent_count": recent_count,
|
|
}
|
|
)
|
|
seen.add(term_key)
|
|
|
|
suggestions.sort(key=_sort_key)
|
|
return suggestions
|
|
|
|
|
|
def _prepare_watch_suggestions(bundle: dict[str, Any], reserved_terms: set[str]) -> list[dict[str, Any]]:
|
|
governance_hints = _require_dict(bundle.get("governance_hints"), "bundle.governance_hints")
|
|
candidates = _require_list(
|
|
governance_hints.get("watch_review_candidates"),
|
|
"bundle.governance_hints.watch_review_candidates",
|
|
)
|
|
top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms")
|
|
recent_counts = _recent_count_map([item for item in top_global_terms if isinstance(item, dict)])
|
|
interest_set, stopword_set, watch_set = _covered_term_sets(bundle)
|
|
|
|
suggestions: list[dict[str, Any]] = []
|
|
seen: set[str] = set(reserved_terms)
|
|
for item in candidates:
|
|
if not isinstance(item, dict):
|
|
continue
|
|
term = item.get("term")
|
|
if not isinstance(term, str) or not term.strip():
|
|
continue
|
|
term_key = _term_key(term)
|
|
if term_key in seen or term_key in interest_set or term_key in stopword_set or term_key in watch_set:
|
|
continue
|
|
total_count = int(item.get("total_count") or 0)
|
|
days_seen = int(item.get("days_seen") or 0)
|
|
recent_count = recent_counts.get(term, 0)
|
|
base_reason = str(item.get("reason") or "Falls into the configured watch-term review range.")
|
|
reason = f"{base_reason} Evidence: total_count={total_count}, days_seen={days_seen}, recent_count={recent_count}."
|
|
suggestions.append(
|
|
{
|
|
"term": term,
|
|
"reason": reason,
|
|
"total_count": total_count,
|
|
"days_seen": days_seen,
|
|
"recent_count": recent_count,
|
|
}
|
|
)
|
|
seen.add(term_key)
|
|
|
|
suggestions.sort(key=_sort_key)
|
|
return suggestions
|
|
|
|
|
|
def _render_table(items: list[dict[str, Any]]) -> str:
|
|
if not items:
|
|
return "_None in this pass._\n"
|
|
lines = [
|
|
"| Term | Total | Days | Recent | Reason |",
|
|
"| --- | ---: | ---: | ---: | --- |",
|
|
]
|
|
for item in items:
|
|
term = str(item.get("term") or "")
|
|
total_count = int(item.get("total_count") or 0)
|
|
days_seen = int(item.get("days_seen") or 0)
|
|
recent_count = int(item.get("recent_count") or 0)
|
|
reason = str(item.get("reason") or "").replace("|", "\\|")
|
|
lines.append(f"| {term} | {total_count} | {days_seen} | {recent_count} | {reason} |")
|
|
return "\n".join(lines) + "\n"
|
|
|
|
|
|
def _render_simple_table(items: list[dict[str, Any]], first_column: str) -> str:
|
|
if not items:
|
|
return "_None in this pass._\n"
|
|
lines = [
|
|
f"| {first_column} | Reason |",
|
|
"| --- | --- |",
|
|
]
|
|
for item in items:
|
|
value = str(item.get(first_column.casefold()) or item.get(first_column) or "")
|
|
reason = str(item.get("reason") or "").replace("|", "\\|")
|
|
lines.append(f"| {value} | {reason} |")
|
|
return "\n".join(lines) + "\n"
|
|
|
|
|
|
def _render_markdown(
|
|
*,
|
|
suggestion_date: str,
|
|
bundle_path: Path,
|
|
json_output_path: Path,
|
|
bundle: dict[str, Any],
|
|
suggestions: dict[str, Any],
|
|
) -> str:
|
|
policy = _require_dict(bundle.get("policy"), "bundle.policy")
|
|
current_config = _require_dict(bundle.get("current_config"), "bundle.current_config")
|
|
top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms")
|
|
uncovered_terms = _require_list(bundle.get("uncovered_terms"), "bundle.uncovered_terms")
|
|
top_preview = [item for item in top_global_terms if isinstance(item, dict)][:5]
|
|
uncovered_preview = [item for item in uncovered_terms if isinstance(item, dict)][:5]
|
|
|
|
interest_items = suggestions["interest_keyword_suggestions"]
|
|
watch_items = suggestions["watch_terms"]
|
|
alias_items = suggestions["alias_suggestions"]
|
|
stopword_items = suggestions["stopword_suggestions"]
|
|
|
|
lines = [
|
|
f"# Term Cleanup Suggestions - {suggestion_date}",
|
|
"",
|
|
"## Review Context",
|
|
"",
|
|
f"- Source bundle: `{bundle_path}`",
|
|
f"- Suggestions JSON: `{json_output_path}`",
|
|
f"- Bundle generated_at: `{bundle.get('generated_at', 'unknown')}`",
|
|
f"- Based on days: `{suggestions['based_on_days']}`",
|
|
f"- Policy schema version: `{policy.get('schema_version', 'unknown')}`",
|
|
"- Scope: implement `interest_keyword_suggestions` and `watch_terms` main path first; keep alias/stopword conservative in this pass.",
|
|
"",
|
|
"## Current State",
|
|
"",
|
|
f"- Interest keywords: `{current_config.get('interest_keyword_count', 0)}`",
|
|
f"- Watch terms: `{current_config.get('watch_term_count', 0)}`",
|
|
f"- Stopwords: `{current_config.get('stopword_count', 0)}`",
|
|
f"- Aliases: `{current_config.get('alias_count', 0)}`",
|
|
f"- Top global terms considered: `{len(top_global_terms)}`",
|
|
f"- Uncovered terms considered: `{len(uncovered_terms)}`",
|
|
"",
|
|
"### Top Terms Snapshot",
|
|
"",
|
|
]
|
|
|
|
if top_preview:
|
|
for item in top_preview:
|
|
lines.append(
|
|
f"- `{item.get('term', '')}`: total_count={item.get('total_count', 0)}, days_seen={item.get('days_seen', 0)}, recent_count={item.get('recent_count', 0)}"
|
|
)
|
|
else:
|
|
lines.append("- No top terms available.")
|
|
|
|
lines.extend([
|
|
"",
|
|
"### Uncovered Terms Snapshot",
|
|
"",
|
|
])
|
|
if uncovered_preview:
|
|
for item in uncovered_preview:
|
|
lines.append(
|
|
f"- `{item.get('term', '')}`: total_count={item.get('total_count', 0)}, days_seen={item.get('days_seen', 0)}, recent_count={item.get('recent_count', 0)}"
|
|
)
|
|
else:
|
|
lines.append("- No uncovered terms available.")
|
|
|
|
lines.extend([
|
|
"",
|
|
"## Suggestion Summary",
|
|
"",
|
|
f"- `interest_keyword_suggestions`: `{len(interest_items)}`",
|
|
f"- `watch_terms`: `{len(watch_items)}`",
|
|
f"- `alias_suggestions`: `{len(alias_items)}`",
|
|
f"- `stopword_suggestions`: `{len(stopword_items)}`",
|
|
"",
|
|
"## Interest Keyword Suggestions",
|
|
"",
|
|
_render_table(interest_items).rstrip(),
|
|
"",
|
|
"## Watch Terms",
|
|
"",
|
|
_render_table(watch_items).rstrip(),
|
|
"",
|
|
"## Alias Suggestions",
|
|
"",
|
|
"_Conservative by design in this minimal version; no automatic alias suggestions are emitted yet._" if not alias_items else _render_simple_table(alias_items, "from").rstrip(),
|
|
"",
|
|
"## Stopword Suggestions",
|
|
"",
|
|
"_Conservative by design in this minimal version; no automatic stopword suggestions are emitted yet._" if not stopword_items else _render_simple_table(stopword_items, "term").rstrip(),
|
|
"",
|
|
"## Apply",
|
|
"",
|
|
"Review the Markdown first, then selectively apply accepted suggestions with the JSON file.",
|
|
"",
|
|
"```bash",
|
|
f"python scripts/apply_term_suggestions.py \\",
|
|
f" --suggestions {json_output_path} \\",
|
|
" --accept-interest \"Claude Code\" \\",
|
|
" --accept-watch \"A2A\" \\",
|
|
" --dry-run",
|
|
"```",
|
|
"",
|
|
])
|
|
return "\n".join(lines)
|
|
|
|
|
|
def _build_output_paths(
|
|
*,
|
|
output_dir: Path,
|
|
suggestion_date: str,
|
|
json_output: Path | None,
|
|
markdown_output: Path | None,
|
|
) -> tuple[Path, Path]:
|
|
stem = f"term-cleanup-suggestions-{suggestion_date}"
|
|
resolved_json = json_output or (output_dir / f"{stem}.json")
|
|
resolved_markdown = markdown_output or (output_dir / f"{stem}.md")
|
|
return resolved_json, resolved_markdown
|
|
|
|
|
|
def main() -> None:
|
|
parser = argparse.ArgumentParser(description="Generate term cleanup suggestions JSON and Markdown from review bundle.")
|
|
parser.add_argument("--bundle", type=Path, default=DEFAULT_BUNDLE_PATH, help="Review bundle JSON file")
|
|
parser.add_argument(
|
|
"--output-dir",
|
|
type=Path,
|
|
default=DEFAULT_OUTPUT_DIR,
|
|
help="Directory for generated suggestions outputs when explicit output paths are not provided",
|
|
)
|
|
parser.add_argument("--date", type=str, default=None, help="Override suggestions date (YYYY-MM-DD)")
|
|
parser.add_argument("--json-output", type=Path, default=None, help="Explicit suggestions JSON output path")
|
|
parser.add_argument("--markdown-output", type=Path, default=None, help="Explicit suggestions Markdown output path")
|
|
parser.add_argument(
|
|
"--emit-markdown",
|
|
action="store_true",
|
|
help="Also write the human-readable Markdown review draft. JSON suggestions are always written.",
|
|
)
|
|
args = parser.parse_args()
|
|
|
|
if not args.bundle.exists():
|
|
raise RuntimeError(f"Bundle file not found: {args.bundle}")
|
|
|
|
bundle = _require_dict(_load_json(args.bundle), "bundle")
|
|
days = bundle.get("days")
|
|
if not isinstance(days, int):
|
|
raise RuntimeError("bundle.days must be an integer.")
|
|
|
|
suggestion_date = args.date or _bundle_date(bundle)
|
|
json_output_path, markdown_output_path = _build_output_paths(
|
|
output_dir=args.output_dir,
|
|
suggestion_date=suggestion_date,
|
|
json_output=args.json_output,
|
|
markdown_output=args.markdown_output,
|
|
)
|
|
|
|
interest_items = _prepare_interest_suggestions(bundle)
|
|
reserved_terms = {_term_key(str(item.get("term") or "")) for item in interest_items}
|
|
watch_items = _prepare_watch_suggestions(bundle, reserved_terms=reserved_terms)
|
|
|
|
suggestions = {
|
|
"date": suggestion_date,
|
|
"based_on_days": days,
|
|
"source_bundle": str(args.bundle),
|
|
"policy_schema_version": _require_dict(bundle.get("policy"), "bundle.policy").get("schema_version", "unknown"),
|
|
"summary": {
|
|
"interest_keyword_suggestions": len(interest_items),
|
|
"watch_terms": len(watch_items),
|
|
"alias_suggestions": 0,
|
|
"stopword_suggestions": 0,
|
|
},
|
|
"alias_suggestions": [],
|
|
"stopword_suggestions": [],
|
|
"interest_keyword_suggestions": interest_items,
|
|
"watch_terms": watch_items,
|
|
}
|
|
markdown = _render_markdown(
|
|
suggestion_date=suggestion_date,
|
|
bundle_path=args.bundle,
|
|
json_output_path=json_output_path,
|
|
bundle=bundle,
|
|
suggestions=suggestions,
|
|
)
|
|
|
|
_save_json(json_output_path, suggestions)
|
|
if args.emit_markdown:
|
|
_save_text(markdown_output_path, markdown)
|
|
|
|
summary = {
|
|
"bundle": str(args.bundle),
|
|
"date": suggestion_date,
|
|
"json_output": str(json_output_path),
|
|
"markdown_output": str(markdown_output_path) if args.emit_markdown else None,
|
|
"interest_keyword_suggestions": len(interest_items),
|
|
"watch_terms": len(watch_items),
|
|
"alias_suggestions": 0,
|
|
"stopword_suggestions": 0,
|
|
"emit_markdown": args.emit_markdown,
|
|
}
|
|
print(json.dumps(summary, ensure_ascii=False, indent=2))
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|