feat: add keyword cleanup docs, skill updates, and delivery compatibility fix
This commit is contained in:
@@ -0,0 +1,411 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||
DEFAULT_BUNDLE_PATH = REPO_ROOT / "outputs" / "term_index" / "review" / "keyword-cleanup-bundle.json"
|
||||
DEFAULT_OUTPUT_DIR = REPO_ROOT / "outputs" / "term_index" / "review"
|
||||
|
||||
|
||||
def _load_json(path: Path) -> Any:
|
||||
return json.loads(path.read_text(encoding="utf-8-sig"))
|
||||
|
||||
|
||||
def _save_json(path: Path, payload: dict[str, Any]) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
||||
|
||||
|
||||
def _save_text(path: Path, content: str) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(content, encoding="utf-8")
|
||||
|
||||
|
||||
def _term_key(value: str) -> str:
|
||||
return value.strip().casefold()
|
||||
|
||||
|
||||
def _utc_today() -> str:
|
||||
return datetime.now(timezone.utc).date().isoformat()
|
||||
|
||||
|
||||
def _require_dict(payload: Any, name: str) -> dict[str, Any]:
|
||||
if not isinstance(payload, dict):
|
||||
raise RuntimeError(f"{name} must be a JSON object.")
|
||||
return payload
|
||||
|
||||
|
||||
def _require_list(payload: Any, name: str) -> list[Any]:
|
||||
if not isinstance(payload, list):
|
||||
raise RuntimeError(f"{name} must be a JSON array.")
|
||||
return payload
|
||||
|
||||
|
||||
def _bundle_date(bundle: dict[str, Any]) -> str:
|
||||
generated_at = bundle.get("generated_at")
|
||||
if isinstance(generated_at, str) and generated_at.strip():
|
||||
normalized = generated_at.replace("Z", "+00:00")
|
||||
try:
|
||||
return datetime.fromisoformat(normalized).date().isoformat()
|
||||
except ValueError:
|
||||
pass
|
||||
return _utc_today()
|
||||
|
||||
|
||||
def _recent_count_map(top_global_terms: list[dict[str, Any]]) -> dict[str, int]:
|
||||
counts: dict[str, int] = {}
|
||||
for item in top_global_terms:
|
||||
term = item.get("term")
|
||||
recent_count = item.get("recent_count")
|
||||
if isinstance(term, str) and isinstance(recent_count, int):
|
||||
counts[term] = recent_count
|
||||
return counts
|
||||
|
||||
|
||||
def _covered_term_sets(bundle: dict[str, Any]) -> tuple[set[str], set[str], set[str]]:
|
||||
current_config = _require_dict(bundle.get("current_config"), "bundle.current_config")
|
||||
interest_keywords = _require_list(current_config.get("interest_keywords"), "bundle.current_config.interest_keywords")
|
||||
stopwords = _require_list(current_config.get("stopwords"), "bundle.current_config.stopwords")
|
||||
watchlist = _require_list(current_config.get("watchlist"), "bundle.current_config.watchlist")
|
||||
|
||||
interest_set = {_term_key(item) for item in interest_keywords if isinstance(item, str) and item.strip()}
|
||||
stopword_set = {_term_key(item) for item in stopwords if isinstance(item, str) and item.strip()}
|
||||
watch_set = {
|
||||
_term_key(str(item.get("term", "")))
|
||||
for item in watchlist
|
||||
if isinstance(item, dict) and isinstance(item.get("term"), str) and str(item.get("term", "")).strip()
|
||||
}
|
||||
return interest_set, stopword_set, watch_set
|
||||
|
||||
|
||||
def _sort_key(item: dict[str, Any]) -> tuple[int, int, int, str, str]:
|
||||
total_count = int(item.get("total_count") or 0)
|
||||
days_seen = int(item.get("days_seen") or 0)
|
||||
recent_count = int(item.get("recent_count") or 0)
|
||||
term = str(item.get("term") or "")
|
||||
return (-total_count, -days_seen, -recent_count, term.casefold(), term)
|
||||
|
||||
|
||||
def _prepare_interest_suggestions(bundle: dict[str, Any]) -> list[dict[str, Any]]:
|
||||
governance_hints = _require_dict(bundle.get("governance_hints"), "bundle.governance_hints")
|
||||
candidates = _require_list(
|
||||
governance_hints.get("interest_review_candidates"),
|
||||
"bundle.governance_hints.interest_review_candidates",
|
||||
)
|
||||
top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms")
|
||||
recent_counts = _recent_count_map([item for item in top_global_terms if isinstance(item, dict)])
|
||||
interest_set, stopword_set, watch_set = _covered_term_sets(bundle)
|
||||
|
||||
suggestions: list[dict[str, Any]] = []
|
||||
seen: set[str] = set()
|
||||
for item in candidates:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
term = item.get("term")
|
||||
if not isinstance(term, str) or not term.strip():
|
||||
continue
|
||||
term_key = _term_key(term)
|
||||
if term_key in seen or term_key in interest_set or term_key in stopword_set:
|
||||
continue
|
||||
total_count = int(item.get("total_count") or 0)
|
||||
days_seen = int(item.get("days_seen") or 0)
|
||||
recent_count = recent_counts.get(term, 0)
|
||||
base_reason = str(item.get("reason") or "Meets the configured interest-keyword review threshold.")
|
||||
if term_key in watch_set:
|
||||
base_reason += " It is currently in watchlist and is ready for promotion."
|
||||
reason = f"{base_reason} Evidence: total_count={total_count}, days_seen={days_seen}, recent_count={recent_count}."
|
||||
suggestions.append(
|
||||
{
|
||||
"term": term,
|
||||
"reason": reason,
|
||||
"total_count": total_count,
|
||||
"days_seen": days_seen,
|
||||
"recent_count": recent_count,
|
||||
}
|
||||
)
|
||||
seen.add(term_key)
|
||||
|
||||
suggestions.sort(key=_sort_key)
|
||||
return suggestions
|
||||
|
||||
|
||||
def _prepare_watch_suggestions(bundle: dict[str, Any], reserved_terms: set[str]) -> list[dict[str, Any]]:
|
||||
governance_hints = _require_dict(bundle.get("governance_hints"), "bundle.governance_hints")
|
||||
candidates = _require_list(
|
||||
governance_hints.get("watch_review_candidates"),
|
||||
"bundle.governance_hints.watch_review_candidates",
|
||||
)
|
||||
top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms")
|
||||
recent_counts = _recent_count_map([item for item in top_global_terms if isinstance(item, dict)])
|
||||
interest_set, stopword_set, watch_set = _covered_term_sets(bundle)
|
||||
|
||||
suggestions: list[dict[str, Any]] = []
|
||||
seen: set[str] = set(reserved_terms)
|
||||
for item in candidates:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
term = item.get("term")
|
||||
if not isinstance(term, str) or not term.strip():
|
||||
continue
|
||||
term_key = _term_key(term)
|
||||
if term_key in seen or term_key in interest_set or term_key in stopword_set or term_key in watch_set:
|
||||
continue
|
||||
total_count = int(item.get("total_count") or 0)
|
||||
days_seen = int(item.get("days_seen") or 0)
|
||||
recent_count = recent_counts.get(term, 0)
|
||||
base_reason = str(item.get("reason") or "Falls into the configured watch-term review range.")
|
||||
reason = f"{base_reason} Evidence: total_count={total_count}, days_seen={days_seen}, recent_count={recent_count}."
|
||||
suggestions.append(
|
||||
{
|
||||
"term": term,
|
||||
"reason": reason,
|
||||
"total_count": total_count,
|
||||
"days_seen": days_seen,
|
||||
"recent_count": recent_count,
|
||||
}
|
||||
)
|
||||
seen.add(term_key)
|
||||
|
||||
suggestions.sort(key=_sort_key)
|
||||
return suggestions
|
||||
|
||||
|
||||
def _render_table(items: list[dict[str, Any]]) -> str:
|
||||
if not items:
|
||||
return "_None in this pass._\n"
|
||||
lines = [
|
||||
"| Term | Total | Days | Recent | Reason |",
|
||||
"| --- | ---: | ---: | ---: | --- |",
|
||||
]
|
||||
for item in items:
|
||||
term = str(item.get("term") or "")
|
||||
total_count = int(item.get("total_count") or 0)
|
||||
days_seen = int(item.get("days_seen") or 0)
|
||||
recent_count = int(item.get("recent_count") or 0)
|
||||
reason = str(item.get("reason") or "").replace("|", "\\|")
|
||||
lines.append(f"| {term} | {total_count} | {days_seen} | {recent_count} | {reason} |")
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
def _render_simple_table(items: list[dict[str, Any]], first_column: str) -> str:
|
||||
if not items:
|
||||
return "_None in this pass._\n"
|
||||
lines = [
|
||||
f"| {first_column} | Reason |",
|
||||
"| --- | --- |",
|
||||
]
|
||||
for item in items:
|
||||
value = str(item.get(first_column.casefold()) or item.get(first_column) or "")
|
||||
reason = str(item.get("reason") or "").replace("|", "\\|")
|
||||
lines.append(f"| {value} | {reason} |")
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
def _render_markdown(
|
||||
*,
|
||||
suggestion_date: str,
|
||||
bundle_path: Path,
|
||||
json_output_path: Path,
|
||||
bundle: dict[str, Any],
|
||||
suggestions: dict[str, Any],
|
||||
) -> str:
|
||||
policy = _require_dict(bundle.get("policy"), "bundle.policy")
|
||||
current_config = _require_dict(bundle.get("current_config"), "bundle.current_config")
|
||||
top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms")
|
||||
uncovered_terms = _require_list(bundle.get("uncovered_terms"), "bundle.uncovered_terms")
|
||||
top_preview = [item for item in top_global_terms if isinstance(item, dict)][:5]
|
||||
uncovered_preview = [item for item in uncovered_terms if isinstance(item, dict)][:5]
|
||||
|
||||
interest_items = suggestions["interest_keyword_suggestions"]
|
||||
watch_items = suggestions["watch_terms"]
|
||||
alias_items = suggestions["alias_suggestions"]
|
||||
stopword_items = suggestions["stopword_suggestions"]
|
||||
|
||||
lines = [
|
||||
f"# Term Cleanup Suggestions - {suggestion_date}",
|
||||
"",
|
||||
"## Review Context",
|
||||
"",
|
||||
f"- Source bundle: `{bundle_path}`",
|
||||
f"- Suggestions JSON: `{json_output_path}`",
|
||||
f"- Bundle generated_at: `{bundle.get('generated_at', 'unknown')}`",
|
||||
f"- Based on days: `{suggestions['based_on_days']}`",
|
||||
f"- Policy schema version: `{policy.get('schema_version', 'unknown')}`",
|
||||
"- Scope: implement `interest_keyword_suggestions` and `watch_terms` main path first; keep alias/stopword conservative in this pass.",
|
||||
"",
|
||||
"## Current State",
|
||||
"",
|
||||
f"- Interest keywords: `{current_config.get('interest_keyword_count', 0)}`",
|
||||
f"- Watch terms: `{current_config.get('watch_term_count', 0)}`",
|
||||
f"- Stopwords: `{current_config.get('stopword_count', 0)}`",
|
||||
f"- Aliases: `{current_config.get('alias_count', 0)}`",
|
||||
f"- Top global terms considered: `{len(top_global_terms)}`",
|
||||
f"- Uncovered terms considered: `{len(uncovered_terms)}`",
|
||||
"",
|
||||
"### Top Terms Snapshot",
|
||||
"",
|
||||
]
|
||||
|
||||
if top_preview:
|
||||
for item in top_preview:
|
||||
lines.append(
|
||||
f"- `{item.get('term', '')}`: total_count={item.get('total_count', 0)}, days_seen={item.get('days_seen', 0)}, recent_count={item.get('recent_count', 0)}"
|
||||
)
|
||||
else:
|
||||
lines.append("- No top terms available.")
|
||||
|
||||
lines.extend([
|
||||
"",
|
||||
"### Uncovered Terms Snapshot",
|
||||
"",
|
||||
])
|
||||
if uncovered_preview:
|
||||
for item in uncovered_preview:
|
||||
lines.append(
|
||||
f"- `{item.get('term', '')}`: total_count={item.get('total_count', 0)}, days_seen={item.get('days_seen', 0)}, recent_count={item.get('recent_count', 0)}"
|
||||
)
|
||||
else:
|
||||
lines.append("- No uncovered terms available.")
|
||||
|
||||
lines.extend([
|
||||
"",
|
||||
"## Suggestion Summary",
|
||||
"",
|
||||
f"- `interest_keyword_suggestions`: `{len(interest_items)}`",
|
||||
f"- `watch_terms`: `{len(watch_items)}`",
|
||||
f"- `alias_suggestions`: `{len(alias_items)}`",
|
||||
f"- `stopword_suggestions`: `{len(stopword_items)}`",
|
||||
"",
|
||||
"## Interest Keyword Suggestions",
|
||||
"",
|
||||
_render_table(interest_items).rstrip(),
|
||||
"",
|
||||
"## Watch Terms",
|
||||
"",
|
||||
_render_table(watch_items).rstrip(),
|
||||
"",
|
||||
"## Alias Suggestions",
|
||||
"",
|
||||
"_Conservative by design in this minimal version; no automatic alias suggestions are emitted yet._" if not alias_items else _render_simple_table(alias_items, "from").rstrip(),
|
||||
"",
|
||||
"## Stopword Suggestions",
|
||||
"",
|
||||
"_Conservative by design in this minimal version; no automatic stopword suggestions are emitted yet._" if not stopword_items else _render_simple_table(stopword_items, "term").rstrip(),
|
||||
"",
|
||||
"## Apply",
|
||||
"",
|
||||
"Review the Markdown first, then selectively apply accepted suggestions with the JSON file.",
|
||||
"",
|
||||
"```bash",
|
||||
f"python scripts/apply_term_suggestions.py \\",
|
||||
f" --suggestions {json_output_path} \\",
|
||||
" --accept-interest \"Claude Code\" \\",
|
||||
" --accept-watch \"A2A\" \\",
|
||||
" --dry-run",
|
||||
"```",
|
||||
"",
|
||||
])
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def _build_output_paths(
|
||||
*,
|
||||
output_dir: Path,
|
||||
suggestion_date: str,
|
||||
json_output: Path | None,
|
||||
markdown_output: Path | None,
|
||||
) -> tuple[Path, Path]:
|
||||
stem = f"term-cleanup-suggestions-{suggestion_date}"
|
||||
resolved_json = json_output or (output_dir / f"{stem}.json")
|
||||
resolved_markdown = markdown_output or (output_dir / f"{stem}.md")
|
||||
return resolved_json, resolved_markdown
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description="Generate term cleanup suggestions JSON and Markdown from review bundle.")
|
||||
parser.add_argument("--bundle", type=Path, default=DEFAULT_BUNDLE_PATH, help="Review bundle JSON file")
|
||||
parser.add_argument(
|
||||
"--output-dir",
|
||||
type=Path,
|
||||
default=DEFAULT_OUTPUT_DIR,
|
||||
help="Directory for generated suggestions outputs when explicit output paths are not provided",
|
||||
)
|
||||
parser.add_argument("--date", type=str, default=None, help="Override suggestions date (YYYY-MM-DD)")
|
||||
parser.add_argument("--json-output", type=Path, default=None, help="Explicit suggestions JSON output path")
|
||||
parser.add_argument("--markdown-output", type=Path, default=None, help="Explicit suggestions Markdown output path")
|
||||
parser.add_argument(
|
||||
"--emit-markdown",
|
||||
action="store_true",
|
||||
help="Also write the human-readable Markdown review draft. JSON suggestions are always written.",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
if not args.bundle.exists():
|
||||
raise RuntimeError(f"Bundle file not found: {args.bundle}")
|
||||
|
||||
bundle = _require_dict(_load_json(args.bundle), "bundle")
|
||||
days = bundle.get("days")
|
||||
if not isinstance(days, int):
|
||||
raise RuntimeError("bundle.days must be an integer.")
|
||||
|
||||
suggestion_date = args.date or _bundle_date(bundle)
|
||||
json_output_path, markdown_output_path = _build_output_paths(
|
||||
output_dir=args.output_dir,
|
||||
suggestion_date=suggestion_date,
|
||||
json_output=args.json_output,
|
||||
markdown_output=args.markdown_output,
|
||||
)
|
||||
|
||||
interest_items = _prepare_interest_suggestions(bundle)
|
||||
reserved_terms = {_term_key(str(item.get("term") or "")) for item in interest_items}
|
||||
watch_items = _prepare_watch_suggestions(bundle, reserved_terms=reserved_terms)
|
||||
|
||||
suggestions = {
|
||||
"date": suggestion_date,
|
||||
"based_on_days": days,
|
||||
"source_bundle": str(args.bundle),
|
||||
"policy_schema_version": _require_dict(bundle.get("policy"), "bundle.policy").get("schema_version", "unknown"),
|
||||
"summary": {
|
||||
"interest_keyword_suggestions": len(interest_items),
|
||||
"watch_terms": len(watch_items),
|
||||
"alias_suggestions": 0,
|
||||
"stopword_suggestions": 0,
|
||||
},
|
||||
"alias_suggestions": [],
|
||||
"stopword_suggestions": [],
|
||||
"interest_keyword_suggestions": interest_items,
|
||||
"watch_terms": watch_items,
|
||||
}
|
||||
markdown = _render_markdown(
|
||||
suggestion_date=suggestion_date,
|
||||
bundle_path=args.bundle,
|
||||
json_output_path=json_output_path,
|
||||
bundle=bundle,
|
||||
suggestions=suggestions,
|
||||
)
|
||||
|
||||
_save_json(json_output_path, suggestions)
|
||||
if args.emit_markdown:
|
||||
_save_text(markdown_output_path, markdown)
|
||||
|
||||
summary = {
|
||||
"bundle": str(args.bundle),
|
||||
"date": suggestion_date,
|
||||
"json_output": str(json_output_path),
|
||||
"markdown_output": str(markdown_output_path) if args.emit_markdown else None,
|
||||
"interest_keyword_suggestions": len(interest_items),
|
||||
"watch_terms": len(watch_items),
|
||||
"alias_suggestions": 0,
|
||||
"stopword_suggestions": 0,
|
||||
"emit_markdown": args.emit_markdown,
|
||||
}
|
||||
print(json.dumps(summary, ensure_ascii=False, indent=2))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user