feat: add keyword cleanup docs, skill updates, and delivery compatibility fix

This commit is contained in:
root
2026-04-14 09:51:57 +08:00
parent a06f2a1d08
commit e3663f681d
5 changed files with 1642 additions and 0 deletions
@@ -0,0 +1,411 @@
from __future__ import annotations
import argparse
import json
from datetime import datetime, timezone
from pathlib import Path
from typing import Any
REPO_ROOT = Path(__file__).resolve().parents[1]
DEFAULT_BUNDLE_PATH = REPO_ROOT / "outputs" / "term_index" / "review" / "keyword-cleanup-bundle.json"
DEFAULT_OUTPUT_DIR = REPO_ROOT / "outputs" / "term_index" / "review"
def _load_json(path: Path) -> Any:
return json.loads(path.read_text(encoding="utf-8-sig"))
def _save_json(path: Path, payload: dict[str, Any]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
def _save_text(path: Path, content: str) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(content, encoding="utf-8")
def _term_key(value: str) -> str:
return value.strip().casefold()
def _utc_today() -> str:
return datetime.now(timezone.utc).date().isoformat()
def _require_dict(payload: Any, name: str) -> dict[str, Any]:
if not isinstance(payload, dict):
raise RuntimeError(f"{name} must be a JSON object.")
return payload
def _require_list(payload: Any, name: str) -> list[Any]:
if not isinstance(payload, list):
raise RuntimeError(f"{name} must be a JSON array.")
return payload
def _bundle_date(bundle: dict[str, Any]) -> str:
generated_at = bundle.get("generated_at")
if isinstance(generated_at, str) and generated_at.strip():
normalized = generated_at.replace("Z", "+00:00")
try:
return datetime.fromisoformat(normalized).date().isoformat()
except ValueError:
pass
return _utc_today()
def _recent_count_map(top_global_terms: list[dict[str, Any]]) -> dict[str, int]:
counts: dict[str, int] = {}
for item in top_global_terms:
term = item.get("term")
recent_count = item.get("recent_count")
if isinstance(term, str) and isinstance(recent_count, int):
counts[term] = recent_count
return counts
def _covered_term_sets(bundle: dict[str, Any]) -> tuple[set[str], set[str], set[str]]:
current_config = _require_dict(bundle.get("current_config"), "bundle.current_config")
interest_keywords = _require_list(current_config.get("interest_keywords"), "bundle.current_config.interest_keywords")
stopwords = _require_list(current_config.get("stopwords"), "bundle.current_config.stopwords")
watchlist = _require_list(current_config.get("watchlist"), "bundle.current_config.watchlist")
interest_set = {_term_key(item) for item in interest_keywords if isinstance(item, str) and item.strip()}
stopword_set = {_term_key(item) for item in stopwords if isinstance(item, str) and item.strip()}
watch_set = {
_term_key(str(item.get("term", "")))
for item in watchlist
if isinstance(item, dict) and isinstance(item.get("term"), str) and str(item.get("term", "")).strip()
}
return interest_set, stopword_set, watch_set
def _sort_key(item: dict[str, Any]) -> tuple[int, int, int, str, str]:
total_count = int(item.get("total_count") or 0)
days_seen = int(item.get("days_seen") or 0)
recent_count = int(item.get("recent_count") or 0)
term = str(item.get("term") or "")
return (-total_count, -days_seen, -recent_count, term.casefold(), term)
def _prepare_interest_suggestions(bundle: dict[str, Any]) -> list[dict[str, Any]]:
governance_hints = _require_dict(bundle.get("governance_hints"), "bundle.governance_hints")
candidates = _require_list(
governance_hints.get("interest_review_candidates"),
"bundle.governance_hints.interest_review_candidates",
)
top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms")
recent_counts = _recent_count_map([item for item in top_global_terms if isinstance(item, dict)])
interest_set, stopword_set, watch_set = _covered_term_sets(bundle)
suggestions: list[dict[str, Any]] = []
seen: set[str] = set()
for item in candidates:
if not isinstance(item, dict):
continue
term = item.get("term")
if not isinstance(term, str) or not term.strip():
continue
term_key = _term_key(term)
if term_key in seen or term_key in interest_set or term_key in stopword_set:
continue
total_count = int(item.get("total_count") or 0)
days_seen = int(item.get("days_seen") or 0)
recent_count = recent_counts.get(term, 0)
base_reason = str(item.get("reason") or "Meets the configured interest-keyword review threshold.")
if term_key in watch_set:
base_reason += " It is currently in watchlist and is ready for promotion."
reason = f"{base_reason} Evidence: total_count={total_count}, days_seen={days_seen}, recent_count={recent_count}."
suggestions.append(
{
"term": term,
"reason": reason,
"total_count": total_count,
"days_seen": days_seen,
"recent_count": recent_count,
}
)
seen.add(term_key)
suggestions.sort(key=_sort_key)
return suggestions
def _prepare_watch_suggestions(bundle: dict[str, Any], reserved_terms: set[str]) -> list[dict[str, Any]]:
governance_hints = _require_dict(bundle.get("governance_hints"), "bundle.governance_hints")
candidates = _require_list(
governance_hints.get("watch_review_candidates"),
"bundle.governance_hints.watch_review_candidates",
)
top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms")
recent_counts = _recent_count_map([item for item in top_global_terms if isinstance(item, dict)])
interest_set, stopword_set, watch_set = _covered_term_sets(bundle)
suggestions: list[dict[str, Any]] = []
seen: set[str] = set(reserved_terms)
for item in candidates:
if not isinstance(item, dict):
continue
term = item.get("term")
if not isinstance(term, str) or not term.strip():
continue
term_key = _term_key(term)
if term_key in seen or term_key in interest_set or term_key in stopword_set or term_key in watch_set:
continue
total_count = int(item.get("total_count") or 0)
days_seen = int(item.get("days_seen") or 0)
recent_count = recent_counts.get(term, 0)
base_reason = str(item.get("reason") or "Falls into the configured watch-term review range.")
reason = f"{base_reason} Evidence: total_count={total_count}, days_seen={days_seen}, recent_count={recent_count}."
suggestions.append(
{
"term": term,
"reason": reason,
"total_count": total_count,
"days_seen": days_seen,
"recent_count": recent_count,
}
)
seen.add(term_key)
suggestions.sort(key=_sort_key)
return suggestions
def _render_table(items: list[dict[str, Any]]) -> str:
if not items:
return "_None in this pass._\n"
lines = [
"| Term | Total | Days | Recent | Reason |",
"| --- | ---: | ---: | ---: | --- |",
]
for item in items:
term = str(item.get("term") or "")
total_count = int(item.get("total_count") or 0)
days_seen = int(item.get("days_seen") or 0)
recent_count = int(item.get("recent_count") or 0)
reason = str(item.get("reason") or "").replace("|", "\\|")
lines.append(f"| {term} | {total_count} | {days_seen} | {recent_count} | {reason} |")
return "\n".join(lines) + "\n"
def _render_simple_table(items: list[dict[str, Any]], first_column: str) -> str:
if not items:
return "_None in this pass._\n"
lines = [
f"| {first_column} | Reason |",
"| --- | --- |",
]
for item in items:
value = str(item.get(first_column.casefold()) or item.get(first_column) or "")
reason = str(item.get("reason") or "").replace("|", "\\|")
lines.append(f"| {value} | {reason} |")
return "\n".join(lines) + "\n"
def _render_markdown(
*,
suggestion_date: str,
bundle_path: Path,
json_output_path: Path,
bundle: dict[str, Any],
suggestions: dict[str, Any],
) -> str:
policy = _require_dict(bundle.get("policy"), "bundle.policy")
current_config = _require_dict(bundle.get("current_config"), "bundle.current_config")
top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms")
uncovered_terms = _require_list(bundle.get("uncovered_terms"), "bundle.uncovered_terms")
top_preview = [item for item in top_global_terms if isinstance(item, dict)][:5]
uncovered_preview = [item for item in uncovered_terms if isinstance(item, dict)][:5]
interest_items = suggestions["interest_keyword_suggestions"]
watch_items = suggestions["watch_terms"]
alias_items = suggestions["alias_suggestions"]
stopword_items = suggestions["stopword_suggestions"]
lines = [
f"# Term Cleanup Suggestions - {suggestion_date}",
"",
"## Review Context",
"",
f"- Source bundle: `{bundle_path}`",
f"- Suggestions JSON: `{json_output_path}`",
f"- Bundle generated_at: `{bundle.get('generated_at', 'unknown')}`",
f"- Based on days: `{suggestions['based_on_days']}`",
f"- Policy schema version: `{policy.get('schema_version', 'unknown')}`",
"- Scope: implement `interest_keyword_suggestions` and `watch_terms` main path first; keep alias/stopword conservative in this pass.",
"",
"## Current State",
"",
f"- Interest keywords: `{current_config.get('interest_keyword_count', 0)}`",
f"- Watch terms: `{current_config.get('watch_term_count', 0)}`",
f"- Stopwords: `{current_config.get('stopword_count', 0)}`",
f"- Aliases: `{current_config.get('alias_count', 0)}`",
f"- Top global terms considered: `{len(top_global_terms)}`",
f"- Uncovered terms considered: `{len(uncovered_terms)}`",
"",
"### Top Terms Snapshot",
"",
]
if top_preview:
for item in top_preview:
lines.append(
f"- `{item.get('term', '')}`: total_count={item.get('total_count', 0)}, days_seen={item.get('days_seen', 0)}, recent_count={item.get('recent_count', 0)}"
)
else:
lines.append("- No top terms available.")
lines.extend([
"",
"### Uncovered Terms Snapshot",
"",
])
if uncovered_preview:
for item in uncovered_preview:
lines.append(
f"- `{item.get('term', '')}`: total_count={item.get('total_count', 0)}, days_seen={item.get('days_seen', 0)}, recent_count={item.get('recent_count', 0)}"
)
else:
lines.append("- No uncovered terms available.")
lines.extend([
"",
"## Suggestion Summary",
"",
f"- `interest_keyword_suggestions`: `{len(interest_items)}`",
f"- `watch_terms`: `{len(watch_items)}`",
f"- `alias_suggestions`: `{len(alias_items)}`",
f"- `stopword_suggestions`: `{len(stopword_items)}`",
"",
"## Interest Keyword Suggestions",
"",
_render_table(interest_items).rstrip(),
"",
"## Watch Terms",
"",
_render_table(watch_items).rstrip(),
"",
"## Alias Suggestions",
"",
"_Conservative by design in this minimal version; no automatic alias suggestions are emitted yet._" if not alias_items else _render_simple_table(alias_items, "from").rstrip(),
"",
"## Stopword Suggestions",
"",
"_Conservative by design in this minimal version; no automatic stopword suggestions are emitted yet._" if not stopword_items else _render_simple_table(stopword_items, "term").rstrip(),
"",
"## Apply",
"",
"Review the Markdown first, then selectively apply accepted suggestions with the JSON file.",
"",
"```bash",
f"python scripts/apply_term_suggestions.py \\",
f" --suggestions {json_output_path} \\",
" --accept-interest \"Claude Code\" \\",
" --accept-watch \"A2A\" \\",
" --dry-run",
"```",
"",
])
return "\n".join(lines)
def _build_output_paths(
*,
output_dir: Path,
suggestion_date: str,
json_output: Path | None,
markdown_output: Path | None,
) -> tuple[Path, Path]:
stem = f"term-cleanup-suggestions-{suggestion_date}"
resolved_json = json_output or (output_dir / f"{stem}.json")
resolved_markdown = markdown_output or (output_dir / f"{stem}.md")
return resolved_json, resolved_markdown
def main() -> None:
parser = argparse.ArgumentParser(description="Generate term cleanup suggestions JSON and Markdown from review bundle.")
parser.add_argument("--bundle", type=Path, default=DEFAULT_BUNDLE_PATH, help="Review bundle JSON file")
parser.add_argument(
"--output-dir",
type=Path,
default=DEFAULT_OUTPUT_DIR,
help="Directory for generated suggestions outputs when explicit output paths are not provided",
)
parser.add_argument("--date", type=str, default=None, help="Override suggestions date (YYYY-MM-DD)")
parser.add_argument("--json-output", type=Path, default=None, help="Explicit suggestions JSON output path")
parser.add_argument("--markdown-output", type=Path, default=None, help="Explicit suggestions Markdown output path")
parser.add_argument(
"--emit-markdown",
action="store_true",
help="Also write the human-readable Markdown review draft. JSON suggestions are always written.",
)
args = parser.parse_args()
if not args.bundle.exists():
raise RuntimeError(f"Bundle file not found: {args.bundle}")
bundle = _require_dict(_load_json(args.bundle), "bundle")
days = bundle.get("days")
if not isinstance(days, int):
raise RuntimeError("bundle.days must be an integer.")
suggestion_date = args.date or _bundle_date(bundle)
json_output_path, markdown_output_path = _build_output_paths(
output_dir=args.output_dir,
suggestion_date=suggestion_date,
json_output=args.json_output,
markdown_output=args.markdown_output,
)
interest_items = _prepare_interest_suggestions(bundle)
reserved_terms = {_term_key(str(item.get("term") or "")) for item in interest_items}
watch_items = _prepare_watch_suggestions(bundle, reserved_terms=reserved_terms)
suggestions = {
"date": suggestion_date,
"based_on_days": days,
"source_bundle": str(args.bundle),
"policy_schema_version": _require_dict(bundle.get("policy"), "bundle.policy").get("schema_version", "unknown"),
"summary": {
"interest_keyword_suggestions": len(interest_items),
"watch_terms": len(watch_items),
"alias_suggestions": 0,
"stopword_suggestions": 0,
},
"alias_suggestions": [],
"stopword_suggestions": [],
"interest_keyword_suggestions": interest_items,
"watch_terms": watch_items,
}
markdown = _render_markdown(
suggestion_date=suggestion_date,
bundle_path=args.bundle,
json_output_path=json_output_path,
bundle=bundle,
suggestions=suggestions,
)
_save_json(json_output_path, suggestions)
if args.emit_markdown:
_save_text(markdown_output_path, markdown)
summary = {
"bundle": str(args.bundle),
"date": suggestion_date,
"json_output": str(json_output_path),
"markdown_output": str(markdown_output_path) if args.emit_markdown else None,
"interest_keyword_suggestions": len(interest_items),
"watch_terms": len(watch_items),
"alias_suggestions": 0,
"stopword_suggestions": 0,
"emit_markdown": args.emit_markdown,
}
print(json.dumps(summary, ensure_ascii=False, indent=2))
if __name__ == "__main__":
main()