Add keyword cleanup governance workflow
This commit is contained in:
@@ -0,0 +1,340 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
from collections import defaultdict
|
||||
from datetime import UTC, datetime
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
DEFAULT_POLICY: dict[str, Any] = {
|
||||
"schema_version": "v1",
|
||||
"interest_keyword_review": {
|
||||
"min_total_count": 3,
|
||||
"min_days_seen": 2,
|
||||
},
|
||||
"watch_term_review": {
|
||||
"min_total_count": 1,
|
||||
"min_days_seen": 1,
|
||||
"max_total_count": 2,
|
||||
"max_days_seen": 2,
|
||||
},
|
||||
"alias_review": {
|
||||
"min_total_count": 2,
|
||||
"min_days_seen": 2,
|
||||
},
|
||||
"stopword_review": {
|
||||
"max_total_count": 2,
|
||||
"max_days_seen": 2,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _load_json(path: Path) -> Any:
|
||||
return json.loads(path.read_text(encoding="utf-8-sig"))
|
||||
|
||||
|
||||
def _save_json(path: Path, payload: dict[str, Any]) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||
|
||||
|
||||
def _load_mapping(path: Path) -> dict[str, str]:
|
||||
if not path.exists():
|
||||
return {}
|
||||
payload = _load_json(path)
|
||||
return payload if isinstance(payload, dict) else {}
|
||||
|
||||
|
||||
def _load_list(path: Path) -> list[str]:
|
||||
if not path.exists():
|
||||
return []
|
||||
payload = _load_json(path)
|
||||
if isinstance(payload, list):
|
||||
return [item for item in payload if isinstance(item, str)]
|
||||
return []
|
||||
|
||||
|
||||
def _load_interest_keywords(path: Path) -> list[str]:
|
||||
if not path.exists():
|
||||
return []
|
||||
payload = _load_json(path)
|
||||
if not isinstance(payload, dict):
|
||||
return []
|
||||
values = payload.get("interest_keywords")
|
||||
if not isinstance(values, list):
|
||||
return []
|
||||
return [item for item in values if isinstance(item, str)]
|
||||
|
||||
|
||||
def _casefold_set(values: list[str]) -> set[str]:
|
||||
return {value.strip().casefold() for value in values if value.strip()}
|
||||
|
||||
|
||||
def _load_policy(path: Path) -> dict[str, Any]:
|
||||
if not path.exists():
|
||||
return DEFAULT_POLICY.copy()
|
||||
payload = _load_json(path)
|
||||
return payload if isinstance(payload, dict) else DEFAULT_POLICY.copy()
|
||||
|
||||
|
||||
def _load_watchlist(path: Path) -> list[dict[str, Any]]:
|
||||
if not path.exists():
|
||||
return []
|
||||
payload = _load_json(path)
|
||||
if not isinstance(payload, dict):
|
||||
return []
|
||||
terms = payload.get("terms")
|
||||
if not isinstance(terms, list):
|
||||
return []
|
||||
return [item for item in terms if isinstance(item, dict) and isinstance(item.get("term"), str)]
|
||||
|
||||
|
||||
def _load_change_log(path: Path) -> list[dict[str, Any]]:
|
||||
if not path.exists():
|
||||
return []
|
||||
payload = _load_json(path)
|
||||
if not isinstance(payload, dict):
|
||||
return []
|
||||
entries = payload.get("entries")
|
||||
if not isinstance(entries, list):
|
||||
return []
|
||||
return [item for item in entries if isinstance(item, dict)]
|
||||
|
||||
|
||||
def _meets_min_thresholds(item: dict[str, Any], thresholds: dict[str, Any]) -> bool:
|
||||
min_total_count = int(thresholds.get("min_total_count", 1))
|
||||
min_days_seen = int(thresholds.get("min_days_seen", 1))
|
||||
total_count = int(item.get("total_count") or 0)
|
||||
days_seen = int(item.get("days_seen") or 0)
|
||||
return total_count >= min_total_count and days_seen >= min_days_seen
|
||||
|
||||
|
||||
def _within_watch_thresholds(item: dict[str, Any], thresholds: dict[str, Any]) -> bool:
|
||||
min_total_count = int(thresholds.get("min_total_count", 1))
|
||||
min_days_seen = int(thresholds.get("min_days_seen", 1))
|
||||
max_total_count = int(thresholds.get("max_total_count", 999999))
|
||||
max_days_seen = int(thresholds.get("max_days_seen", 999999))
|
||||
total_count = int(item.get("total_count") or 0)
|
||||
days_seen = int(item.get("days_seen") or 0)
|
||||
return (
|
||||
total_count >= min_total_count
|
||||
and days_seen >= min_days_seen
|
||||
and total_count <= max_total_count
|
||||
and days_seen <= max_days_seen
|
||||
)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Build a compact review bundle for the keyword-cleanup-review skill."
|
||||
)
|
||||
repo_root = Path(__file__).resolve().parents[3]
|
||||
parser.add_argument(
|
||||
"--stats",
|
||||
type=Path,
|
||||
default=repo_root / "data" / "term_index" / "term_stats.json",
|
||||
help="Global keyword stats file",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--daily-dir",
|
||||
type=Path,
|
||||
default=repo_root / "data" / "term_index" / "daily",
|
||||
help="Directory containing daily keyword index JSON files",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--aliases",
|
||||
type=Path,
|
||||
default=repo_root / "configs" / "term_aliases.json",
|
||||
help="Term aliases JSON file",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--stopwords",
|
||||
type=Path,
|
||||
default=repo_root / "configs" / "term_stopwords.json",
|
||||
help="Term stopwords JSON file",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--context",
|
||||
type=Path,
|
||||
default=repo_root / "configs" / "filter_context.personal.json",
|
||||
help="Personal filter context JSON file",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--policy",
|
||||
type=Path,
|
||||
default=repo_root / "configs" / "term_cleanup_policy.json",
|
||||
help="Keyword cleanup policy JSON file",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--watchlist",
|
||||
type=Path,
|
||||
default=repo_root / "configs" / "term_watchlist.json",
|
||||
help="Tracked watch terms JSON file",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--change-log",
|
||||
type=Path,
|
||||
default=repo_root / "configs" / "term_change_log.json",
|
||||
help="Applied keyword cleanup change log JSON file",
|
||||
)
|
||||
parser.add_argument("--days", type=int, default=7, help="How many recent daily files to include")
|
||||
parser.add_argument("--top", type=int, default=50, help="How many top global terms to include")
|
||||
parser.add_argument(
|
||||
"--output",
|
||||
type=Path,
|
||||
default=repo_root / "outputs" / "term_index" / "review" / "keyword-cleanup-bundle.json",
|
||||
help="Output JSON file",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
stats_payload = _load_json(args.stats) if args.stats.exists() else {"terms": []}
|
||||
stats_terms = stats_payload.get("terms") if isinstance(stats_payload, dict) else []
|
||||
if not isinstance(stats_terms, list):
|
||||
stats_terms = []
|
||||
|
||||
aliases = _load_mapping(args.aliases)
|
||||
stopwords = _load_list(args.stopwords)
|
||||
interest_keywords = _load_interest_keywords(args.context)
|
||||
policy = _load_policy(args.policy)
|
||||
watchlist = _load_watchlist(args.watchlist)
|
||||
change_log = _load_change_log(args.change_log)
|
||||
|
||||
recent_daily_paths = sorted(args.daily_dir.glob("*.json"))[-args.days :] if args.daily_dir.exists() else []
|
||||
recent_daily: list[dict[str, Any]] = []
|
||||
recent_counter: defaultdict[str, int] = defaultdict(int)
|
||||
for path in recent_daily_paths:
|
||||
payload = _load_json(path)
|
||||
if not isinstance(payload, dict):
|
||||
continue
|
||||
terms = payload.get("terms")
|
||||
if not isinstance(terms, list):
|
||||
terms = []
|
||||
compact_terms = []
|
||||
for term in terms:
|
||||
if not isinstance(term, dict):
|
||||
continue
|
||||
normalized = term.get("normalized_term")
|
||||
count = term.get("count")
|
||||
if not isinstance(normalized, str) or not isinstance(count, int):
|
||||
continue
|
||||
compact_terms.append({"term": normalized, "count": count})
|
||||
recent_counter[normalized] += count
|
||||
recent_daily.append(
|
||||
{
|
||||
"date": payload.get("date"),
|
||||
"candidate_count": payload.get("candidate_count"),
|
||||
"top_terms": compact_terms[:20],
|
||||
}
|
||||
)
|
||||
|
||||
interest_set = _casefold_set(interest_keywords)
|
||||
stopword_set = _casefold_set(stopwords)
|
||||
alias_keys = _casefold_set(list(aliases.keys()))
|
||||
alias_values = _casefold_set(list(aliases.values()))
|
||||
watch_set = _casefold_set([str(item.get("term", "")) for item in watchlist])
|
||||
|
||||
top_global_terms = []
|
||||
for item in stats_terms[: args.top]:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
term = item.get("term")
|
||||
if not isinstance(term, str):
|
||||
continue
|
||||
folded = term.strip().casefold()
|
||||
top_global_terms.append(
|
||||
{
|
||||
"term": term,
|
||||
"total_count": item.get("total_count"),
|
||||
"days_seen": item.get("days_seen"),
|
||||
"first_seen": item.get("first_seen"),
|
||||
"last_seen": item.get("last_seen"),
|
||||
"in_interest_keywords": folded in interest_set,
|
||||
"is_stopword": folded in stopword_set,
|
||||
"is_alias_source": folded in alias_keys,
|
||||
"is_alias_target": folded in alias_values,
|
||||
"in_watchlist": folded in watch_set,
|
||||
"recent_count": recent_counter.get(term, 0),
|
||||
}
|
||||
)
|
||||
|
||||
uncovered_terms = [
|
||||
item for item in top_global_terms if not item["in_interest_keywords"] and not item["is_stopword"]
|
||||
][:20]
|
||||
interest_thresholds = policy.get("interest_keyword_review") if isinstance(policy, dict) else {}
|
||||
watch_thresholds = policy.get("watch_term_review") if isinstance(policy, dict) else {}
|
||||
interest_review_candidates = [
|
||||
{
|
||||
"term": item["term"],
|
||||
"total_count": item["total_count"],
|
||||
"days_seen": item["days_seen"],
|
||||
"reason": (
|
||||
"Meets the configured interest-keyword review threshold and is not yet covered "
|
||||
"by interest keywords or stopwords."
|
||||
),
|
||||
}
|
||||
for item in uncovered_terms
|
||||
if not item["in_watchlist"] and _meets_min_thresholds(item, interest_thresholds)
|
||||
][:20]
|
||||
watch_review_candidates = [
|
||||
{
|
||||
"term": item["term"],
|
||||
"total_count": item["total_count"],
|
||||
"days_seen": item["days_seen"],
|
||||
"reason": (
|
||||
"Falls into the configured watch-term review range and should be observed "
|
||||
"before promotion into interest keywords."
|
||||
),
|
||||
}
|
||||
for item in uncovered_terms
|
||||
if not item["in_watchlist"]
|
||||
and not _meets_min_thresholds(item, interest_thresholds)
|
||||
and _within_watch_thresholds(item, watch_thresholds)
|
||||
][:20]
|
||||
recent_hot_terms = sorted(
|
||||
({"term": term, "recent_count": count} for term, count in recent_counter.items()),
|
||||
key=lambda item: (-item["recent_count"], item["term"].casefold(), item["term"]),
|
||||
)[:20]
|
||||
|
||||
bundle = {
|
||||
"generated_at": datetime.now(tz=UTC).isoformat(),
|
||||
"days": args.days,
|
||||
"top": args.top,
|
||||
"sources": {
|
||||
"stats": str(args.stats),
|
||||
"daily_dir": str(args.daily_dir),
|
||||
"aliases": str(args.aliases),
|
||||
"stopwords": str(args.stopwords),
|
||||
"context": str(args.context),
|
||||
"policy": str(args.policy),
|
||||
"watchlist": str(args.watchlist),
|
||||
"change_log": str(args.change_log),
|
||||
},
|
||||
"policy": policy,
|
||||
"current_config": {
|
||||
"alias_count": len(aliases),
|
||||
"stopword_count": len(stopwords),
|
||||
"interest_keyword_count": len(interest_keywords),
|
||||
"watch_term_count": len(watchlist),
|
||||
"aliases": aliases,
|
||||
"stopwords": stopwords,
|
||||
"interest_keywords": interest_keywords,
|
||||
"watchlist": watchlist,
|
||||
},
|
||||
"change_log_tail": change_log[-10:],
|
||||
"top_global_terms": top_global_terms,
|
||||
"recent_daily": recent_daily,
|
||||
"recent_hot_terms": recent_hot_terms,
|
||||
"uncovered_terms": uncovered_terms,
|
||||
"governance_hints": {
|
||||
"interest_review_candidates": interest_review_candidates,
|
||||
"watch_review_candidates": watch_review_candidates,
|
||||
},
|
||||
}
|
||||
_save_json(args.output, bundle)
|
||||
print(f"Saved keyword cleanup bundle to {args.output}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user