Files
reader/scripts/apply_term_suggestions.py

402 lines
16 KiB
Python

from __future__ import annotations
import argparse
import json
from datetime import UTC, datetime
from pathlib import Path
from typing import Any
REPO_ROOT = Path(__file__).resolve().parents[1]
DEFAULT_ALIASES_PATH = REPO_ROOT / "configs" / "term_aliases.json"
DEFAULT_STOPWORDS_PATH = REPO_ROOT / "configs" / "term_stopwords.json"
DEFAULT_CONTEXT_PATH = REPO_ROOT / "configs" / "filter_context.personal.json"
DEFAULT_WATCHLIST_PATH = REPO_ROOT / "configs" / "term_watchlist.json"
DEFAULT_CHANGE_LOG_PATH = REPO_ROOT / "configs" / "term_change_log.json"
def _load_json(path: Path, default: Any) -> Any:
if not path.exists():
return default
return json.loads(path.read_text(encoding="utf-8-sig"))
def _save_json(path: Path, payload: dict[str, Any] | list[Any]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
def _term_key(value: str) -> str:
return value.strip().casefold()
def _utc_now() -> str:
return datetime.now(tz=UTC).isoformat().replace("+00:00", "Z")
def _load_suggestions(path: Path) -> dict[str, Any]:
payload = _load_json(path, {})
if not isinstance(payload, dict):
raise RuntimeError("Suggestions file must contain a JSON object.")
return payload
def _load_aliases(path: Path) -> dict[str, str]:
payload = _load_json(path, {})
if not isinstance(payload, dict):
raise RuntimeError("Aliases file must contain a JSON object.")
result: dict[str, str] = {}
for key, value in payload.items():
if isinstance(key, str) and isinstance(value, str) and key.strip() and value.strip():
result[key] = value
return result
def _load_stopwords(path: Path) -> list[str]:
payload = _load_json(path, [])
if not isinstance(payload, list):
raise RuntimeError("Stopwords file must contain a JSON array.")
return [item for item in payload if isinstance(item, str) and item.strip()]
def _load_context(path: Path) -> dict[str, Any]:
payload = _load_json(path, {})
if not isinstance(payload, dict):
raise RuntimeError("Context file must contain a JSON object.")
payload.setdefault("interest_keywords", [])
if not isinstance(payload["interest_keywords"], list):
raise RuntimeError("filter_context.personal.json interest_keywords must be a JSON array.")
payload["interest_keywords"] = [item for item in payload["interest_keywords"] if isinstance(item, str) and item.strip()]
return payload
def _load_watchlist(path: Path) -> dict[str, Any]:
payload = _load_json(
path,
{
"schema_version": "v1",
"updated_at": _utc_now(),
"terms": [],
},
)
if not isinstance(payload, dict):
raise RuntimeError("Watchlist file must contain a JSON object.")
payload.setdefault("schema_version", "v1")
payload.setdefault("updated_at", _utc_now())
terms = payload.get("terms", [])
if not isinstance(terms, list):
raise RuntimeError("Watchlist terms must be a JSON array.")
payload["terms"] = [item for item in terms if isinstance(item, dict) and isinstance(item.get("term"), str)]
return payload
def _load_change_log(path: Path) -> dict[str, Any]:
payload = _load_json(
path,
{
"schema_version": "v1",
"entries": [],
},
)
if not isinstance(payload, dict):
raise RuntimeError("Change log file must contain a JSON object.")
payload.setdefault("schema_version", "v1")
entries = payload.get("entries", [])
if not isinstance(entries, list):
raise RuntimeError("Change log entries must be a JSON array.")
payload["entries"] = [item for item in entries if isinstance(item, dict)]
return payload
def _build_index(entries: list[dict[str, Any]], field: str) -> dict[str, dict[str, Any]]:
index: dict[str, dict[str, Any]] = {}
for entry in entries:
value = entry.get(field)
if isinstance(value, str) and value.strip():
index[_term_key(value)] = entry
return index
def _append_change(
change_log: dict[str, Any],
*,
applied_at: str,
action: str,
term: str,
value: str | None,
reason: str,
suggestions_path: Path,
suggestion_date: str | None,
based_on_days: int | None,
) -> None:
entry: dict[str, Any] = {
"applied_at": applied_at,
"action": action,
"term": term,
"reason": reason,
"suggestions_path": str(suggestions_path),
}
if value is not None:
entry["value"] = value
if suggestion_date is not None:
entry["suggestion_date"] = suggestion_date
if based_on_days is not None:
entry["based_on_days"] = based_on_days
change_log["entries"].append(entry)
def _remove_from_watchlist(
watchlist: dict[str, Any],
*,
term: str,
) -> bool:
folded = _term_key(term)
original_len = len(watchlist["terms"])
watchlist["terms"] = [item for item in watchlist["terms"] if _term_key(str(item.get("term", ""))) != folded]
return len(watchlist["terms"]) != original_len
def main() -> None:
parser = argparse.ArgumentParser(description="Apply accepted term cleanup suggestions into repo config files.")
parser.add_argument("--suggestions", type=Path, required=True, help="Suggestion JSON file")
parser.add_argument("--aliases", type=Path, default=DEFAULT_ALIASES_PATH, help="Term aliases JSON file")
parser.add_argument("--stopwords", type=Path, default=DEFAULT_STOPWORDS_PATH, help="Term stopwords JSON file")
parser.add_argument("--context", type=Path, default=DEFAULT_CONTEXT_PATH, help="Personal filter context JSON file")
parser.add_argument("--watchlist", type=Path, default=DEFAULT_WATCHLIST_PATH, help="Watchlist JSON file")
parser.add_argument("--change-log", type=Path, default=DEFAULT_CHANGE_LOG_PATH, help="Change log JSON file")
parser.add_argument("--accept-watch", nargs="*", default=[], help="Accept watch_terms by term name")
parser.add_argument(
"--accept-interest", nargs="*", default=[], help="Accept interest_keyword_suggestions by term name"
)
parser.add_argument("--accept-stopword", nargs="*", default=[], help="Accept stopword_suggestions by term name")
parser.add_argument("--accept-alias", nargs="*", default=[], help="Accept alias_suggestions by source term")
parser.add_argument("--dry-run", action="store_true", help="Preview changes without writing files")
args = parser.parse_args()
suggestions = _load_suggestions(args.suggestions)
aliases = _load_aliases(args.aliases)
stopwords = _load_stopwords(args.stopwords)
context = _load_context(args.context)
watchlist = _load_watchlist(args.watchlist)
change_log = _load_change_log(args.change_log)
alias_suggestions = suggestions.get("alias_suggestions", [])
stopword_suggestions = suggestions.get("stopword_suggestions", [])
interest_suggestions = suggestions.get("interest_keyword_suggestions", [])
watch_suggestions = suggestions.get("watch_terms", [])
if not all(isinstance(bucket, list) for bucket in [alias_suggestions, stopword_suggestions, interest_suggestions, watch_suggestions]):
raise RuntimeError("Suggestion JSON buckets must all be arrays.")
alias_index = _build_index([item for item in alias_suggestions if isinstance(item, dict)], "from")
stopword_index = _build_index([item for item in stopword_suggestions if isinstance(item, dict)], "term")
interest_index = _build_index([item for item in interest_suggestions if isinstance(item, dict)], "term")
watch_index = _build_index([item for item in watch_suggestions if isinstance(item, dict)], "term")
watch_map = {_term_key(str(item.get("term", ""))): item for item in watchlist["terms"]}
stopword_set = {_term_key(item) for item in stopwords}
interest_set = {_term_key(item) for item in context["interest_keywords"]}
alias_source_set = {_term_key(key) for key in aliases}
applied_at = _utc_now()
suggestion_date = suggestions.get("date") if isinstance(suggestions.get("date"), str) else None
based_on_days = suggestions.get("based_on_days") if isinstance(suggestions.get("based_on_days"), int) else None
applied: list[dict[str, Any]] = []
skipped: list[dict[str, Any]] = []
def skip(action: str, term: str, reason: str) -> None:
skipped.append({"action": action, "term": term, "reason": reason})
def applied_entry(action: str, term: str, reason: str, value: str | None = None) -> None:
item: dict[str, Any] = {"action": action, "term": term, "reason": reason}
if value is not None:
item["value"] = value
applied.append(item)
for raw_term in args.accept_watch:
term = raw_term.strip()
suggestion = watch_index.get(_term_key(term))
if suggestion is None:
skip("add_watch_term", term, "Term not found in watch_terms suggestions.")
continue
if _term_key(term) in watch_map:
skip("add_watch_term", term, "Term already exists in watchlist.")
continue
reason = str(suggestion.get("reason") or "Accepted from watch_terms suggestion.")
watch_item = {
"term": str(suggestion.get("term") or term),
"added_at": applied_at,
"source": str(args.suggestions),
"reason": reason,
"status": "watching",
}
watchlist["terms"].append(watch_item)
watch_map[_term_key(watch_item["term"])] = watch_item
_append_change(
change_log,
applied_at=applied_at,
action="add_watch_term",
term=watch_item["term"],
value=None,
reason=reason,
suggestions_path=args.suggestions,
suggestion_date=suggestion_date,
based_on_days=based_on_days,
)
applied_entry("add_watch_term", watch_item["term"], reason)
for raw_term in args.accept_interest:
term = raw_term.strip()
suggestion = interest_index.get(_term_key(term))
if suggestion is None:
skip("add_interest_keyword", term, "Term not found in interest_keyword_suggestions.")
continue
resolved_term = str(suggestion.get("term") or term)
if _term_key(resolved_term) in interest_set:
skip("add_interest_keyword", resolved_term, "Term already exists in interest_keywords.")
continue
reason = str(suggestion.get("reason") or "Accepted from interest_keyword_suggestions.")
context["interest_keywords"].append(resolved_term)
interest_set.add(_term_key(resolved_term))
_append_change(
change_log,
applied_at=applied_at,
action="add_interest_keyword",
term=resolved_term,
value=None,
reason=reason,
suggestions_path=args.suggestions,
suggestion_date=suggestion_date,
based_on_days=based_on_days,
)
if _remove_from_watchlist(watchlist, term=resolved_term):
_append_change(
change_log,
applied_at=applied_at,
action="remove_watch_term",
term=resolved_term,
value="promoted_to_interest_keyword",
reason="Removed from watchlist after promotion into interest_keywords.",
suggestions_path=args.suggestions,
suggestion_date=suggestion_date,
based_on_days=based_on_days,
)
applied_entry("add_interest_keyword", resolved_term, reason)
for raw_term in args.accept_stopword:
term = raw_term.strip()
suggestion = stopword_index.get(_term_key(term))
if suggestion is None:
skip("add_stopword", term, "Term not found in stopword_suggestions.")
continue
resolved_term = str(suggestion.get("term") or term)
if _term_key(resolved_term) in stopword_set:
skip("add_stopword", resolved_term, "Term already exists in stopwords.")
continue
reason = str(suggestion.get("reason") or "Accepted from stopword_suggestions.")
stopwords.append(resolved_term)
stopword_set.add(_term_key(resolved_term))
_append_change(
change_log,
applied_at=applied_at,
action="add_stopword",
term=resolved_term,
value=None,
reason=reason,
suggestions_path=args.suggestions,
suggestion_date=suggestion_date,
based_on_days=based_on_days,
)
if _remove_from_watchlist(watchlist, term=resolved_term):
_append_change(
change_log,
applied_at=applied_at,
action="remove_watch_term",
term=resolved_term,
value="promoted_to_stopword",
reason="Removed from watchlist after being added to stopwords.",
suggestions_path=args.suggestions,
suggestion_date=suggestion_date,
based_on_days=based_on_days,
)
applied_entry("add_stopword", resolved_term, reason)
for raw_term in args.accept_alias:
term = raw_term.strip()
suggestion = alias_index.get(_term_key(term))
if suggestion is None:
skip("add_alias", term, "Source term not found in alias_suggestions.")
continue
source_term = str(suggestion.get("from") or term)
target_term = str(suggestion.get("to") or "").strip()
if not target_term:
skip("add_alias", source_term, "Alias suggestion target is empty.")
continue
existing_target = aliases.get(source_term)
if existing_target == target_term:
skip("add_alias", source_term, "Alias already exists with the same target.")
continue
if _term_key(source_term) in alias_source_set and existing_target != target_term:
skip("add_alias", source_term, f"Alias source already exists with a different target: {existing_target}")
continue
reason = str(suggestion.get("reason") or "Accepted from alias_suggestions.")
aliases[source_term] = target_term
alias_source_set.add(_term_key(source_term))
_append_change(
change_log,
applied_at=applied_at,
action="add_alias",
term=source_term,
value=target_term,
reason=reason,
suggestions_path=args.suggestions,
suggestion_date=suggestion_date,
based_on_days=based_on_days,
)
if _remove_from_watchlist(watchlist, term=source_term):
_append_change(
change_log,
applied_at=applied_at,
action="remove_watch_term",
term=source_term,
value="resolved_as_alias_source",
reason="Removed from watchlist after alias mapping was accepted.",
suggestions_path=args.suggestions,
suggestion_date=suggestion_date,
based_on_days=based_on_days,
)
applied_entry("add_alias", source_term, reason, value=target_term)
watchlist["terms"] = sorted(watchlist["terms"], key=lambda item: (_term_key(str(item.get("term", ""))), str(item.get("term", ""))))
watchlist["updated_at"] = applied_at
stopwords = sorted(stopwords, key=lambda item: (_term_key(item), item))
context["interest_keywords"] = sorted(context["interest_keywords"], key=lambda item: (_term_key(item), item))
summary = {
"dry_run": args.dry_run,
"suggestions": str(args.suggestions),
"applied_count": len(applied),
"skipped_count": len(skipped),
"applied": applied,
"skipped": skipped,
"output_paths": {
"aliases": str(args.aliases),
"stopwords": str(args.stopwords),
"context": str(args.context),
"watchlist": str(args.watchlist),
"change_log": str(args.change_log),
},
}
if not args.dry_run:
_save_json(args.aliases, aliases)
_save_json(args.stopwords, stopwords)
_save_json(args.context, context)
_save_json(args.watchlist, watchlist)
_save_json(args.change_log, change_log)
print(json.dumps(summary, ensure_ascii=False, indent=2))
if __name__ == "__main__":
main()