#!/usr/bin/env python3 """ Generate semantic keyword suggestions using LLM. Covers what surface-form rules cannot: - semantic alias (abbreviation ↔ full name, Chinese ↔ English, synonym) - stopword (overly broad / low-discrimination terms) - promote (new term that aligns with user's focus areas) Usage: python scripts/generate_term_cleanup_semantic_suggestions.py \ --bundle outputs/term_index/review/keyword-cleanup-bundle.json \ --output outputs/term_index/review/term-cleanup-semantic-suggestions-2026-05-14.json """ from __future__ import annotations import argparse import json import os import re import sys import time from datetime import datetime, timezone from pathlib import Path from typing import Any from urllib.request import Request, urlopen REPO_ROOT = Path(__file__).resolve().parents[1] DEFAULT_BUNDLE_PATH = REPO_ROOT / "outputs" / "term_index" / "review" / "keyword-cleanup-bundle.json" DEFAULT_OUTPUT_DIR = REPO_ROOT / "outputs" / "term_index" / "review" def _load_json(path: Path) -> Any: return json.loads(path.read_text(encoding="utf-8-sig")) def _save_json(path: Path, payload: dict[str, Any]) -> None: path.parent.mkdir(parents=True, exist_ok=True) path.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") def _load_env(path: Path) -> dict[str, str]: """Load key=value pairs from .env file.""" env: dict[str, str] = {} if not path.exists(): return env for line in path.read_text(encoding="utf-8").splitlines(): line = line.strip() if not line or line.startswith("#") or "=" not in line: continue key, _, value = line.partition("=") env[key.strip()] = value.strip().strip("\"'") return env def _build_prompt( interest_keywords: list[str], rule_alias_suggestions: list[dict[str, str]], candidate_terms: list[dict[str, Any]], relevant_watch_terms: list[dict[str, Any]], ) -> str: """Build the LLM prompt for semantic suggestions.""" interest_bullets = "\n".join(f" - {t}" for t in sorted(interest_keywords)) candidate_bullets = "\n".join( f" - {t['term']} (count={t['total_count']}, days={t['days_seen']})" for t in candidate_terms[:40] ) # Alias from rule layer (for LLM to build on, not duplicate) rule_alias_text = "" if rule_alias_suggestions: rule_alias_text = "\nSurface-form alias (already identified, skip these):\n" + "\n".join( f" {a['from']} → {a['to']} ({a['reason']})" for a in rule_alias_suggestions ) watch_text = "" if relevant_watch_terms: watch_text = "\nWatch terms (low-frequency but potentially relevant):\n" + "\n".join( f" {t['term']} (count={t['total_count']}, days={t['days_seen']})" for t in relevant_watch_terms[:20] ) return f"""You are a keyword governance assistant for an AI engineer. Your job is to analyze keyword data and produce structured suggestions. ## User's focus areas - AI Agent engineering (Skills, Harness, MCP, Agent architecture) - Backend engineering (Java, Go, Kubernetes, MySQL, distributed systems) - Open source AI tools and practices (Claude Code, Cursor, DeepSeek, OpenClaw) - LLM application engineering (context engineering, RAG, prompt engineering) ## Interest keywords (52 already configured) {interest_bullets} ## Uncovered candidate terms (sorted by frequency) {candidate_bullets} {watch_text}{rule_alias_text} ## Task Analyze the candidate terms and output a JSON object with exactly three keys: 1. "semantic_alias": array of alias suggestions that SURFACE RULES CAN'T CATCH (e.g. abbreviation↔full name, Chinese↔English, different naming for the same concept). Format: [{{"from": "", "to": "", "reason": ""}}] 2. "stopword": array of terms that are too broad/generic to be useful as filters. A stopword is a term that appears frequently but has LOW DISCRIMINATION — it matches too many unrelated articles and clutters the keyword index. Format: [{{"term": "", "reason": ""}}] 3. "promote_to_interest": array of uncovered terms that align well with the user's focus areas and should be added as interest keywords. Format: [{{"term": "", "reason": ""}}] ## Rules - Be conservative. When in doubt, leave it out. - Only suggest alias for terms that clearly refer to the SAME concept as an existing interest keyword. - Only suggest stopword for terms that are genuinely too broad (appear in many unrelated contexts). - Only suggest promote for terms that clearly match the user's stated focus areas. - Output valid JSON only, no markdown, no explanation outside the JSON.""" def _call_llm(prompt: str, api_url: str, model: str, api_key: str) -> str: """Call LLM API and return the response text.""" payload = json.dumps({ "model": model, "messages": [{"role": "user", "content": prompt}], "temperature": 0.1, "max_tokens": 2048, }).encode("utf-8") req = Request( api_url.rstrip("/") + "/chat/completions", data=payload, headers={ "Content-Type": "application/json", "Authorization": f"Bearer {api_key}", }, ) max_retries = 3 for attempt in range(max_retries): try: with urlopen(req, timeout=120) as resp: result = json.loads(resp.read().decode("utf-8")) return result["choices"][0]["message"]["content"] except Exception as e: if attempt < max_retries - 1: wait = 2 ** attempt print(f" LLM call failed (attempt {attempt+1}/{max_retries}): {e}", file=sys.stderr) print(f" Retrying in {wait}s...", file=sys.stderr) time.sleep(wait) else: raise def _parse_llm_response(text: str) -> dict[str, list[dict[str, str]]]: """Extract JSON from LLM response (may contain markdown fences).""" # Try to find JSON block json_match = re.search(r"```(?:json)?\s*\n?(\{.*?\})\s*\n?```", text, re.DOTALL) if json_match: text = json_match.group(1) # Clean up: remove any text before { or after } start = text.find("{") end = text.rfind("}") if start >= 0 and end > start: text = text[start : end + 1] try: result = json.loads(text) except json.JSONDecodeError: # Try partial recovery print(f" Warning: LLM response not clean JSON, attempting recovery", file=sys.stderr) print(f" Raw: {text[:500]}", file=sys.stderr) return {"semantic_alias": [], "stopword": [], "promote_to_interest": []} # Normalize keys normalized = { "semantic_alias": result.get("semantic_alias", result.get("alias", [])), "stopword": result.get("stopword", result.get("stopword_suggestions", [])), "promote_to_interest": result.get("promote_to_interest", result.get("promote", [])), } # Ensure each is a list for key in normalized: if not isinstance(normalized[key], list): normalized[key] = [] return normalized def main() -> None: parser = argparse.ArgumentParser(description="Generate semantic keyword suggestions via LLM.") parser.add_argument("--bundle", type=Path, default=DEFAULT_BUNDLE_PATH, help="Review bundle JSON path") parser.add_argument("--suggestions", type=Path, default=None, help="Existing suggestions JSON (for rule alias context)") parser.add_argument("--output", type=Path, default=None, help="Output JSON path (auto-generated if omitted)") parser.add_argument("--llm-api-url", type=str, default=None, help="LLM API base URL") parser.add_argument("--llm-model", type=str, default=None, help="LLM model name") parser.add_argument("--llm-api-key", type=str, default=None, help="LLM API key") parser.add_argument("--dry-run", action="store_true", help="Print prompt and exit without calling LLM") args = parser.parse_args() # Load config env_path = REPO_ROOT / ".env" env = _load_env(env_path) if env_path.exists() else {} api_url = args.llm_api_url or os.environ.get("LLM_API_URL") or env.get("LLM_API_URL", "https://api.deepseek.com") # Map OpenClaw model aliases to actual API model names model_raw = args.llm_model or os.environ.get("LLM_MODEL") or env.get("LLM_MODEL", "deepseek-chat") MODEL_ALIAS_MAP = { "deepseek/deepseek-v4-flash": "deepseek-chat", "deepseek/deepseek-chat": "deepseek-chat", "deepseek-v4-flash": "deepseek-chat", "deepseek-chat": "deepseek-chat", } model = MODEL_ALIAS_MAP.get(model_raw, model_raw) api_key = args.llm_api_key or os.environ.get("LLM_API_KEY") or env.get("LLM_API_KEY", "") if not api_key: print("Error: No LLM API key found. Set LLM_API_KEY in .env or pass --llm-api-key.", file=sys.stderr) sys.exit(1) # Load bundle if not args.bundle.exists(): print(f"Error: Bundle not found: {args.bundle}", file=sys.stderr) sys.exit(1) bundle = _load_json(args.bundle) current_config = bundle.get("current_config", {}) interest_keywords = current_config.get("interest_keywords", []) top_global_terms = bundle.get("top_global_terms", []) governance_hints = bundle.get("governance_hints", {}) # Build candidate list (uncovered terms from interest + watch candidates) candidate_terms = [] for item in governance_hints.get("interest_review_candidates", []): if isinstance(item, dict): candidate_terms.append({ "term": item.get("term", ""), "total_count": item.get("total_count", 0), "days_seen": item.get("days_seen", 0), "percentile": item.get("percentile", 0), "growth": item.get("growth", 0), }) for item in governance_hints.get("watch_review_candidates", []): if isinstance(item, dict): # Avoid duplicates if not any(c["term"] == item.get("term") for c in candidate_terms): candidate_terms.append({ "term": item.get("term", ""), "total_count": item.get("total_count", 0), "days_seen": item.get("days_seen", 0), "percentile": item.get("percentile", 0), "growth": item.get("growth", 0), }) # Sort by total_count descending candidate_terms.sort(key=lambda x: -x["total_count"]) relevant_watch_terms = governance_hints.get("watch_review_candidates", [])[:20] # Load rule-layer alias suggestions if available rule_alias = [] if args.suggestions and args.suggestions.exists(): s = _load_json(args.suggestions) rule_alias = s.get("alias_suggestions", []) # Build prompt prompt = _build_prompt( interest_keywords=interest_keywords, rule_alias_suggestions=rule_alias, candidate_terms=candidate_terms, relevant_watch_terms=relevant_watch_terms, ) # Determine output path suggestion_date = datetime.now(timezone.utc).date().isoformat() output_path = args.output or (DEFAULT_OUTPUT_DIR / f"term-cleanup-semantic-suggestions-{suggestion_date}.json") if args.dry_run: print("=== DRY RUN: Prompt ===") print(prompt) print("\n=== END ===") print(f"\nWould write to: {output_path}") return # Call LLM print(f"Calling LLM ({model})...", file=sys.stderr) response = _call_llm(prompt, api_url, model, api_key) print(f"LLM response received ({len(response)} chars)", file=sys.stderr) # Parse parsed = _parse_llm_response(response) # Build output output = { "date": suggestion_date, "source_bundle": str(args.bundle), "model": model, "interest_keyword_count": len(interest_keywords), "candidate_count": len(candidate_terms), **parsed, } _save_json(output_path, output) summary = { "output": str(output_path), "semantic_alias": len(output.get("semantic_alias", [])), "stopword": len(output.get("stopword", [])), "promote_to_interest": len(output.get("promote_to_interest", [])), } print(json.dumps(summary, ensure_ascii=False, indent=2)) if __name__ == "__main__": main()