Files
reader/scripts/generate_term_cleanup_semantic_suggestions.py
T
root 590d050218 keyword cleanup: v2 engine, alias rule layer, LLM semantic suggestions
- build_review_bundle.py: 新增 _compute_percentile/_compute_growth,
  候选池从固定阈值改为百分位排名 + 增速因子 (v2 policy)
- term_cleanup_policy.json: 升级 v2 schema
- generate_term_cleanup_suggestions.py: 新增 _prepare_alias_suggestions,
  规则层输出 alias (大小写/单复数/分词变体)
- generate_term_cleanup_semantic_suggestions.py: 新增 LLM 语义建议脚本
  (DeepSeek API, 产出 semantic alias/stopword/promote)
- SKILL.md: 更新为 5 Phase 工作流程
- 首轮清洗 apply: interest 54, aliases 17组, stopwords 17个
- docs/design/keyword-cleanup-flow-overview.md: 流程文档
- plans/: 引擎设计方案
2026-05-14 17:17:49 +08:00

315 lines
12 KiB
Python
Executable File

#!/usr/bin/env python3
"""
Generate semantic keyword suggestions using LLM.
Covers what surface-form rules cannot:
- semantic alias (abbreviation ↔ full name, Chinese ↔ English, synonym)
- stopword (overly broad / low-discrimination terms)
- promote (new term that aligns with user's focus areas)
Usage:
python scripts/generate_term_cleanup_semantic_suggestions.py \
--bundle outputs/term_index/review/keyword-cleanup-bundle.json \
--output outputs/term_index/review/term-cleanup-semantic-suggestions-2026-05-14.json
"""
from __future__ import annotations
import argparse
import json
import os
import re
import sys
import time
from datetime import datetime, timezone
from pathlib import Path
from typing import Any
from urllib.request import Request, urlopen
REPO_ROOT = Path(__file__).resolve().parents[1]
DEFAULT_BUNDLE_PATH = REPO_ROOT / "outputs" / "term_index" / "review" / "keyword-cleanup-bundle.json"
DEFAULT_OUTPUT_DIR = REPO_ROOT / "outputs" / "term_index" / "review"
def _load_json(path: Path) -> Any:
return json.loads(path.read_text(encoding="utf-8-sig"))
def _save_json(path: Path, payload: dict[str, Any]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
def _load_env(path: Path) -> dict[str, str]:
"""Load key=value pairs from .env file."""
env: dict[str, str] = {}
if not path.exists():
return env
for line in path.read_text(encoding="utf-8").splitlines():
line = line.strip()
if not line or line.startswith("#") or "=" not in line:
continue
key, _, value = line.partition("=")
env[key.strip()] = value.strip().strip("\"'")
return env
def _build_prompt(
interest_keywords: list[str],
rule_alias_suggestions: list[dict[str, str]],
candidate_terms: list[dict[str, Any]],
relevant_watch_terms: list[dict[str, Any]],
) -> str:
"""Build the LLM prompt for semantic suggestions."""
interest_bullets = "\n".join(f" - {t}" for t in sorted(interest_keywords))
candidate_bullets = "\n".join(
f" - {t['term']} (count={t['total_count']}, days={t['days_seen']})"
for t in candidate_terms[:40]
)
# Alias from rule layer (for LLM to build on, not duplicate)
rule_alias_text = ""
if rule_alias_suggestions:
rule_alias_text = "\nSurface-form alias (already identified, skip these):\n" + "\n".join(
f" {a['from']} → {a['to']} ({a['reason']})"
for a in rule_alias_suggestions
)
watch_text = ""
if relevant_watch_terms:
watch_text = "\nWatch terms (low-frequency but potentially relevant):\n" + "\n".join(
f" {t['term']} (count={t['total_count']}, days={t['days_seen']})"
for t in relevant_watch_terms[:20]
)
return f"""You are a keyword governance assistant for an AI engineer. Your job is to analyze keyword data and produce structured suggestions.
## User's focus areas
- AI Agent engineering (Skills, Harness, MCP, Agent architecture)
- Backend engineering (Java, Go, Kubernetes, MySQL, distributed systems)
- Open source AI tools and practices (Claude Code, Cursor, DeepSeek, OpenClaw)
- LLM application engineering (context engineering, RAG, prompt engineering)
## Interest keywords (52 already configured)
{interest_bullets}
## Uncovered candidate terms (sorted by frequency)
{candidate_bullets}
{watch_text}{rule_alias_text}
## Task
Analyze the candidate terms and output a JSON object with exactly three keys:
1. "semantic_alias": array of alias suggestions that SURFACE RULES CAN'T CATCH (e.g. abbreviation↔full name, Chinese↔English, different naming for the same concept).
Format: [{{"from": "<variant>", "to": "<canonical interest keyword>", "reason": "<why>"}}]
2. "stopword": array of terms that are too broad/generic to be useful as filters. A stopword is a term that appears frequently but has LOW DISCRIMINATION — it matches too many unrelated articles and clutters the keyword index.
Format: [{{"term": "<term>", "reason": "<why it should be a stopword>"}}]
3. "promote_to_interest": array of uncovered terms that align well with the user's focus areas and should be added as interest keywords.
Format: [{{"term": "<term>", "reason": "<why it fits>"}}]
## Rules
- Be conservative. When in doubt, leave it out.
- Only suggest alias for terms that clearly refer to the SAME concept as an existing interest keyword.
- Only suggest stopword for terms that are genuinely too broad (appear in many unrelated contexts).
- Only suggest promote for terms that clearly match the user's stated focus areas.
- Output valid JSON only, no markdown, no explanation outside the JSON."""
def _call_llm(prompt: str, api_url: str, model: str, api_key: str) -> str:
"""Call LLM API and return the response text."""
payload = json.dumps({
"model": model,
"messages": [{"role": "user", "content": prompt}],
"temperature": 0.1,
"max_tokens": 2048,
}).encode("utf-8")
req = Request(
api_url.rstrip("/") + "/chat/completions",
data=payload,
headers={
"Content-Type": "application/json",
"Authorization": f"Bearer {api_key}",
},
)
max_retries = 3
for attempt in range(max_retries):
try:
with urlopen(req, timeout=120) as resp:
result = json.loads(resp.read().decode("utf-8"))
return result["choices"][0]["message"]["content"]
except Exception as e:
if attempt < max_retries - 1:
wait = 2 ** attempt
print(f" LLM call failed (attempt {attempt+1}/{max_retries}): {e}", file=sys.stderr)
print(f" Retrying in {wait}s...", file=sys.stderr)
time.sleep(wait)
else:
raise
def _parse_llm_response(text: str) -> dict[str, list[dict[str, str]]]:
"""Extract JSON from LLM response (may contain markdown fences)."""
# Try to find JSON block
json_match = re.search(r"```(?:json)?\s*\n?(\{.*?\})\s*\n?```", text, re.DOTALL)
if json_match:
text = json_match.group(1)
# Clean up: remove any text before { or after }
start = text.find("{")
end = text.rfind("}")
if start >= 0 and end > start:
text = text[start : end + 1]
try:
result = json.loads(text)
except json.JSONDecodeError:
# Try partial recovery
print(f" Warning: LLM response not clean JSON, attempting recovery", file=sys.stderr)
print(f" Raw: {text[:500]}", file=sys.stderr)
return {"semantic_alias": [], "stopword": [], "promote_to_interest": []}
# Normalize keys
normalized = {
"semantic_alias": result.get("semantic_alias", result.get("alias", [])),
"stopword": result.get("stopword", result.get("stopword_suggestions", [])),
"promote_to_interest": result.get("promote_to_interest", result.get("promote", [])),
}
# Ensure each is a list
for key in normalized:
if not isinstance(normalized[key], list):
normalized[key] = []
return normalized
def main() -> None:
parser = argparse.ArgumentParser(description="Generate semantic keyword suggestions via LLM.")
parser.add_argument("--bundle", type=Path, default=DEFAULT_BUNDLE_PATH, help="Review bundle JSON path")
parser.add_argument("--suggestions", type=Path, default=None, help="Existing suggestions JSON (for rule alias context)")
parser.add_argument("--output", type=Path, default=None, help="Output JSON path (auto-generated if omitted)")
parser.add_argument("--llm-api-url", type=str, default=None, help="LLM API base URL")
parser.add_argument("--llm-model", type=str, default=None, help="LLM model name")
parser.add_argument("--llm-api-key", type=str, default=None, help="LLM API key")
parser.add_argument("--dry-run", action="store_true", help="Print prompt and exit without calling LLM")
args = parser.parse_args()
# Load config
env_path = REPO_ROOT / ".env"
env = _load_env(env_path) if env_path.exists() else {}
api_url = args.llm_api_url or os.environ.get("LLM_API_URL") or env.get("LLM_API_URL", "https://api.deepseek.com")
# Map OpenClaw model aliases to actual API model names
model_raw = args.llm_model or os.environ.get("LLM_MODEL") or env.get("LLM_MODEL", "deepseek-chat")
MODEL_ALIAS_MAP = {
"deepseek/deepseek-v4-flash": "deepseek-chat",
"deepseek/deepseek-chat": "deepseek-chat",
"deepseek-v4-flash": "deepseek-chat",
"deepseek-chat": "deepseek-chat",
}
model = MODEL_ALIAS_MAP.get(model_raw, model_raw)
api_key = args.llm_api_key or os.environ.get("LLM_API_KEY") or env.get("LLM_API_KEY", "")
if not api_key:
print("Error: No LLM API key found. Set LLM_API_KEY in .env or pass --llm-api-key.", file=sys.stderr)
sys.exit(1)
# Load bundle
if not args.bundle.exists():
print(f"Error: Bundle not found: {args.bundle}", file=sys.stderr)
sys.exit(1)
bundle = _load_json(args.bundle)
current_config = bundle.get("current_config", {})
interest_keywords = current_config.get("interest_keywords", [])
top_global_terms = bundle.get("top_global_terms", [])
governance_hints = bundle.get("governance_hints", {})
# Build candidate list (uncovered terms from interest + watch candidates)
candidate_terms = []
for item in governance_hints.get("interest_review_candidates", []):
if isinstance(item, dict):
candidate_terms.append({
"term": item.get("term", ""),
"total_count": item.get("total_count", 0),
"days_seen": item.get("days_seen", 0),
"percentile": item.get("percentile", 0),
"growth": item.get("growth", 0),
})
for item in governance_hints.get("watch_review_candidates", []):
if isinstance(item, dict):
# Avoid duplicates
if not any(c["term"] == item.get("term") for c in candidate_terms):
candidate_terms.append({
"term": item.get("term", ""),
"total_count": item.get("total_count", 0),
"days_seen": item.get("days_seen", 0),
"percentile": item.get("percentile", 0),
"growth": item.get("growth", 0),
})
# Sort by total_count descending
candidate_terms.sort(key=lambda x: -x["total_count"])
relevant_watch_terms = governance_hints.get("watch_review_candidates", [])[:20]
# Load rule-layer alias suggestions if available
rule_alias = []
if args.suggestions and args.suggestions.exists():
s = _load_json(args.suggestions)
rule_alias = s.get("alias_suggestions", [])
# Build prompt
prompt = _build_prompt(
interest_keywords=interest_keywords,
rule_alias_suggestions=rule_alias,
candidate_terms=candidate_terms,
relevant_watch_terms=relevant_watch_terms,
)
# Determine output path
suggestion_date = datetime.now(timezone.utc).date().isoformat()
output_path = args.output or (DEFAULT_OUTPUT_DIR / f"term-cleanup-semantic-suggestions-{suggestion_date}.json")
if args.dry_run:
print("=== DRY RUN: Prompt ===")
print(prompt)
print("\n=== END ===")
print(f"\nWould write to: {output_path}")
return
# Call LLM
print(f"Calling LLM ({model})...", file=sys.stderr)
response = _call_llm(prompt, api_url, model, api_key)
print(f"LLM response received ({len(response)} chars)", file=sys.stderr)
# Parse
parsed = _parse_llm_response(response)
# Build output
output = {
"date": suggestion_date,
"source_bundle": str(args.bundle),
"model": model,
"interest_keyword_count": len(interest_keywords),
"candidate_count": len(candidate_terms),
**parsed,
}
_save_json(output_path, output)
summary = {
"output": str(output_path),
"semantic_alias": len(output.get("semantic_alias", [])),
"stopword": len(output.get("stopword", [])),
"promote_to_interest": len(output.get("promote_to_interest", [])),
}
print(json.dumps(summary, ensure_ascii=False, indent=2))
if __name__ == "__main__":
main()