- build_review_bundle.py: 新增 _compute_percentile/_compute_growth, 候选池从固定阈值改为百分位排名 + 增速因子 (v2 policy) - term_cleanup_policy.json: 升级 v2 schema - generate_term_cleanup_suggestions.py: 新增 _prepare_alias_suggestions, 规则层输出 alias (大小写/单复数/分词变体) - generate_term_cleanup_semantic_suggestions.py: 新增 LLM 语义建议脚本 (DeepSeek API, 产出 semantic alias/stopword/promote) - SKILL.md: 更新为 5 Phase 工作流程 - 首轮清洗 apply: interest 54, aliases 17组, stopwords 17个 - docs/design/keyword-cleanup-flow-overview.md: 流程文档 - plans/: 引擎设计方案
315 lines
12 KiB
Python
Executable File
315 lines
12 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""
|
|
Generate semantic keyword suggestions using LLM.
|
|
|
|
Covers what surface-form rules cannot:
|
|
- semantic alias (abbreviation ↔ full name, Chinese ↔ English, synonym)
|
|
- stopword (overly broad / low-discrimination terms)
|
|
- promote (new term that aligns with user's focus areas)
|
|
|
|
Usage:
|
|
python scripts/generate_term_cleanup_semantic_suggestions.py \
|
|
--bundle outputs/term_index/review/keyword-cleanup-bundle.json \
|
|
--output outputs/term_index/review/term-cleanup-semantic-suggestions-2026-05-14.json
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import os
|
|
import re
|
|
import sys
|
|
import time
|
|
from datetime import datetime, timezone
|
|
from pathlib import Path
|
|
from typing import Any
|
|
from urllib.request import Request, urlopen
|
|
|
|
|
|
REPO_ROOT = Path(__file__).resolve().parents[1]
|
|
DEFAULT_BUNDLE_PATH = REPO_ROOT / "outputs" / "term_index" / "review" / "keyword-cleanup-bundle.json"
|
|
DEFAULT_OUTPUT_DIR = REPO_ROOT / "outputs" / "term_index" / "review"
|
|
|
|
|
|
def _load_json(path: Path) -> Any:
|
|
return json.loads(path.read_text(encoding="utf-8-sig"))
|
|
|
|
|
|
def _save_json(path: Path, payload: dict[str, Any]) -> None:
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
|
|
|
|
|
def _load_env(path: Path) -> dict[str, str]:
|
|
"""Load key=value pairs from .env file."""
|
|
env: dict[str, str] = {}
|
|
if not path.exists():
|
|
return env
|
|
for line in path.read_text(encoding="utf-8").splitlines():
|
|
line = line.strip()
|
|
if not line or line.startswith("#") or "=" not in line:
|
|
continue
|
|
key, _, value = line.partition("=")
|
|
env[key.strip()] = value.strip().strip("\"'")
|
|
return env
|
|
|
|
|
|
def _build_prompt(
|
|
interest_keywords: list[str],
|
|
rule_alias_suggestions: list[dict[str, str]],
|
|
candidate_terms: list[dict[str, Any]],
|
|
relevant_watch_terms: list[dict[str, Any]],
|
|
) -> str:
|
|
"""Build the LLM prompt for semantic suggestions."""
|
|
|
|
interest_bullets = "\n".join(f" - {t}" for t in sorted(interest_keywords))
|
|
candidate_bullets = "\n".join(
|
|
f" - {t['term']} (count={t['total_count']}, days={t['days_seen']})"
|
|
for t in candidate_terms[:40]
|
|
)
|
|
|
|
# Alias from rule layer (for LLM to build on, not duplicate)
|
|
rule_alias_text = ""
|
|
if rule_alias_suggestions:
|
|
rule_alias_text = "\nSurface-form alias (already identified, skip these):\n" + "\n".join(
|
|
f" {a['from']} → {a['to']} ({a['reason']})"
|
|
for a in rule_alias_suggestions
|
|
)
|
|
|
|
watch_text = ""
|
|
if relevant_watch_terms:
|
|
watch_text = "\nWatch terms (low-frequency but potentially relevant):\n" + "\n".join(
|
|
f" {t['term']} (count={t['total_count']}, days={t['days_seen']})"
|
|
for t in relevant_watch_terms[:20]
|
|
)
|
|
|
|
return f"""You are a keyword governance assistant for an AI engineer. Your job is to analyze keyword data and produce structured suggestions.
|
|
|
|
## User's focus areas
|
|
- AI Agent engineering (Skills, Harness, MCP, Agent architecture)
|
|
- Backend engineering (Java, Go, Kubernetes, MySQL, distributed systems)
|
|
- Open source AI tools and practices (Claude Code, Cursor, DeepSeek, OpenClaw)
|
|
- LLM application engineering (context engineering, RAG, prompt engineering)
|
|
|
|
## Interest keywords (52 already configured)
|
|
{interest_bullets}
|
|
|
|
## Uncovered candidate terms (sorted by frequency)
|
|
{candidate_bullets}
|
|
{watch_text}{rule_alias_text}
|
|
|
|
## Task
|
|
Analyze the candidate terms and output a JSON object with exactly three keys:
|
|
|
|
1. "semantic_alias": array of alias suggestions that SURFACE RULES CAN'T CATCH (e.g. abbreviation↔full name, Chinese↔English, different naming for the same concept).
|
|
Format: [{{"from": "<variant>", "to": "<canonical interest keyword>", "reason": "<why>"}}]
|
|
|
|
2. "stopword": array of terms that are too broad/generic to be useful as filters. A stopword is a term that appears frequently but has LOW DISCRIMINATION — it matches too many unrelated articles and clutters the keyword index.
|
|
Format: [{{"term": "<term>", "reason": "<why it should be a stopword>"}}]
|
|
|
|
3. "promote_to_interest": array of uncovered terms that align well with the user's focus areas and should be added as interest keywords.
|
|
Format: [{{"term": "<term>", "reason": "<why it fits>"}}]
|
|
|
|
## Rules
|
|
- Be conservative. When in doubt, leave it out.
|
|
- Only suggest alias for terms that clearly refer to the SAME concept as an existing interest keyword.
|
|
- Only suggest stopword for terms that are genuinely too broad (appear in many unrelated contexts).
|
|
- Only suggest promote for terms that clearly match the user's stated focus areas.
|
|
- Output valid JSON only, no markdown, no explanation outside the JSON."""
|
|
|
|
|
|
def _call_llm(prompt: str, api_url: str, model: str, api_key: str) -> str:
|
|
"""Call LLM API and return the response text."""
|
|
payload = json.dumps({
|
|
"model": model,
|
|
"messages": [{"role": "user", "content": prompt}],
|
|
"temperature": 0.1,
|
|
"max_tokens": 2048,
|
|
}).encode("utf-8")
|
|
|
|
req = Request(
|
|
api_url.rstrip("/") + "/chat/completions",
|
|
data=payload,
|
|
headers={
|
|
"Content-Type": "application/json",
|
|
"Authorization": f"Bearer {api_key}",
|
|
},
|
|
)
|
|
|
|
max_retries = 3
|
|
for attempt in range(max_retries):
|
|
try:
|
|
with urlopen(req, timeout=120) as resp:
|
|
result = json.loads(resp.read().decode("utf-8"))
|
|
return result["choices"][0]["message"]["content"]
|
|
except Exception as e:
|
|
if attempt < max_retries - 1:
|
|
wait = 2 ** attempt
|
|
print(f" LLM call failed (attempt {attempt+1}/{max_retries}): {e}", file=sys.stderr)
|
|
print(f" Retrying in {wait}s...", file=sys.stderr)
|
|
time.sleep(wait)
|
|
else:
|
|
raise
|
|
|
|
|
|
def _parse_llm_response(text: str) -> dict[str, list[dict[str, str]]]:
|
|
"""Extract JSON from LLM response (may contain markdown fences)."""
|
|
# Try to find JSON block
|
|
json_match = re.search(r"```(?:json)?\s*\n?(\{.*?\})\s*\n?```", text, re.DOTALL)
|
|
if json_match:
|
|
text = json_match.group(1)
|
|
|
|
# Clean up: remove any text before { or after }
|
|
start = text.find("{")
|
|
end = text.rfind("}")
|
|
if start >= 0 and end > start:
|
|
text = text[start : end + 1]
|
|
|
|
try:
|
|
result = json.loads(text)
|
|
except json.JSONDecodeError:
|
|
# Try partial recovery
|
|
print(f" Warning: LLM response not clean JSON, attempting recovery", file=sys.stderr)
|
|
print(f" Raw: {text[:500]}", file=sys.stderr)
|
|
return {"semantic_alias": [], "stopword": [], "promote_to_interest": []}
|
|
|
|
# Normalize keys
|
|
normalized = {
|
|
"semantic_alias": result.get("semantic_alias", result.get("alias", [])),
|
|
"stopword": result.get("stopword", result.get("stopword_suggestions", [])),
|
|
"promote_to_interest": result.get("promote_to_interest", result.get("promote", [])),
|
|
}
|
|
# Ensure each is a list
|
|
for key in normalized:
|
|
if not isinstance(normalized[key], list):
|
|
normalized[key] = []
|
|
return normalized
|
|
|
|
|
|
def main() -> None:
|
|
parser = argparse.ArgumentParser(description="Generate semantic keyword suggestions via LLM.")
|
|
parser.add_argument("--bundle", type=Path, default=DEFAULT_BUNDLE_PATH, help="Review bundle JSON path")
|
|
parser.add_argument("--suggestions", type=Path, default=None, help="Existing suggestions JSON (for rule alias context)")
|
|
parser.add_argument("--output", type=Path, default=None, help="Output JSON path (auto-generated if omitted)")
|
|
parser.add_argument("--llm-api-url", type=str, default=None, help="LLM API base URL")
|
|
parser.add_argument("--llm-model", type=str, default=None, help="LLM model name")
|
|
parser.add_argument("--llm-api-key", type=str, default=None, help="LLM API key")
|
|
parser.add_argument("--dry-run", action="store_true", help="Print prompt and exit without calling LLM")
|
|
args = parser.parse_args()
|
|
|
|
# Load config
|
|
env_path = REPO_ROOT / ".env"
|
|
env = _load_env(env_path) if env_path.exists() else {}
|
|
|
|
api_url = args.llm_api_url or os.environ.get("LLM_API_URL") or env.get("LLM_API_URL", "https://api.deepseek.com")
|
|
# Map OpenClaw model aliases to actual API model names
|
|
model_raw = args.llm_model or os.environ.get("LLM_MODEL") or env.get("LLM_MODEL", "deepseek-chat")
|
|
MODEL_ALIAS_MAP = {
|
|
"deepseek/deepseek-v4-flash": "deepseek-chat",
|
|
"deepseek/deepseek-chat": "deepseek-chat",
|
|
"deepseek-v4-flash": "deepseek-chat",
|
|
"deepseek-chat": "deepseek-chat",
|
|
}
|
|
model = MODEL_ALIAS_MAP.get(model_raw, model_raw)
|
|
api_key = args.llm_api_key or os.environ.get("LLM_API_KEY") or env.get("LLM_API_KEY", "")
|
|
|
|
if not api_key:
|
|
print("Error: No LLM API key found. Set LLM_API_KEY in .env or pass --llm-api-key.", file=sys.stderr)
|
|
sys.exit(1)
|
|
|
|
# Load bundle
|
|
if not args.bundle.exists():
|
|
print(f"Error: Bundle not found: {args.bundle}", file=sys.stderr)
|
|
sys.exit(1)
|
|
|
|
bundle = _load_json(args.bundle)
|
|
current_config = bundle.get("current_config", {})
|
|
interest_keywords = current_config.get("interest_keywords", [])
|
|
top_global_terms = bundle.get("top_global_terms", [])
|
|
governance_hints = bundle.get("governance_hints", {})
|
|
|
|
# Build candidate list (uncovered terms from interest + watch candidates)
|
|
candidate_terms = []
|
|
for item in governance_hints.get("interest_review_candidates", []):
|
|
if isinstance(item, dict):
|
|
candidate_terms.append({
|
|
"term": item.get("term", ""),
|
|
"total_count": item.get("total_count", 0),
|
|
"days_seen": item.get("days_seen", 0),
|
|
"percentile": item.get("percentile", 0),
|
|
"growth": item.get("growth", 0),
|
|
})
|
|
for item in governance_hints.get("watch_review_candidates", []):
|
|
if isinstance(item, dict):
|
|
# Avoid duplicates
|
|
if not any(c["term"] == item.get("term") for c in candidate_terms):
|
|
candidate_terms.append({
|
|
"term": item.get("term", ""),
|
|
"total_count": item.get("total_count", 0),
|
|
"days_seen": item.get("days_seen", 0),
|
|
"percentile": item.get("percentile", 0),
|
|
"growth": item.get("growth", 0),
|
|
})
|
|
|
|
# Sort by total_count descending
|
|
candidate_terms.sort(key=lambda x: -x["total_count"])
|
|
relevant_watch_terms = governance_hints.get("watch_review_candidates", [])[:20]
|
|
|
|
# Load rule-layer alias suggestions if available
|
|
rule_alias = []
|
|
if args.suggestions and args.suggestions.exists():
|
|
s = _load_json(args.suggestions)
|
|
rule_alias = s.get("alias_suggestions", [])
|
|
|
|
# Build prompt
|
|
prompt = _build_prompt(
|
|
interest_keywords=interest_keywords,
|
|
rule_alias_suggestions=rule_alias,
|
|
candidate_terms=candidate_terms,
|
|
relevant_watch_terms=relevant_watch_terms,
|
|
)
|
|
|
|
# Determine output path
|
|
suggestion_date = datetime.now(timezone.utc).date().isoformat()
|
|
output_path = args.output or (DEFAULT_OUTPUT_DIR / f"term-cleanup-semantic-suggestions-{suggestion_date}.json")
|
|
|
|
if args.dry_run:
|
|
print("=== DRY RUN: Prompt ===")
|
|
print(prompt)
|
|
print("\n=== END ===")
|
|
print(f"\nWould write to: {output_path}")
|
|
return
|
|
|
|
# Call LLM
|
|
print(f"Calling LLM ({model})...", file=sys.stderr)
|
|
response = _call_llm(prompt, api_url, model, api_key)
|
|
print(f"LLM response received ({len(response)} chars)", file=sys.stderr)
|
|
|
|
# Parse
|
|
parsed = _parse_llm_response(response)
|
|
|
|
# Build output
|
|
output = {
|
|
"date": suggestion_date,
|
|
"source_bundle": str(args.bundle),
|
|
"model": model,
|
|
"interest_keyword_count": len(interest_keywords),
|
|
"candidate_count": len(candidate_terms),
|
|
**parsed,
|
|
}
|
|
|
|
_save_json(output_path, output)
|
|
|
|
summary = {
|
|
"output": str(output_path),
|
|
"semantic_alias": len(output.get("semantic_alias", [])),
|
|
"stopword": len(output.get("stopword", [])),
|
|
"promote_to_interest": len(output.get("promote_to_interest", [])),
|
|
}
|
|
print(json.dumps(summary, ensure_ascii=False, indent=2))
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|