keyword cleanup: v2 engine, alias rule layer, LLM semantic suggestions

- build_review_bundle.py: 新增 _compute_percentile/_compute_growth,
  候选池从固定阈值改为百分位排名 + 增速因子 (v2 policy)
- term_cleanup_policy.json: 升级 v2 schema
- generate_term_cleanup_suggestions.py: 新增 _prepare_alias_suggestions,
  规则层输出 alias (大小写/单复数/分词变体)
- generate_term_cleanup_semantic_suggestions.py: 新增 LLM 语义建议脚本
  (DeepSeek API, 产出 semantic alias/stopword/promote)
- SKILL.md: 更新为 5 Phase 工作流程
- 首轮清洗 apply: interest 54, aliases 17组, stopwords 17个
- docs/design/keyword-cleanup-flow-overview.md: 流程文档
- plans/: 引擎设计方案
This commit is contained in:
root
2026-05-14 17:17:49 +08:00
parent 4399c9ca90
commit 590d050218
13 changed files with 1592 additions and 57 deletions
+314
View File
@@ -0,0 +1,314 @@
#!/usr/bin/env python3
"""
Generate semantic keyword suggestions using LLM.
Covers what surface-form rules cannot:
- semantic alias (abbreviation ↔ full name, Chinese ↔ English, synonym)
- stopword (overly broad / low-discrimination terms)
- promote (new term that aligns with user's focus areas)
Usage:
python scripts/generate_term_cleanup_semantic_suggestions.py \
--bundle outputs/term_index/review/keyword-cleanup-bundle.json \
--output outputs/term_index/review/term-cleanup-semantic-suggestions-2026-05-14.json
"""
from __future__ import annotations
import argparse
import json
import os
import re
import sys
import time
from datetime import datetime, timezone
from pathlib import Path
from typing import Any
from urllib.request import Request, urlopen
REPO_ROOT = Path(__file__).resolve().parents[1]
DEFAULT_BUNDLE_PATH = REPO_ROOT / "outputs" / "term_index" / "review" / "keyword-cleanup-bundle.json"
DEFAULT_OUTPUT_DIR = REPO_ROOT / "outputs" / "term_index" / "review"
def _load_json(path: Path) -> Any:
return json.loads(path.read_text(encoding="utf-8-sig"))
def _save_json(path: Path, payload: dict[str, Any]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
def _load_env(path: Path) -> dict[str, str]:
"""Load key=value pairs from .env file."""
env: dict[str, str] = {}
if not path.exists():
return env
for line in path.read_text(encoding="utf-8").splitlines():
line = line.strip()
if not line or line.startswith("#") or "=" not in line:
continue
key, _, value = line.partition("=")
env[key.strip()] = value.strip().strip("\"'")
return env
def _build_prompt(
interest_keywords: list[str],
rule_alias_suggestions: list[dict[str, str]],
candidate_terms: list[dict[str, Any]],
relevant_watch_terms: list[dict[str, Any]],
) -> str:
"""Build the LLM prompt for semantic suggestions."""
interest_bullets = "\n".join(f" - {t}" for t in sorted(interest_keywords))
candidate_bullets = "\n".join(
f" - {t['term']} (count={t['total_count']}, days={t['days_seen']})"
for t in candidate_terms[:40]
)
# Alias from rule layer (for LLM to build on, not duplicate)
rule_alias_text = ""
if rule_alias_suggestions:
rule_alias_text = "\nSurface-form alias (already identified, skip these):\n" + "\n".join(
f" {a['from']} → {a['to']} ({a['reason']})"
for a in rule_alias_suggestions
)
watch_text = ""
if relevant_watch_terms:
watch_text = "\nWatch terms (low-frequency but potentially relevant):\n" + "\n".join(
f" {t['term']} (count={t['total_count']}, days={t['days_seen']})"
for t in relevant_watch_terms[:20]
)
return f"""You are a keyword governance assistant for an AI engineer. Your job is to analyze keyword data and produce structured suggestions.
## User's focus areas
- AI Agent engineering (Skills, Harness, MCP, Agent architecture)
- Backend engineering (Java, Go, Kubernetes, MySQL, distributed systems)
- Open source AI tools and practices (Claude Code, Cursor, DeepSeek, OpenClaw)
- LLM application engineering (context engineering, RAG, prompt engineering)
## Interest keywords (52 already configured)
{interest_bullets}
## Uncovered candidate terms (sorted by frequency)
{candidate_bullets}
{watch_text}{rule_alias_text}
## Task
Analyze the candidate terms and output a JSON object with exactly three keys:
1. "semantic_alias": array of alias suggestions that SURFACE RULES CAN'T CATCH (e.g. abbreviation↔full name, Chinese↔English, different naming for the same concept).
Format: [{{"from": "<variant>", "to": "<canonical interest keyword>", "reason": "<why>"}}]
2. "stopword": array of terms that are too broad/generic to be useful as filters. A stopword is a term that appears frequently but has LOW DISCRIMINATION — it matches too many unrelated articles and clutters the keyword index.
Format: [{{"term": "<term>", "reason": "<why it should be a stopword>"}}]
3. "promote_to_interest": array of uncovered terms that align well with the user's focus areas and should be added as interest keywords.
Format: [{{"term": "<term>", "reason": "<why it fits>"}}]
## Rules
- Be conservative. When in doubt, leave it out.
- Only suggest alias for terms that clearly refer to the SAME concept as an existing interest keyword.
- Only suggest stopword for terms that are genuinely too broad (appear in many unrelated contexts).
- Only suggest promote for terms that clearly match the user's stated focus areas.
- Output valid JSON only, no markdown, no explanation outside the JSON."""
def _call_llm(prompt: str, api_url: str, model: str, api_key: str) -> str:
"""Call LLM API and return the response text."""
payload = json.dumps({
"model": model,
"messages": [{"role": "user", "content": prompt}],
"temperature": 0.1,
"max_tokens": 2048,
}).encode("utf-8")
req = Request(
api_url.rstrip("/") + "/chat/completions",
data=payload,
headers={
"Content-Type": "application/json",
"Authorization": f"Bearer {api_key}",
},
)
max_retries = 3
for attempt in range(max_retries):
try:
with urlopen(req, timeout=120) as resp:
result = json.loads(resp.read().decode("utf-8"))
return result["choices"][0]["message"]["content"]
except Exception as e:
if attempt < max_retries - 1:
wait = 2 ** attempt
print(f" LLM call failed (attempt {attempt+1}/{max_retries}): {e}", file=sys.stderr)
print(f" Retrying in {wait}s...", file=sys.stderr)
time.sleep(wait)
else:
raise
def _parse_llm_response(text: str) -> dict[str, list[dict[str, str]]]:
"""Extract JSON from LLM response (may contain markdown fences)."""
# Try to find JSON block
json_match = re.search(r"```(?:json)?\s*\n?(\{.*?\})\s*\n?```", text, re.DOTALL)
if json_match:
text = json_match.group(1)
# Clean up: remove any text before { or after }
start = text.find("{")
end = text.rfind("}")
if start >= 0 and end > start:
text = text[start : end + 1]
try:
result = json.loads(text)
except json.JSONDecodeError:
# Try partial recovery
print(f" Warning: LLM response not clean JSON, attempting recovery", file=sys.stderr)
print(f" Raw: {text[:500]}", file=sys.stderr)
return {"semantic_alias": [], "stopword": [], "promote_to_interest": []}
# Normalize keys
normalized = {
"semantic_alias": result.get("semantic_alias", result.get("alias", [])),
"stopword": result.get("stopword", result.get("stopword_suggestions", [])),
"promote_to_interest": result.get("promote_to_interest", result.get("promote", [])),
}
# Ensure each is a list
for key in normalized:
if not isinstance(normalized[key], list):
normalized[key] = []
return normalized
def main() -> None:
parser = argparse.ArgumentParser(description="Generate semantic keyword suggestions via LLM.")
parser.add_argument("--bundle", type=Path, default=DEFAULT_BUNDLE_PATH, help="Review bundle JSON path")
parser.add_argument("--suggestions", type=Path, default=None, help="Existing suggestions JSON (for rule alias context)")
parser.add_argument("--output", type=Path, default=None, help="Output JSON path (auto-generated if omitted)")
parser.add_argument("--llm-api-url", type=str, default=None, help="LLM API base URL")
parser.add_argument("--llm-model", type=str, default=None, help="LLM model name")
parser.add_argument("--llm-api-key", type=str, default=None, help="LLM API key")
parser.add_argument("--dry-run", action="store_true", help="Print prompt and exit without calling LLM")
args = parser.parse_args()
# Load config
env_path = REPO_ROOT / ".env"
env = _load_env(env_path) if env_path.exists() else {}
api_url = args.llm_api_url or os.environ.get("LLM_API_URL") or env.get("LLM_API_URL", "https://api.deepseek.com")
# Map OpenClaw model aliases to actual API model names
model_raw = args.llm_model or os.environ.get("LLM_MODEL") or env.get("LLM_MODEL", "deepseek-chat")
MODEL_ALIAS_MAP = {
"deepseek/deepseek-v4-flash": "deepseek-chat",
"deepseek/deepseek-chat": "deepseek-chat",
"deepseek-v4-flash": "deepseek-chat",
"deepseek-chat": "deepseek-chat",
}
model = MODEL_ALIAS_MAP.get(model_raw, model_raw)
api_key = args.llm_api_key or os.environ.get("LLM_API_KEY") or env.get("LLM_API_KEY", "")
if not api_key:
print("Error: No LLM API key found. Set LLM_API_KEY in .env or pass --llm-api-key.", file=sys.stderr)
sys.exit(1)
# Load bundle
if not args.bundle.exists():
print(f"Error: Bundle not found: {args.bundle}", file=sys.stderr)
sys.exit(1)
bundle = _load_json(args.bundle)
current_config = bundle.get("current_config", {})
interest_keywords = current_config.get("interest_keywords", [])
top_global_terms = bundle.get("top_global_terms", [])
governance_hints = bundle.get("governance_hints", {})
# Build candidate list (uncovered terms from interest + watch candidates)
candidate_terms = []
for item in governance_hints.get("interest_review_candidates", []):
if isinstance(item, dict):
candidate_terms.append({
"term": item.get("term", ""),
"total_count": item.get("total_count", 0),
"days_seen": item.get("days_seen", 0),
"percentile": item.get("percentile", 0),
"growth": item.get("growth", 0),
})
for item in governance_hints.get("watch_review_candidates", []):
if isinstance(item, dict):
# Avoid duplicates
if not any(c["term"] == item.get("term") for c in candidate_terms):
candidate_terms.append({
"term": item.get("term", ""),
"total_count": item.get("total_count", 0),
"days_seen": item.get("days_seen", 0),
"percentile": item.get("percentile", 0),
"growth": item.get("growth", 0),
})
# Sort by total_count descending
candidate_terms.sort(key=lambda x: -x["total_count"])
relevant_watch_terms = governance_hints.get("watch_review_candidates", [])[:20]
# Load rule-layer alias suggestions if available
rule_alias = []
if args.suggestions and args.suggestions.exists():
s = _load_json(args.suggestions)
rule_alias = s.get("alias_suggestions", [])
# Build prompt
prompt = _build_prompt(
interest_keywords=interest_keywords,
rule_alias_suggestions=rule_alias,
candidate_terms=candidate_terms,
relevant_watch_terms=relevant_watch_terms,
)
# Determine output path
suggestion_date = datetime.now(timezone.utc).date().isoformat()
output_path = args.output or (DEFAULT_OUTPUT_DIR / f"term-cleanup-semantic-suggestions-{suggestion_date}.json")
if args.dry_run:
print("=== DRY RUN: Prompt ===")
print(prompt)
print("\n=== END ===")
print(f"\nWould write to: {output_path}")
return
# Call LLM
print(f"Calling LLM ({model})...", file=sys.stderr)
response = _call_llm(prompt, api_url, model, api_key)
print(f"LLM response received ({len(response)} chars)", file=sys.stderr)
# Parse
parsed = _parse_llm_response(response)
# Build output
output = {
"date": suggestion_date,
"source_bundle": str(args.bundle),
"model": model,
"interest_keyword_count": len(interest_keywords),
"candidate_count": len(candidate_terms),
**parsed,
}
_save_json(output_path, output)
summary = {
"output": str(output_path),
"semantic_alias": len(output.get("semantic_alias", [])),
"stopword": len(output.get("stopword", [])),
"promote_to_interest": len(output.get("promote_to_interest", [])),
}
print(json.dumps(summary, ensure_ascii=False, indent=2))
if __name__ == "__main__":
main()
+128 -3
View File
@@ -175,6 +175,121 @@ def _prepare_watch_suggestions(bundle: dict[str, Any], reserved_terms: set[str])
return suggestions
def _prepare_alias_suggestions(
bundle: dict[str, Any],
all_terms: list[dict[str, Any]] | None = None,
) -> list[dict[str, Any]]:
"""
Generate alias suggestions using surface-form rules (no LLM).
Rules:
1. casefold match — same normalized form, different original casing
2. trailing-s singularization — singular/plural variants
3. whitespace/hyphen normalization — word boundary variants
Scans all_terms (full term_stats) if provided; otherwise falls back
to top_global_terms from the bundle.
"""
current_config = _require_dict(bundle.get("current_config"), "bundle.current_config")
interest_keywords = _require_list(
current_config.get("interest_keywords"), "bundle.current_config.interest_keywords"
)
top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms")
source_terms = all_terms if all_terms is not None else top_global_terms
interest_set = {_term_key(t) for t in interest_keywords if isinstance(t, str)}
interest_originals: set[str] = {t for t in interest_keywords if isinstance(t, str)}
# Build full casefold → [original forms] map
cf_map: dict[str, list[str]] = {}
for item in source_terms:
term = None
if isinstance(item, dict):
term = item.get("term")
elif isinstance(item, str):
term = item
if not isinstance(term, str) or not term.strip():
continue
key = _term_key(term)
if key not in cf_map:
cf_map[key] = []
if term not in cf_map[key]:
cf_map[key].append(term)
suggestions: list[dict[str, Any]] = []
seen_pairs: set[tuple[str, str]] = set()
def _add(from_term: str, to_term: str, reason: str) -> None:
pair = (_term_key(from_term), _term_key(to_term))
if pair in seen_pairs:
return
seen_pairs.add(pair)
suggestions.append({"from": from_term, "to": to_term, "reason": reason})
# Build a set of all term keys from source for quick lookup
source_keys = set(cf_map.keys())
# Rule 1: casefold match — same normalized form, different casing
for key, variants in cf_map.items():
if len(variants) < 2:
continue
canonical = None
alt_forms = []
for v in variants:
if v in interest_originals:
canonical = v
else:
alt_forms.append(v)
if canonical and alt_forms:
for alt in alt_forms:
_add(alt, canonical, "Case variant")
elif len(variants) >= 2 and not canonical:
# None is canonical — suggest the highest-frequency form
ranked = sorted(variants, key=lambda t: -(
next(
(it.get("total_count", 0) for it in top_global_terms if it.get("term") == t),
0,
)
))
for alt in ranked[1:]:
_add(alt, ranked[0], "Case variant (auto-ranked)")
# Rule 2: singular/plural — trailing-s normalization
# Check all source terms (not just interest keys) for bidirectional matching
for key in source_keys:
if key in interest_set:
continue
if key.endswith("s") and len(key) > 2:
singular_key = key.rstrip("s")
if singular_key in interest_set and singular_key != key:
# Find canonical interest keyword
canon = next((t for t in interest_keywords if _term_key(t) == singular_key), None)
from_form = cf_map[key][0]
if canon:
_add(from_form, canon, "Plural variant")
# singular form → interest has plural
plural_key = key + "s"
if plural_key in interest_set and plural_key != key:
canon = next((t for t in interest_keywords if _term_key(t) == plural_key), None)
from_form = cf_map[key][0]
if canon:
_add(from_form, canon, "Singular variant")
# Rule 3: whitespace/hyphen normalization
for key in source_keys:
if key in interest_set:
continue
normalized = key.replace("-", "").replace("_", "").replace(" ", "")
if normalized in interest_set and normalized != key:
canon = next((t for t in interest_keywords if _term_key(t) == normalized), None)
from_form = cf_map[key][0]
if canon:
_add(from_form, canon, "Whitespace/punctuation variant")
suggestions.sort(key=lambda x: (x["from"].casefold(), x["to"].casefold()))
return suggestions
def _render_table(items: list[dict[str, Any]]) -> str:
if not items:
return "_None in this pass._\n"
@@ -361,9 +476,19 @@ def main() -> None:
markdown_output=args.markdown_output,
)
# Load full term_stats for alias scanning (bundle only has top N)
stats_path = REPO_ROOT / "data" / "term_index" / "term_stats.json"
all_stats_terms: list[str] = []
if stats_path.exists():
stats_payload = _load_json(stats_path)
raw_terms = stats_payload.get("terms") if isinstance(stats_payload, dict) else []
if isinstance(raw_terms, list):
all_stats_terms = [str(t["term"]) for t in raw_terms if isinstance(t, dict) and isinstance(t.get("term"), str)]
interest_items = _prepare_interest_suggestions(bundle)
reserved_terms = {_term_key(str(item.get("term") or "")) for item in interest_items}
watch_items = _prepare_watch_suggestions(bundle, reserved_terms=reserved_terms)
alias_items = _prepare_alias_suggestions(bundle, all_terms=all_stats_terms)
suggestions = {
"date": suggestion_date,
@@ -373,10 +498,10 @@ def main() -> None:
"summary": {
"interest_keyword_suggestions": len(interest_items),
"watch_terms": len(watch_items),
"alias_suggestions": 0,
"alias_suggestions": len(alias_items),
"stopword_suggestions": 0,
},
"alias_suggestions": [],
"alias_suggestions": alias_items,
"stopword_suggestions": [],
"interest_keyword_suggestions": interest_items,
"watch_terms": watch_items,
@@ -400,7 +525,7 @@ def main() -> None:
"markdown_output": str(markdown_output_path) if args.emit_markdown else None,
"interest_keyword_suggestions": len(interest_items),
"watch_terms": len(watch_items),
"alias_suggestions": 0,
"alias_suggestions": len(alias_items),
"stopword_suggestions": 0,
"emit_markdown": args.emit_markdown,
}