keyword cleanup: v2 engine, alias rule layer, LLM semantic suggestions
- build_review_bundle.py: 新增 _compute_percentile/_compute_growth, 候选池从固定阈值改为百分位排名 + 增速因子 (v2 policy) - term_cleanup_policy.json: 升级 v2 schema - generate_term_cleanup_suggestions.py: 新增 _prepare_alias_suggestions, 规则层输出 alias (大小写/单复数/分词变体) - generate_term_cleanup_semantic_suggestions.py: 新增 LLM 语义建议脚本 (DeepSeek API, 产出 semantic alias/stopword/promote) - SKILL.md: 更新为 5 Phase 工作流程 - 首轮清洗 apply: interest 54, aliases 17组, stopwords 17个 - docs/design/keyword-cleanup-flow-overview.md: 流程文档 - plans/: 引擎设计方案
This commit is contained in:
@@ -320,6 +320,23 @@
|
||||
|
||||
---
|
||||
|
||||
### [DONE][P1] interest/watch 候选引擎从固定阈值改为百分位排名 + 增速因子
|
||||
|
||||
目标:
|
||||
- 解决固定阈值(total_count>=3)不随数据量自适应的问题
|
||||
- 引入趋势信号(growth 因子),识别近期集中爆发的词
|
||||
- 支持 7 天、41 天、200 天数据量下取同样的 top 5%/5%-20% 而不需调阈值
|
||||
|
||||
要求:
|
||||
- `build_review_bundle.py`:新增 percentile 和 growth 计算函数;候选池从固定阈值改为百分位 + 增速
|
||||
- `configs/term_cleanup_policy.json`:升级为 v2 schema,percentile/growth 替代绝对阈值
|
||||
- 不改 `generate_term_cleanup_suggestions.py` 和 `apply_term_suggestions.py`
|
||||
- 全量跑一次对比新旧产出,确认差异合理
|
||||
|
||||
方案文档:`plans/keyword-cleanup-interest-watch-engine-improvement.md`
|
||||
|
||||
---
|
||||
|
||||
### [DONE][P3] 更新 README / handoff / docs,明确 MCP 为正式入口
|
||||
|
||||
目标:
|
||||
|
||||
@@ -27,13 +27,21 @@
|
||||
"Agent Skills",
|
||||
"AgentScope",
|
||||
"AI Agent",
|
||||
"AI Coding Agent",
|
||||
"AliSQL",
|
||||
"Anthropic",
|
||||
"Claude",
|
||||
"Claude Code",
|
||||
"CLAUDE.md",
|
||||
"Context Engineering",
|
||||
"Cursor",
|
||||
"DeepSeek",
|
||||
"FastAPI",
|
||||
"Gin",
|
||||
"Go",
|
||||
"gRPC",
|
||||
"Harness Engineering",
|
||||
"Hermes Agent",
|
||||
"Java",
|
||||
"Kafka",
|
||||
"Kubernetes",
|
||||
@@ -49,13 +57,25 @@
|
||||
"RAG",
|
||||
"ReActAgent",
|
||||
"Redis",
|
||||
"Skill",
|
||||
"SKILL.md",
|
||||
"Skills",
|
||||
"Spring",
|
||||
"SubAgent",
|
||||
"TypeScript",
|
||||
"Vibe Coding",
|
||||
"Workflow",
|
||||
"上下文压缩",
|
||||
"上下文工程",
|
||||
"上下文管理",
|
||||
"云原生",
|
||||
"代码审查",
|
||||
"可观测性",
|
||||
"向量数据库",
|
||||
"多Agent协作",
|
||||
"大模型",
|
||||
"微服务",
|
||||
"渐进式披露",
|
||||
"知识库"
|
||||
]
|
||||
}
|
||||
|
||||
@@ -1,5 +1,19 @@
|
||||
{
|
||||
"AI助手": "AI Agent",
|
||||
"图文RAG": "RAG",
|
||||
"Prompt架构": "Prompt Engineering"
|
||||
"Prompt架构": "Prompt Engineering",
|
||||
"Agent Skill": "Agent Skills",
|
||||
"Binlog": "binlog",
|
||||
"Coding Agent": "AI Coding Agent",
|
||||
"Subagent": "SubAgent",
|
||||
"Subagents": "SubAgent",
|
||||
"vibe coding": "Vibe Coding",
|
||||
"Agent架构": "AI Agent",
|
||||
"Agent专业化": "AI Agent",
|
||||
"Agent Teams": "多Agent协作",
|
||||
"Agentic Engineering": "AI Agent",
|
||||
"CLI工具": "CLI",
|
||||
"AI编程": "AI Coding Agent",
|
||||
"记忆管理": "上下文管理",
|
||||
"会话管理": "上下文管理"
|
||||
}
|
||||
|
||||
@@ -99,6 +99,443 @@
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json",
|
||||
"suggestion_date": "2026-04-08",
|
||||
"based_on_days": 7
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||
"action": "add_interest_keyword",
|
||||
"term": "Anthropic",
|
||||
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=13, days_seen=10, recent_count=13.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||
"action": "add_interest_keyword",
|
||||
"term": "Harness Engineering",
|
||||
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=12, days_seen=11, recent_count=12.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||
"action": "add_interest_keyword",
|
||||
"term": "Skill",
|
||||
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=11, days_seen=9, recent_count=11.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||
"action": "add_interest_keyword",
|
||||
"term": "上下文工程",
|
||||
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=6, days_seen=6, recent_count=6.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||
"action": "add_interest_keyword",
|
||||
"term": "多Agent协作",
|
||||
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=6, days_seen=6, recent_count=6.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||
"action": "add_interest_keyword",
|
||||
"term": "Claude",
|
||||
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=6, days_seen=5, recent_count=6.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||
"action": "add_interest_keyword",
|
||||
"term": "上下文管理",
|
||||
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=6, days_seen=5, recent_count=6.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||
"action": "add_interest_keyword",
|
||||
"term": "渐进式披露",
|
||||
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=5, days_seen=5, recent_count=5.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||
"action": "add_interest_keyword",
|
||||
"term": "Skills",
|
||||
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=5, days_seen=4, recent_count=5.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||
"action": "add_interest_keyword",
|
||||
"term": "SKILL.md",
|
||||
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=4, days_seen=4, recent_count=4.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||
"action": "add_interest_keyword",
|
||||
"term": "CLAUDE.md",
|
||||
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=4, days_seen=4, recent_count=4.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||
"action": "add_interest_keyword",
|
||||
"term": "上下文压缩",
|
||||
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=4, days_seen=4, recent_count=4.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||
"action": "add_interest_keyword",
|
||||
"term": "AI Coding Agent",
|
||||
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=3, days_seen=3, recent_count=3.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||
"action": "add_interest_keyword",
|
||||
"term": "Hermes Agent",
|
||||
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=3, days_seen=3, recent_count=3.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||
"action": "add_interest_keyword",
|
||||
"term": "Vibe Coding",
|
||||
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=5, days_seen=4, recent_count=5.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||
"action": "add_interest_keyword",
|
||||
"term": "Context Engineering",
|
||||
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=3, days_seen=3, recent_count=3.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||
"action": "add_interest_keyword",
|
||||
"term": "Cursor",
|
||||
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=3, days_seen=3, recent_count=3.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||
"action": "add_interest_keyword",
|
||||
"term": "大模型",
|
||||
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=4, days_seen=3, recent_count=4.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_interest_keyword",
|
||||
"term": "TypeScript",
|
||||
"reason": "Core language for AI agent development (e.g., Claude Code, Cursor) and backend engineering, complements existing Python/Java/Go keywords.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_interest_keyword",
|
||||
"term": "代码审查",
|
||||
"reason": "Chinese term for 'code review', a key practice in backend engineering and AI agent development workflows.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_stopword",
|
||||
"term": "Channels",
|
||||
"reason": "Too generic; could refer to communication channels, YouTube channels, or software channels, not specific to user's focus areas.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_stopword",
|
||||
"term": "Memory",
|
||||
"reason": "Extremely broad term; could refer to computer memory, human memory, or memory in various contexts, not discriminative enough.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_stopword",
|
||||
"term": "Prompt",
|
||||
"reason": "Already covered by 'Prompt Engineering' as a more specific term; 'Prompt' alone is too broad and matches many unrelated articles.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_stopword",
|
||||
"term": "AGI",
|
||||
"reason": "Too broad and speculative; not directly actionable for the user's practical engineering focus areas.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_stopword",
|
||||
"term": "AI日报",
|
||||
"reason": "Generic news term; not a technical concept or tool, would add noise to the keyword index.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_stopword",
|
||||
"term": "AIHOT",
|
||||
"reason": "Unclear meaning, likely a brand or aggregator, not a specific technical term.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_stopword",
|
||||
"term": "All In Code",
|
||||
"reason": "Too vague; could refer to a podcast, a philosophy, or a project, not a specific technical concept.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_stopword",
|
||||
"term": "auto-twitter-campaign",
|
||||
"reason": "Too specific to a single project/tool, not a general interest keyword for the user's focus areas.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_stopword",
|
||||
"term": "ChangeSet",
|
||||
"reason": "Generic term used in version control and databases; too broad to be a useful filter.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_stopword",
|
||||
"term": "Lumina",
|
||||
"reason": "Unclear reference; could be a product, framework, or brand, not clearly aligned with user's focus.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_stopword",
|
||||
"term": "OpenViking",
|
||||
"reason": "Unclear reference; not a known tool or concept in the user's stated focus areas.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_stopword",
|
||||
"term": "Seedance 2.0",
|
||||
"reason": "Unclear reference; likely a product or version, not a general technical term.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_stopword",
|
||||
"term": "质量门禁",
|
||||
"reason": "Chinese term for 'quality gate', too generic in software engineering; not specific to user's focus areas.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_alias",
|
||||
"term": "Agent Skill",
|
||||
"reason": "Singular variant",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"value": "Agent Skills",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_alias",
|
||||
"term": "Binlog",
|
||||
"reason": "Case variant (auto-ranked)",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"value": "binlog",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_alias",
|
||||
"term": "Coding Agent",
|
||||
"reason": "Abbreviated form of 'AI Coding Agent', referring to the same concept.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"value": "AI Coding Agent",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_alias",
|
||||
"term": "Subagent",
|
||||
"reason": "Case variant",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"value": "SubAgent",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_alias",
|
||||
"term": "Subagents",
|
||||
"reason": "Plural variant",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"value": "SubAgent",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_alias",
|
||||
"term": "vibe coding",
|
||||
"reason": "Case variant",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"value": "Vibe Coding",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_alias",
|
||||
"term": "Agent架构",
|
||||
"reason": "Chinese translation of 'Agent architecture', a core concept in AI Agent engineering.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"value": "AI Agent",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_alias",
|
||||
"term": "Agent专业化",
|
||||
"reason": "Chinese term for 'Agent specialization', directly related to Agent engineering.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"value": "AI Agent",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_alias",
|
||||
"term": "Agent Teams",
|
||||
"reason": "English equivalent of 'Multi-Agent collaboration', same concept.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"value": "多Agent协作",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_alias",
|
||||
"term": "Agentic Engineering",
|
||||
"reason": "Broader term for engineering with AI agents, closely related to Agent engineering focus.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"value": "AI Agent",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_alias",
|
||||
"term": "CLI工具",
|
||||
"reason": "Chinese translation of 'CLI tool', same concept.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"value": "CLI",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_alias",
|
||||
"term": "AI编程",
|
||||
"reason": "Chinese term for 'AI programming', closely related to AI Coding Agent.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"value": "AI Coding Agent",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_alias",
|
||||
"term": "记忆管理",
|
||||
"reason": "Chinese term for 'memory management', closely related to context management in LLM applications.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"value": "上下文管理",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
},
|
||||
{
|
||||
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||
"action": "add_alias",
|
||||
"term": "会话管理",
|
||||
"reason": "Chinese term for 'session management', related to context management in LLM applications.",
|
||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||
"value": "上下文管理",
|
||||
"suggestion_date": "2026-05-14",
|
||||
"based_on_days": 365
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
@@ -1,14 +1,12 @@
|
||||
{
|
||||
"schema_version": "v1",
|
||||
"schema_version": "v2",
|
||||
"interest_keyword_review": {
|
||||
"min_total_count": 3,
|
||||
"min_days_seen": 2
|
||||
"percentile_max": 0.05,
|
||||
"growth_promotion": 0.5
|
||||
},
|
||||
"watch_term_review": {
|
||||
"min_total_count": 1,
|
||||
"min_days_seen": 1,
|
||||
"max_total_count": 2,
|
||||
"max_days_seen": 2
|
||||
"percentile_min": 0.05,
|
||||
"percentile_max": 0.20
|
||||
},
|
||||
"alias_review": {
|
||||
"min_total_count": 2,
|
||||
@@ -19,7 +17,10 @@
|
||||
"max_days_seen": 2
|
||||
},
|
||||
"notes": [
|
||||
"当前阶段采用保守阈值,避免在低样本条件下直接扩充 interest_keywords。",
|
||||
"watch_terms 先用于观察,后续再决定是否升格为 interest_keywords 或进入 alias/stopword 配置。"
|
||||
"v2: interest/watch 使用百分位排名 + 增速因子替代固定阈值",
|
||||
"percentile 越小表示排名越高(top 5% = percentile 0.05)",
|
||||
"growth = recent_count / total_count,衡量近期活跃度",
|
||||
"watch_term_review 的 percentile_min 可理解为兴趣边界下限,低于此值的词归入 interest 候选",
|
||||
"growth_promotion(默认 0.5)用于识别近期集中爆发词,即使排位不高也主动推荐确认"
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,6 +1,19 @@
|
||||
[
|
||||
"AGI",
|
||||
"AIHOT",
|
||||
"AI日报",
|
||||
"All In Code",
|
||||
"auto-twitter-campaign",
|
||||
"ChangeSet",
|
||||
"Channels",
|
||||
"Lumina",
|
||||
"Memory",
|
||||
"OpenViking",
|
||||
"Prompt",
|
||||
"Seedance 2.0",
|
||||
"奋斗文化",
|
||||
"小银",
|
||||
"质量门禁",
|
||||
"银行客户经理",
|
||||
"飞盘物理"
|
||||
]
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"schema_version": "v1",
|
||||
"updated_at": "2026-04-08T02:34:14.194320Z",
|
||||
"updated_at": "2026-05-14T09:12:50.749877Z",
|
||||
"terms": [
|
||||
{
|
||||
"term": "A2A",
|
||||
|
||||
@@ -0,0 +1,151 @@
|
||||
# 关键词清洗流程概述
|
||||
|
||||
> 2026-05-14 初版
|
||||
> 从"数据记录"到"人工确认落盘"的完整链路
|
||||
|
||||
---
|
||||
|
||||
## 整体数据流
|
||||
|
||||
```
|
||||
每日日报 pipeline
|
||||
│
|
||||
▼
|
||||
term_index/daily/YYYY-MM-DD.json ← 每天一篇候选文章的热词统计
|
||||
│
|
||||
▼
|
||||
term_index/term_stats.json ← 所有 daily 的汇总(1070 个词)
|
||||
│
|
||||
├──── build_review_bundle.py ← 打包为审查数据包
|
||||
│ │
|
||||
│ ▼
|
||||
│ review/keyword-cleanup-bundle.json
|
||||
│ │
|
||||
│ ▼
|
||||
│ generate_term_cleanup_suggestions.py
|
||||
│ │
|
||||
│ ▼
|
||||
│ review/term-cleanup-suggestions-YYYY-MM-DD.json ← 正式建议产物
|
||||
│ │
|
||||
│ ▼
|
||||
│ (可选) review/term-cleanup-suggestions-YYYY-MM-DD.md ← 展示稿
|
||||
│
|
||||
├──── 人工确认哪些建议 accept
|
||||
│
|
||||
▼
|
||||
apply_term_suggestions.py ← 写入配置
|
||||
│
|
||||
├── configs/filter_context.personal.json ← interest_keywords
|
||||
├── configs/term_aliases.json ← alias
|
||||
├── configs/term_stopwords.json ← stopword
|
||||
├── configs/term_watchlist.json ← watch
|
||||
└── configs/term_change_log.json ← 变更日志
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 各环节说明
|
||||
|
||||
### 阶段 1:数据记录(每日自动)
|
||||
|
||||
```bash
|
||||
# FreshRSS pipeline 跑完后自动产出
|
||||
data/term_index/daily/2026-05-14.json
|
||||
```
|
||||
|
||||
- 每天一篇,记录当天候选文章中出现的热词
|
||||
- 包含 term、total_count、days_seen 等信息
|
||||
- 目前累计 **41 天**,共 **1070 个独立词**
|
||||
|
||||
### 阶段 2:全量汇总(每日自动)
|
||||
|
||||
```bash
|
||||
data/term_index/term_stats.json
|
||||
```
|
||||
|
||||
- 从所有 daily 文件重建,会覆盖重跑
|
||||
- 按 total_count 排序,前 5 名:OpenClaw(30)、Claude Code(25)、AI Agent(17)、Anthropic(13)、MCP(13)
|
||||
|
||||
### 阶段 3:构建审查数据包(手动触发)
|
||||
|
||||
```bash
|
||||
python skills/keyword-cleanup-review/scripts/build_review_bundle.py \
|
||||
--days 365 \
|
||||
--top 100 \
|
||||
--output outputs/term_index/review/keyword-cleanup-bundle.json
|
||||
```
|
||||
|
||||
- 把 term_stats + 当前配置打成一包,方便后续处理
|
||||
- 输出:`review/keyword-cleanup-bundle.json`
|
||||
|
||||
### 阶段 4:生成建议(手动触发)
|
||||
|
||||
```bash
|
||||
python scripts/generate_term_cleanup_suggestions.py \
|
||||
--bundle outputs/term_index/review/keyword-cleanup-bundle.json \
|
||||
--emit-markdown
|
||||
```
|
||||
|
||||
#### 当前产出能力
|
||||
|
||||
| 建议类型 | 状态 | 当前阈值 | 说明 |
|
||||
|---------|------|----------|------|
|
||||
| interest_keyword_suggestions | ✅ **已实现** | total≥3, days≥2 | 产出 20 条 |
|
||||
| watch_terms | ✅ **已实现** | total≤2, days≤2 | 本次 0 条 |
|
||||
| alias_suggestions | ❌ **硬编码为空** | policy 有阈值(total≥2, days≥2)但脚本未实现 | |
|
||||
| stopword_suggestions | ❌ **硬编码为空** | policy 有阈值(total≤2, days≤2)但脚本未实现 | |
|
||||
|
||||
**关键发现:** alias 和 stopword 不是"阈值太保守",是 **generate 脚本里压根没写对应的生成函数**。policy 文件里阈值已经配好了(alias: min_total=2/min_days=2,stopword: max_total=2/max_days=2),但脚本第 376-380 行直接硬编码为 `[]` 和 `0`。
|
||||
|
||||
### 阶段 5:人工确认(手动)
|
||||
|
||||
```
|
||||
OpenClaw 把建议列给你 → 你确认哪些 accept → 我执行 apply
|
||||
```
|
||||
|
||||
本次模式:
|
||||
- 高频(≥5次/5天以上)→ 强烈推荐 ✅
|
||||
- 中频(3-4次)→ 附带建议 ✅
|
||||
- 泛词 → 建议跳过 ❌
|
||||
|
||||
### 阶段 6:落盘配置(手动)
|
||||
|
||||
```bash
|
||||
python scripts/apply_term_suggestions.py \
|
||||
--suggestions outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json \
|
||||
--accept-interest 词1 词2 ...
|
||||
```
|
||||
|
||||
- dry-run 预览 → 确认后正式 apply
|
||||
- 写入 `configs/filter_context.personal.json`
|
||||
- 同步记录到 `term_change_log.json`
|
||||
- **不备份原始配置**(待优化)
|
||||
- **apply 后不自动清理 review 目录**(待优化)
|
||||
|
||||
### 阶段 7:维护清理(按需)
|
||||
|
||||
由 OpenClaw 侧 `reader-keyword-maintenance` skill 处理:
|
||||
- 删除旧 markdown 展示稿
|
||||
- 保留最近一份 bundle
|
||||
- 保守保留 suggestions JSON
|
||||
|
||||
---
|
||||
|
||||
## 当前配置资产
|
||||
|
||||
| 文件 | 内容 | 数据量 |
|
||||
|------|------|--------|
|
||||
| `filter_context.personal.json` | interest_keywords | 52 个 |
|
||||
| `term_aliases.json` | 别名映射 | 0 组(未启用) |
|
||||
| `term_stopwords.json` | 停用词 | 0 个(未启用) |
|
||||
| `term_watchlist.json` | 观察词 | 6 个 |
|
||||
| `term_change_log.json` | 所有变更记录 | 已记录 |
|
||||
|
||||
---
|
||||
|
||||
## 待优化项
|
||||
|
||||
1. **alias/stopword 建议生成为空** — generate 脚本硬编码缺实现,policy 已有阈值,需要补函数
|
||||
2. **apply 前无配置备份** — 建议 apply 前自动 cp 备份
|
||||
3. **apply 后无自动收尾** — 建议 apply 后自动删旧 markdown 和 bundle
|
||||
4. **alias 识别依赖规则而非 LLM** — 当前全靠统计阈值,无法做语义级判断(如中英文映射、缩写展开)。如果需要高级 alias 识别,可以用 LLM 生成候选,规则脚本做 apply
|
||||
@@ -0,0 +1,290 @@
|
||||
# interest/watch 候选引擎改进方案
|
||||
|
||||
> 从固定阈值到自适应排位 + 趋势因子的演进
|
||||
|
||||
## 1. 背景
|
||||
|
||||
### 1.1 当前实现
|
||||
|
||||
`build_review_bundle.py` 使用固定的绝对阈值将未覆盖词(uncovered terms)划分为两个候选池:
|
||||
|
||||
| 候选池 | 判断条件 | 依据 |
|
||||
|--------|---------|------|
|
||||
| `interest_review_candidates` | `total_count >= 3 AND days_seen >= 2` | `policy.interest_keyword_review` |
|
||||
| `watch_review_candidates` | `total_count <= 2 AND days_seen <= 2` | `policy.watch_term_review` |
|
||||
|
||||
`generate_term_cleanup_suggestions.py` 则直接从这两个候选池过滤、去重、排序后输出。
|
||||
|
||||
### 1.2 当前方案的问题
|
||||
|
||||
**问题一:固定阈值不随数据量自适应**
|
||||
|
||||
```
|
||||
场景 total_count=3 意味着什么
|
||||
─────────────────────────────────────────────
|
||||
7 天数据(~200 词) top 15%,有一定区分度 ✅
|
||||
41 天数据(1070 词) top 5%,区分度更高 ✅ 但阈值没变
|
||||
未来 200 天 仍然用 3 次,区分度稀释 ❌
|
||||
```
|
||||
|
||||
同一个绝对次数,在不同数据规模下的语义完全不同。手工调阈值不可持续。
|
||||
|
||||
**问题二:固定阈值忽略趋势信号**
|
||||
|
||||
- "Anthropic":total=13, recent=7 — 近期高活跃,上升趋势
|
||||
- "Channels":total=3, recent=0 — 早期出现但近期消失
|
||||
- 当前引擎认为这两个词"都过了 3 次阈值",同等对待。实际一个是强烈买入信号,一个是过气词。
|
||||
|
||||
**问题三:interest 和 watch 的分界线是硬的**
|
||||
|
||||
total=3 → interest,total=2 → watch。一个词从 2 次变成 3 次就自动"升级",没有过渡、没有缓冲。
|
||||
|
||||
### 1.3 讨论结论
|
||||
|
||||
与老大讨论后确认:
|
||||
|
||||
1. interest/watch 是**统计判断**,不需要大模型介入,纯算法可以解决
|
||||
2. 当前引擎缺的不是大模型,而是**算法本身没写完**——自适应维度(排位、趋势)还没实现
|
||||
3. alias 和 stopword 需要语义判断,与 interest/watch 分属不同阶段,不在本方案范围内
|
||||
4. 修改量小,可以在 1 小时内落地
|
||||
|
||||
---
|
||||
|
||||
## 2. 设计方案
|
||||
|
||||
### 2.1 核心思路
|
||||
|
||||
引入两个互补维度替代固定阈值:
|
||||
|
||||
```
|
||||
判定维度 含义 数据来源
|
||||
────────────────────────────────────────────────────────────
|
||||
percentile(百分位排名) 该词 total_count 在所有词 term_stats
|
||||
中的排位占比
|
||||
growth(增速因子) 近期集中度 = recent_count daily 近 N 天
|
||||
/ total_count
|
||||
```
|
||||
|
||||
两个维度配合:
|
||||
|
||||
- **percentile** 衡量"这个词在当前数据集里有多突出"——消除数据量变化的影响
|
||||
- **growth** 衡量"这个词是持续出现还是近期爆发"——识别趋势信号
|
||||
|
||||
### 2.2 候选池划分逻辑
|
||||
|
||||
```
|
||||
percentile
|
||||
│
|
||||
┌─────────────────────┐
|
||||
│ top 5% │
|
||||
│ → 建议 interest │ ← 高频稳定词
|
||||
├─────────────────────┤
|
||||
│ top 5%-20% │
|
||||
│ → 建议 watch │ ← 有信号但未达 threshold
|
||||
├─────────────────────┤
|
||||
│ bottom 80% │
|
||||
│ → 暂不处理 │ ← 噪声/低频
|
||||
└─────────────────────┘
|
||||
|
||||
额外规则:
|
||||
如果词在 top 20% 之外,但 growth > 0.5(近期集中度高)
|
||||
→ 主动提升到 watch / 主动推 confirm
|
||||
```
|
||||
|
||||
这样就不需要关心"total_count 是 3 还是 5",只看数据自己说话。
|
||||
|
||||
### 2.3 接口变化
|
||||
|
||||
**`configs/term_cleanup_policy.json`**:
|
||||
|
||||
```json
|
||||
{
|
||||
"schema_version": "v2",
|
||||
"interest_keyword_review": {
|
||||
"percentile_max": 0.05,
|
||||
"growth_promotion": 0.5
|
||||
},
|
||||
"watch_term_review": {
|
||||
"percentile_min": 0.05,
|
||||
"percentile_max": 0.20
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
`v1` 的 `min_total_count`/`min_days_seen` 等绝对阈值字段不再使用。
|
||||
|
||||
**`build_review_bundle.py` 输出的候选项**:
|
||||
|
||||
```json
|
||||
{
|
||||
"term": "Anthropic",
|
||||
"total_count": 13,
|
||||
"days_seen": 10,
|
||||
"percentile": 0.012,
|
||||
"growth": 0.54,
|
||||
"reason": "top 1.2% by frequency, 54% of occurrences in recent window — strong signal."
|
||||
}
|
||||
```
|
||||
|
||||
### 2.4 不需要改动的部分
|
||||
|
||||
- `generate_term_cleanup_suggestions.py` — 它只消费候选池,不用改
|
||||
- `apply_term_suggestions.py` — 消费 suggestions JSON,不用改
|
||||
- `keyword-cleanup-bundle.json` 结构 — 向后兼容,新增 percentile/growth 字段
|
||||
|
||||
---
|
||||
|
||||
## 3. 实施计划
|
||||
|
||||
### 3.1 改动范围
|
||||
|
||||
| 文件 | 改动量 | 内容 |
|
||||
|------|--------|------|
|
||||
| `skills/keyword-cleanup-review/scripts/build_review_bundle.py` | ~40 行 | 新增 `_compute_percentile()` 和 `_compute_growth()` 函数;修改候选池生成逻辑;候选项中增加 percentile/growth |
|
||||
| `configs/term_cleanup_policy.json` | ~10 行 | schema v2:percentile/growth 替代绝对阈值 |
|
||||
|
||||
### 3.2 实施步骤
|
||||
|
||||
1. **build_review_bundle.py**:在 `top_global_terms` 生成后,增加 percentile 计算函数和 growth 计算函数
|
||||
2. **build_review_bundle.py**:修改 `interest_review_candidates` 和 `watch_review_candidates` 的生成逻辑,从固定阈值改为 percentile + growth
|
||||
3. **build_review_bundle.py**:候选项增加 `percentile` 和 `growth` 字段,更新 `reason` 文案
|
||||
4. **term_cleanup_policy.json**:更新为 v2 schema
|
||||
5. **验证**:全量跑一次(`--days 365 --top 100`),对比新旧两份输出的差异
|
||||
|
||||
### 3.3 验证方法
|
||||
|
||||
```bash
|
||||
# 1. 用旧版生成 baseline
|
||||
cd /home/ubuntu/zhu/github/reader
|
||||
python3.11 skills/keyword-cleanup-review/scripts/build_review_bundle.py \
|
||||
--days 365 --top 100 \
|
||||
--output /tmp/bundle-baseline.json
|
||||
|
||||
# 2. 改代码后用新版生成
|
||||
python3.11 skills/keyword-cleanup-review/scripts/build_review_bundle.py \
|
||||
--days 365 --top 100 \
|
||||
--output /tmp/bundle-new.json
|
||||
|
||||
# 3. 对比 governance_hints
|
||||
python3 -c "
|
||||
import json
|
||||
a = json.load(open('/tmp/bundle-baseline.json'))
|
||||
b = json.load(open('/tmp/bundle-new.json'))
|
||||
for key in ['interest_review_candidates', 'watch_review_candidates']:
|
||||
old = set(i['term'] for i in a['governance_hints'][key])
|
||||
new = set(i['term'] for i in b['governance_hints'][key])
|
||||
print(f'{key}: 新增={new-old}, 减少={old-new}')
|
||||
"
|
||||
```
|
||||
|
||||
### 3.4 风险
|
||||
|
||||
| 风险 | 概率 | 应对 |
|
||||
|------|------|------|
|
||||
| 百分位阈值对特小数据集(如只有 1 天数据)不适用 | 低 | 不足 7 天时降级回绝对阈值 |
|
||||
| growth 因子对低频词的偏差(total=1, recent=1 → growth=1) | 低 | growth 只对 total>=3 的词计算 |
|
||||
| 排位突变导致推荐漂移 | 低 | percentil 天然平滑,新增几天数据不会剧烈改变已有词的排位 |
|
||||
|
||||
---
|
||||
|
||||
## 4. alias/stopword 设计方案
|
||||
|
||||
### 4.1 核心判断
|
||||
|
||||
alias 和 stopword 需要语义理解,与 interest/watch(纯统计)性质不同。
|
||||
|
||||
| 类型 | 需要什么 | 判断方式 |
|
||||
|------|---------|----------|
|
||||
| 大小写变体 | 表层 | 规则:casefold 去重 |
|
||||
| 单复数 | 表层 | 规则:去/加 s 后缀匹配 |
|
||||
| 分词变体(空格/连字符) | 表层 | 规则:去空格归一 |
|
||||
| 简写全称(MCP→Model Context Protocol) | **语义** | LLM |
|
||||
| 中英文(上下文工程→Context Engineering) | **语义** | LLM |
|
||||
| 同义不同名(Rush→猿辅导 Rush 平台) | **语义** | LLM |
|
||||
| stopword(大模型、AI 太泛) | **语义** | LLM |
|
||||
|
||||
### 4.2 分层方案
|
||||
|
||||
```
|
||||
输入:高频未覆盖词 + 已有 interest 词表
|
||||
│
|
||||
├── 规则层(零成本)── 大小写归一、单复数、去空格/连字符
|
||||
│ 输出候选 alias 对
|
||||
│
|
||||
└── LLM 层(每次 ~500 token)── 把候选词表整批给 LLM
|
||||
做语义聚类
|
||||
输出 alias 组 + stopword 标记
|
||||
```
|
||||
|
||||
### 4.3 规则层设计
|
||||
|
||||
在 `generate_term_cleanup_suggestions.py` 中新增 `_prepare_alias_suggestions()` 函数:
|
||||
|
||||
```python
|
||||
def _prepare_alias_suggestions(top_terms, interest_keywords):
|
||||
"""
|
||||
基于表层规则生成 alias 建议。
|
||||
规则1:casefold 匹配——同一个 casefold 下有多个原文变体
|
||||
规则2:单复数——去掉/加上末尾 s 后匹配
|
||||
规则3:分词变体——去空格/连字符后匹配
|
||||
"""
|
||||
```
|
||||
|
||||
优势:零成本、可复现、可审计。直接写入 suggestions JSON,随 generate 一起输出。
|
||||
|
||||
### 4.4 LLM 层设计
|
||||
|
||||
单独脚本,非 generate 主链路的一部分。
|
||||
|
||||
```bash
|
||||
python scripts/generate_term_cleanup_semantic_suggestions.py \
|
||||
--suggestions outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json \
|
||||
--output outputs/term_index/review/term-cleanup-semantic-suggestions-YYYY-MM-DD.json
|
||||
```
|
||||
|
||||
LLM prompt 设计:
|
||||
|
||||
```
|
||||
你是一个关键词治理助手。以下是一个用户的 interest 关键词列表和一批未覆盖的高频词。
|
||||
请做三件事:
|
||||
|
||||
1. ALIAS:判断哪些未覆盖词是已有 interest 关键词的别名/变体
|
||||
2. STOPWORD:标记哪些词太宽泛/通用,建议排除
|
||||
3. PROMOTE:标记哪些新词与用户关注方向一致,建议加入 interest
|
||||
|
||||
用户关注方向:AI Agent 工程化、后端工程、开源工具、大模型落地
|
||||
```
|
||||
|
||||
LLM 层输出格式:
|
||||
|
||||
```json
|
||||
{
|
||||
"alias_suggestions": [
|
||||
{"from": "Context Engineering", "to": "上下文工程", "reason": "中英文对应同一概念"}
|
||||
],
|
||||
"stopword_suggestions": [
|
||||
{"term": "大模型", "reason": "过于宽泛,高频率但低区分度"}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
### 4.5 预期效果
|
||||
|
||||
| 覆盖类型 | 规则层 | LLM 层 |
|
||||
|---------|--------|--------|
|
||||
| 大小写变体 | ✅ | — |
|
||||
| 单复数 | ✅ | — |
|
||||
| 分词变体 | ✅ | — |
|
||||
| 简写全称 | — | ✅ |
|
||||
| 中英文映射 | — | ✅ |
|
||||
| 同义不同名 | — | ✅ |
|
||||
| stopword 判断 | — | ✅ |
|
||||
|
||||
---
|
||||
|
||||
## 5. 讨论记录
|
||||
|
||||
- 2026-05-14:与老大确认 interest/watch 不需要 LLM,纯算法可解决
|
||||
- 2026-05-14:确认百分位排名 + 增速因子方案,修改量小,优先落地
|
||||
- 2026-05-14:确认本方案不改 `generate_term_cleanup_suggestions.py` 和 `apply_term_suggestions.py`
|
||||
- 2026-05-14:确认 alias/stopword 采用规则层 + LLM 层分层方案,规则层零成本优先
|
||||
+314
@@ -0,0 +1,314 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Generate semantic keyword suggestions using LLM.
|
||||
|
||||
Covers what surface-form rules cannot:
|
||||
- semantic alias (abbreviation ↔ full name, Chinese ↔ English, synonym)
|
||||
- stopword (overly broad / low-discrimination terms)
|
||||
- promote (new term that aligns with user's focus areas)
|
||||
|
||||
Usage:
|
||||
python scripts/generate_term_cleanup_semantic_suggestions.py \
|
||||
--bundle outputs/term_index/review/keyword-cleanup-bundle.json \
|
||||
--output outputs/term_index/review/term-cleanup-semantic-suggestions-2026-05-14.json
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
from urllib.request import Request, urlopen
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||
DEFAULT_BUNDLE_PATH = REPO_ROOT / "outputs" / "term_index" / "review" / "keyword-cleanup-bundle.json"
|
||||
DEFAULT_OUTPUT_DIR = REPO_ROOT / "outputs" / "term_index" / "review"
|
||||
|
||||
|
||||
def _load_json(path: Path) -> Any:
|
||||
return json.loads(path.read_text(encoding="utf-8-sig"))
|
||||
|
||||
|
||||
def _save_json(path: Path, payload: dict[str, Any]) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
||||
|
||||
|
||||
def _load_env(path: Path) -> dict[str, str]:
|
||||
"""Load key=value pairs from .env file."""
|
||||
env: dict[str, str] = {}
|
||||
if not path.exists():
|
||||
return env
|
||||
for line in path.read_text(encoding="utf-8").splitlines():
|
||||
line = line.strip()
|
||||
if not line or line.startswith("#") or "=" not in line:
|
||||
continue
|
||||
key, _, value = line.partition("=")
|
||||
env[key.strip()] = value.strip().strip("\"'")
|
||||
return env
|
||||
|
||||
|
||||
def _build_prompt(
|
||||
interest_keywords: list[str],
|
||||
rule_alias_suggestions: list[dict[str, str]],
|
||||
candidate_terms: list[dict[str, Any]],
|
||||
relevant_watch_terms: list[dict[str, Any]],
|
||||
) -> str:
|
||||
"""Build the LLM prompt for semantic suggestions."""
|
||||
|
||||
interest_bullets = "\n".join(f" - {t}" for t in sorted(interest_keywords))
|
||||
candidate_bullets = "\n".join(
|
||||
f" - {t['term']} (count={t['total_count']}, days={t['days_seen']})"
|
||||
for t in candidate_terms[:40]
|
||||
)
|
||||
|
||||
# Alias from rule layer (for LLM to build on, not duplicate)
|
||||
rule_alias_text = ""
|
||||
if rule_alias_suggestions:
|
||||
rule_alias_text = "\nSurface-form alias (already identified, skip these):\n" + "\n".join(
|
||||
f" {a['from']} → {a['to']} ({a['reason']})"
|
||||
for a in rule_alias_suggestions
|
||||
)
|
||||
|
||||
watch_text = ""
|
||||
if relevant_watch_terms:
|
||||
watch_text = "\nWatch terms (low-frequency but potentially relevant):\n" + "\n".join(
|
||||
f" {t['term']} (count={t['total_count']}, days={t['days_seen']})"
|
||||
for t in relevant_watch_terms[:20]
|
||||
)
|
||||
|
||||
return f"""You are a keyword governance assistant for an AI engineer. Your job is to analyze keyword data and produce structured suggestions.
|
||||
|
||||
## User's focus areas
|
||||
- AI Agent engineering (Skills, Harness, MCP, Agent architecture)
|
||||
- Backend engineering (Java, Go, Kubernetes, MySQL, distributed systems)
|
||||
- Open source AI tools and practices (Claude Code, Cursor, DeepSeek, OpenClaw)
|
||||
- LLM application engineering (context engineering, RAG, prompt engineering)
|
||||
|
||||
## Interest keywords (52 already configured)
|
||||
{interest_bullets}
|
||||
|
||||
## Uncovered candidate terms (sorted by frequency)
|
||||
{candidate_bullets}
|
||||
{watch_text}{rule_alias_text}
|
||||
|
||||
## Task
|
||||
Analyze the candidate terms and output a JSON object with exactly three keys:
|
||||
|
||||
1. "semantic_alias": array of alias suggestions that SURFACE RULES CAN'T CATCH (e.g. abbreviation↔full name, Chinese↔English, different naming for the same concept).
|
||||
Format: [{{"from": "<variant>", "to": "<canonical interest keyword>", "reason": "<why>"}}]
|
||||
|
||||
2. "stopword": array of terms that are too broad/generic to be useful as filters. A stopword is a term that appears frequently but has LOW DISCRIMINATION — it matches too many unrelated articles and clutters the keyword index.
|
||||
Format: [{{"term": "<term>", "reason": "<why it should be a stopword>"}}]
|
||||
|
||||
3. "promote_to_interest": array of uncovered terms that align well with the user's focus areas and should be added as interest keywords.
|
||||
Format: [{{"term": "<term>", "reason": "<why it fits>"}}]
|
||||
|
||||
## Rules
|
||||
- Be conservative. When in doubt, leave it out.
|
||||
- Only suggest alias for terms that clearly refer to the SAME concept as an existing interest keyword.
|
||||
- Only suggest stopword for terms that are genuinely too broad (appear in many unrelated contexts).
|
||||
- Only suggest promote for terms that clearly match the user's stated focus areas.
|
||||
- Output valid JSON only, no markdown, no explanation outside the JSON."""
|
||||
|
||||
|
||||
def _call_llm(prompt: str, api_url: str, model: str, api_key: str) -> str:
|
||||
"""Call LLM API and return the response text."""
|
||||
payload = json.dumps({
|
||||
"model": model,
|
||||
"messages": [{"role": "user", "content": prompt}],
|
||||
"temperature": 0.1,
|
||||
"max_tokens": 2048,
|
||||
}).encode("utf-8")
|
||||
|
||||
req = Request(
|
||||
api_url.rstrip("/") + "/chat/completions",
|
||||
data=payload,
|
||||
headers={
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": f"Bearer {api_key}",
|
||||
},
|
||||
)
|
||||
|
||||
max_retries = 3
|
||||
for attempt in range(max_retries):
|
||||
try:
|
||||
with urlopen(req, timeout=120) as resp:
|
||||
result = json.loads(resp.read().decode("utf-8"))
|
||||
return result["choices"][0]["message"]["content"]
|
||||
except Exception as e:
|
||||
if attempt < max_retries - 1:
|
||||
wait = 2 ** attempt
|
||||
print(f" LLM call failed (attempt {attempt+1}/{max_retries}): {e}", file=sys.stderr)
|
||||
print(f" Retrying in {wait}s...", file=sys.stderr)
|
||||
time.sleep(wait)
|
||||
else:
|
||||
raise
|
||||
|
||||
|
||||
def _parse_llm_response(text: str) -> dict[str, list[dict[str, str]]]:
|
||||
"""Extract JSON from LLM response (may contain markdown fences)."""
|
||||
# Try to find JSON block
|
||||
json_match = re.search(r"```(?:json)?\s*\n?(\{.*?\})\s*\n?```", text, re.DOTALL)
|
||||
if json_match:
|
||||
text = json_match.group(1)
|
||||
|
||||
# Clean up: remove any text before { or after }
|
||||
start = text.find("{")
|
||||
end = text.rfind("}")
|
||||
if start >= 0 and end > start:
|
||||
text = text[start : end + 1]
|
||||
|
||||
try:
|
||||
result = json.loads(text)
|
||||
except json.JSONDecodeError:
|
||||
# Try partial recovery
|
||||
print(f" Warning: LLM response not clean JSON, attempting recovery", file=sys.stderr)
|
||||
print(f" Raw: {text[:500]}", file=sys.stderr)
|
||||
return {"semantic_alias": [], "stopword": [], "promote_to_interest": []}
|
||||
|
||||
# Normalize keys
|
||||
normalized = {
|
||||
"semantic_alias": result.get("semantic_alias", result.get("alias", [])),
|
||||
"stopword": result.get("stopword", result.get("stopword_suggestions", [])),
|
||||
"promote_to_interest": result.get("promote_to_interest", result.get("promote", [])),
|
||||
}
|
||||
# Ensure each is a list
|
||||
for key in normalized:
|
||||
if not isinstance(normalized[key], list):
|
||||
normalized[key] = []
|
||||
return normalized
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description="Generate semantic keyword suggestions via LLM.")
|
||||
parser.add_argument("--bundle", type=Path, default=DEFAULT_BUNDLE_PATH, help="Review bundle JSON path")
|
||||
parser.add_argument("--suggestions", type=Path, default=None, help="Existing suggestions JSON (for rule alias context)")
|
||||
parser.add_argument("--output", type=Path, default=None, help="Output JSON path (auto-generated if omitted)")
|
||||
parser.add_argument("--llm-api-url", type=str, default=None, help="LLM API base URL")
|
||||
parser.add_argument("--llm-model", type=str, default=None, help="LLM model name")
|
||||
parser.add_argument("--llm-api-key", type=str, default=None, help="LLM API key")
|
||||
parser.add_argument("--dry-run", action="store_true", help="Print prompt and exit without calling LLM")
|
||||
args = parser.parse_args()
|
||||
|
||||
# Load config
|
||||
env_path = REPO_ROOT / ".env"
|
||||
env = _load_env(env_path) if env_path.exists() else {}
|
||||
|
||||
api_url = args.llm_api_url or os.environ.get("LLM_API_URL") or env.get("LLM_API_URL", "https://api.deepseek.com")
|
||||
# Map OpenClaw model aliases to actual API model names
|
||||
model_raw = args.llm_model or os.environ.get("LLM_MODEL") or env.get("LLM_MODEL", "deepseek-chat")
|
||||
MODEL_ALIAS_MAP = {
|
||||
"deepseek/deepseek-v4-flash": "deepseek-chat",
|
||||
"deepseek/deepseek-chat": "deepseek-chat",
|
||||
"deepseek-v4-flash": "deepseek-chat",
|
||||
"deepseek-chat": "deepseek-chat",
|
||||
}
|
||||
model = MODEL_ALIAS_MAP.get(model_raw, model_raw)
|
||||
api_key = args.llm_api_key or os.environ.get("LLM_API_KEY") or env.get("LLM_API_KEY", "")
|
||||
|
||||
if not api_key:
|
||||
print("Error: No LLM API key found. Set LLM_API_KEY in .env or pass --llm-api-key.", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
# Load bundle
|
||||
if not args.bundle.exists():
|
||||
print(f"Error: Bundle not found: {args.bundle}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
bundle = _load_json(args.bundle)
|
||||
current_config = bundle.get("current_config", {})
|
||||
interest_keywords = current_config.get("interest_keywords", [])
|
||||
top_global_terms = bundle.get("top_global_terms", [])
|
||||
governance_hints = bundle.get("governance_hints", {})
|
||||
|
||||
# Build candidate list (uncovered terms from interest + watch candidates)
|
||||
candidate_terms = []
|
||||
for item in governance_hints.get("interest_review_candidates", []):
|
||||
if isinstance(item, dict):
|
||||
candidate_terms.append({
|
||||
"term": item.get("term", ""),
|
||||
"total_count": item.get("total_count", 0),
|
||||
"days_seen": item.get("days_seen", 0),
|
||||
"percentile": item.get("percentile", 0),
|
||||
"growth": item.get("growth", 0),
|
||||
})
|
||||
for item in governance_hints.get("watch_review_candidates", []):
|
||||
if isinstance(item, dict):
|
||||
# Avoid duplicates
|
||||
if not any(c["term"] == item.get("term") for c in candidate_terms):
|
||||
candidate_terms.append({
|
||||
"term": item.get("term", ""),
|
||||
"total_count": item.get("total_count", 0),
|
||||
"days_seen": item.get("days_seen", 0),
|
||||
"percentile": item.get("percentile", 0),
|
||||
"growth": item.get("growth", 0),
|
||||
})
|
||||
|
||||
# Sort by total_count descending
|
||||
candidate_terms.sort(key=lambda x: -x["total_count"])
|
||||
relevant_watch_terms = governance_hints.get("watch_review_candidates", [])[:20]
|
||||
|
||||
# Load rule-layer alias suggestions if available
|
||||
rule_alias = []
|
||||
if args.suggestions and args.suggestions.exists():
|
||||
s = _load_json(args.suggestions)
|
||||
rule_alias = s.get("alias_suggestions", [])
|
||||
|
||||
# Build prompt
|
||||
prompt = _build_prompt(
|
||||
interest_keywords=interest_keywords,
|
||||
rule_alias_suggestions=rule_alias,
|
||||
candidate_terms=candidate_terms,
|
||||
relevant_watch_terms=relevant_watch_terms,
|
||||
)
|
||||
|
||||
# Determine output path
|
||||
suggestion_date = datetime.now(timezone.utc).date().isoformat()
|
||||
output_path = args.output or (DEFAULT_OUTPUT_DIR / f"term-cleanup-semantic-suggestions-{suggestion_date}.json")
|
||||
|
||||
if args.dry_run:
|
||||
print("=== DRY RUN: Prompt ===")
|
||||
print(prompt)
|
||||
print("\n=== END ===")
|
||||
print(f"\nWould write to: {output_path}")
|
||||
return
|
||||
|
||||
# Call LLM
|
||||
print(f"Calling LLM ({model})...", file=sys.stderr)
|
||||
response = _call_llm(prompt, api_url, model, api_key)
|
||||
print(f"LLM response received ({len(response)} chars)", file=sys.stderr)
|
||||
|
||||
# Parse
|
||||
parsed = _parse_llm_response(response)
|
||||
|
||||
# Build output
|
||||
output = {
|
||||
"date": suggestion_date,
|
||||
"source_bundle": str(args.bundle),
|
||||
"model": model,
|
||||
"interest_keyword_count": len(interest_keywords),
|
||||
"candidate_count": len(candidate_terms),
|
||||
**parsed,
|
||||
}
|
||||
|
||||
_save_json(output_path, output)
|
||||
|
||||
summary = {
|
||||
"output": str(output_path),
|
||||
"semantic_alias": len(output.get("semantic_alias", [])),
|
||||
"stopword": len(output.get("stopword", [])),
|
||||
"promote_to_interest": len(output.get("promote_to_interest", [])),
|
||||
}
|
||||
print(json.dumps(summary, ensure_ascii=False, indent=2))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -175,6 +175,121 @@ def _prepare_watch_suggestions(bundle: dict[str, Any], reserved_terms: set[str])
|
||||
return suggestions
|
||||
|
||||
|
||||
def _prepare_alias_suggestions(
|
||||
bundle: dict[str, Any],
|
||||
all_terms: list[dict[str, Any]] | None = None,
|
||||
) -> list[dict[str, Any]]:
|
||||
"""
|
||||
Generate alias suggestions using surface-form rules (no LLM).
|
||||
|
||||
Rules:
|
||||
1. casefold match — same normalized form, different original casing
|
||||
2. trailing-s singularization — singular/plural variants
|
||||
3. whitespace/hyphen normalization — word boundary variants
|
||||
|
||||
Scans all_terms (full term_stats) if provided; otherwise falls back
|
||||
to top_global_terms from the bundle.
|
||||
"""
|
||||
current_config = _require_dict(bundle.get("current_config"), "bundle.current_config")
|
||||
interest_keywords = _require_list(
|
||||
current_config.get("interest_keywords"), "bundle.current_config.interest_keywords"
|
||||
)
|
||||
top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms")
|
||||
source_terms = all_terms if all_terms is not None else top_global_terms
|
||||
|
||||
interest_set = {_term_key(t) for t in interest_keywords if isinstance(t, str)}
|
||||
interest_originals: set[str] = {t for t in interest_keywords if isinstance(t, str)}
|
||||
|
||||
# Build full casefold → [original forms] map
|
||||
cf_map: dict[str, list[str]] = {}
|
||||
for item in source_terms:
|
||||
term = None
|
||||
if isinstance(item, dict):
|
||||
term = item.get("term")
|
||||
elif isinstance(item, str):
|
||||
term = item
|
||||
if not isinstance(term, str) or not term.strip():
|
||||
continue
|
||||
key = _term_key(term)
|
||||
if key not in cf_map:
|
||||
cf_map[key] = []
|
||||
if term not in cf_map[key]:
|
||||
cf_map[key].append(term)
|
||||
|
||||
suggestions: list[dict[str, Any]] = []
|
||||
seen_pairs: set[tuple[str, str]] = set()
|
||||
|
||||
def _add(from_term: str, to_term: str, reason: str) -> None:
|
||||
pair = (_term_key(from_term), _term_key(to_term))
|
||||
if pair in seen_pairs:
|
||||
return
|
||||
seen_pairs.add(pair)
|
||||
suggestions.append({"from": from_term, "to": to_term, "reason": reason})
|
||||
|
||||
# Build a set of all term keys from source for quick lookup
|
||||
source_keys = set(cf_map.keys())
|
||||
|
||||
# Rule 1: casefold match — same normalized form, different casing
|
||||
for key, variants in cf_map.items():
|
||||
if len(variants) < 2:
|
||||
continue
|
||||
canonical = None
|
||||
alt_forms = []
|
||||
for v in variants:
|
||||
if v in interest_originals:
|
||||
canonical = v
|
||||
else:
|
||||
alt_forms.append(v)
|
||||
if canonical and alt_forms:
|
||||
for alt in alt_forms:
|
||||
_add(alt, canonical, "Case variant")
|
||||
elif len(variants) >= 2 and not canonical:
|
||||
# None is canonical — suggest the highest-frequency form
|
||||
ranked = sorted(variants, key=lambda t: -(
|
||||
next(
|
||||
(it.get("total_count", 0) for it in top_global_terms if it.get("term") == t),
|
||||
0,
|
||||
)
|
||||
))
|
||||
for alt in ranked[1:]:
|
||||
_add(alt, ranked[0], "Case variant (auto-ranked)")
|
||||
|
||||
# Rule 2: singular/plural — trailing-s normalization
|
||||
# Check all source terms (not just interest keys) for bidirectional matching
|
||||
for key in source_keys:
|
||||
if key in interest_set:
|
||||
continue
|
||||
if key.endswith("s") and len(key) > 2:
|
||||
singular_key = key.rstrip("s")
|
||||
if singular_key in interest_set and singular_key != key:
|
||||
# Find canonical interest keyword
|
||||
canon = next((t for t in interest_keywords if _term_key(t) == singular_key), None)
|
||||
from_form = cf_map[key][0]
|
||||
if canon:
|
||||
_add(from_form, canon, "Plural variant")
|
||||
# singular form → interest has plural
|
||||
plural_key = key + "s"
|
||||
if plural_key in interest_set and plural_key != key:
|
||||
canon = next((t for t in interest_keywords if _term_key(t) == plural_key), None)
|
||||
from_form = cf_map[key][0]
|
||||
if canon:
|
||||
_add(from_form, canon, "Singular variant")
|
||||
|
||||
# Rule 3: whitespace/hyphen normalization
|
||||
for key in source_keys:
|
||||
if key in interest_set:
|
||||
continue
|
||||
normalized = key.replace("-", "").replace("_", "").replace(" ", "")
|
||||
if normalized in interest_set and normalized != key:
|
||||
canon = next((t for t in interest_keywords if _term_key(t) == normalized), None)
|
||||
from_form = cf_map[key][0]
|
||||
if canon:
|
||||
_add(from_form, canon, "Whitespace/punctuation variant")
|
||||
|
||||
suggestions.sort(key=lambda x: (x["from"].casefold(), x["to"].casefold()))
|
||||
return suggestions
|
||||
|
||||
|
||||
def _render_table(items: list[dict[str, Any]]) -> str:
|
||||
if not items:
|
||||
return "_None in this pass._\n"
|
||||
@@ -361,9 +476,19 @@ def main() -> None:
|
||||
markdown_output=args.markdown_output,
|
||||
)
|
||||
|
||||
# Load full term_stats for alias scanning (bundle only has top N)
|
||||
stats_path = REPO_ROOT / "data" / "term_index" / "term_stats.json"
|
||||
all_stats_terms: list[str] = []
|
||||
if stats_path.exists():
|
||||
stats_payload = _load_json(stats_path)
|
||||
raw_terms = stats_payload.get("terms") if isinstance(stats_payload, dict) else []
|
||||
if isinstance(raw_terms, list):
|
||||
all_stats_terms = [str(t["term"]) for t in raw_terms if isinstance(t, dict) and isinstance(t.get("term"), str)]
|
||||
|
||||
interest_items = _prepare_interest_suggestions(bundle)
|
||||
reserved_terms = {_term_key(str(item.get("term") or "")) for item in interest_items}
|
||||
watch_items = _prepare_watch_suggestions(bundle, reserved_terms=reserved_terms)
|
||||
alias_items = _prepare_alias_suggestions(bundle, all_terms=all_stats_terms)
|
||||
|
||||
suggestions = {
|
||||
"date": suggestion_date,
|
||||
@@ -373,10 +498,10 @@ def main() -> None:
|
||||
"summary": {
|
||||
"interest_keyword_suggestions": len(interest_items),
|
||||
"watch_terms": len(watch_items),
|
||||
"alias_suggestions": 0,
|
||||
"alias_suggestions": len(alias_items),
|
||||
"stopword_suggestions": 0,
|
||||
},
|
||||
"alias_suggestions": [],
|
||||
"alias_suggestions": alias_items,
|
||||
"stopword_suggestions": [],
|
||||
"interest_keyword_suggestions": interest_items,
|
||||
"watch_terms": watch_items,
|
||||
@@ -400,7 +525,7 @@ def main() -> None:
|
||||
"markdown_output": str(markdown_output_path) if args.emit_markdown else None,
|
||||
"interest_keyword_suggestions": len(interest_items),
|
||||
"watch_terms": len(watch_items),
|
||||
"alias_suggestions": 0,
|
||||
"alias_suggestions": len(alias_items),
|
||||
"stopword_suggestions": 0,
|
||||
"emit_markdown": args.emit_markdown,
|
||||
}
|
||||
|
||||
@@ -26,7 +26,7 @@ description: 生成 reader 项目的正式关键词 review 输入。当用户需
|
||||
|
||||
## 工作流程
|
||||
|
||||
1. 构建精简的审查数据包(临时工作文件):
|
||||
### Phase 1:构建审查数据包
|
||||
|
||||
```bash
|
||||
python skills/keyword-cleanup-review/scripts/build_review_bundle.py
|
||||
@@ -34,51 +34,92 @@ python skills/keyword-cleanup-review/scripts/build_review_bundle.py
|
||||
|
||||
可选参数:
|
||||
|
||||
- `--days 7`
|
||||
- `--top 50`
|
||||
- `--days 7`(默认 7,建议传 365 覆盖全量)
|
||||
- `--top 100`(考虑的词数)
|
||||
- `--output outputs/term_index/review/keyword-cleanup-bundle.json`
|
||||
|
||||
2. 阅读建议模式:
|
||||
#### 候选引擎策略
|
||||
|
||||
- `skills/keyword-cleanup-review/references/suggestion-schema.md`
|
||||
根据 `configs/term_cleanup_policy.json` 的 `schema_version` 自动切换:
|
||||
|
||||
3. 运行 suggestions 生成脚本:
|
||||
| 版本 | 策略 | 说明 |
|
||||
|------|------|------|
|
||||
| v1(旧) | 固定阈值(total≥3/days≥2 → interest) | 小数据集兼容 |
|
||||
| v2(当前默认) | 百分位排名 + 增速因子 | 自适应数据量,不需要手工调阈值 |
|
||||
|
||||
v2 策略说明:
|
||||
- **percentile**:total_count 在所有词里的排位占比。top 5% → interest 候选,5%-20% → watch 候选
|
||||
- **growth**:recent_count / total_count,衡量近期活跃度。growth≥0.5 的排位外词也会主动推荐
|
||||
|
||||
### Phase 2:生成建议(规则层)
|
||||
|
||||
```bash
|
||||
python scripts/generate_term_cleanup_suggestions.py ^
|
||||
python scripts/generate_term_cleanup_suggestions.py \
|
||||
--bundle outputs/term_index/review/keyword-cleanup-bundle.json
|
||||
```
|
||||
|
||||
默认生成:
|
||||
|
||||
- 一份符合模式的 JSON 建议文件(正式建议产物,也是 review / apply 之间唯一正式输入)
|
||||
|
||||
如需人工审阅展示稿,再显式加:
|
||||
如需人工审阅展示稿:
|
||||
|
||||
```bash
|
||||
python scripts/generate_term_cleanup_suggestions.py ^
|
||||
--bundle outputs/term_index/review/keyword-cleanup-bundle.json ^
|
||||
python scripts/generate_term_cleanup_suggestions.py \
|
||||
--bundle outputs/term_index/review/keyword-cleanup-bundle.json \
|
||||
--emit-markdown
|
||||
```
|
||||
|
||||
这时才会额外生成:
|
||||
#### 产出能力
|
||||
|
||||
- 一份简短的供人工审阅的 Markdown 报告(临时展示稿)
|
||||
| 建议类型 | 状态 | 方法 |
|
||||
|---------|------|------|
|
||||
| interest 建议 | ✅ 已实现 | 百分位 top 5% + 增速促活 |
|
||||
| watch 建议 | ✅ 已实现 | 百分位 5%-20% |
|
||||
| alias 建议 | ✅ 已实现 | 规则层:大小写归一、单复数、去空格/连字符 |
|
||||
| stopword 建议 | ❌ 规则层空缺 | 见 Phase 3(LLM 层) |
|
||||
|
||||
4. 严格保持边界:
|
||||
默认生成:
|
||||
- `term-cleanup-suggestions-YYYY-MM-DD.json`(正式建议产物)
|
||||
|
||||
显式加 `--emit-markdown` 额外生成:
|
||||
- `term-cleanup-suggestions-YYYY-MM-DD.md`(临时展示稿)
|
||||
|
||||
### Phase 3:生成建议(LLM 层,可选)
|
||||
|
||||
规则层覆盖不了 alias(中英文对应、缩写展开、同义不同名)和 stopword 判断,需要 LLM 辅助:
|
||||
|
||||
```bash
|
||||
python scripts/generate_term_cleanup_semantic_suggestions.py \
|
||||
--bundle outputs/term_index/review/keyword-cleanup-bundle.json \
|
||||
--suggestions outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json \
|
||||
--output outputs/term_index/review/term-cleanup-semantic-suggestions-YYYY-MM-DD.json
|
||||
```
|
||||
|
||||
从 `.env` 读取 LLM 配置(`LLM_API_URL` / `LLM_MODEL` / `LLM_API_KEY`),使用 DeepSeek API。
|
||||
|
||||
输出三部分:
|
||||
|
||||
| 输出 | 说明 |
|
||||
|------|------|
|
||||
| `semantic_alias` | 语义级别名(中英文、缩写、同义不同名) |
|
||||
| `stopword` | 泛词过滤建议(规则层做不了的需要语义判断的) |
|
||||
| `promote_to_interest` | 与用户关注方向一致的新词,建议加入 interest |
|
||||
|
||||
**注:LLM 层产物是候选,不应自动 apply,需要人工确认后由 OpenClaw 编排 apply。**
|
||||
|
||||
### Phase 4:输出给 OpenClaw 编排
|
||||
|
||||
- `suggestions JSON` = review / apply 之间唯一正式建议输入
|
||||
- `semantic-suggestions JSON` = LLM 补充建议,需要人工筛选后合并到 suggestions JSON 再 apply
|
||||
- Markdown = 临时展示层
|
||||
- 后续汇报、确认、dry-run、apply、收尾清理由 OpenClaw 编排层执行
|
||||
|
||||
### Phase 5:严格保持边界
|
||||
|
||||
- 建议 `configs/term_aliases.json` 的修改
|
||||
- 建议 `configs/term_stopwords.json` 的修改
|
||||
- 建议 `configs/filter_context.personal.json` 的新增
|
||||
- **LLM 层产出(semantic-suggestions)不自动 apply**,需人工确认后由 OpenClaw 编排层执行
|
||||
- 除非用户明确要求,否则不要直接编辑这些文件
|
||||
- 除非用户要求修改规则逻辑,否则不要建议直接编辑 `configs/filter_rules.json`
|
||||
|
||||
5. 输出交接口径:
|
||||
|
||||
- 将 JSON suggestions 视为正式 review 输入
|
||||
- 将 Markdown 视为可选展示层
|
||||
- 后续汇报、确认、dry-run apply、正式 apply、收尾清理应由 OpenClaw 编排层继续执行
|
||||
|
||||
## 审查启发式规则
|
||||
|
||||
优先考虑以下决策:
|
||||
@@ -141,6 +182,7 @@ JSON 输出应遵循:
|
||||
短期保留:
|
||||
|
||||
- `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json`
|
||||
- `outputs/term_index/review/term-cleanup-semantic-suggestions-YYYY-MM-DD.json`
|
||||
|
||||
临时产物:
|
||||
|
||||
@@ -173,7 +215,9 @@ JSON 输出应遵循:
|
||||
## 资源
|
||||
|
||||
- 脚本:
|
||||
- `scripts/build_review_bundle.py`
|
||||
- `skills/keyword-cleanup-review/scripts/build_review_bundle.py`
|
||||
- `scripts/generate_term_cleanup_suggestions.py`
|
||||
- `scripts/generate_term_cleanup_semantic_suggestions.py`(LLM 层)
|
||||
- 参考文档:
|
||||
- `references/suggestion-schema.md`
|
||||
- `plans/keyword-cleanup-interest-watch-engine-improvement.md`(v2 引擎设计)
|
||||
|
||||
@@ -9,16 +9,15 @@ from typing import Any
|
||||
|
||||
|
||||
DEFAULT_POLICY: dict[str, Any] = {
|
||||
"schema_version": "v1",
|
||||
"schema_version": "v2",
|
||||
"interest_keyword_review": {
|
||||
"min_total_count": 3,
|
||||
"min_days_seen": 2,
|
||||
"percentile_min": 0.0,
|
||||
"percentile_max": 0.05,
|
||||
"growth_promotion": 0.5,
|
||||
},
|
||||
"watch_term_review": {
|
||||
"min_total_count": 1,
|
||||
"min_days_seen": 1,
|
||||
"max_total_count": 2,
|
||||
"max_days_seen": 2,
|
||||
"percentile_min": 0.05,
|
||||
"percentile_max": 0.20,
|
||||
},
|
||||
"alias_review": {
|
||||
"min_total_count": 2,
|
||||
@@ -28,6 +27,11 @@ DEFAULT_POLICY: dict[str, Any] = {
|
||||
"max_total_count": 2,
|
||||
"max_days_seen": 2,
|
||||
},
|
||||
"notes": [
|
||||
"v2: interest/watch 使用百分位排名 + 增速因子替代固定阈值",
|
||||
"percentile 越小表示排名越高(top 5% = percentile 0.05)",
|
||||
"growth = recent_count / total_count,衡量近期活跃度",
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
@@ -126,6 +130,37 @@ def _within_watch_thresholds(item: dict[str, Any], thresholds: dict[str, Any]) -
|
||||
)
|
||||
|
||||
|
||||
def _compute_percentile(value: int, sorted_values: list[int]) -> float:
|
||||
"""
|
||||
Return the percentile rank of `value` in `sorted_values` (ascending).
|
||||
0.0 = highest frequency (top rank), 1.0 = lowest frequency (bottom rank).
|
||||
"""
|
||||
if not sorted_values:
|
||||
return 1.0
|
||||
# bisect_left — count of values strictly less than `value`
|
||||
lo, hi = 0, len(sorted_values)
|
||||
while lo < hi:
|
||||
mid = (lo + hi) // 2
|
||||
if sorted_values[mid] < value:
|
||||
lo = mid + 1
|
||||
else:
|
||||
hi = mid
|
||||
rank = lo
|
||||
# invert: smallest value → rank=0 → 1.0 (bottom)
|
||||
# largest value → rank=len → 0.0 (top)
|
||||
return 1.0 - (rank / len(sorted_values))
|
||||
|
||||
|
||||
def _compute_growth(recent_count: int, total_count: int) -> float:
|
||||
"""
|
||||
Return growth factor: recent_count / total_count.
|
||||
Only meaningful when total_count >= 3; returns 0.0 for small counts.
|
||||
"""
|
||||
if total_count < 3:
|
||||
return 0.0
|
||||
return recent_count / total_count
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Build a compact review bundle for the keyword-cleanup-review skill."
|
||||
@@ -235,6 +270,13 @@ def main() -> None:
|
||||
alias_values = _casefold_set(list(aliases.values()))
|
||||
watch_set = _casefold_set([str(item.get("term", "")) for item in watchlist])
|
||||
|
||||
# Build a sorted list of all total_counts for percentile computation
|
||||
all_total_counts = sorted(
|
||||
int(item.get("total_count") or 0)
|
||||
for item in stats_terms
|
||||
if isinstance(item, dict) and isinstance(item.get("term"), str)
|
||||
)
|
||||
|
||||
top_global_terms = []
|
||||
for item in stats_terms[: args.top]:
|
||||
if not isinstance(item, dict):
|
||||
@@ -256,42 +298,108 @@ def main() -> None:
|
||||
"is_alias_target": folded in alias_values,
|
||||
"in_watchlist": folded in watch_set,
|
||||
"recent_count": recent_counter.get(term, 0),
|
||||
"percentile": _compute_percentile(
|
||||
int(item.get("total_count") or 0), all_total_counts
|
||||
),
|
||||
"growth": _compute_growth(
|
||||
recent_counter.get(term, 0),
|
||||
int(item.get("total_count") or 0),
|
||||
),
|
||||
}
|
||||
)
|
||||
|
||||
# Keep more uncovered terms for percentile-based selection
|
||||
uncovered_terms = [
|
||||
item for item in top_global_terms if not item["in_interest_keywords"] and not item["is_stopword"]
|
||||
][:20]
|
||||
][:100]
|
||||
|
||||
policy_version = (policy.get("schema_version") if isinstance(policy, dict) else None) or "v1"
|
||||
interest_thresholds = policy.get("interest_keyword_review") if isinstance(policy, dict) else {}
|
||||
watch_thresholds = policy.get("watch_term_review") if isinstance(policy, dict) else {}
|
||||
|
||||
if policy_version == "v2" or "percentile_max" in interest_thresholds:
|
||||
# v2: percentile + growth based selection
|
||||
pct_min_interest = float(interest_thresholds.get("percentile_min", 0.0))
|
||||
pct_max_interest = float(interest_thresholds.get("percentile_max", 0.05))
|
||||
growth_promo = float(interest_thresholds.get("growth_promotion", 0.5))
|
||||
pct_min_watch = float(watch_thresholds.get("percentile_min", 0.05))
|
||||
pct_max_watch = float(watch_thresholds.get("percentile_max", 0.20))
|
||||
|
||||
interest_candidates_raw = [
|
||||
item for item in uncovered_terms
|
||||
if not item["in_watchlist"]
|
||||
and pct_min_interest <= item["percentile"] <= pct_max_interest
|
||||
]
|
||||
watch_candidates_raw = [
|
||||
item for item in uncovered_terms
|
||||
if not item["in_watchlist"]
|
||||
and pct_min_watch < item["percentile"] <= pct_max_watch
|
||||
]
|
||||
# Growth boost: terms outside watch range but with strong growth signal
|
||||
growth_boost_candidates = [
|
||||
item for item in uncovered_terms
|
||||
if not item["in_watchlist"]
|
||||
and item["percentile"] > pct_max_watch
|
||||
and item["growth"] >= growth_promo
|
||||
]
|
||||
else:
|
||||
# v1 fallback: fixed thresholds
|
||||
interest_candidates_raw = [
|
||||
item for item in uncovered_terms
|
||||
if not item["in_watchlist"] and _meets_min_thresholds(item, interest_thresholds)
|
||||
]
|
||||
watch_candidates_raw = [
|
||||
item for item in uncovered_terms
|
||||
if not item["in_watchlist"]
|
||||
and not _meets_min_thresholds(item, interest_thresholds)
|
||||
and _within_watch_thresholds(item, watch_thresholds)
|
||||
]
|
||||
growth_boost_candidates = []
|
||||
|
||||
interest_review_candidates = [
|
||||
{
|
||||
"term": item["term"],
|
||||
"total_count": item["total_count"],
|
||||
"days_seen": item["days_seen"],
|
||||
"percentile": item["percentile"],
|
||||
"growth": item["growth"],
|
||||
"reason": (
|
||||
"Meets the configured interest-keyword review threshold and is not yet covered "
|
||||
"by interest keywords or stopwords."
|
||||
f"top {item['percentile']:.1%} by frequency,"
|
||||
f"growth={item['growth']:.0%},"
|
||||
"not yet covered by interest keywords or stopwords."
|
||||
),
|
||||
}
|
||||
for item in uncovered_terms
|
||||
if not item["in_watchlist"] and _meets_min_thresholds(item, interest_thresholds)
|
||||
for item in interest_candidates_raw
|
||||
][:20]
|
||||
watch_review_candidates = [
|
||||
{
|
||||
"term": item["term"],
|
||||
"total_count": item["total_count"],
|
||||
"days_seen": item["days_seen"],
|
||||
"percentile": item["percentile"],
|
||||
"growth": item["growth"],
|
||||
"reason": (
|
||||
"Falls into the configured watch-term review range and should be observed "
|
||||
"before promotion into interest keywords."
|
||||
f"top {item['percentile']:.1%} by frequency,"
|
||||
f"growth={item['growth']:.0%},"
|
||||
"fell into watch-review range."
|
||||
),
|
||||
}
|
||||
for item in uncovered_terms
|
||||
if not item["in_watchlist"]
|
||||
and not _meets_min_thresholds(item, interest_thresholds)
|
||||
and _within_watch_thresholds(item, watch_thresholds)
|
||||
for item in watch_candidates_raw
|
||||
][:20]
|
||||
growth_boost_review_items = [
|
||||
{
|
||||
"term": item["term"],
|
||||
"total_count": item["total_count"],
|
||||
"days_seen": item["days_seen"],
|
||||
"percentile": item["percentile"],
|
||||
"growth": item["growth"],
|
||||
"reason": (
|
||||
f"growth spike: {item['growth']:.0%} of occurrences in recent window "
|
||||
f"(total={item['total_count']}, days={item['days_seen']})."
|
||||
),
|
||||
}
|
||||
for item in growth_boost_candidates
|
||||
][:5]
|
||||
recent_hot_terms = sorted(
|
||||
({"term": term, "recent_count": count} for term, count in recent_counter.items()),
|
||||
key=lambda item: (-item["recent_count"], item["term"].casefold(), item["term"]),
|
||||
@@ -330,6 +438,7 @@ def main() -> None:
|
||||
"governance_hints": {
|
||||
"interest_review_candidates": interest_review_candidates,
|
||||
"watch_review_candidates": watch_review_candidates,
|
||||
"growth_boost_review_items": growth_boost_review_items,
|
||||
},
|
||||
}
|
||||
_save_json(args.output, bundle)
|
||||
|
||||
Reference in New Issue
Block a user