Compare commits

...
2 Commits
Author SHA1 Message Date
root cdbcdcd485 feat: pipeline 并行摘要 + 子进程 env 注入 + 循环导入修复
源码:
- runtime/__init__.py: resume_jobs/resume_service 改为懒加载,打破循环导入
- freshrss_pipeline_jobs.py / resume_jobs.py: 子进程注入 .env 环境变量
- freshrss_pipeline.py: LLM 摘要串行改并行 (ThreadPoolExecutor, max_workers=4)

配置:
- term_aliases: 19→149 条,大幅扩充别名映射
- term_stopwords: 19→132 条,增加过滤规则
- filter_context.personal.json: +7 个兴趣关键词
2026-07-28 16:24:31 +08:00
root 590d050218 keyword cleanup: v2 engine, alias rule layer, LLM semantic suggestions
- build_review_bundle.py: 新增 _compute_percentile/_compute_growth,
  候选池从固定阈值改为百分位排名 + 增速因子 (v2 policy)
- term_cleanup_policy.json: 升级 v2 schema
- generate_term_cleanup_suggestions.py: 新增 _prepare_alias_suggestions,
  规则层输出 alias (大小写/单复数/分词变体)
- generate_term_cleanup_semantic_suggestions.py: 新增 LLM 语义建议脚本
  (DeepSeek API, 产出 semantic alias/stopword/promote)
- SKILL.md: 更新为 5 Phase 工作流程
- 首轮清洗 apply: interest 54, aliases 17组, stopwords 17个
- docs/design/keyword-cleanup-flow-overview.md: 流程文档
- plans/: 引擎设计方案
2026-05-14 17:17:49 +08:00
17 changed files with 2003 additions and 93 deletions
+17
View File
@@ -320,6 +320,23 @@
---
### [DONE][P1] interest/watch 候选引擎从固定阈值改为百分位排名 + 增速因子
目标:
- 解决固定阈值(total_count>=3)不随数据量自适应的问题
- 引入趋势信号(growth 因子),识别近期集中爆发的词
- 支持 7 天、41 天、200 天数据量下取同样的 top 5%/5%-20% 而不需调阈值
要求:
- `build_review_bundle.py`:新增 percentile 和 growth 计算函数;候选池从固定阈值改为百分位 + 增速
- `configs/term_cleanup_policy.json`:升级为 v2 schema,percentile/growth 替代绝对阈值
- 不改 `generate_term_cleanup_suggestions.py` 和 `apply_term_suggestions.py`
- 全量跑一次对比新旧产出,确认差异合理
方案文档:`plans/keyword-cleanup-interest-watch-engine-improvement.md`
---
### [DONE][P3] 更新 README / handoff / docs,明确 MCP 为正式入口
目标:
+27
View File
@@ -27,18 +27,30 @@
"Agent Skills",
"AgentScope",
"AI Agent",
"AI Coding Agent",
"AliSQL",
"Anthropic",
"Claude",
"Claude Code",
"CLAUDE.md",
"CLI",
"Context Engineering",
"Cursor",
"ChatGPT",
"DeepSeek",
"FastAPI",
"Gin",
"Go",
"gRPC",
"Harness Engineering",
"Hermes Agent",
"Java",
"Kafka",
"Kubernetes",
"LLM",
"Loop Engineering",
"MCP",
"MoE",
"MySQL",
"MySQL复制延迟",
"OpenAI",
@@ -47,15 +59,30 @@
"Prompt Engineering",
"Python",
"RAG",
"ReAct",
"ReActAgent",
"Redis",
"Skill",
"SKILL.md",
"Skills",
"Spring",
"SubAgent",
"TypeScript",
"Vibe Coding",
"Workflow",
"上下文压缩",
"上下文工程",
"上下文管理",
"云原生",
"代码审查",
"可观测性",
"向量数据库",
"多Agent协作",
"大模型",
"子Agent",
"强化学习",
"微服务",
"渐进式披露",
"知识库"
]
}
+146 -2
View File
@@ -1,5 +1,149 @@
{
"AI助手": "AI Agent",
"Agent": "Agent",
"Agent框架": "Agent",
"Agent能力": "Agent Skills",
"智能体": "AI Agent",
"Agentic架构": "Agentic架构",
"多Agent协作": "多Agent协作",
"多智能体架构": "多Agent协作",
"Multi-Agent": "多Agent",
"Subagent": "子Agent",
"Sub Agents验证": "子Agent",
"子Agent": "子Agent",
"子智能体": "子Agent",
"Coding Agent": "AI Coding Agent",
"AI编程": "AI Coding Agent",
"AI辅助编程": "AI Coding Agent",
"代码生成": "AI代码生成",
"代码审查": "Code Review",
"Prompt": "Prompt Engineering",
"Prompt Caching": "提示缓存",
"RAG": "RAG",
"图文RAG": "RAG",
"Prompt架构": "Prompt Engineering"
"向量检索": "向量检索",
"向量嵌入": "向量嵌入",
"Multi-Token Prediction": "多Token预测",
"Pair-In Pair-Out": "PIPO架构",
"PIPO": "PIPO架构",
"上下文管理": "上下文管理",
"上下文卸载": "上下文卸载",
"Self-GC": "上下文压缩",
"记忆管理": "上下文管理",
"会话管理": "上下文管理",
"Harness Engineering": "Harness工程化",
"Harness架构": "Harness工程化",
"Harness": "Harness工程化",
"Loop Engineering": "Loop Engineering",
"推理加速": "推理加速",
"推理深度": "推理深度",
"长链路推理": "长链路推理",
"RLVR": "RLVR",
"GRPO": "GRPO",
"强化学习": "强化学习",
"Multi-Agent RL": "多Agent强化学习",
"Viking AI搜索": "AI搜索",
"Viking AI Search": "AI搜索",
"智能搜索": "AI搜索",
"SearchCLI": "CLI搜索",
"视频生成": "AI视频生成",
"视频生成模型": "AI视频生成",
"LingBot-Video": "AI视频生成",
"视觉自回归模型": "AI视频生成",
"火山云数据库PostgreSQL Serverless版": "Serverless数据库",
"PostgreSQL": "PostgreSQL",
"MySQL": "MySQL",
"OceanBase": "OceanBase",
"StarRocks": "StarRocks",
"Milvus": "Milvus",
"Seal AI Zone": "AI安全",
"NEX沙箱": "沙箱隔离",
"MicroVM": "沙箱隔离",
"安全左移": "安全左移",
"安全中台": "AI安全",
"成本降低": "成本优化",
"成本杠杆": "成本优化",
"Scale-to-Zero": "弹性伸缩",
"Data as Git": "数据分支管理",
"Schema Diff": "Schema对比",
"Time Travel": "数据回溯",
"多端架构": "多端架构",
"契约化": "契约化架构",
"大仓": "大仓工程化",
"Vibe Coding": "Vibe Coding",
"LLM Judge": "LLM评估",
"SWE-Bench": "SWE-Bench",
"SWE Bench Pro": "SWE-Bench",
"SWE-Bench Pro": "SWE-Bench",
"Verification Agent": "验证Agent",
"CLI工具": "CLI",
"CLI": "CLI",
"漏桶算法": "限流架构",
"固定窗口限流": "限流架构",
"Suspend消费控制": "限流架构",
"RocketMQ LiteTopic": "消息队列",
"LLM Wiki": "LLM知识库",
"知识工程": "知识工程",
"语义资产": "语义资产管理",
"知识图谱": "知识图谱",
"知识库沉淀": "知识管理",
"Skill": "Skill",
"Skill Hub": "技能生态",
"具身智能": "具身智能",
"Open X-Embodiment": "具身智能",
"YOLO Classifier": "目标检测",
"MCP": "MCP",
"MCP连接器": "MCP",
"缓存击穿": "缓存优化",
"GPU算力调度": "算力调度",
"异构资源": "异构计算",
"XPU": "异构计算",
"弹性RDMA": "RDMA网络",
"国内主流GPU": "国产芯片",
"国产AI芯片": "国产芯片",
"Paxos协议": "分布式一致性",
"Token": "Token管理",
"Token效率": "Token管理",
"百万token上下文": "长上下文",
"MoE": "MoE架构",
"MoE架构": "MoE架构",
"思维链": "思维链",
"CoT Distillation": "思维链蒸馏",
"自然语言驱动": "自然语言交互",
"NL2SQL": "NL2SQL",
"AI对齐": "AI对齐",
"注意力机制": "注意力机制",
"多模态": "多模态",
"音视频工作台": "音视频处理",
"AI助手": "AI Agent",
"Agent架构": "AI Agent",
"Agent专业化": "AI Agent",
"Agent Teams": "多Agent协作",
"Agentic Engineering": "AI Agent",
"AI智能体": "AI Agent",
"LLM Agent": "AI Agent",
"AI Harness": "Harness Engineering",
"AI代码生成": "AI Coding Agent",
"Memory管理": "上下文管理",
"Agent Skill": "Agent Skills",
"Binlog": "binlog",
"vibe coding": "Vibe Coding",
"Agent组织化协作平台": "Agent协作平台",
"Anthropic": "Anthropic",
"OpenClaw": "OpenClaw",
"WorkBuddy": "WorkBuddy",
"Claude": "Claude",
"ChatGPT": "ChatGPT",
"GPT": "GPT",
"Opus": "Opus",
"Sonnet": "Sonnet",
"Grok": "Grok",
"Qwen": "Qwen",
"GLM": "GLM",
"Claude Code": "Claude Code",
"Cursor": "Cursor",
"Codex": "Codex",
"Pi": "Pi",
"CoT": "CoT",
"SVG": "SVG",
"TTS": "TTS"
}
+491
View File
@@ -99,6 +99,497 @@
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json",
"suggestion_date": "2026-04-08",
"based_on_days": 7
},
{
"applied_at": "2026-05-14T08:20:35.134979Z",
"action": "add_interest_keyword",
"term": "Anthropic",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=13, days_seen=10, recent_count=13.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T08:20:35.134979Z",
"action": "add_interest_keyword",
"term": "Harness Engineering",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=12, days_seen=11, recent_count=12.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T08:20:35.134979Z",
"action": "add_interest_keyword",
"term": "Skill",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=11, days_seen=9, recent_count=11.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T08:20:35.134979Z",
"action": "add_interest_keyword",
"term": "上下文工程",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=6, days_seen=6, recent_count=6.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T08:20:35.134979Z",
"action": "add_interest_keyword",
"term": "多Agent协作",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=6, days_seen=6, recent_count=6.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T08:20:35.134979Z",
"action": "add_interest_keyword",
"term": "Claude",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=6, days_seen=5, recent_count=6.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T08:20:35.134979Z",
"action": "add_interest_keyword",
"term": "上下文管理",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=6, days_seen=5, recent_count=6.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T08:20:35.134979Z",
"action": "add_interest_keyword",
"term": "渐进式披露",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=5, days_seen=5, recent_count=5.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T08:20:35.134979Z",
"action": "add_interest_keyword",
"term": "Skills",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=5, days_seen=4, recent_count=5.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T08:20:35.134979Z",
"action": "add_interest_keyword",
"term": "SKILL.md",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=4, days_seen=4, recent_count=4.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T08:20:35.134979Z",
"action": "add_interest_keyword",
"term": "CLAUDE.md",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=4, days_seen=4, recent_count=4.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T08:20:35.134979Z",
"action": "add_interest_keyword",
"term": "上下文压缩",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=4, days_seen=4, recent_count=4.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T08:20:35.134979Z",
"action": "add_interest_keyword",
"term": "AI Coding Agent",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=3, days_seen=3, recent_count=3.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T08:20:35.134979Z",
"action": "add_interest_keyword",
"term": "Hermes Agent",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=3, days_seen=3, recent_count=3.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T08:20:35.134979Z",
"action": "add_interest_keyword",
"term": "Vibe Coding",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=5, days_seen=4, recent_count=5.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T08:20:35.134979Z",
"action": "add_interest_keyword",
"term": "Context Engineering",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=3, days_seen=3, recent_count=3.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T08:20:35.134979Z",
"action": "add_interest_keyword",
"term": "Cursor",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=3, days_seen=3, recent_count=3.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T08:20:35.134979Z",
"action": "add_interest_keyword",
"term": "大模型",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=4, days_seen=3, recent_count=4.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_interest_keyword",
"term": "TypeScript",
"reason": "Core language for AI agent development (e.g., Claude Code, Cursor) and backend engineering, complements existing Python/Java/Go keywords.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_interest_keyword",
"term": "代码审查",
"reason": "Chinese term for 'code review', a key practice in backend engineering and AI agent development workflows.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_stopword",
"term": "Channels",
"reason": "Too generic; could refer to communication channels, YouTube channels, or software channels, not specific to user's focus areas.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_stopword",
"term": "Memory",
"reason": "Extremely broad term; could refer to computer memory, human memory, or memory in various contexts, not discriminative enough.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_stopword",
"term": "Prompt",
"reason": "Already covered by 'Prompt Engineering' as a more specific term; 'Prompt' alone is too broad and matches many unrelated articles.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_stopword",
"term": "AGI",
"reason": "Too broad and speculative; not directly actionable for the user's practical engineering focus areas.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_stopword",
"term": "AI日报",
"reason": "Generic news term; not a technical concept or tool, would add noise to the keyword index.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_stopword",
"term": "AIHOT",
"reason": "Unclear meaning, likely a brand or aggregator, not a specific technical term.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_stopword",
"term": "All In Code",
"reason": "Too vague; could refer to a podcast, a philosophy, or a project, not a specific technical concept.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_stopword",
"term": "auto-twitter-campaign",
"reason": "Too specific to a single project/tool, not a general interest keyword for the user's focus areas.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_stopword",
"term": "ChangeSet",
"reason": "Generic term used in version control and databases; too broad to be a useful filter.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_stopword",
"term": "Lumina",
"reason": "Unclear reference; could be a product, framework, or brand, not clearly aligned with user's focus.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_stopword",
"term": "OpenViking",
"reason": "Unclear reference; not a known tool or concept in the user's stated focus areas.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_stopword",
"term": "Seedance 2.0",
"reason": "Unclear reference; likely a product or version, not a general technical term.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_stopword",
"term": "质量门禁",
"reason": "Chinese term for 'quality gate', too generic in software engineering; not specific to user's focus areas.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_alias",
"term": "Agent Skill",
"reason": "Singular variant",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"value": "Agent Skills",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_alias",
"term": "Binlog",
"reason": "Case variant (auto-ranked)",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"value": "binlog",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_alias",
"term": "Coding Agent",
"reason": "Abbreviated form of 'AI Coding Agent', referring to the same concept.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"value": "AI Coding Agent",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_alias",
"term": "Subagent",
"reason": "Case variant",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"value": "SubAgent",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_alias",
"term": "Subagents",
"reason": "Plural variant",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"value": "SubAgent",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_alias",
"term": "vibe coding",
"reason": "Case variant",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"value": "Vibe Coding",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_alias",
"term": "Agent架构",
"reason": "Chinese translation of 'Agent architecture', a core concept in AI Agent engineering.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"value": "AI Agent",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_alias",
"term": "Agent专业化",
"reason": "Chinese term for 'Agent specialization', directly related to Agent engineering.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"value": "AI Agent",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_alias",
"term": "Agent Teams",
"reason": "English equivalent of 'Multi-Agent collaboration', same concept.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"value": "多Agent协作",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_alias",
"term": "Agentic Engineering",
"reason": "Broader term for engineering with AI agents, closely related to Agent engineering focus.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"value": "AI Agent",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_alias",
"term": "CLI工具",
"reason": "Chinese translation of 'CLI tool', same concept.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"value": "CLI",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_alias",
"term": "AI编程",
"reason": "Chinese term for 'AI programming', closely related to AI Coding Agent.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"value": "AI Coding Agent",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_alias",
"term": "记忆管理",
"reason": "Chinese term for 'memory management', closely related to context management in LLM applications.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"value": "上下文管理",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-05-14T09:12:50.749877Z",
"action": "add_alias",
"term": "会话管理",
"reason": "Chinese term for 'session management', related to context management in LLM applications.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
"value": "上下文管理",
"suggestion_date": "2026-05-14",
"based_on_days": 365
},
{
"applied_at": "2026-07-15T02:17:50.155586Z",
"action": "add_interest_keyword",
"term": "强化学习",
"reason": "top 0.8% by frequency,growth=100%,not yet covered by interest keywords or stopwords. Evidence: total_count=10, days_seen=10, recent_count=10.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-07-15.json",
"suggestion_date": "2026-07-15",
"based_on_days": 365
},
{
"applied_at": "2026-07-15T02:17:50.155586Z",
"action": "add_interest_keyword",
"term": "ReAct",
"reason": "top 1.1% by frequency,growth=100%,not yet covered by interest keywords or stopwords. Evidence: total_count=8, days_seen=8, recent_count=8.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-07-15.json",
"suggestion_date": "2026-07-15",
"based_on_days": 365
},
{
"applied_at": "2026-07-15T02:17:50.155586Z",
"action": "add_interest_keyword",
"term": "CLI",
"reason": "top 1.1% by frequency,growth=100%,not yet covered by interest keywords or stopwords. Evidence: total_count=8, days_seen=7, recent_count=8.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-07-15.json",
"suggestion_date": "2026-07-15",
"based_on_days": 365
},
{
"applied_at": "2026-07-15T02:17:50.155586Z",
"action": "add_interest_keyword",
"term": "子Agent",
"reason": "top 1.3% by frequency,growth=100%,not yet covered by interest keywords or stopwords. Evidence: total_count=7, days_seen=6, recent_count=7.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-07-15.json",
"suggestion_date": "2026-07-15",
"based_on_days": 365
},
{
"applied_at": "2026-07-15T02:17:50.155586Z",
"action": "add_interest_keyword",
"term": "Loop Engineering",
"reason": "top 1.5% by frequency,growth=100%,not yet covered by interest keywords or stopwords. Evidence: total_count=6, days_seen=6, recent_count=6.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-07-15.json",
"suggestion_date": "2026-07-15",
"based_on_days": 365
},
{
"applied_at": "2026-07-15T02:17:50.155586Z",
"action": "add_interest_keyword",
"term": "MoE",
"reason": "top 2.0% by frequency,growth=100%,not yet covered by interest keywords or stopwords. Evidence: total_count=5, days_seen=4, recent_count=5.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-07-15.json",
"suggestion_date": "2026-07-15",
"based_on_days": 365
}
]
}
+10 -9
View File
@@ -1,14 +1,12 @@
{
"schema_version": "v1",
"schema_version": "v2",
"interest_keyword_review": {
"min_total_count": 3,
"min_days_seen": 2
"percentile_max": 0.05,
"growth_promotion": 0.5
},
"watch_term_review": {
"min_total_count": 1,
"min_days_seen": 1,
"max_total_count": 2,
"max_days_seen": 2
"percentile_min": 0.05,
"percentile_max": 0.20
},
"alias_review": {
"min_total_count": 2,
@@ -19,7 +17,10 @@
"max_days_seen": 2
},
"notes": [
"当前阶段采用保守阈值,避免在低样本条件下直接扩充 interest_keywords。",
"watch_terms 先用于观察,后续再决定是否升格为 interest_keywords 或进入 alias/stopword 配置。"
"v2: interest/watch 使用百分位排名 + 增速因子替代固定阈值",
"percentile 越小表示排名越高(top 5% = percentile 0.05)",
"growth = recent_count / total_count,衡量近期活跃度",
"watch_term_review 的 percentile_min 可理解为兴趣边界下限,低于此值的词归入 interest 候选",
"growth_promotion(默认 0.5)用于识别近期集中爆发词,即使排位不高也主动推荐确认"
]
}
+121 -1
View File
@@ -1,6 +1,126 @@
[
"1688",
"AGI",
"AIHOT",
"AI日报",
"All In Code",
"Andrej Karpathy",
"Anthropic",
"auto-twitter-campaign",
"Boundaries",
"ChangeSet",
"Channels",
"Claude Fable 5",
"Claude Mythos",
"Cohere",
"Confidence Head",
"Cosmos 3",
"Databricks",
"DINOv2",
"domain-mapping",
"Dropbox",
"EchoGen",
"FLUX.1-dev VAE",
"GB300 GPU",
"GLM 5.2",
"GLM5.0",
"GPT-5.5",
"GPT-Live",
"GPT5.5",
"Grok 4.5",
"GrowBrain",
"iMedImage",
"iMedLoop",
"iMedMaaS",
"iMedStudio",
"J-space",
"JLens",
"John Jumper",
"J空间",
"KAIROS",
"KubeRay",
"LibTV Agent",
"LingBot-Video",
"Lumina",
"Markdown",
"Marvis",
"MDASH",
"Meta Superintelligence Labs",
"MTS",
"Muse Image",
"Muse Video",
"N-gram Embedding",
"OCP China",
"OCP China 2026",
"On-Policy Distillation",
"OPC训练营",
"OpenAI",
"OpenBMC",
"OpenClaw",
"OpenViking",
"Opus 4.8",
"Qwen3",
"Qwen3-30B-A3B",
"RAS API",
"Redfish",
"ScMoE",
"Seal AI Zone",
"SealRouter",
"Seedance 2.0",
"Sonnet 5",
"Spec模式",
"STE固件团队",
"Three.js",
"Unity AI Gateway",
"Vant Weapp",
"WeTV",
"WorkBuddy",
"wpc",
"YOLO Classifier",
"一人公司",
"中国科学技术大学",
"五大扶持体系",
"出门问问",
"分镜",
"剧本",
"奋斗文化",
"字节跳动",
"小银",
"得力",
"德适科技",
"成都天府长岛",
"扣子",
"星云平台",
"火山引擎",
"百度百舸",
"百炼网关",
"科大讯飞",
"腾讯云开发者社区",
"腾讯混元Hy3",
"蚂蚁灵波",
"贝尔实验室",
"质量门禁",
"配乐",
"配音",
"银行客户经理",
"飞盘物理"
"飞书妙搭",
"飞盘物理",
"自动化",
"定时任务",
"开源模型",
"陌生化",
"AlphaFold",
"Brand Kit",
"DataWorks",
"Enhance-Nanocodec",
"IRIS Codec",
"Lovart",
"MiniMax M3",
"Gemini 3.5 Flash",
"Codex",
"CodeBuddy",
"Claude Cowork",
"AGENTS.md",
"Claude",
"RLVR"
]
+1 -1
View File
@@ -1,6 +1,6 @@
{
"schema_version": "v1",
"updated_at": "2026-04-08T02:34:14.194320Z",
"updated_at": "2026-07-15T02:17:50.155586Z",
"terms": [
{
"term": "A2A",
@@ -0,0 +1,151 @@
# 关键词清洗流程概述
> 2026-05-14 初版
> 从"数据记录"到"人工确认落盘"的完整链路
---
## 整体数据流
```
每日日报 pipeline
│
▼
term_index/daily/YYYY-MM-DD.json ← 每天一篇候选文章的热词统计
│
▼
term_index/term_stats.json ← 所有 daily 的汇总(1070 个词)
│
├──── build_review_bundle.py ← 打包为审查数据包
│ │
│ ▼
│ review/keyword-cleanup-bundle.json
│ │
│ ▼
│ generate_term_cleanup_suggestions.py
│ │
│ ▼
│ review/term-cleanup-suggestions-YYYY-MM-DD.json ← 正式建议产物
│ │
│ ▼
│ (可选) review/term-cleanup-suggestions-YYYY-MM-DD.md ← 展示稿
│
├──── 人工确认哪些建议 accept
│
▼
apply_term_suggestions.py ← 写入配置
│
├── configs/filter_context.personal.json ← interest_keywords
├── configs/term_aliases.json ← alias
├── configs/term_stopwords.json ← stopword
├── configs/term_watchlist.json ← watch
└── configs/term_change_log.json ← 变更日志
```
---
## 各环节说明
### 阶段 1:数据记录(每日自动)
```bash
# FreshRSS pipeline 跑完后自动产出
data/term_index/daily/2026-05-14.json
```
- 每天一篇,记录当天候选文章中出现的热词
- 包含 term、total_count、days_seen 等信息
- 目前累计 **41 天**,共 **1070 个独立词**
### 阶段 2:全量汇总(每日自动)
```bash
data/term_index/term_stats.json
```
- 从所有 daily 文件重建,会覆盖重跑
- 按 total_count 排序,前 5 名:OpenClaw(30)、Claude Code(25)、AI Agent(17)、Anthropic(13)、MCP(13)
### 阶段 3:构建审查数据包(手动触发)
```bash
python skills/keyword-cleanup-review/scripts/build_review_bundle.py \
--days 365 \
--top 100 \
--output outputs/term_index/review/keyword-cleanup-bundle.json
```
- 把 term_stats + 当前配置打成一包,方便后续处理
- 输出:`review/keyword-cleanup-bundle.json`
### 阶段 4:生成建议(手动触发)
```bash
python scripts/generate_term_cleanup_suggestions.py \
--bundle outputs/term_index/review/keyword-cleanup-bundle.json \
--emit-markdown
```
#### 当前产出能力
| 建议类型 | 状态 | 当前阈值 | 说明 |
|---------|------|----------|------|
| interest_keyword_suggestions | ✅ **已实现** | total≥3, days≥2 | 产出 20 条 |
| watch_terms | ✅ **已实现** | total≤2, days≤2 | 本次 0 条 |
| alias_suggestions | ❌ **硬编码为空** | policy 有阈值(total≥2, days≥2)但脚本未实现 | |
| stopword_suggestions | ❌ **硬编码为空** | policy 有阈值(total≤2, days≤2)但脚本未实现 | |
**关键发现:** alias 和 stopword 不是"阈值太保守",是 **generate 脚本里压根没写对应的生成函数**。policy 文件里阈值已经配好了(alias: min_total=2/min_days=2,stopword: max_total=2/max_days=2),但脚本第 376-380 行直接硬编码为 `[]` 和 `0`。
### 阶段 5:人工确认(手动)
```
OpenClaw 把建议列给你 → 你确认哪些 accept → 我执行 apply
```
本次模式:
- 高频(≥5次/5天以上)→ 强烈推荐 ✅
- 中频(3-4次)→ 附带建议 ✅
- 泛词 → 建议跳过 ❌
### 阶段 6:落盘配置(手动)
```bash
python scripts/apply_term_suggestions.py \
--suggestions outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json \
--accept-interest 词1 词2 ...
```
- dry-run 预览 → 确认后正式 apply
- 写入 `configs/filter_context.personal.json`
- 同步记录到 `term_change_log.json`
- **不备份原始配置**(待优化)
- **apply 后不自动清理 review 目录**(待优化)
### 阶段 7:维护清理(按需)
由 OpenClaw 侧 `reader-keyword-maintenance` skill 处理:
- 删除旧 markdown 展示稿
- 保留最近一份 bundle
- 保守保留 suggestions JSON
---
## 当前配置资产
| 文件 | 内容 | 数据量 |
|------|------|--------|
| `filter_context.personal.json` | interest_keywords | 52 个 |
| `term_aliases.json` | 别名映射 | 0 组(未启用) |
| `term_stopwords.json` | 停用词 | 0 个(未启用) |
| `term_watchlist.json` | 观察词 | 6 个 |
| `term_change_log.json` | 所有变更记录 | 已记录 |
---
## 待优化项
1. **alias/stopword 建议生成为空** — generate 脚本硬编码缺实现,policy 已有阈值,需要补函数
2. **apply 前无配置备份** — 建议 apply 前自动 cp 备份
3. **apply 后无自动收尾** — 建议 apply 后自动删旧 markdown 和 bundle
4. **alias 识别依赖规则而非 LLM** — 当前全靠统计阈值,无法做语义级判断(如中英文映射、缩写展开)。如果需要高级 alias 识别,可以用 LLM 生成候选,规则脚本做 apply
@@ -0,0 +1,290 @@
# interest/watch 候选引擎改进方案
> 从固定阈值到自适应排位 + 趋势因子的演进
## 1. 背景
### 1.1 当前实现
`build_review_bundle.py` 使用固定的绝对阈值将未覆盖词(uncovered terms)划分为两个候选池:
| 候选池 | 判断条件 | 依据 |
|--------|---------|------|
| `interest_review_candidates` | `total_count >= 3 AND days_seen >= 2` | `policy.interest_keyword_review` |
| `watch_review_candidates` | `total_count <= 2 AND days_seen <= 2` | `policy.watch_term_review` |
`generate_term_cleanup_suggestions.py` 则直接从这两个候选池过滤、去重、排序后输出。
### 1.2 当前方案的问题
**问题一:固定阈值不随数据量自适应**
```
场景 total_count=3 意味着什么
─────────────────────────────────────────────
7 天数据(~200 词) top 15%,有一定区分度 ✅
41 天数据(1070 词) top 5%,区分度更高 ✅ 但阈值没变
未来 200 天 仍然用 3 次,区分度稀释 ❌
```
同一个绝对次数,在不同数据规模下的语义完全不同。手工调阈值不可持续。
**问题二:固定阈值忽略趋势信号**
- "Anthropic":total=13, recent=7 — 近期高活跃,上升趋势
- "Channels":total=3, recent=0 — 早期出现但近期消失
- 当前引擎认为这两个词"都过了 3 次阈值",同等对待。实际一个是强烈买入信号,一个是过气词。
**问题三:interest 和 watch 的分界线是硬的**
total=3 → interest,total=2 → watch。一个词从 2 次变成 3 次就自动"升级",没有过渡、没有缓冲。
### 1.3 讨论结论
与老大讨论后确认:
1. interest/watch 是**统计判断**,不需要大模型介入,纯算法可以解决
2. 当前引擎缺的不是大模型,而是**算法本身没写完**——自适应维度(排位、趋势)还没实现
3. alias 和 stopword 需要语义判断,与 interest/watch 分属不同阶段,不在本方案范围内
4. 修改量小,可以在 1 小时内落地
---
## 2. 设计方案
### 2.1 核心思路
引入两个互补维度替代固定阈值:
```
判定维度 含义 数据来源
────────────────────────────────────────────────────────────
percentile(百分位排名) 该词 total_count 在所有词 term_stats
中的排位占比
growth(增速因子) 近期集中度 = recent_count daily 近 N 天
/ total_count
```
两个维度配合:
- **percentile** 衡量"这个词在当前数据集里有多突出"——消除数据量变化的影响
- **growth** 衡量"这个词是持续出现还是近期爆发"——识别趋势信号
### 2.2 候选池划分逻辑
```
percentile
│
┌─────────────────────┐
│ top 5% │
│ → 建议 interest │ ← 高频稳定词
├─────────────────────┤
│ top 5%-20% │
│ → 建议 watch │ ← 有信号但未达 threshold
├─────────────────────┤
│ bottom 80% │
│ → 暂不处理 │ ← 噪声/低频
└─────────────────────┘
额外规则:
如果词在 top 20% 之外,但 growth > 0.5(近期集中度高)
→ 主动提升到 watch / 主动推 confirm
```
这样就不需要关心"total_count 是 3 还是 5",只看数据自己说话。
### 2.3 接口变化
**`configs/term_cleanup_policy.json`**:
```json
{
"schema_version": "v2",
"interest_keyword_review": {
"percentile_max": 0.05,
"growth_promotion": 0.5
},
"watch_term_review": {
"percentile_min": 0.05,
"percentile_max": 0.20
}
}
```
`v1` 的 `min_total_count`/`min_days_seen` 等绝对阈值字段不再使用。
**`build_review_bundle.py` 输出的候选项**:
```json
{
"term": "Anthropic",
"total_count": 13,
"days_seen": 10,
"percentile": 0.012,
"growth": 0.54,
"reason": "top 1.2% by frequency, 54% of occurrences in recent window — strong signal."
}
```
### 2.4 不需要改动的部分
- `generate_term_cleanup_suggestions.py` — 它只消费候选池,不用改
- `apply_term_suggestions.py` — 消费 suggestions JSON,不用改
- `keyword-cleanup-bundle.json` 结构 — 向后兼容,新增 percentile/growth 字段
---
## 3. 实施计划
### 3.1 改动范围
| 文件 | 改动量 | 内容 |
|------|--------|------|
| `skills/keyword-cleanup-review/scripts/build_review_bundle.py` | ~40 行 | 新增 `_compute_percentile()` 和 `_compute_growth()` 函数;修改候选池生成逻辑;候选项中增加 percentile/growth |
| `configs/term_cleanup_policy.json` | ~10 行 | schema v2:percentile/growth 替代绝对阈值 |
### 3.2 实施步骤
1. **build_review_bundle.py**:在 `top_global_terms` 生成后,增加 percentile 计算函数和 growth 计算函数
2. **build_review_bundle.py**:修改 `interest_review_candidates` 和 `watch_review_candidates` 的生成逻辑,从固定阈值改为 percentile + growth
3. **build_review_bundle.py**:候选项增加 `percentile` 和 `growth` 字段,更新 `reason` 文案
4. **term_cleanup_policy.json**:更新为 v2 schema
5. **验证**:全量跑一次(`--days 365 --top 100`),对比新旧两份输出的差异
### 3.3 验证方法
```bash
# 1. 用旧版生成 baseline
cd /home/ubuntu/zhu/github/reader
python3.11 skills/keyword-cleanup-review/scripts/build_review_bundle.py \
--days 365 --top 100 \
--output /tmp/bundle-baseline.json
# 2. 改代码后用新版生成
python3.11 skills/keyword-cleanup-review/scripts/build_review_bundle.py \
--days 365 --top 100 \
--output /tmp/bundle-new.json
# 3. 对比 governance_hints
python3 -c "
import json
a = json.load(open('/tmp/bundle-baseline.json'))
b = json.load(open('/tmp/bundle-new.json'))
for key in ['interest_review_candidates', 'watch_review_candidates']:
old = set(i['term'] for i in a['governance_hints'][key])
new = set(i['term'] for i in b['governance_hints'][key])
print(f'{key}: 新增={new-old}, 减少={old-new}')
"
```
### 3.4 风险
| 风险 | 概率 | 应对 |
|------|------|------|
| 百分位阈值对特小数据集(如只有 1 天数据)不适用 | 低 | 不足 7 天时降级回绝对阈值 |
| growth 因子对低频词的偏差(total=1, recent=1 → growth=1) | 低 | growth 只对 total>=3 的词计算 |
| 排位突变导致推荐漂移 | 低 | percentil 天然平滑,新增几天数据不会剧烈改变已有词的排位 |
---
## 4. alias/stopword 设计方案
### 4.1 核心判断
alias 和 stopword 需要语义理解,与 interest/watch(纯统计)性质不同。
| 类型 | 需要什么 | 判断方式 |
|------|---------|----------|
| 大小写变体 | 表层 | 规则:casefold 去重 |
| 单复数 | 表层 | 规则:去/加 s 后缀匹配 |
| 分词变体(空格/连字符) | 表层 | 规则:去空格归一 |
| 简写全称(MCP→Model Context Protocol) | **语义** | LLM |
| 中英文(上下文工程→Context Engineering) | **语义** | LLM |
| 同义不同名(Rush→猿辅导 Rush 平台) | **语义** | LLM |
| stopword(大模型、AI 太泛) | **语义** | LLM |
### 4.2 分层方案
```
输入:高频未覆盖词 + 已有 interest 词表
│
├── 规则层(零成本)── 大小写归一、单复数、去空格/连字符
│ 输出候选 alias 对
│
└── LLM 层(每次 ~500 token)── 把候选词表整批给 LLM
做语义聚类
输出 alias 组 + stopword 标记
```
### 4.3 规则层设计
在 `generate_term_cleanup_suggestions.py` 中新增 `_prepare_alias_suggestions()` 函数:
```python
def _prepare_alias_suggestions(top_terms, interest_keywords):
"""
基于表层规则生成 alias 建议。
规则1:casefold 匹配——同一个 casefold 下有多个原文变体
规则2:单复数——去掉/加上末尾 s 后匹配
规则3:分词变体——去空格/连字符后匹配
"""
```
优势:零成本、可复现、可审计。直接写入 suggestions JSON,随 generate 一起输出。
### 4.4 LLM 层设计
单独脚本,非 generate 主链路的一部分。
```bash
python scripts/generate_term_cleanup_semantic_suggestions.py \
--suggestions outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json \
--output outputs/term_index/review/term-cleanup-semantic-suggestions-YYYY-MM-DD.json
```
LLM prompt 设计:
```
你是一个关键词治理助手。以下是一个用户的 interest 关键词列表和一批未覆盖的高频词。
请做三件事:
1. ALIAS:判断哪些未覆盖词是已有 interest 关键词的别名/变体
2. STOPWORD:标记哪些词太宽泛/通用,建议排除
3. PROMOTE:标记哪些新词与用户关注方向一致,建议加入 interest
用户关注方向:AI Agent 工程化、后端工程、开源工具、大模型落地
```
LLM 层输出格式:
```json
{
"alias_suggestions": [
{"from": "Context Engineering", "to": "上下文工程", "reason": "中英文对应同一概念"}
],
"stopword_suggestions": [
{"term": "大模型", "reason": "过于宽泛,高频率但低区分度"}
]
}
```
### 4.5 预期效果
| 覆盖类型 | 规则层 | LLM 层 |
|---------|--------|--------|
| 大小写变体 | ✅ | — |
| 单复数 | ✅ | — |
| 分词变体 | ✅ | — |
| 简写全称 | — | ✅ |
| 中英文映射 | — | ✅ |
| 同义不同名 | — | ✅ |
| stopword 判断 | — | ✅ |
---
## 5. 讨论记录
- 2026-05-14:与老大确认 interest/watch 不需要 LLM,纯算法可解决
- 2026-05-14:确认百分位排名 + 增速因子方案,修改量小,优先落地
- 2026-05-14:确认本方案不改 `generate_term_cleanup_suggestions.py` 和 `apply_term_suggestions.py`
- 2026-05-14:确认 alias/stopword 采用规则层 + LLM 层分层方案,规则层零成本优先
+314
View File
@@ -0,0 +1,314 @@
#!/usr/bin/env python3
"""
Generate semantic keyword suggestions using LLM.
Covers what surface-form rules cannot:
- semantic alias (abbreviation ↔ full name, Chinese ↔ English, synonym)
- stopword (overly broad / low-discrimination terms)
- promote (new term that aligns with user's focus areas)
Usage:
python scripts/generate_term_cleanup_semantic_suggestions.py \
--bundle outputs/term_index/review/keyword-cleanup-bundle.json \
--output outputs/term_index/review/term-cleanup-semantic-suggestions-2026-05-14.json
"""
from __future__ import annotations
import argparse
import json
import os
import re
import sys
import time
from datetime import datetime, timezone
from pathlib import Path
from typing import Any
from urllib.request import Request, urlopen
REPO_ROOT = Path(__file__).resolve().parents[1]
DEFAULT_BUNDLE_PATH = REPO_ROOT / "outputs" / "term_index" / "review" / "keyword-cleanup-bundle.json"
DEFAULT_OUTPUT_DIR = REPO_ROOT / "outputs" / "term_index" / "review"
def _load_json(path: Path) -> Any:
return json.loads(path.read_text(encoding="utf-8-sig"))
def _save_json(path: Path, payload: dict[str, Any]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
def _load_env(path: Path) -> dict[str, str]:
"""Load key=value pairs from .env file."""
env: dict[str, str] = {}
if not path.exists():
return env
for line in path.read_text(encoding="utf-8").splitlines():
line = line.strip()
if not line or line.startswith("#") or "=" not in line:
continue
key, _, value = line.partition("=")
env[key.strip()] = value.strip().strip("\"'")
return env
def _build_prompt(
interest_keywords: list[str],
rule_alias_suggestions: list[dict[str, str]],
candidate_terms: list[dict[str, Any]],
relevant_watch_terms: list[dict[str, Any]],
) -> str:
"""Build the LLM prompt for semantic suggestions."""
interest_bullets = "\n".join(f" - {t}" for t in sorted(interest_keywords))
candidate_bullets = "\n".join(
f" - {t['term']} (count={t['total_count']}, days={t['days_seen']})"
for t in candidate_terms[:40]
)
# Alias from rule layer (for LLM to build on, not duplicate)
rule_alias_text = ""
if rule_alias_suggestions:
rule_alias_text = "\nSurface-form alias (already identified, skip these):\n" + "\n".join(
f" {a['from']} → {a['to']} ({a['reason']})"
for a in rule_alias_suggestions
)
watch_text = ""
if relevant_watch_terms:
watch_text = "\nWatch terms (low-frequency but potentially relevant):\n" + "\n".join(
f" {t['term']} (count={t['total_count']}, days={t['days_seen']})"
for t in relevant_watch_terms[:20]
)
return f"""You are a keyword governance assistant for an AI engineer. Your job is to analyze keyword data and produce structured suggestions.
## User's focus areas
- AI Agent engineering (Skills, Harness, MCP, Agent architecture)
- Backend engineering (Java, Go, Kubernetes, MySQL, distributed systems)
- Open source AI tools and practices (Claude Code, Cursor, DeepSeek, OpenClaw)
- LLM application engineering (context engineering, RAG, prompt engineering)
## Interest keywords (52 already configured)
{interest_bullets}
## Uncovered candidate terms (sorted by frequency)
{candidate_bullets}
{watch_text}{rule_alias_text}
## Task
Analyze the candidate terms and output a JSON object with exactly three keys:
1. "semantic_alias": array of alias suggestions that SURFACE RULES CAN'T CATCH (e.g. abbreviation↔full name, Chinese↔English, different naming for the same concept).
Format: [{{"from": "<variant>", "to": "<canonical interest keyword>", "reason": "<why>"}}]
2. "stopword": array of terms that are too broad/generic to be useful as filters. A stopword is a term that appears frequently but has LOW DISCRIMINATION — it matches too many unrelated articles and clutters the keyword index.
Format: [{{"term": "<term>", "reason": "<why it should be a stopword>"}}]
3. "promote_to_interest": array of uncovered terms that align well with the user's focus areas and should be added as interest keywords.
Format: [{{"term": "<term>", "reason": "<why it fits>"}}]
## Rules
- Be conservative. When in doubt, leave it out.
- Only suggest alias for terms that clearly refer to the SAME concept as an existing interest keyword.
- Only suggest stopword for terms that are genuinely too broad (appear in many unrelated contexts).
- Only suggest promote for terms that clearly match the user's stated focus areas.
- Output valid JSON only, no markdown, no explanation outside the JSON."""
def _call_llm(prompt: str, api_url: str, model: str, api_key: str) -> str:
"""Call LLM API and return the response text."""
payload = json.dumps({
"model": model,
"messages": [{"role": "user", "content": prompt}],
"temperature": 0.1,
"max_tokens": 2048,
}).encode("utf-8")
req = Request(
api_url.rstrip("/") + "/chat/completions",
data=payload,
headers={
"Content-Type": "application/json",
"Authorization": f"Bearer {api_key}",
},
)
max_retries = 3
for attempt in range(max_retries):
try:
with urlopen(req, timeout=120) as resp:
result = json.loads(resp.read().decode("utf-8"))
return result["choices"][0]["message"]["content"]
except Exception as e:
if attempt < max_retries - 1:
wait = 2 ** attempt
print(f" LLM call failed (attempt {attempt+1}/{max_retries}): {e}", file=sys.stderr)
print(f" Retrying in {wait}s...", file=sys.stderr)
time.sleep(wait)
else:
raise
def _parse_llm_response(text: str) -> dict[str, list[dict[str, str]]]:
"""Extract JSON from LLM response (may contain markdown fences)."""
# Try to find JSON block
json_match = re.search(r"```(?:json)?\s*\n?(\{.*?\})\s*\n?```", text, re.DOTALL)
if json_match:
text = json_match.group(1)
# Clean up: remove any text before { or after }
start = text.find("{")
end = text.rfind("}")
if start >= 0 and end > start:
text = text[start : end + 1]
try:
result = json.loads(text)
except json.JSONDecodeError:
# Try partial recovery
print(f" Warning: LLM response not clean JSON, attempting recovery", file=sys.stderr)
print(f" Raw: {text[:500]}", file=sys.stderr)
return {"semantic_alias": [], "stopword": [], "promote_to_interest": []}
# Normalize keys
normalized = {
"semantic_alias": result.get("semantic_alias", result.get("alias", [])),
"stopword": result.get("stopword", result.get("stopword_suggestions", [])),
"promote_to_interest": result.get("promote_to_interest", result.get("promote", [])),
}
# Ensure each is a list
for key in normalized:
if not isinstance(normalized[key], list):
normalized[key] = []
return normalized
def main() -> None:
parser = argparse.ArgumentParser(description="Generate semantic keyword suggestions via LLM.")
parser.add_argument("--bundle", type=Path, default=DEFAULT_BUNDLE_PATH, help="Review bundle JSON path")
parser.add_argument("--suggestions", type=Path, default=None, help="Existing suggestions JSON (for rule alias context)")
parser.add_argument("--output", type=Path, default=None, help="Output JSON path (auto-generated if omitted)")
parser.add_argument("--llm-api-url", type=str, default=None, help="LLM API base URL")
parser.add_argument("--llm-model", type=str, default=None, help="LLM model name")
parser.add_argument("--llm-api-key", type=str, default=None, help="LLM API key")
parser.add_argument("--dry-run", action="store_true", help="Print prompt and exit without calling LLM")
args = parser.parse_args()
# Load config
env_path = REPO_ROOT / ".env"
env = _load_env(env_path) if env_path.exists() else {}
api_url = args.llm_api_url or os.environ.get("LLM_API_URL") or env.get("LLM_API_URL", "https://api.deepseek.com")
# Map OpenClaw model aliases to actual API model names
model_raw = args.llm_model or os.environ.get("LLM_MODEL") or env.get("LLM_MODEL", "deepseek-chat")
MODEL_ALIAS_MAP = {
"deepseek/deepseek-v4-flash": "deepseek-chat",
"deepseek/deepseek-chat": "deepseek-chat",
"deepseek-v4-flash": "deepseek-chat",
"deepseek-chat": "deepseek-chat",
}
model = MODEL_ALIAS_MAP.get(model_raw, model_raw)
api_key = args.llm_api_key or os.environ.get("LLM_API_KEY") or env.get("LLM_API_KEY", "")
if not api_key:
print("Error: No LLM API key found. Set LLM_API_KEY in .env or pass --llm-api-key.", file=sys.stderr)
sys.exit(1)
# Load bundle
if not args.bundle.exists():
print(f"Error: Bundle not found: {args.bundle}", file=sys.stderr)
sys.exit(1)
bundle = _load_json(args.bundle)
current_config = bundle.get("current_config", {})
interest_keywords = current_config.get("interest_keywords", [])
top_global_terms = bundle.get("top_global_terms", [])
governance_hints = bundle.get("governance_hints", {})
# Build candidate list (uncovered terms from interest + watch candidates)
candidate_terms = []
for item in governance_hints.get("interest_review_candidates", []):
if isinstance(item, dict):
candidate_terms.append({
"term": item.get("term", ""),
"total_count": item.get("total_count", 0),
"days_seen": item.get("days_seen", 0),
"percentile": item.get("percentile", 0),
"growth": item.get("growth", 0),
})
for item in governance_hints.get("watch_review_candidates", []):
if isinstance(item, dict):
# Avoid duplicates
if not any(c["term"] == item.get("term") for c in candidate_terms):
candidate_terms.append({
"term": item.get("term", ""),
"total_count": item.get("total_count", 0),
"days_seen": item.get("days_seen", 0),
"percentile": item.get("percentile", 0),
"growth": item.get("growth", 0),
})
# Sort by total_count descending
candidate_terms.sort(key=lambda x: -x["total_count"])
relevant_watch_terms = governance_hints.get("watch_review_candidates", [])[:20]
# Load rule-layer alias suggestions if available
rule_alias = []
if args.suggestions and args.suggestions.exists():
s = _load_json(args.suggestions)
rule_alias = s.get("alias_suggestions", [])
# Build prompt
prompt = _build_prompt(
interest_keywords=interest_keywords,
rule_alias_suggestions=rule_alias,
candidate_terms=candidate_terms,
relevant_watch_terms=relevant_watch_terms,
)
# Determine output path
suggestion_date = datetime.now(timezone.utc).date().isoformat()
output_path = args.output or (DEFAULT_OUTPUT_DIR / f"term-cleanup-semantic-suggestions-{suggestion_date}.json")
if args.dry_run:
print("=== DRY RUN: Prompt ===")
print(prompt)
print("\n=== END ===")
print(f"\nWould write to: {output_path}")
return
# Call LLM
print(f"Calling LLM ({model})...", file=sys.stderr)
response = _call_llm(prompt, api_url, model, api_key)
print(f"LLM response received ({len(response)} chars)", file=sys.stderr)
# Parse
parsed = _parse_llm_response(response)
# Build output
output = {
"date": suggestion_date,
"source_bundle": str(args.bundle),
"model": model,
"interest_keyword_count": len(interest_keywords),
"candidate_count": len(candidate_terms),
**parsed,
}
_save_json(output_path, output)
summary = {
"output": str(output_path),
"semantic_alias": len(output.get("semantic_alias", [])),
"stopword": len(output.get("stopword", [])),
"promote_to_interest": len(output.get("promote_to_interest", [])),
}
print(json.dumps(summary, ensure_ascii=False, indent=2))
if __name__ == "__main__":
main()
+128 -3
View File
@@ -175,6 +175,121 @@ def _prepare_watch_suggestions(bundle: dict[str, Any], reserved_terms: set[str])
return suggestions
def _prepare_alias_suggestions(
bundle: dict[str, Any],
all_terms: list[dict[str, Any]] | None = None,
) -> list[dict[str, Any]]:
"""
Generate alias suggestions using surface-form rules (no LLM).
Rules:
1. casefold match — same normalized form, different original casing
2. trailing-s singularization — singular/plural variants
3. whitespace/hyphen normalization — word boundary variants
Scans all_terms (full term_stats) if provided; otherwise falls back
to top_global_terms from the bundle.
"""
current_config = _require_dict(bundle.get("current_config"), "bundle.current_config")
interest_keywords = _require_list(
current_config.get("interest_keywords"), "bundle.current_config.interest_keywords"
)
top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms")
source_terms = all_terms if all_terms is not None else top_global_terms
interest_set = {_term_key(t) for t in interest_keywords if isinstance(t, str)}
interest_originals: set[str] = {t for t in interest_keywords if isinstance(t, str)}
# Build full casefold → [original forms] map
cf_map: dict[str, list[str]] = {}
for item in source_terms:
term = None
if isinstance(item, dict):
term = item.get("term")
elif isinstance(item, str):
term = item
if not isinstance(term, str) or not term.strip():
continue
key = _term_key(term)
if key not in cf_map:
cf_map[key] = []
if term not in cf_map[key]:
cf_map[key].append(term)
suggestions: list[dict[str, Any]] = []
seen_pairs: set[tuple[str, str]] = set()
def _add(from_term: str, to_term: str, reason: str) -> None:
pair = (_term_key(from_term), _term_key(to_term))
if pair in seen_pairs:
return
seen_pairs.add(pair)
suggestions.append({"from": from_term, "to": to_term, "reason": reason})
# Build a set of all term keys from source for quick lookup
source_keys = set(cf_map.keys())
# Rule 1: casefold match — same normalized form, different casing
for key, variants in cf_map.items():
if len(variants) < 2:
continue
canonical = None
alt_forms = []
for v in variants:
if v in interest_originals:
canonical = v
else:
alt_forms.append(v)
if canonical and alt_forms:
for alt in alt_forms:
_add(alt, canonical, "Case variant")
elif len(variants) >= 2 and not canonical:
# None is canonical — suggest the highest-frequency form
ranked = sorted(variants, key=lambda t: -(
next(
(it.get("total_count", 0) for it in top_global_terms if it.get("term") == t),
0,
)
))
for alt in ranked[1:]:
_add(alt, ranked[0], "Case variant (auto-ranked)")
# Rule 2: singular/plural — trailing-s normalization
# Check all source terms (not just interest keys) for bidirectional matching
for key in source_keys:
if key in interest_set:
continue
if key.endswith("s") and len(key) > 2:
singular_key = key.rstrip("s")
if singular_key in interest_set and singular_key != key:
# Find canonical interest keyword
canon = next((t for t in interest_keywords if _term_key(t) == singular_key), None)
from_form = cf_map[key][0]
if canon:
_add(from_form, canon, "Plural variant")
# singular form → interest has plural
plural_key = key + "s"
if plural_key in interest_set and plural_key != key:
canon = next((t for t in interest_keywords if _term_key(t) == plural_key), None)
from_form = cf_map[key][0]
if canon:
_add(from_form, canon, "Singular variant")
# Rule 3: whitespace/hyphen normalization
for key in source_keys:
if key in interest_set:
continue
normalized = key.replace("-", "").replace("_", "").replace(" ", "")
if normalized in interest_set and normalized != key:
canon = next((t for t in interest_keywords if _term_key(t) == normalized), None)
from_form = cf_map[key][0]
if canon:
_add(from_form, canon, "Whitespace/punctuation variant")
suggestions.sort(key=lambda x: (x["from"].casefold(), x["to"].casefold()))
return suggestions
def _render_table(items: list[dict[str, Any]]) -> str:
if not items:
return "_None in this pass._\n"
@@ -361,9 +476,19 @@ def main() -> None:
markdown_output=args.markdown_output,
)
# Load full term_stats for alias scanning (bundle only has top N)
stats_path = REPO_ROOT / "data" / "term_index" / "term_stats.json"
all_stats_terms: list[str] = []
if stats_path.exists():
stats_payload = _load_json(stats_path)
raw_terms = stats_payload.get("terms") if isinstance(stats_payload, dict) else []
if isinstance(raw_terms, list):
all_stats_terms = [str(t["term"]) for t in raw_terms if isinstance(t, dict) and isinstance(t.get("term"), str)]
interest_items = _prepare_interest_suggestions(bundle)
reserved_terms = {_term_key(str(item.get("term") or "")) for item in interest_items}
watch_items = _prepare_watch_suggestions(bundle, reserved_terms=reserved_terms)
alias_items = _prepare_alias_suggestions(bundle, all_terms=all_stats_terms)
suggestions = {
"date": suggestion_date,
@@ -373,10 +498,10 @@ def main() -> None:
"summary": {
"interest_keyword_suggestions": len(interest_items),
"watch_terms": len(watch_items),
"alias_suggestions": 0,
"alias_suggestions": len(alias_items),
"stopword_suggestions": 0,
},
"alias_suggestions": [],
"alias_suggestions": alias_items,
"stopword_suggestions": [],
"interest_keyword_suggestions": interest_items,
"watch_terms": watch_items,
@@ -400,7 +525,7 @@ def main() -> None:
"markdown_output": str(markdown_output_path) if args.emit_markdown else None,
"interest_keyword_suggestions": len(interest_items),
"watch_terms": len(watch_items),
"alias_suggestions": 0,
"alias_suggestions": len(alias_items),
"stopword_suggestions": 0,
"emit_markdown": args.emit_markdown,
}
+68 -24
View File
@@ -26,7 +26,7 @@ description: 生成 reader 项目的正式关键词 review 输入。当用户需
## 工作流程
1. 构建精简的审查数据包(临时工作文件):
### Phase 1:构建审查数据包
```bash
python skills/keyword-cleanup-review/scripts/build_review_bundle.py
@@ -34,51 +34,92 @@ python skills/keyword-cleanup-review/scripts/build_review_bundle.py
可选参数:
- `--days 7`
- `--top 50`
- `--days 7`(默认 7,建议传 365 覆盖全量)
- `--top 100`(考虑的词数)
- `--output outputs/term_index/review/keyword-cleanup-bundle.json`
2. 阅读建议模式:
#### 候选引擎策略
- `skills/keyword-cleanup-review/references/suggestion-schema.md`
根据 `configs/term_cleanup_policy.json` 的 `schema_version` 自动切换:
3. 运行 suggestions 生成脚本:
| 版本 | 策略 | 说明 |
|------|------|------|
| v1(旧) | 固定阈值(total≥3/days≥2 → interest) | 小数据集兼容 |
| v2(当前默认) | 百分位排名 + 增速因子 | 自适应数据量,不需要手工调阈值 |
v2 策略说明:
- **percentile**:total_count 在所有词里的排位占比。top 5% → interest 候选,5%-20% → watch 候选
- **growth**:recent_count / total_count,衡量近期活跃度。growth≥0.5 的排位外词也会主动推荐
### Phase 2:生成建议(规则层)
```bash
python scripts/generate_term_cleanup_suggestions.py ^
python scripts/generate_term_cleanup_suggestions.py \
--bundle outputs/term_index/review/keyword-cleanup-bundle.json
```
默认生成:
- 一份符合模式的 JSON 建议文件(正式建议产物,也是 review / apply 之间唯一正式输入)
如需人工审阅展示稿,再显式加:
如需人工审阅展示稿:
```bash
python scripts/generate_term_cleanup_suggestions.py ^
--bundle outputs/term_index/review/keyword-cleanup-bundle.json ^
python scripts/generate_term_cleanup_suggestions.py \
--bundle outputs/term_index/review/keyword-cleanup-bundle.json \
--emit-markdown
```
这时才会额外生成:
#### 产出能力
- 一份简短的供人工审阅的 Markdown 报告(临时展示稿)
| 建议类型 | 状态 | 方法 |
|---------|------|------|
| interest 建议 | ✅ 已实现 | 百分位 top 5% + 增速促活 |
| watch 建议 | ✅ 已实现 | 百分位 5%-20% |
| alias 建议 | ✅ 已实现 | 规则层:大小写归一、单复数、去空格/连字符 |
| stopword 建议 | ❌ 规则层空缺 | 见 Phase 3(LLM 层) |
4. 严格保持边界:
默认生成:
- `term-cleanup-suggestions-YYYY-MM-DD.json`(正式建议产物)
显式加 `--emit-markdown` 额外生成:
- `term-cleanup-suggestions-YYYY-MM-DD.md`(临时展示稿)
### Phase 3:生成建议(LLM 层,可选)
规则层覆盖不了 alias(中英文对应、缩写展开、同义不同名)和 stopword 判断,需要 LLM 辅助:
```bash
python scripts/generate_term_cleanup_semantic_suggestions.py \
--bundle outputs/term_index/review/keyword-cleanup-bundle.json \
--suggestions outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json \
--output outputs/term_index/review/term-cleanup-semantic-suggestions-YYYY-MM-DD.json
```
从 `.env` 读取 LLM 配置(`LLM_API_URL` / `LLM_MODEL` / `LLM_API_KEY`),使用 DeepSeek API。
输出三部分:
| 输出 | 说明 |
|------|------|
| `semantic_alias` | 语义级别名(中英文、缩写、同义不同名) |
| `stopword` | 泛词过滤建议(规则层做不了的需要语义判断的) |
| `promote_to_interest` | 与用户关注方向一致的新词,建议加入 interest |
**注:LLM 层产物是候选,不应自动 apply,需要人工确认后由 OpenClaw 编排 apply。**
### Phase 4:输出给 OpenClaw 编排
- `suggestions JSON` = review / apply 之间唯一正式建议输入
- `semantic-suggestions JSON` = LLM 补充建议,需要人工筛选后合并到 suggestions JSON 再 apply
- Markdown = 临时展示层
- 后续汇报、确认、dry-run、apply、收尾清理由 OpenClaw 编排层执行
### Phase 5:严格保持边界
- 建议 `configs/term_aliases.json` 的修改
- 建议 `configs/term_stopwords.json` 的修改
- 建议 `configs/filter_context.personal.json` 的新增
- **LLM 层产出(semantic-suggestions)不自动 apply**,需人工确认后由 OpenClaw 编排层执行
- 除非用户明确要求,否则不要直接编辑这些文件
- 除非用户要求修改规则逻辑,否则不要建议直接编辑 `configs/filter_rules.json`
5. 输出交接口径:
- 将 JSON suggestions 视为正式 review 输入
- 将 Markdown 视为可选展示层
- 后续汇报、确认、dry-run apply、正式 apply、收尾清理应由 OpenClaw 编排层继续执行
## 审查启发式规则
优先考虑以下决策:
@@ -141,6 +182,7 @@ JSON 输出应遵循:
短期保留:
- `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json`
- `outputs/term_index/review/term-cleanup-semantic-suggestions-YYYY-MM-DD.json`
临时产物:
@@ -173,7 +215,9 @@ JSON 输出应遵循:
## 资源
- 脚本:
- `scripts/build_review_bundle.py`
- `skills/keyword-cleanup-review/scripts/build_review_bundle.py`
- `scripts/generate_term_cleanup_suggestions.py`
- `scripts/generate_term_cleanup_semantic_suggestions.py`(LLM 层)
- 参考文档:
- `references/suggestion-schema.md`
- `plans/keyword-cleanup-interest-watch-engine-improvement.md`(v2 引擎设计)
@@ -9,16 +9,15 @@ from typing import Any
DEFAULT_POLICY: dict[str, Any] = {
"schema_version": "v1",
"schema_version": "v2",
"interest_keyword_review": {
"min_total_count": 3,
"min_days_seen": 2,
"percentile_min": 0.0,
"percentile_max": 0.05,
"growth_promotion": 0.5,
},
"watch_term_review": {
"min_total_count": 1,
"min_days_seen": 1,
"max_total_count": 2,
"max_days_seen": 2,
"percentile_min": 0.05,
"percentile_max": 0.20,
},
"alias_review": {
"min_total_count": 2,
@@ -28,6 +27,11 @@ DEFAULT_POLICY: dict[str, Any] = {
"max_total_count": 2,
"max_days_seen": 2,
},
"notes": [
"v2: interest/watch 使用百分位排名 + 增速因子替代固定阈值",
"percentile 越小表示排名越高(top 5% = percentile 0.05)",
"growth = recent_count / total_count,衡量近期活跃度",
],
}
@@ -126,6 +130,37 @@ def _within_watch_thresholds(item: dict[str, Any], thresholds: dict[str, Any]) -
)
def _compute_percentile(value: int, sorted_values: list[int]) -> float:
"""
Return the percentile rank of `value` in `sorted_values` (ascending).
0.0 = highest frequency (top rank), 1.0 = lowest frequency (bottom rank).
"""
if not sorted_values:
return 1.0
# bisect_left — count of values strictly less than `value`
lo, hi = 0, len(sorted_values)
while lo < hi:
mid = (lo + hi) // 2
if sorted_values[mid] < value:
lo = mid + 1
else:
hi = mid
rank = lo
# invert: smallest value → rank=0 → 1.0 (bottom)
# largest value → rank=len → 0.0 (top)
return 1.0 - (rank / len(sorted_values))
def _compute_growth(recent_count: int, total_count: int) -> float:
"""
Return growth factor: recent_count / total_count.
Only meaningful when total_count >= 3; returns 0.0 for small counts.
"""
if total_count < 3:
return 0.0
return recent_count / total_count
def main() -> None:
parser = argparse.ArgumentParser(
description="Build a compact review bundle for the keyword-cleanup-review skill."
@@ -235,6 +270,13 @@ def main() -> None:
alias_values = _casefold_set(list(aliases.values()))
watch_set = _casefold_set([str(item.get("term", "")) for item in watchlist])
# Build a sorted list of all total_counts for percentile computation
all_total_counts = sorted(
int(item.get("total_count") or 0)
for item in stats_terms
if isinstance(item, dict) and isinstance(item.get("term"), str)
)
top_global_terms = []
for item in stats_terms[: args.top]:
if not isinstance(item, dict):
@@ -256,42 +298,108 @@ def main() -> None:
"is_alias_target": folded in alias_values,
"in_watchlist": folded in watch_set,
"recent_count": recent_counter.get(term, 0),
"percentile": _compute_percentile(
int(item.get("total_count") or 0), all_total_counts
),
"growth": _compute_growth(
recent_counter.get(term, 0),
int(item.get("total_count") or 0),
),
}
)
# Keep more uncovered terms for percentile-based selection
uncovered_terms = [
item for item in top_global_terms if not item["in_interest_keywords"] and not item["is_stopword"]
][:20]
][:100]
policy_version = (policy.get("schema_version") if isinstance(policy, dict) else None) or "v1"
interest_thresholds = policy.get("interest_keyword_review") if isinstance(policy, dict) else {}
watch_thresholds = policy.get("watch_term_review") if isinstance(policy, dict) else {}
if policy_version == "v2" or "percentile_max" in interest_thresholds:
# v2: percentile + growth based selection
pct_min_interest = float(interest_thresholds.get("percentile_min", 0.0))
pct_max_interest = float(interest_thresholds.get("percentile_max", 0.05))
growth_promo = float(interest_thresholds.get("growth_promotion", 0.5))
pct_min_watch = float(watch_thresholds.get("percentile_min", 0.05))
pct_max_watch = float(watch_thresholds.get("percentile_max", 0.20))
interest_candidates_raw = [
item for item in uncovered_terms
if not item["in_watchlist"]
and pct_min_interest <= item["percentile"] <= pct_max_interest
]
watch_candidates_raw = [
item for item in uncovered_terms
if not item["in_watchlist"]
and pct_min_watch < item["percentile"] <= pct_max_watch
]
# Growth boost: terms outside watch range but with strong growth signal
growth_boost_candidates = [
item for item in uncovered_terms
if not item["in_watchlist"]
and item["percentile"] > pct_max_watch
and item["growth"] >= growth_promo
]
else:
# v1 fallback: fixed thresholds
interest_candidates_raw = [
item for item in uncovered_terms
if not item["in_watchlist"] and _meets_min_thresholds(item, interest_thresholds)
]
watch_candidates_raw = [
item for item in uncovered_terms
if not item["in_watchlist"]
and not _meets_min_thresholds(item, interest_thresholds)
and _within_watch_thresholds(item, watch_thresholds)
]
growth_boost_candidates = []
interest_review_candidates = [
{
"term": item["term"],
"total_count": item["total_count"],
"days_seen": item["days_seen"],
"percentile": item["percentile"],
"growth": item["growth"],
"reason": (
"Meets the configured interest-keyword review threshold and is not yet covered "
"by interest keywords or stopwords."
f"top {item['percentile']:.1%} by frequency,"
f"growth={item['growth']:.0%},"
"not yet covered by interest keywords or stopwords."
),
}
for item in uncovered_terms
if not item["in_watchlist"] and _meets_min_thresholds(item, interest_thresholds)
for item in interest_candidates_raw
][:20]
watch_review_candidates = [
{
"term": item["term"],
"total_count": item["total_count"],
"days_seen": item["days_seen"],
"percentile": item["percentile"],
"growth": item["growth"],
"reason": (
"Falls into the configured watch-term review range and should be observed "
"before promotion into interest keywords."
f"top {item['percentile']:.1%} by frequency,"
f"growth={item['growth']:.0%},"
"fell into watch-review range."
),
}
for item in uncovered_terms
if not item["in_watchlist"]
and not _meets_min_thresholds(item, interest_thresholds)
and _within_watch_thresholds(item, watch_thresholds)
for item in watch_candidates_raw
][:20]
growth_boost_review_items = [
{
"term": item["term"],
"total_count": item["total_count"],
"days_seen": item["days_seen"],
"percentile": item["percentile"],
"growth": item["growth"],
"reason": (
f"growth spike: {item['growth']:.0%} of occurrences in recent window "
f"(total={item['total_count']}, days={item['days_seen']})."
),
}
for item in growth_boost_candidates
][:5]
recent_hot_terms = sorted(
({"term": term, "recent_count": count} for term, count in recent_counter.items()),
key=lambda item: (-item["recent_count"], item["term"].casefold(), item["term"]),
@@ -330,6 +438,7 @@ def main() -> None:
"governance_hints": {
"interest_review_candidates": interest_review_candidates,
"watch_review_candidates": watch_review_candidates,
"growth_boost_review_items": growth_boost_review_items,
},
}
_save_json(args.output, bundle)
+24 -2
View File
@@ -6,8 +6,30 @@ from .freshrss_pipeline_jobs import (
start_freshrss_pipeline_job,
)
from .query_service import get_delivery_payload, get_run_report, get_run_status, list_run_artifacts, list_runs
from .resume_jobs import get_resume_job_result, get_resume_job_status, start_resume_job
from .resume_service import inspect_resume_plan, resume_run
# NOTE: resume_jobs and resume_service are NOT eagerly imported here to avoid
# a circular import chain:
# workflows/freshrss_pipeline.py -> runtime -> resume_jobs -> resume_service
# -> workflows/freshrss_pipeline.py (circular!)
# They are lazy-loaded via __getattr__ when accessed as summary_mcp.runtime.*
def __getattr__(name):
import importlib
_LAZY = {
"get_resume_job_result": ("resume_jobs", "get_resume_job_result"),
"get_resume_job_status": ("resume_jobs", "get_resume_job_status"),
"start_resume_job": ("resume_jobs", "start_resume_job"),
"inspect_resume_plan": ("resume_service", "inspect_resume_plan"),
"resume_run": ("resume_service", "resume_run"),
}
if name in _LAZY:
mod_name, attr_name = _LAZY[name]
mod = importlib.import_module(f".{mod_name}", __package__)
return getattr(mod, attr_name)
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
__all__ = [
"ArtifactRecord",
@@ -9,6 +9,8 @@ from pathlib import Path
from typing import Any
from uuid import uuid4
from dotenv import dotenv_values
from .run_store import RunStore
from .query_service import _resolve_run_record
@@ -29,6 +31,25 @@ DEFAULT_STAGES = [
]
MIN_JOB_STALE_SECONDS = 30 * 60
MAX_JOB_STALE_SECONDS = 6 * 60 * 60
DEFAULT_DOTENV_PATH = REPO_ROOT / ".env"
def _build_subprocess_env() -> dict[str, str]:
"""Build an env dict for subprocess, merging parent env with .env values.
The subprocess inherits the Hermes MCP server's environment, but .env values
may not be in os.environ at the time the subprocess is spawned. This function
loads them from .env and merges them in so the child process sees all needed
variables (LLM_API_KEY, LLM_MODEL, LLM_API_URL, FRESHRSS_*, etc.) directly
in os.environ, avoiding any dotenv-loading timing issues inside the subprocess.
"""
env = os.environ.copy()
if DEFAULT_DOTENV_PATH.exists():
for key, value in dotenv_values(DEFAULT_DOTENV_PATH).items():
if isinstance(key, str) and isinstance(value, str) and value:
# Only set if not already present in parent env
env.setdefault(key, value)
return env
def _now() -> datetime:
@@ -218,6 +239,7 @@ def start_freshrss_pipeline_job(
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL,
start_new_session=True,
env=_build_subprocess_env(),
)
except Exception as exc:
report_file = _write_job_report(
+19
View File
@@ -9,6 +9,8 @@ from pathlib import Path
from typing import Any
from uuid import uuid4
from dotenv import dotenv_values
from .query_service import _resolve_run_record
from .resume_service import (
SUPPORTED_RESUME_STAGES,
@@ -38,6 +40,22 @@ DEFAULT_STAGES = [
]
MIN_JOB_STALE_SECONDS = 30 * 60
MAX_JOB_STALE_SECONDS = 6 * 60 * 60
DEFAULT_DOTENV_PATH = REPO_ROOT / ".env"
def _build_subprocess_env() -> dict[str, str]:
"""Build an env dict for subprocess, merging parent env with .env values.
Ensures the subprocess sees all needed variables (LLM_API_KEY, LLM_MODEL,
LLM_API_URL, FRESHRSS_*, etc.) directly in os.environ, avoiding dotenv
timing issues in the child process.
"""
env = os.environ.copy()
if DEFAULT_DOTENV_PATH.exists():
for key, value in dotenv_values(DEFAULT_DOTENV_PATH).items():
if isinstance(key, str) and isinstance(value, str) and value:
env.setdefault(key, value)
return env
def _now() -> datetime:
@@ -276,6 +294,7 @@ def start_resume_job(*, run_id: str) -> dict[str, Any]:
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL,
start_new_session=True,
env=_build_subprocess_env(),
)
except Exception as exc:
report_file = _write_job_report(
+46 -32
View File
@@ -2,6 +2,7 @@ from __future__ import annotations
import json
import os
from concurrent.futures import ThreadPoolExecutor, as_completed
from datetime import date, datetime, timezone
from pathlib import Path
from typing import Any
@@ -192,7 +193,7 @@ def _build_candidate_batch_payload(*, run_id: str, item_contexts: list[dict[str,
}
def _persist_summary_batch_artifact(*, run_store: RunStore, run_dir: Path, item_contexts: list[dict[str, Any]]) -> Path:
def _persist_summary_batch_artifact(*, run_store, run_dir: Path, item_contexts: list[dict[str, Any]]) -> Path:
output_path = _summary_batch_output(run_dir)
_save_json(
output_path,
@@ -202,7 +203,7 @@ def _persist_summary_batch_artifact(*, run_store: RunStore, run_dir: Path, item_
return output_path
def _persist_candidate_batch_artifact(*, run_store: RunStore, run_dir: Path, item_contexts: list[dict[str, Any]]) -> Path:
def _persist_candidate_batch_artifact(*, run_store, run_dir: Path, item_contexts: list[dict[str, Any]]) -> Path:
output_path = _candidate_batch_output(run_dir)
_save_json(
output_path,
@@ -461,37 +462,50 @@ def run_freshrss_pipeline(
summary_success_count = 0
summary_failed_count = 0
summary_candidates = [ctx for ctx in item_contexts if ctx["extraction"] is not None and ctx["extraction"].success]
for item_context in summary_candidates:
item_report = item_context["item_report"]
summary_exit_code, summary_payload, summary_report = run_loop_payload(
extracted_payload=item_context["extracted_payload"],
prompt_path=resolved_prompt_path,
output_path=item_context["summary_output"],
max_retries=max_retries,
timeout_seconds=timeout_seconds,
api_key=resolved_llm_api_key,
model=resolved_llm_model,
api_url=resolved_llm_api_url,
)
if summary_exit_code != 0 or summary_payload is None:
item_report["status"] = "summary_failed"
if summary_report is not None:
item_report["summary_errors"] = summary_report.errors
summary_failed_count += 1
else:
item_context["summary_payload"] = summary_payload
item_report["status"] = "summarized"
summary_success_count += 1
# Parallelize LLM summaries — I/O bound calls, independent per article
with ThreadPoolExecutor(max_workers=min(len(summary_candidates) or 1, 4)) as pool:
fut_map = {}
for item_context in summary_candidates:
fut = pool.submit(
run_loop_payload,
extracted_payload=item_context["extracted_payload"],
prompt_path=resolved_prompt_path,
output_path=item_context["summary_output"],
max_retries=max_retries,
timeout_seconds=timeout_seconds,
api_key=resolved_llm_api_key,
model=resolved_llm_model,
api_url=resolved_llm_api_url,
)
fut_map[fut] = item_context
run_store.update_stage(
SUMMARY_STAGE,
outputs={
"expected_items": extracted_success_count,
"completed_items": summary_success_count + summary_failed_count,
"success_count": summary_success_count,
"failed_count": summary_failed_count,
},
)
for fut in as_completed(fut_map):
item_context = fut_map[fut]
item_report = item_context["item_report"]
try:
summary_exit_code, summary_payload, summary_report = fut.result()
except Exception as exc:
summary_exit_code, summary_payload, summary_report = 1, None, None
if summary_exit_code != 0 or summary_payload is None:
item_report["status"] = "summary_failed"
if summary_report is not None:
item_report["summary_errors"] = summary_report.errors
summary_failed_count += 1
else:
item_context["summary_payload"] = summary_payload
item_report["status"] = "summarized"
summary_success_count += 1
run_store.update_stage(
SUMMARY_STAGE,
outputs={
"expected_items": extracted_success_count,
"completed_items": summary_success_count + summary_failed_count,
"success_count": summary_success_count,
"failed_count": summary_failed_count,
},
)
summary_batch_output = _persist_summary_batch_artifact(
run_store=run_store,