Compare commits
2
Commits
4399c9ca90
...
cdbcdcd485
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
cdbcdcd485 | ||
|
|
590d050218 |
@@ -320,6 +320,23 @@
|
|||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
|
### [DONE][P1] interest/watch 候选引擎从固定阈值改为百分位排名 + 增速因子
|
||||||
|
|
||||||
|
目标:
|
||||||
|
- 解决固定阈值(total_count>=3)不随数据量自适应的问题
|
||||||
|
- 引入趋势信号(growth 因子),识别近期集中爆发的词
|
||||||
|
- 支持 7 天、41 天、200 天数据量下取同样的 top 5%/5%-20% 而不需调阈值
|
||||||
|
|
||||||
|
要求:
|
||||||
|
- `build_review_bundle.py`:新增 percentile 和 growth 计算函数;候选池从固定阈值改为百分位 + 增速
|
||||||
|
- `configs/term_cleanup_policy.json`:升级为 v2 schema,percentile/growth 替代绝对阈值
|
||||||
|
- 不改 `generate_term_cleanup_suggestions.py` 和 `apply_term_suggestions.py`
|
||||||
|
- 全量跑一次对比新旧产出,确认差异合理
|
||||||
|
|
||||||
|
方案文档:`plans/keyword-cleanup-interest-watch-engine-improvement.md`
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
### [DONE][P3] 更新 README / handoff / docs,明确 MCP 为正式入口
|
### [DONE][P3] 更新 README / handoff / docs,明确 MCP 为正式入口
|
||||||
|
|
||||||
目标:
|
目标:
|
||||||
|
|||||||
@@ -27,18 +27,30 @@
|
|||||||
"Agent Skills",
|
"Agent Skills",
|
||||||
"AgentScope",
|
"AgentScope",
|
||||||
"AI Agent",
|
"AI Agent",
|
||||||
|
"AI Coding Agent",
|
||||||
"AliSQL",
|
"AliSQL",
|
||||||
|
"Anthropic",
|
||||||
|
"Claude",
|
||||||
"Claude Code",
|
"Claude Code",
|
||||||
|
"CLAUDE.md",
|
||||||
|
"CLI",
|
||||||
|
"Context Engineering",
|
||||||
|
"Cursor",
|
||||||
|
"ChatGPT",
|
||||||
"DeepSeek",
|
"DeepSeek",
|
||||||
"FastAPI",
|
"FastAPI",
|
||||||
"Gin",
|
"Gin",
|
||||||
"Go",
|
"Go",
|
||||||
"gRPC",
|
"gRPC",
|
||||||
|
"Harness Engineering",
|
||||||
|
"Hermes Agent",
|
||||||
"Java",
|
"Java",
|
||||||
"Kafka",
|
"Kafka",
|
||||||
"Kubernetes",
|
"Kubernetes",
|
||||||
"LLM",
|
"LLM",
|
||||||
|
"Loop Engineering",
|
||||||
"MCP",
|
"MCP",
|
||||||
|
"MoE",
|
||||||
"MySQL",
|
"MySQL",
|
||||||
"MySQL复制延迟",
|
"MySQL复制延迟",
|
||||||
"OpenAI",
|
"OpenAI",
|
||||||
@@ -47,15 +59,30 @@
|
|||||||
"Prompt Engineering",
|
"Prompt Engineering",
|
||||||
"Python",
|
"Python",
|
||||||
"RAG",
|
"RAG",
|
||||||
|
"ReAct",
|
||||||
"ReActAgent",
|
"ReActAgent",
|
||||||
"Redis",
|
"Redis",
|
||||||
|
"Skill",
|
||||||
|
"SKILL.md",
|
||||||
|
"Skills",
|
||||||
"Spring",
|
"Spring",
|
||||||
"SubAgent",
|
"SubAgent",
|
||||||
|
"TypeScript",
|
||||||
|
"Vibe Coding",
|
||||||
"Workflow",
|
"Workflow",
|
||||||
|
"上下文压缩",
|
||||||
|
"上下文工程",
|
||||||
|
"上下文管理",
|
||||||
"云原生",
|
"云原生",
|
||||||
|
"代码审查",
|
||||||
"可观测性",
|
"可观测性",
|
||||||
"向量数据库",
|
"向量数据库",
|
||||||
|
"多Agent协作",
|
||||||
|
"大模型",
|
||||||
|
"子Agent",
|
||||||
|
"强化学习",
|
||||||
"微服务",
|
"微服务",
|
||||||
|
"渐进式披露",
|
||||||
"知识库"
|
"知识库"
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
|
|||||||
+146
-2
@@ -1,5 +1,149 @@
|
|||||||
{
|
{
|
||||||
"AI助手": "AI Agent",
|
"Agent": "Agent",
|
||||||
|
"Agent框架": "Agent",
|
||||||
|
"Agent能力": "Agent Skills",
|
||||||
|
"智能体": "AI Agent",
|
||||||
|
"Agentic架构": "Agentic架构",
|
||||||
|
"多Agent协作": "多Agent协作",
|
||||||
|
"多智能体架构": "多Agent协作",
|
||||||
|
"Multi-Agent": "多Agent",
|
||||||
|
"Subagent": "子Agent",
|
||||||
|
"Sub Agents验证": "子Agent",
|
||||||
|
"子Agent": "子Agent",
|
||||||
|
"子智能体": "子Agent",
|
||||||
|
"Coding Agent": "AI Coding Agent",
|
||||||
|
"AI编程": "AI Coding Agent",
|
||||||
|
"AI辅助编程": "AI Coding Agent",
|
||||||
|
"代码生成": "AI代码生成",
|
||||||
|
"代码审查": "Code Review",
|
||||||
|
"Prompt": "Prompt Engineering",
|
||||||
|
"Prompt Caching": "提示缓存",
|
||||||
|
"RAG": "RAG",
|
||||||
"图文RAG": "RAG",
|
"图文RAG": "RAG",
|
||||||
"Prompt架构": "Prompt Engineering"
|
"向量检索": "向量检索",
|
||||||
|
"向量嵌入": "向量嵌入",
|
||||||
|
"Multi-Token Prediction": "多Token预测",
|
||||||
|
"Pair-In Pair-Out": "PIPO架构",
|
||||||
|
"PIPO": "PIPO架构",
|
||||||
|
"上下文管理": "上下文管理",
|
||||||
|
"上下文卸载": "上下文卸载",
|
||||||
|
"Self-GC": "上下文压缩",
|
||||||
|
"记忆管理": "上下文管理",
|
||||||
|
"会话管理": "上下文管理",
|
||||||
|
"Harness Engineering": "Harness工程化",
|
||||||
|
"Harness架构": "Harness工程化",
|
||||||
|
"Harness": "Harness工程化",
|
||||||
|
"Loop Engineering": "Loop Engineering",
|
||||||
|
"推理加速": "推理加速",
|
||||||
|
"推理深度": "推理深度",
|
||||||
|
"长链路推理": "长链路推理",
|
||||||
|
"RLVR": "RLVR",
|
||||||
|
"GRPO": "GRPO",
|
||||||
|
"强化学习": "强化学习",
|
||||||
|
"Multi-Agent RL": "多Agent强化学习",
|
||||||
|
"Viking AI搜索": "AI搜索",
|
||||||
|
"Viking AI Search": "AI搜索",
|
||||||
|
"智能搜索": "AI搜索",
|
||||||
|
"SearchCLI": "CLI搜索",
|
||||||
|
"视频生成": "AI视频生成",
|
||||||
|
"视频生成模型": "AI视频生成",
|
||||||
|
"LingBot-Video": "AI视频生成",
|
||||||
|
"视觉自回归模型": "AI视频生成",
|
||||||
|
"火山云数据库PostgreSQL Serverless版": "Serverless数据库",
|
||||||
|
"PostgreSQL": "PostgreSQL",
|
||||||
|
"MySQL": "MySQL",
|
||||||
|
"OceanBase": "OceanBase",
|
||||||
|
"StarRocks": "StarRocks",
|
||||||
|
"Milvus": "Milvus",
|
||||||
|
"Seal AI Zone": "AI安全",
|
||||||
|
"NEX沙箱": "沙箱隔离",
|
||||||
|
"MicroVM": "沙箱隔离",
|
||||||
|
"安全左移": "安全左移",
|
||||||
|
"安全中台": "AI安全",
|
||||||
|
"成本降低": "成本优化",
|
||||||
|
"成本杠杆": "成本优化",
|
||||||
|
"Scale-to-Zero": "弹性伸缩",
|
||||||
|
"Data as Git": "数据分支管理",
|
||||||
|
"Schema Diff": "Schema对比",
|
||||||
|
"Time Travel": "数据回溯",
|
||||||
|
"多端架构": "多端架构",
|
||||||
|
"契约化": "契约化架构",
|
||||||
|
"大仓": "大仓工程化",
|
||||||
|
"Vibe Coding": "Vibe Coding",
|
||||||
|
"LLM Judge": "LLM评估",
|
||||||
|
"SWE-Bench": "SWE-Bench",
|
||||||
|
"SWE Bench Pro": "SWE-Bench",
|
||||||
|
"SWE-Bench Pro": "SWE-Bench",
|
||||||
|
"Verification Agent": "验证Agent",
|
||||||
|
"CLI工具": "CLI",
|
||||||
|
"CLI": "CLI",
|
||||||
|
"漏桶算法": "限流架构",
|
||||||
|
"固定窗口限流": "限流架构",
|
||||||
|
"Suspend消费控制": "限流架构",
|
||||||
|
"RocketMQ LiteTopic": "消息队列",
|
||||||
|
"LLM Wiki": "LLM知识库",
|
||||||
|
"知识工程": "知识工程",
|
||||||
|
"语义资产": "语义资产管理",
|
||||||
|
"知识图谱": "知识图谱",
|
||||||
|
"知识库沉淀": "知识管理",
|
||||||
|
"Skill": "Skill",
|
||||||
|
"Skill Hub": "技能生态",
|
||||||
|
"具身智能": "具身智能",
|
||||||
|
"Open X-Embodiment": "具身智能",
|
||||||
|
"YOLO Classifier": "目标检测",
|
||||||
|
"MCP": "MCP",
|
||||||
|
"MCP连接器": "MCP",
|
||||||
|
"缓存击穿": "缓存优化",
|
||||||
|
"GPU算力调度": "算力调度",
|
||||||
|
"异构资源": "异构计算",
|
||||||
|
"XPU": "异构计算",
|
||||||
|
"弹性RDMA": "RDMA网络",
|
||||||
|
"国内主流GPU": "国产芯片",
|
||||||
|
"国产AI芯片": "国产芯片",
|
||||||
|
"Paxos协议": "分布式一致性",
|
||||||
|
"Token": "Token管理",
|
||||||
|
"Token效率": "Token管理",
|
||||||
|
"百万token上下文": "长上下文",
|
||||||
|
"MoE": "MoE架构",
|
||||||
|
"MoE架构": "MoE架构",
|
||||||
|
"思维链": "思维链",
|
||||||
|
"CoT Distillation": "思维链蒸馏",
|
||||||
|
"自然语言驱动": "自然语言交互",
|
||||||
|
"NL2SQL": "NL2SQL",
|
||||||
|
"AI对齐": "AI对齐",
|
||||||
|
"注意力机制": "注意力机制",
|
||||||
|
"多模态": "多模态",
|
||||||
|
"音视频工作台": "音视频处理",
|
||||||
|
"AI助手": "AI Agent",
|
||||||
|
"Agent架构": "AI Agent",
|
||||||
|
"Agent专业化": "AI Agent",
|
||||||
|
"Agent Teams": "多Agent协作",
|
||||||
|
"Agentic Engineering": "AI Agent",
|
||||||
|
"AI智能体": "AI Agent",
|
||||||
|
"LLM Agent": "AI Agent",
|
||||||
|
"AI Harness": "Harness Engineering",
|
||||||
|
"AI代码生成": "AI Coding Agent",
|
||||||
|
"Memory管理": "上下文管理",
|
||||||
|
"Agent Skill": "Agent Skills",
|
||||||
|
"Binlog": "binlog",
|
||||||
|
"vibe coding": "Vibe Coding",
|
||||||
|
"Agent组织化协作平台": "Agent协作平台",
|
||||||
|
"Anthropic": "Anthropic",
|
||||||
|
"OpenClaw": "OpenClaw",
|
||||||
|
"WorkBuddy": "WorkBuddy",
|
||||||
|
"Claude": "Claude",
|
||||||
|
"ChatGPT": "ChatGPT",
|
||||||
|
"GPT": "GPT",
|
||||||
|
"Opus": "Opus",
|
||||||
|
"Sonnet": "Sonnet",
|
||||||
|
"Grok": "Grok",
|
||||||
|
"Qwen": "Qwen",
|
||||||
|
"GLM": "GLM",
|
||||||
|
"Claude Code": "Claude Code",
|
||||||
|
"Cursor": "Cursor",
|
||||||
|
"Codex": "Codex",
|
||||||
|
"Pi": "Pi",
|
||||||
|
"CoT": "CoT",
|
||||||
|
"SVG": "SVG",
|
||||||
|
"TTS": "TTS"
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -99,6 +99,497 @@
|
|||||||
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json",
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json",
|
||||||
"suggestion_date": "2026-04-08",
|
"suggestion_date": "2026-04-08",
|
||||||
"based_on_days": 7
|
"based_on_days": 7
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "Anthropic",
|
||||||
|
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=13, days_seen=10, recent_count=13.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "Harness Engineering",
|
||||||
|
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=12, days_seen=11, recent_count=12.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "Skill",
|
||||||
|
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=11, days_seen=9, recent_count=11.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "上下文工程",
|
||||||
|
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=6, days_seen=6, recent_count=6.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "多Agent协作",
|
||||||
|
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=6, days_seen=6, recent_count=6.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "Claude",
|
||||||
|
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=6, days_seen=5, recent_count=6.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "上下文管理",
|
||||||
|
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=6, days_seen=5, recent_count=6.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "渐进式披露",
|
||||||
|
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=5, days_seen=5, recent_count=5.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "Skills",
|
||||||
|
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=5, days_seen=4, recent_count=5.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "SKILL.md",
|
||||||
|
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=4, days_seen=4, recent_count=4.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "CLAUDE.md",
|
||||||
|
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=4, days_seen=4, recent_count=4.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "上下文压缩",
|
||||||
|
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=4, days_seen=4, recent_count=4.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "AI Coding Agent",
|
||||||
|
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=3, days_seen=3, recent_count=3.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "Hermes Agent",
|
||||||
|
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=3, days_seen=3, recent_count=3.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "Vibe Coding",
|
||||||
|
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=5, days_seen=4, recent_count=5.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "Context Engineering",
|
||||||
|
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=3, days_seen=3, recent_count=3.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "Cursor",
|
||||||
|
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=3, days_seen=3, recent_count=3.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T08:20:35.134979Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "大模型",
|
||||||
|
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=4, days_seen=3, recent_count=4.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "TypeScript",
|
||||||
|
"reason": "Core language for AI agent development (e.g., Claude Code, Cursor) and backend engineering, complements existing Python/Java/Go keywords.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "代码审查",
|
||||||
|
"reason": "Chinese term for 'code review', a key practice in backend engineering and AI agent development workflows.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_stopword",
|
||||||
|
"term": "Channels",
|
||||||
|
"reason": "Too generic; could refer to communication channels, YouTube channels, or software channels, not specific to user's focus areas.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_stopword",
|
||||||
|
"term": "Memory",
|
||||||
|
"reason": "Extremely broad term; could refer to computer memory, human memory, or memory in various contexts, not discriminative enough.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_stopword",
|
||||||
|
"term": "Prompt",
|
||||||
|
"reason": "Already covered by 'Prompt Engineering' as a more specific term; 'Prompt' alone is too broad and matches many unrelated articles.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_stopword",
|
||||||
|
"term": "AGI",
|
||||||
|
"reason": "Too broad and speculative; not directly actionable for the user's practical engineering focus areas.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_stopword",
|
||||||
|
"term": "AI日报",
|
||||||
|
"reason": "Generic news term; not a technical concept or tool, would add noise to the keyword index.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_stopword",
|
||||||
|
"term": "AIHOT",
|
||||||
|
"reason": "Unclear meaning, likely a brand or aggregator, not a specific technical term.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_stopword",
|
||||||
|
"term": "All In Code",
|
||||||
|
"reason": "Too vague; could refer to a podcast, a philosophy, or a project, not a specific technical concept.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_stopword",
|
||||||
|
"term": "auto-twitter-campaign",
|
||||||
|
"reason": "Too specific to a single project/tool, not a general interest keyword for the user's focus areas.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_stopword",
|
||||||
|
"term": "ChangeSet",
|
||||||
|
"reason": "Generic term used in version control and databases; too broad to be a useful filter.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_stopword",
|
||||||
|
"term": "Lumina",
|
||||||
|
"reason": "Unclear reference; could be a product, framework, or brand, not clearly aligned with user's focus.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_stopword",
|
||||||
|
"term": "OpenViking",
|
||||||
|
"reason": "Unclear reference; not a known tool or concept in the user's stated focus areas.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_stopword",
|
||||||
|
"term": "Seedance 2.0",
|
||||||
|
"reason": "Unclear reference; likely a product or version, not a general technical term.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_stopword",
|
||||||
|
"term": "质量门禁",
|
||||||
|
"reason": "Chinese term for 'quality gate', too generic in software engineering; not specific to user's focus areas.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_alias",
|
||||||
|
"term": "Agent Skill",
|
||||||
|
"reason": "Singular variant",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"value": "Agent Skills",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_alias",
|
||||||
|
"term": "Binlog",
|
||||||
|
"reason": "Case variant (auto-ranked)",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"value": "binlog",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_alias",
|
||||||
|
"term": "Coding Agent",
|
||||||
|
"reason": "Abbreviated form of 'AI Coding Agent', referring to the same concept.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"value": "AI Coding Agent",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_alias",
|
||||||
|
"term": "Subagent",
|
||||||
|
"reason": "Case variant",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"value": "SubAgent",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_alias",
|
||||||
|
"term": "Subagents",
|
||||||
|
"reason": "Plural variant",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"value": "SubAgent",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_alias",
|
||||||
|
"term": "vibe coding",
|
||||||
|
"reason": "Case variant",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"value": "Vibe Coding",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_alias",
|
||||||
|
"term": "Agent架构",
|
||||||
|
"reason": "Chinese translation of 'Agent architecture', a core concept in AI Agent engineering.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"value": "AI Agent",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_alias",
|
||||||
|
"term": "Agent专业化",
|
||||||
|
"reason": "Chinese term for 'Agent specialization', directly related to Agent engineering.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"value": "AI Agent",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_alias",
|
||||||
|
"term": "Agent Teams",
|
||||||
|
"reason": "English equivalent of 'Multi-Agent collaboration', same concept.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"value": "多Agent协作",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_alias",
|
||||||
|
"term": "Agentic Engineering",
|
||||||
|
"reason": "Broader term for engineering with AI agents, closely related to Agent engineering focus.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"value": "AI Agent",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_alias",
|
||||||
|
"term": "CLI工具",
|
||||||
|
"reason": "Chinese translation of 'CLI tool', same concept.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"value": "CLI",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_alias",
|
||||||
|
"term": "AI编程",
|
||||||
|
"reason": "Chinese term for 'AI programming', closely related to AI Coding Agent.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"value": "AI Coding Agent",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_alias",
|
||||||
|
"term": "记忆管理",
|
||||||
|
"reason": "Chinese term for 'memory management', closely related to context management in LLM applications.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"value": "上下文管理",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-05-14T09:12:50.749877Z",
|
||||||
|
"action": "add_alias",
|
||||||
|
"term": "会话管理",
|
||||||
|
"reason": "Chinese term for 'session management', related to context management in LLM applications.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json",
|
||||||
|
"value": "上下文管理",
|
||||||
|
"suggestion_date": "2026-05-14",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-07-15T02:17:50.155586Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "强化学习",
|
||||||
|
"reason": "top 0.8% by frequency,growth=100%,not yet covered by interest keywords or stopwords. Evidence: total_count=10, days_seen=10, recent_count=10.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-07-15.json",
|
||||||
|
"suggestion_date": "2026-07-15",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-07-15T02:17:50.155586Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "ReAct",
|
||||||
|
"reason": "top 1.1% by frequency,growth=100%,not yet covered by interest keywords or stopwords. Evidence: total_count=8, days_seen=8, recent_count=8.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-07-15.json",
|
||||||
|
"suggestion_date": "2026-07-15",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-07-15T02:17:50.155586Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "CLI",
|
||||||
|
"reason": "top 1.1% by frequency,growth=100%,not yet covered by interest keywords or stopwords. Evidence: total_count=8, days_seen=7, recent_count=8.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-07-15.json",
|
||||||
|
"suggestion_date": "2026-07-15",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-07-15T02:17:50.155586Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "子Agent",
|
||||||
|
"reason": "top 1.3% by frequency,growth=100%,not yet covered by interest keywords or stopwords. Evidence: total_count=7, days_seen=6, recent_count=7.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-07-15.json",
|
||||||
|
"suggestion_date": "2026-07-15",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-07-15T02:17:50.155586Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "Loop Engineering",
|
||||||
|
"reason": "top 1.5% by frequency,growth=100%,not yet covered by interest keywords or stopwords. Evidence: total_count=6, days_seen=6, recent_count=6.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-07-15.json",
|
||||||
|
"suggestion_date": "2026-07-15",
|
||||||
|
"based_on_days": 365
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"applied_at": "2026-07-15T02:17:50.155586Z",
|
||||||
|
"action": "add_interest_keyword",
|
||||||
|
"term": "MoE",
|
||||||
|
"reason": "top 2.0% by frequency,growth=100%,not yet covered by interest keywords or stopwords. Evidence: total_count=5, days_seen=4, recent_count=5.",
|
||||||
|
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-07-15.json",
|
||||||
|
"suggestion_date": "2026-07-15",
|
||||||
|
"based_on_days": 365
|
||||||
}
|
}
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,14 +1,12 @@
|
|||||||
{
|
{
|
||||||
"schema_version": "v1",
|
"schema_version": "v2",
|
||||||
"interest_keyword_review": {
|
"interest_keyword_review": {
|
||||||
"min_total_count": 3,
|
"percentile_max": 0.05,
|
||||||
"min_days_seen": 2
|
"growth_promotion": 0.5
|
||||||
},
|
},
|
||||||
"watch_term_review": {
|
"watch_term_review": {
|
||||||
"min_total_count": 1,
|
"percentile_min": 0.05,
|
||||||
"min_days_seen": 1,
|
"percentile_max": 0.20
|
||||||
"max_total_count": 2,
|
|
||||||
"max_days_seen": 2
|
|
||||||
},
|
},
|
||||||
"alias_review": {
|
"alias_review": {
|
||||||
"min_total_count": 2,
|
"min_total_count": 2,
|
||||||
@@ -19,7 +17,10 @@
|
|||||||
"max_days_seen": 2
|
"max_days_seen": 2
|
||||||
},
|
},
|
||||||
"notes": [
|
"notes": [
|
||||||
"当前阶段采用保守阈值,避免在低样本条件下直接扩充 interest_keywords。",
|
"v2: interest/watch 使用百分位排名 + 增速因子替代固定阈值",
|
||||||
"watch_terms 先用于观察,后续再决定是否升格为 interest_keywords 或进入 alias/stopword 配置。"
|
"percentile 越小表示排名越高(top 5% = percentile 0.05)",
|
||||||
|
"growth = recent_count / total_count,衡量近期活跃度",
|
||||||
|
"watch_term_review 的 percentile_min 可理解为兴趣边界下限,低于此值的词归入 interest 候选",
|
||||||
|
"growth_promotion(默认 0.5)用于识别近期集中爆发词,即使排位不高也主动推荐确认"
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
|
|||||||
+121
-1
@@ -1,6 +1,126 @@
|
|||||||
[
|
[
|
||||||
|
"1688",
|
||||||
|
"AGI",
|
||||||
|
"AIHOT",
|
||||||
|
"AI日报",
|
||||||
|
"All In Code",
|
||||||
|
"Andrej Karpathy",
|
||||||
|
"Anthropic",
|
||||||
|
"auto-twitter-campaign",
|
||||||
|
"Boundaries",
|
||||||
|
"ChangeSet",
|
||||||
|
"Channels",
|
||||||
|
"Claude Fable 5",
|
||||||
|
"Claude Mythos",
|
||||||
|
"Cohere",
|
||||||
|
"Confidence Head",
|
||||||
|
"Cosmos 3",
|
||||||
|
"Databricks",
|
||||||
|
"DINOv2",
|
||||||
|
"domain-mapping",
|
||||||
|
"Dropbox",
|
||||||
|
"EchoGen",
|
||||||
|
"FLUX.1-dev VAE",
|
||||||
|
"GB300 GPU",
|
||||||
|
"GLM 5.2",
|
||||||
|
"GLM5.0",
|
||||||
|
"GPT-5.5",
|
||||||
|
"GPT-Live",
|
||||||
|
"GPT5.5",
|
||||||
|
"Grok 4.5",
|
||||||
|
"GrowBrain",
|
||||||
|
"iMedImage",
|
||||||
|
"iMedLoop",
|
||||||
|
"iMedMaaS",
|
||||||
|
"iMedStudio",
|
||||||
|
"J-space",
|
||||||
|
"JLens",
|
||||||
|
"John Jumper",
|
||||||
|
"J空间",
|
||||||
|
"KAIROS",
|
||||||
|
"KubeRay",
|
||||||
|
"LibTV Agent",
|
||||||
|
"LingBot-Video",
|
||||||
|
"Lumina",
|
||||||
|
"Markdown",
|
||||||
|
"Marvis",
|
||||||
|
"MDASH",
|
||||||
|
"Meta Superintelligence Labs",
|
||||||
|
"MTS",
|
||||||
|
"Muse Image",
|
||||||
|
"Muse Video",
|
||||||
|
"N-gram Embedding",
|
||||||
|
"OCP China",
|
||||||
|
"OCP China 2026",
|
||||||
|
"On-Policy Distillation",
|
||||||
|
"OPC训练营",
|
||||||
|
"OpenAI",
|
||||||
|
"OpenBMC",
|
||||||
|
"OpenClaw",
|
||||||
|
"OpenViking",
|
||||||
|
"Opus 4.8",
|
||||||
|
"Qwen3",
|
||||||
|
"Qwen3-30B-A3B",
|
||||||
|
"RAS API",
|
||||||
|
"Redfish",
|
||||||
|
"ScMoE",
|
||||||
|
"Seal AI Zone",
|
||||||
|
"SealRouter",
|
||||||
|
"Seedance 2.0",
|
||||||
|
"Sonnet 5",
|
||||||
|
"Spec模式",
|
||||||
|
"STE固件团队",
|
||||||
|
"Three.js",
|
||||||
|
"Unity AI Gateway",
|
||||||
|
"Vant Weapp",
|
||||||
|
"WeTV",
|
||||||
|
"WorkBuddy",
|
||||||
|
"wpc",
|
||||||
|
"YOLO Classifier",
|
||||||
|
"一人公司",
|
||||||
|
"中国科学技术大学",
|
||||||
|
"五大扶持体系",
|
||||||
|
"出门问问",
|
||||||
|
"分镜",
|
||||||
|
"剧本",
|
||||||
"奋斗文化",
|
"奋斗文化",
|
||||||
|
"字节跳动",
|
||||||
"小银",
|
"小银",
|
||||||
|
"得力",
|
||||||
|
"德适科技",
|
||||||
|
"成都天府长岛",
|
||||||
|
"扣子",
|
||||||
|
"星云平台",
|
||||||
|
"火山引擎",
|
||||||
|
"百度百舸",
|
||||||
|
"百炼网关",
|
||||||
|
"科大讯飞",
|
||||||
|
"腾讯云开发者社区",
|
||||||
|
"腾讯混元Hy3",
|
||||||
|
"蚂蚁灵波",
|
||||||
|
"贝尔实验室",
|
||||||
|
"质量门禁",
|
||||||
|
"配乐",
|
||||||
|
"配音",
|
||||||
"银行客户经理",
|
"银行客户经理",
|
||||||
"飞盘物理"
|
"飞书妙搭",
|
||||||
|
"飞盘物理",
|
||||||
|
"自动化",
|
||||||
|
"定时任务",
|
||||||
|
"开源模型",
|
||||||
|
"陌生化",
|
||||||
|
"AlphaFold",
|
||||||
|
"Brand Kit",
|
||||||
|
"DataWorks",
|
||||||
|
"Enhance-Nanocodec",
|
||||||
|
"IRIS Codec",
|
||||||
|
"Lovart",
|
||||||
|
"MiniMax M3",
|
||||||
|
"Gemini 3.5 Flash",
|
||||||
|
"Codex",
|
||||||
|
"CodeBuddy",
|
||||||
|
"Claude Cowork",
|
||||||
|
"AGENTS.md",
|
||||||
|
"Claude",
|
||||||
|
"RLVR"
|
||||||
]
|
]
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"schema_version": "v1",
|
"schema_version": "v1",
|
||||||
"updated_at": "2026-04-08T02:34:14.194320Z",
|
"updated_at": "2026-07-15T02:17:50.155586Z",
|
||||||
"terms": [
|
"terms": [
|
||||||
{
|
{
|
||||||
"term": "A2A",
|
"term": "A2A",
|
||||||
|
|||||||
@@ -0,0 +1,151 @@
|
|||||||
|
# 关键词清洗流程概述
|
||||||
|
|
||||||
|
> 2026-05-14 初版
|
||||||
|
> 从"数据记录"到"人工确认落盘"的完整链路
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 整体数据流
|
||||||
|
|
||||||
|
```
|
||||||
|
每日日报 pipeline
|
||||||
|
│
|
||||||
|
▼
|
||||||
|
term_index/daily/YYYY-MM-DD.json ← 每天一篇候选文章的热词统计
|
||||||
|
│
|
||||||
|
▼
|
||||||
|
term_index/term_stats.json ← 所有 daily 的汇总(1070 个词)
|
||||||
|
│
|
||||||
|
├──── build_review_bundle.py ← 打包为审查数据包
|
||||||
|
│ │
|
||||||
|
│ ▼
|
||||||
|
│ review/keyword-cleanup-bundle.json
|
||||||
|
│ │
|
||||||
|
│ ▼
|
||||||
|
│ generate_term_cleanup_suggestions.py
|
||||||
|
│ │
|
||||||
|
│ ▼
|
||||||
|
│ review/term-cleanup-suggestions-YYYY-MM-DD.json ← 正式建议产物
|
||||||
|
│ │
|
||||||
|
│ ▼
|
||||||
|
│ (可选) review/term-cleanup-suggestions-YYYY-MM-DD.md ← 展示稿
|
||||||
|
│
|
||||||
|
├──── 人工确认哪些建议 accept
|
||||||
|
│
|
||||||
|
▼
|
||||||
|
apply_term_suggestions.py ← 写入配置
|
||||||
|
│
|
||||||
|
├── configs/filter_context.personal.json ← interest_keywords
|
||||||
|
├── configs/term_aliases.json ← alias
|
||||||
|
├── configs/term_stopwords.json ← stopword
|
||||||
|
├── configs/term_watchlist.json ← watch
|
||||||
|
└── configs/term_change_log.json ← 变更日志
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 各环节说明
|
||||||
|
|
||||||
|
### 阶段 1:数据记录(每日自动)
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# FreshRSS pipeline 跑完后自动产出
|
||||||
|
data/term_index/daily/2026-05-14.json
|
||||||
|
```
|
||||||
|
|
||||||
|
- 每天一篇,记录当天候选文章中出现的热词
|
||||||
|
- 包含 term、total_count、days_seen 等信息
|
||||||
|
- 目前累计 **41 天**,共 **1070 个独立词**
|
||||||
|
|
||||||
|
### 阶段 2:全量汇总(每日自动)
|
||||||
|
|
||||||
|
```bash
|
||||||
|
data/term_index/term_stats.json
|
||||||
|
```
|
||||||
|
|
||||||
|
- 从所有 daily 文件重建,会覆盖重跑
|
||||||
|
- 按 total_count 排序,前 5 名:OpenClaw(30)、Claude Code(25)、AI Agent(17)、Anthropic(13)、MCP(13)
|
||||||
|
|
||||||
|
### 阶段 3:构建审查数据包(手动触发)
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python skills/keyword-cleanup-review/scripts/build_review_bundle.py \
|
||||||
|
--days 365 \
|
||||||
|
--top 100 \
|
||||||
|
--output outputs/term_index/review/keyword-cleanup-bundle.json
|
||||||
|
```
|
||||||
|
|
||||||
|
- 把 term_stats + 当前配置打成一包,方便后续处理
|
||||||
|
- 输出:`review/keyword-cleanup-bundle.json`
|
||||||
|
|
||||||
|
### 阶段 4:生成建议(手动触发)
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python scripts/generate_term_cleanup_suggestions.py \
|
||||||
|
--bundle outputs/term_index/review/keyword-cleanup-bundle.json \
|
||||||
|
--emit-markdown
|
||||||
|
```
|
||||||
|
|
||||||
|
#### 当前产出能力
|
||||||
|
|
||||||
|
| 建议类型 | 状态 | 当前阈值 | 说明 |
|
||||||
|
|---------|------|----------|------|
|
||||||
|
| interest_keyword_suggestions | ✅ **已实现** | total≥3, days≥2 | 产出 20 条 |
|
||||||
|
| watch_terms | ✅ **已实现** | total≤2, days≤2 | 本次 0 条 |
|
||||||
|
| alias_suggestions | ❌ **硬编码为空** | policy 有阈值(total≥2, days≥2)但脚本未实现 | |
|
||||||
|
| stopword_suggestions | ❌ **硬编码为空** | policy 有阈值(total≤2, days≤2)但脚本未实现 | |
|
||||||
|
|
||||||
|
**关键发现:** alias 和 stopword 不是"阈值太保守",是 **generate 脚本里压根没写对应的生成函数**。policy 文件里阈值已经配好了(alias: min_total=2/min_days=2,stopword: max_total=2/max_days=2),但脚本第 376-380 行直接硬编码为 `[]` 和 `0`。
|
||||||
|
|
||||||
|
### 阶段 5:人工确认(手动)
|
||||||
|
|
||||||
|
```
|
||||||
|
OpenClaw 把建议列给你 → 你确认哪些 accept → 我执行 apply
|
||||||
|
```
|
||||||
|
|
||||||
|
本次模式:
|
||||||
|
- 高频(≥5次/5天以上)→ 强烈推荐 ✅
|
||||||
|
- 中频(3-4次)→ 附带建议 ✅
|
||||||
|
- 泛词 → 建议跳过 ❌
|
||||||
|
|
||||||
|
### 阶段 6:落盘配置(手动)
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python scripts/apply_term_suggestions.py \
|
||||||
|
--suggestions outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json \
|
||||||
|
--accept-interest 词1 词2 ...
|
||||||
|
```
|
||||||
|
|
||||||
|
- dry-run 预览 → 确认后正式 apply
|
||||||
|
- 写入 `configs/filter_context.personal.json`
|
||||||
|
- 同步记录到 `term_change_log.json`
|
||||||
|
- **不备份原始配置**(待优化)
|
||||||
|
- **apply 后不自动清理 review 目录**(待优化)
|
||||||
|
|
||||||
|
### 阶段 7:维护清理(按需)
|
||||||
|
|
||||||
|
由 OpenClaw 侧 `reader-keyword-maintenance` skill 处理:
|
||||||
|
- 删除旧 markdown 展示稿
|
||||||
|
- 保留最近一份 bundle
|
||||||
|
- 保守保留 suggestions JSON
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 当前配置资产
|
||||||
|
|
||||||
|
| 文件 | 内容 | 数据量 |
|
||||||
|
|------|------|--------|
|
||||||
|
| `filter_context.personal.json` | interest_keywords | 52 个 |
|
||||||
|
| `term_aliases.json` | 别名映射 | 0 组(未启用) |
|
||||||
|
| `term_stopwords.json` | 停用词 | 0 个(未启用) |
|
||||||
|
| `term_watchlist.json` | 观察词 | 6 个 |
|
||||||
|
| `term_change_log.json` | 所有变更记录 | 已记录 |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 待优化项
|
||||||
|
|
||||||
|
1. **alias/stopword 建议生成为空** — generate 脚本硬编码缺实现,policy 已有阈值,需要补函数
|
||||||
|
2. **apply 前无配置备份** — 建议 apply 前自动 cp 备份
|
||||||
|
3. **apply 后无自动收尾** — 建议 apply 后自动删旧 markdown 和 bundle
|
||||||
|
4. **alias 识别依赖规则而非 LLM** — 当前全靠统计阈值,无法做语义级判断(如中英文映射、缩写展开)。如果需要高级 alias 识别,可以用 LLM 生成候选,规则脚本做 apply
|
||||||
@@ -0,0 +1,290 @@
|
|||||||
|
# interest/watch 候选引擎改进方案
|
||||||
|
|
||||||
|
> 从固定阈值到自适应排位 + 趋势因子的演进
|
||||||
|
|
||||||
|
## 1. 背景
|
||||||
|
|
||||||
|
### 1.1 当前实现
|
||||||
|
|
||||||
|
`build_review_bundle.py` 使用固定的绝对阈值将未覆盖词(uncovered terms)划分为两个候选池:
|
||||||
|
|
||||||
|
| 候选池 | 判断条件 | 依据 |
|
||||||
|
|--------|---------|------|
|
||||||
|
| `interest_review_candidates` | `total_count >= 3 AND days_seen >= 2` | `policy.interest_keyword_review` |
|
||||||
|
| `watch_review_candidates` | `total_count <= 2 AND days_seen <= 2` | `policy.watch_term_review` |
|
||||||
|
|
||||||
|
`generate_term_cleanup_suggestions.py` 则直接从这两个候选池过滤、去重、排序后输出。
|
||||||
|
|
||||||
|
### 1.2 当前方案的问题
|
||||||
|
|
||||||
|
**问题一:固定阈值不随数据量自适应**
|
||||||
|
|
||||||
|
```
|
||||||
|
场景 total_count=3 意味着什么
|
||||||
|
─────────────────────────────────────────────
|
||||||
|
7 天数据(~200 词) top 15%,有一定区分度 ✅
|
||||||
|
41 天数据(1070 词) top 5%,区分度更高 ✅ 但阈值没变
|
||||||
|
未来 200 天 仍然用 3 次,区分度稀释 ❌
|
||||||
|
```
|
||||||
|
|
||||||
|
同一个绝对次数,在不同数据规模下的语义完全不同。手工调阈值不可持续。
|
||||||
|
|
||||||
|
**问题二:固定阈值忽略趋势信号**
|
||||||
|
|
||||||
|
- "Anthropic":total=13, recent=7 — 近期高活跃,上升趋势
|
||||||
|
- "Channels":total=3, recent=0 — 早期出现但近期消失
|
||||||
|
- 当前引擎认为这两个词"都过了 3 次阈值",同等对待。实际一个是强烈买入信号,一个是过气词。
|
||||||
|
|
||||||
|
**问题三:interest 和 watch 的分界线是硬的**
|
||||||
|
|
||||||
|
total=3 → interest,total=2 → watch。一个词从 2 次变成 3 次就自动"升级",没有过渡、没有缓冲。
|
||||||
|
|
||||||
|
### 1.3 讨论结论
|
||||||
|
|
||||||
|
与老大讨论后确认:
|
||||||
|
|
||||||
|
1. interest/watch 是**统计判断**,不需要大模型介入,纯算法可以解决
|
||||||
|
2. 当前引擎缺的不是大模型,而是**算法本身没写完**——自适应维度(排位、趋势)还没实现
|
||||||
|
3. alias 和 stopword 需要语义判断,与 interest/watch 分属不同阶段,不在本方案范围内
|
||||||
|
4. 修改量小,可以在 1 小时内落地
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 2. 设计方案
|
||||||
|
|
||||||
|
### 2.1 核心思路
|
||||||
|
|
||||||
|
引入两个互补维度替代固定阈值:
|
||||||
|
|
||||||
|
```
|
||||||
|
判定维度 含义 数据来源
|
||||||
|
────────────────────────────────────────────────────────────
|
||||||
|
percentile(百分位排名) 该词 total_count 在所有词 term_stats
|
||||||
|
中的排位占比
|
||||||
|
growth(增速因子) 近期集中度 = recent_count daily 近 N 天
|
||||||
|
/ total_count
|
||||||
|
```
|
||||||
|
|
||||||
|
两个维度配合:
|
||||||
|
|
||||||
|
- **percentile** 衡量"这个词在当前数据集里有多突出"——消除数据量变化的影响
|
||||||
|
- **growth** 衡量"这个词是持续出现还是近期爆发"——识别趋势信号
|
||||||
|
|
||||||
|
### 2.2 候选池划分逻辑
|
||||||
|
|
||||||
|
```
|
||||||
|
percentile
|
||||||
|
│
|
||||||
|
┌─────────────────────┐
|
||||||
|
│ top 5% │
|
||||||
|
│ → 建议 interest │ ← 高频稳定词
|
||||||
|
├─────────────────────┤
|
||||||
|
│ top 5%-20% │
|
||||||
|
│ → 建议 watch │ ← 有信号但未达 threshold
|
||||||
|
├─────────────────────┤
|
||||||
|
│ bottom 80% │
|
||||||
|
│ → 暂不处理 │ ← 噪声/低频
|
||||||
|
└─────────────────────┘
|
||||||
|
|
||||||
|
额外规则:
|
||||||
|
如果词在 top 20% 之外,但 growth > 0.5(近期集中度高)
|
||||||
|
→ 主动提升到 watch / 主动推 confirm
|
||||||
|
```
|
||||||
|
|
||||||
|
这样就不需要关心"total_count 是 3 还是 5",只看数据自己说话。
|
||||||
|
|
||||||
|
### 2.3 接口变化
|
||||||
|
|
||||||
|
**`configs/term_cleanup_policy.json`**:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"schema_version": "v2",
|
||||||
|
"interest_keyword_review": {
|
||||||
|
"percentile_max": 0.05,
|
||||||
|
"growth_promotion": 0.5
|
||||||
|
},
|
||||||
|
"watch_term_review": {
|
||||||
|
"percentile_min": 0.05,
|
||||||
|
"percentile_max": 0.20
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
`v1` 的 `min_total_count`/`min_days_seen` 等绝对阈值字段不再使用。
|
||||||
|
|
||||||
|
**`build_review_bundle.py` 输出的候选项**:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"term": "Anthropic",
|
||||||
|
"total_count": 13,
|
||||||
|
"days_seen": 10,
|
||||||
|
"percentile": 0.012,
|
||||||
|
"growth": 0.54,
|
||||||
|
"reason": "top 1.2% by frequency, 54% of occurrences in recent window — strong signal."
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
### 2.4 不需要改动的部分
|
||||||
|
|
||||||
|
- `generate_term_cleanup_suggestions.py` — 它只消费候选池,不用改
|
||||||
|
- `apply_term_suggestions.py` — 消费 suggestions JSON,不用改
|
||||||
|
- `keyword-cleanup-bundle.json` 结构 — 向后兼容,新增 percentile/growth 字段
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 3. 实施计划
|
||||||
|
|
||||||
|
### 3.1 改动范围
|
||||||
|
|
||||||
|
| 文件 | 改动量 | 内容 |
|
||||||
|
|------|--------|------|
|
||||||
|
| `skills/keyword-cleanup-review/scripts/build_review_bundle.py` | ~40 行 | 新增 `_compute_percentile()` 和 `_compute_growth()` 函数;修改候选池生成逻辑;候选项中增加 percentile/growth |
|
||||||
|
| `configs/term_cleanup_policy.json` | ~10 行 | schema v2:percentile/growth 替代绝对阈值 |
|
||||||
|
|
||||||
|
### 3.2 实施步骤
|
||||||
|
|
||||||
|
1. **build_review_bundle.py**:在 `top_global_terms` 生成后,增加 percentile 计算函数和 growth 计算函数
|
||||||
|
2. **build_review_bundle.py**:修改 `interest_review_candidates` 和 `watch_review_candidates` 的生成逻辑,从固定阈值改为 percentile + growth
|
||||||
|
3. **build_review_bundle.py**:候选项增加 `percentile` 和 `growth` 字段,更新 `reason` 文案
|
||||||
|
4. **term_cleanup_policy.json**:更新为 v2 schema
|
||||||
|
5. **验证**:全量跑一次(`--days 365 --top 100`),对比新旧两份输出的差异
|
||||||
|
|
||||||
|
### 3.3 验证方法
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# 1. 用旧版生成 baseline
|
||||||
|
cd /home/ubuntu/zhu/github/reader
|
||||||
|
python3.11 skills/keyword-cleanup-review/scripts/build_review_bundle.py \
|
||||||
|
--days 365 --top 100 \
|
||||||
|
--output /tmp/bundle-baseline.json
|
||||||
|
|
||||||
|
# 2. 改代码后用新版生成
|
||||||
|
python3.11 skills/keyword-cleanup-review/scripts/build_review_bundle.py \
|
||||||
|
--days 365 --top 100 \
|
||||||
|
--output /tmp/bundle-new.json
|
||||||
|
|
||||||
|
# 3. 对比 governance_hints
|
||||||
|
python3 -c "
|
||||||
|
import json
|
||||||
|
a = json.load(open('/tmp/bundle-baseline.json'))
|
||||||
|
b = json.load(open('/tmp/bundle-new.json'))
|
||||||
|
for key in ['interest_review_candidates', 'watch_review_candidates']:
|
||||||
|
old = set(i['term'] for i in a['governance_hints'][key])
|
||||||
|
new = set(i['term'] for i in b['governance_hints'][key])
|
||||||
|
print(f'{key}: 新增={new-old}, 减少={old-new}')
|
||||||
|
"
|
||||||
|
```
|
||||||
|
|
||||||
|
### 3.4 风险
|
||||||
|
|
||||||
|
| 风险 | 概率 | 应对 |
|
||||||
|
|------|------|------|
|
||||||
|
| 百分位阈值对特小数据集(如只有 1 天数据)不适用 | 低 | 不足 7 天时降级回绝对阈值 |
|
||||||
|
| growth 因子对低频词的偏差(total=1, recent=1 → growth=1) | 低 | growth 只对 total>=3 的词计算 |
|
||||||
|
| 排位突变导致推荐漂移 | 低 | percentil 天然平滑,新增几天数据不会剧烈改变已有词的排位 |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 4. alias/stopword 设计方案
|
||||||
|
|
||||||
|
### 4.1 核心判断
|
||||||
|
|
||||||
|
alias 和 stopword 需要语义理解,与 interest/watch(纯统计)性质不同。
|
||||||
|
|
||||||
|
| 类型 | 需要什么 | 判断方式 |
|
||||||
|
|------|---------|----------|
|
||||||
|
| 大小写变体 | 表层 | 规则:casefold 去重 |
|
||||||
|
| 单复数 | 表层 | 规则:去/加 s 后缀匹配 |
|
||||||
|
| 分词变体(空格/连字符) | 表层 | 规则:去空格归一 |
|
||||||
|
| 简写全称(MCP→Model Context Protocol) | **语义** | LLM |
|
||||||
|
| 中英文(上下文工程→Context Engineering) | **语义** | LLM |
|
||||||
|
| 同义不同名(Rush→猿辅导 Rush 平台) | **语义** | LLM |
|
||||||
|
| stopword(大模型、AI 太泛) | **语义** | LLM |
|
||||||
|
|
||||||
|
### 4.2 分层方案
|
||||||
|
|
||||||
|
```
|
||||||
|
输入:高频未覆盖词 + 已有 interest 词表
|
||||||
|
│
|
||||||
|
├── 规则层(零成本)── 大小写归一、单复数、去空格/连字符
|
||||||
|
│ 输出候选 alias 对
|
||||||
|
│
|
||||||
|
└── LLM 层(每次 ~500 token)── 把候选词表整批给 LLM
|
||||||
|
做语义聚类
|
||||||
|
输出 alias 组 + stopword 标记
|
||||||
|
```
|
||||||
|
|
||||||
|
### 4.3 规则层设计
|
||||||
|
|
||||||
|
在 `generate_term_cleanup_suggestions.py` 中新增 `_prepare_alias_suggestions()` 函数:
|
||||||
|
|
||||||
|
```python
|
||||||
|
def _prepare_alias_suggestions(top_terms, interest_keywords):
|
||||||
|
"""
|
||||||
|
基于表层规则生成 alias 建议。
|
||||||
|
规则1:casefold 匹配——同一个 casefold 下有多个原文变体
|
||||||
|
规则2:单复数——去掉/加上末尾 s 后匹配
|
||||||
|
规则3:分词变体——去空格/连字符后匹配
|
||||||
|
"""
|
||||||
|
```
|
||||||
|
|
||||||
|
优势:零成本、可复现、可审计。直接写入 suggestions JSON,随 generate 一起输出。
|
||||||
|
|
||||||
|
### 4.4 LLM 层设计
|
||||||
|
|
||||||
|
单独脚本,非 generate 主链路的一部分。
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python scripts/generate_term_cleanup_semantic_suggestions.py \
|
||||||
|
--suggestions outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json \
|
||||||
|
--output outputs/term_index/review/term-cleanup-semantic-suggestions-YYYY-MM-DD.json
|
||||||
|
```
|
||||||
|
|
||||||
|
LLM prompt 设计:
|
||||||
|
|
||||||
|
```
|
||||||
|
你是一个关键词治理助手。以下是一个用户的 interest 关键词列表和一批未覆盖的高频词。
|
||||||
|
请做三件事:
|
||||||
|
|
||||||
|
1. ALIAS:判断哪些未覆盖词是已有 interest 关键词的别名/变体
|
||||||
|
2. STOPWORD:标记哪些词太宽泛/通用,建议排除
|
||||||
|
3. PROMOTE:标记哪些新词与用户关注方向一致,建议加入 interest
|
||||||
|
|
||||||
|
用户关注方向:AI Agent 工程化、后端工程、开源工具、大模型落地
|
||||||
|
```
|
||||||
|
|
||||||
|
LLM 层输出格式:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"alias_suggestions": [
|
||||||
|
{"from": "Context Engineering", "to": "上下文工程", "reason": "中英文对应同一概念"}
|
||||||
|
],
|
||||||
|
"stopword_suggestions": [
|
||||||
|
{"term": "大模型", "reason": "过于宽泛,高频率但低区分度"}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
### 4.5 预期效果
|
||||||
|
|
||||||
|
| 覆盖类型 | 规则层 | LLM 层 |
|
||||||
|
|---------|--------|--------|
|
||||||
|
| 大小写变体 | ✅ | — |
|
||||||
|
| 单复数 | ✅ | — |
|
||||||
|
| 分词变体 | ✅ | — |
|
||||||
|
| 简写全称 | — | ✅ |
|
||||||
|
| 中英文映射 | — | ✅ |
|
||||||
|
| 同义不同名 | — | ✅ |
|
||||||
|
| stopword 判断 | — | ✅ |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 5. 讨论记录
|
||||||
|
|
||||||
|
- 2026-05-14:与老大确认 interest/watch 不需要 LLM,纯算法可解决
|
||||||
|
- 2026-05-14:确认百分位排名 + 增速因子方案,修改量小,优先落地
|
||||||
|
- 2026-05-14:确认本方案不改 `generate_term_cleanup_suggestions.py` 和 `apply_term_suggestions.py`
|
||||||
|
- 2026-05-14:确认 alias/stopword 采用规则层 + LLM 层分层方案,规则层零成本优先
|
||||||
+314
@@ -0,0 +1,314 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""
|
||||||
|
Generate semantic keyword suggestions using LLM.
|
||||||
|
|
||||||
|
Covers what surface-form rules cannot:
|
||||||
|
- semantic alias (abbreviation ↔ full name, Chinese ↔ English, synonym)
|
||||||
|
- stopword (overly broad / low-discrimination terms)
|
||||||
|
- promote (new term that aligns with user's focus areas)
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
python scripts/generate_term_cleanup_semantic_suggestions.py \
|
||||||
|
--bundle outputs/term_index/review/keyword-cleanup-bundle.json \
|
||||||
|
--output outputs/term_index/review/term-cleanup-semantic-suggestions-2026-05-14.json
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
from urllib.request import Request, urlopen
|
||||||
|
|
||||||
|
|
||||||
|
REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||||
|
DEFAULT_BUNDLE_PATH = REPO_ROOT / "outputs" / "term_index" / "review" / "keyword-cleanup-bundle.json"
|
||||||
|
DEFAULT_OUTPUT_DIR = REPO_ROOT / "outputs" / "term_index" / "review"
|
||||||
|
|
||||||
|
|
||||||
|
def _load_json(path: Path) -> Any:
|
||||||
|
return json.loads(path.read_text(encoding="utf-8-sig"))
|
||||||
|
|
||||||
|
|
||||||
|
def _save_json(path: Path, payload: dict[str, Any]) -> None:
|
||||||
|
path.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
||||||
|
|
||||||
|
|
||||||
|
def _load_env(path: Path) -> dict[str, str]:
|
||||||
|
"""Load key=value pairs from .env file."""
|
||||||
|
env: dict[str, str] = {}
|
||||||
|
if not path.exists():
|
||||||
|
return env
|
||||||
|
for line in path.read_text(encoding="utf-8").splitlines():
|
||||||
|
line = line.strip()
|
||||||
|
if not line or line.startswith("#") or "=" not in line:
|
||||||
|
continue
|
||||||
|
key, _, value = line.partition("=")
|
||||||
|
env[key.strip()] = value.strip().strip("\"'")
|
||||||
|
return env
|
||||||
|
|
||||||
|
|
||||||
|
def _build_prompt(
|
||||||
|
interest_keywords: list[str],
|
||||||
|
rule_alias_suggestions: list[dict[str, str]],
|
||||||
|
candidate_terms: list[dict[str, Any]],
|
||||||
|
relevant_watch_terms: list[dict[str, Any]],
|
||||||
|
) -> str:
|
||||||
|
"""Build the LLM prompt for semantic suggestions."""
|
||||||
|
|
||||||
|
interest_bullets = "\n".join(f" - {t}" for t in sorted(interest_keywords))
|
||||||
|
candidate_bullets = "\n".join(
|
||||||
|
f" - {t['term']} (count={t['total_count']}, days={t['days_seen']})"
|
||||||
|
for t in candidate_terms[:40]
|
||||||
|
)
|
||||||
|
|
||||||
|
# Alias from rule layer (for LLM to build on, not duplicate)
|
||||||
|
rule_alias_text = ""
|
||||||
|
if rule_alias_suggestions:
|
||||||
|
rule_alias_text = "\nSurface-form alias (already identified, skip these):\n" + "\n".join(
|
||||||
|
f" {a['from']} → {a['to']} ({a['reason']})"
|
||||||
|
for a in rule_alias_suggestions
|
||||||
|
)
|
||||||
|
|
||||||
|
watch_text = ""
|
||||||
|
if relevant_watch_terms:
|
||||||
|
watch_text = "\nWatch terms (low-frequency but potentially relevant):\n" + "\n".join(
|
||||||
|
f" {t['term']} (count={t['total_count']}, days={t['days_seen']})"
|
||||||
|
for t in relevant_watch_terms[:20]
|
||||||
|
)
|
||||||
|
|
||||||
|
return f"""You are a keyword governance assistant for an AI engineer. Your job is to analyze keyword data and produce structured suggestions.
|
||||||
|
|
||||||
|
## User's focus areas
|
||||||
|
- AI Agent engineering (Skills, Harness, MCP, Agent architecture)
|
||||||
|
- Backend engineering (Java, Go, Kubernetes, MySQL, distributed systems)
|
||||||
|
- Open source AI tools and practices (Claude Code, Cursor, DeepSeek, OpenClaw)
|
||||||
|
- LLM application engineering (context engineering, RAG, prompt engineering)
|
||||||
|
|
||||||
|
## Interest keywords (52 already configured)
|
||||||
|
{interest_bullets}
|
||||||
|
|
||||||
|
## Uncovered candidate terms (sorted by frequency)
|
||||||
|
{candidate_bullets}
|
||||||
|
{watch_text}{rule_alias_text}
|
||||||
|
|
||||||
|
## Task
|
||||||
|
Analyze the candidate terms and output a JSON object with exactly three keys:
|
||||||
|
|
||||||
|
1. "semantic_alias": array of alias suggestions that SURFACE RULES CAN'T CATCH (e.g. abbreviation↔full name, Chinese↔English, different naming for the same concept).
|
||||||
|
Format: [{{"from": "<variant>", "to": "<canonical interest keyword>", "reason": "<why>"}}]
|
||||||
|
|
||||||
|
2. "stopword": array of terms that are too broad/generic to be useful as filters. A stopword is a term that appears frequently but has LOW DISCRIMINATION — it matches too many unrelated articles and clutters the keyword index.
|
||||||
|
Format: [{{"term": "<term>", "reason": "<why it should be a stopword>"}}]
|
||||||
|
|
||||||
|
3. "promote_to_interest": array of uncovered terms that align well with the user's focus areas and should be added as interest keywords.
|
||||||
|
Format: [{{"term": "<term>", "reason": "<why it fits>"}}]
|
||||||
|
|
||||||
|
## Rules
|
||||||
|
- Be conservative. When in doubt, leave it out.
|
||||||
|
- Only suggest alias for terms that clearly refer to the SAME concept as an existing interest keyword.
|
||||||
|
- Only suggest stopword for terms that are genuinely too broad (appear in many unrelated contexts).
|
||||||
|
- Only suggest promote for terms that clearly match the user's stated focus areas.
|
||||||
|
- Output valid JSON only, no markdown, no explanation outside the JSON."""
|
||||||
|
|
||||||
|
|
||||||
|
def _call_llm(prompt: str, api_url: str, model: str, api_key: str) -> str:
|
||||||
|
"""Call LLM API and return the response text."""
|
||||||
|
payload = json.dumps({
|
||||||
|
"model": model,
|
||||||
|
"messages": [{"role": "user", "content": prompt}],
|
||||||
|
"temperature": 0.1,
|
||||||
|
"max_tokens": 2048,
|
||||||
|
}).encode("utf-8")
|
||||||
|
|
||||||
|
req = Request(
|
||||||
|
api_url.rstrip("/") + "/chat/completions",
|
||||||
|
data=payload,
|
||||||
|
headers={
|
||||||
|
"Content-Type": "application/json",
|
||||||
|
"Authorization": f"Bearer {api_key}",
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
max_retries = 3
|
||||||
|
for attempt in range(max_retries):
|
||||||
|
try:
|
||||||
|
with urlopen(req, timeout=120) as resp:
|
||||||
|
result = json.loads(resp.read().decode("utf-8"))
|
||||||
|
return result["choices"][0]["message"]["content"]
|
||||||
|
except Exception as e:
|
||||||
|
if attempt < max_retries - 1:
|
||||||
|
wait = 2 ** attempt
|
||||||
|
print(f" LLM call failed (attempt {attempt+1}/{max_retries}): {e}", file=sys.stderr)
|
||||||
|
print(f" Retrying in {wait}s...", file=sys.stderr)
|
||||||
|
time.sleep(wait)
|
||||||
|
else:
|
||||||
|
raise
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_llm_response(text: str) -> dict[str, list[dict[str, str]]]:
|
||||||
|
"""Extract JSON from LLM response (may contain markdown fences)."""
|
||||||
|
# Try to find JSON block
|
||||||
|
json_match = re.search(r"```(?:json)?\s*\n?(\{.*?\})\s*\n?```", text, re.DOTALL)
|
||||||
|
if json_match:
|
||||||
|
text = json_match.group(1)
|
||||||
|
|
||||||
|
# Clean up: remove any text before { or after }
|
||||||
|
start = text.find("{")
|
||||||
|
end = text.rfind("}")
|
||||||
|
if start >= 0 and end > start:
|
||||||
|
text = text[start : end + 1]
|
||||||
|
|
||||||
|
try:
|
||||||
|
result = json.loads(text)
|
||||||
|
except json.JSONDecodeError:
|
||||||
|
# Try partial recovery
|
||||||
|
print(f" Warning: LLM response not clean JSON, attempting recovery", file=sys.stderr)
|
||||||
|
print(f" Raw: {text[:500]}", file=sys.stderr)
|
||||||
|
return {"semantic_alias": [], "stopword": [], "promote_to_interest": []}
|
||||||
|
|
||||||
|
# Normalize keys
|
||||||
|
normalized = {
|
||||||
|
"semantic_alias": result.get("semantic_alias", result.get("alias", [])),
|
||||||
|
"stopword": result.get("stopword", result.get("stopword_suggestions", [])),
|
||||||
|
"promote_to_interest": result.get("promote_to_interest", result.get("promote", [])),
|
||||||
|
}
|
||||||
|
# Ensure each is a list
|
||||||
|
for key in normalized:
|
||||||
|
if not isinstance(normalized[key], list):
|
||||||
|
normalized[key] = []
|
||||||
|
return normalized
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
parser = argparse.ArgumentParser(description="Generate semantic keyword suggestions via LLM.")
|
||||||
|
parser.add_argument("--bundle", type=Path, default=DEFAULT_BUNDLE_PATH, help="Review bundle JSON path")
|
||||||
|
parser.add_argument("--suggestions", type=Path, default=None, help="Existing suggestions JSON (for rule alias context)")
|
||||||
|
parser.add_argument("--output", type=Path, default=None, help="Output JSON path (auto-generated if omitted)")
|
||||||
|
parser.add_argument("--llm-api-url", type=str, default=None, help="LLM API base URL")
|
||||||
|
parser.add_argument("--llm-model", type=str, default=None, help="LLM model name")
|
||||||
|
parser.add_argument("--llm-api-key", type=str, default=None, help="LLM API key")
|
||||||
|
parser.add_argument("--dry-run", action="store_true", help="Print prompt and exit without calling LLM")
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
# Load config
|
||||||
|
env_path = REPO_ROOT / ".env"
|
||||||
|
env = _load_env(env_path) if env_path.exists() else {}
|
||||||
|
|
||||||
|
api_url = args.llm_api_url or os.environ.get("LLM_API_URL") or env.get("LLM_API_URL", "https://api.deepseek.com")
|
||||||
|
# Map OpenClaw model aliases to actual API model names
|
||||||
|
model_raw = args.llm_model or os.environ.get("LLM_MODEL") or env.get("LLM_MODEL", "deepseek-chat")
|
||||||
|
MODEL_ALIAS_MAP = {
|
||||||
|
"deepseek/deepseek-v4-flash": "deepseek-chat",
|
||||||
|
"deepseek/deepseek-chat": "deepseek-chat",
|
||||||
|
"deepseek-v4-flash": "deepseek-chat",
|
||||||
|
"deepseek-chat": "deepseek-chat",
|
||||||
|
}
|
||||||
|
model = MODEL_ALIAS_MAP.get(model_raw, model_raw)
|
||||||
|
api_key = args.llm_api_key or os.environ.get("LLM_API_KEY") or env.get("LLM_API_KEY", "")
|
||||||
|
|
||||||
|
if not api_key:
|
||||||
|
print("Error: No LLM API key found. Set LLM_API_KEY in .env or pass --llm-api-key.", file=sys.stderr)
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
# Load bundle
|
||||||
|
if not args.bundle.exists():
|
||||||
|
print(f"Error: Bundle not found: {args.bundle}", file=sys.stderr)
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
bundle = _load_json(args.bundle)
|
||||||
|
current_config = bundle.get("current_config", {})
|
||||||
|
interest_keywords = current_config.get("interest_keywords", [])
|
||||||
|
top_global_terms = bundle.get("top_global_terms", [])
|
||||||
|
governance_hints = bundle.get("governance_hints", {})
|
||||||
|
|
||||||
|
# Build candidate list (uncovered terms from interest + watch candidates)
|
||||||
|
candidate_terms = []
|
||||||
|
for item in governance_hints.get("interest_review_candidates", []):
|
||||||
|
if isinstance(item, dict):
|
||||||
|
candidate_terms.append({
|
||||||
|
"term": item.get("term", ""),
|
||||||
|
"total_count": item.get("total_count", 0),
|
||||||
|
"days_seen": item.get("days_seen", 0),
|
||||||
|
"percentile": item.get("percentile", 0),
|
||||||
|
"growth": item.get("growth", 0),
|
||||||
|
})
|
||||||
|
for item in governance_hints.get("watch_review_candidates", []):
|
||||||
|
if isinstance(item, dict):
|
||||||
|
# Avoid duplicates
|
||||||
|
if not any(c["term"] == item.get("term") for c in candidate_terms):
|
||||||
|
candidate_terms.append({
|
||||||
|
"term": item.get("term", ""),
|
||||||
|
"total_count": item.get("total_count", 0),
|
||||||
|
"days_seen": item.get("days_seen", 0),
|
||||||
|
"percentile": item.get("percentile", 0),
|
||||||
|
"growth": item.get("growth", 0),
|
||||||
|
})
|
||||||
|
|
||||||
|
# Sort by total_count descending
|
||||||
|
candidate_terms.sort(key=lambda x: -x["total_count"])
|
||||||
|
relevant_watch_terms = governance_hints.get("watch_review_candidates", [])[:20]
|
||||||
|
|
||||||
|
# Load rule-layer alias suggestions if available
|
||||||
|
rule_alias = []
|
||||||
|
if args.suggestions and args.suggestions.exists():
|
||||||
|
s = _load_json(args.suggestions)
|
||||||
|
rule_alias = s.get("alias_suggestions", [])
|
||||||
|
|
||||||
|
# Build prompt
|
||||||
|
prompt = _build_prompt(
|
||||||
|
interest_keywords=interest_keywords,
|
||||||
|
rule_alias_suggestions=rule_alias,
|
||||||
|
candidate_terms=candidate_terms,
|
||||||
|
relevant_watch_terms=relevant_watch_terms,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Determine output path
|
||||||
|
suggestion_date = datetime.now(timezone.utc).date().isoformat()
|
||||||
|
output_path = args.output or (DEFAULT_OUTPUT_DIR / f"term-cleanup-semantic-suggestions-{suggestion_date}.json")
|
||||||
|
|
||||||
|
if args.dry_run:
|
||||||
|
print("=== DRY RUN: Prompt ===")
|
||||||
|
print(prompt)
|
||||||
|
print("\n=== END ===")
|
||||||
|
print(f"\nWould write to: {output_path}")
|
||||||
|
return
|
||||||
|
|
||||||
|
# Call LLM
|
||||||
|
print(f"Calling LLM ({model})...", file=sys.stderr)
|
||||||
|
response = _call_llm(prompt, api_url, model, api_key)
|
||||||
|
print(f"LLM response received ({len(response)} chars)", file=sys.stderr)
|
||||||
|
|
||||||
|
# Parse
|
||||||
|
parsed = _parse_llm_response(response)
|
||||||
|
|
||||||
|
# Build output
|
||||||
|
output = {
|
||||||
|
"date": suggestion_date,
|
||||||
|
"source_bundle": str(args.bundle),
|
||||||
|
"model": model,
|
||||||
|
"interest_keyword_count": len(interest_keywords),
|
||||||
|
"candidate_count": len(candidate_terms),
|
||||||
|
**parsed,
|
||||||
|
}
|
||||||
|
|
||||||
|
_save_json(output_path, output)
|
||||||
|
|
||||||
|
summary = {
|
||||||
|
"output": str(output_path),
|
||||||
|
"semantic_alias": len(output.get("semantic_alias", [])),
|
||||||
|
"stopword": len(output.get("stopword", [])),
|
||||||
|
"promote_to_interest": len(output.get("promote_to_interest", [])),
|
||||||
|
}
|
||||||
|
print(json.dumps(summary, ensure_ascii=False, indent=2))
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -175,6 +175,121 @@ def _prepare_watch_suggestions(bundle: dict[str, Any], reserved_terms: set[str])
|
|||||||
return suggestions
|
return suggestions
|
||||||
|
|
||||||
|
|
||||||
|
def _prepare_alias_suggestions(
|
||||||
|
bundle: dict[str, Any],
|
||||||
|
all_terms: list[dict[str, Any]] | None = None,
|
||||||
|
) -> list[dict[str, Any]]:
|
||||||
|
"""
|
||||||
|
Generate alias suggestions using surface-form rules (no LLM).
|
||||||
|
|
||||||
|
Rules:
|
||||||
|
1. casefold match — same normalized form, different original casing
|
||||||
|
2. trailing-s singularization — singular/plural variants
|
||||||
|
3. whitespace/hyphen normalization — word boundary variants
|
||||||
|
|
||||||
|
Scans all_terms (full term_stats) if provided; otherwise falls back
|
||||||
|
to top_global_terms from the bundle.
|
||||||
|
"""
|
||||||
|
current_config = _require_dict(bundle.get("current_config"), "bundle.current_config")
|
||||||
|
interest_keywords = _require_list(
|
||||||
|
current_config.get("interest_keywords"), "bundle.current_config.interest_keywords"
|
||||||
|
)
|
||||||
|
top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms")
|
||||||
|
source_terms = all_terms if all_terms is not None else top_global_terms
|
||||||
|
|
||||||
|
interest_set = {_term_key(t) for t in interest_keywords if isinstance(t, str)}
|
||||||
|
interest_originals: set[str] = {t for t in interest_keywords if isinstance(t, str)}
|
||||||
|
|
||||||
|
# Build full casefold → [original forms] map
|
||||||
|
cf_map: dict[str, list[str]] = {}
|
||||||
|
for item in source_terms:
|
||||||
|
term = None
|
||||||
|
if isinstance(item, dict):
|
||||||
|
term = item.get("term")
|
||||||
|
elif isinstance(item, str):
|
||||||
|
term = item
|
||||||
|
if not isinstance(term, str) or not term.strip():
|
||||||
|
continue
|
||||||
|
key = _term_key(term)
|
||||||
|
if key not in cf_map:
|
||||||
|
cf_map[key] = []
|
||||||
|
if term not in cf_map[key]:
|
||||||
|
cf_map[key].append(term)
|
||||||
|
|
||||||
|
suggestions: list[dict[str, Any]] = []
|
||||||
|
seen_pairs: set[tuple[str, str]] = set()
|
||||||
|
|
||||||
|
def _add(from_term: str, to_term: str, reason: str) -> None:
|
||||||
|
pair = (_term_key(from_term), _term_key(to_term))
|
||||||
|
if pair in seen_pairs:
|
||||||
|
return
|
||||||
|
seen_pairs.add(pair)
|
||||||
|
suggestions.append({"from": from_term, "to": to_term, "reason": reason})
|
||||||
|
|
||||||
|
# Build a set of all term keys from source for quick lookup
|
||||||
|
source_keys = set(cf_map.keys())
|
||||||
|
|
||||||
|
# Rule 1: casefold match — same normalized form, different casing
|
||||||
|
for key, variants in cf_map.items():
|
||||||
|
if len(variants) < 2:
|
||||||
|
continue
|
||||||
|
canonical = None
|
||||||
|
alt_forms = []
|
||||||
|
for v in variants:
|
||||||
|
if v in interest_originals:
|
||||||
|
canonical = v
|
||||||
|
else:
|
||||||
|
alt_forms.append(v)
|
||||||
|
if canonical and alt_forms:
|
||||||
|
for alt in alt_forms:
|
||||||
|
_add(alt, canonical, "Case variant")
|
||||||
|
elif len(variants) >= 2 and not canonical:
|
||||||
|
# None is canonical — suggest the highest-frequency form
|
||||||
|
ranked = sorted(variants, key=lambda t: -(
|
||||||
|
next(
|
||||||
|
(it.get("total_count", 0) for it in top_global_terms if it.get("term") == t),
|
||||||
|
0,
|
||||||
|
)
|
||||||
|
))
|
||||||
|
for alt in ranked[1:]:
|
||||||
|
_add(alt, ranked[0], "Case variant (auto-ranked)")
|
||||||
|
|
||||||
|
# Rule 2: singular/plural — trailing-s normalization
|
||||||
|
# Check all source terms (not just interest keys) for bidirectional matching
|
||||||
|
for key in source_keys:
|
||||||
|
if key in interest_set:
|
||||||
|
continue
|
||||||
|
if key.endswith("s") and len(key) > 2:
|
||||||
|
singular_key = key.rstrip("s")
|
||||||
|
if singular_key in interest_set and singular_key != key:
|
||||||
|
# Find canonical interest keyword
|
||||||
|
canon = next((t for t in interest_keywords if _term_key(t) == singular_key), None)
|
||||||
|
from_form = cf_map[key][0]
|
||||||
|
if canon:
|
||||||
|
_add(from_form, canon, "Plural variant")
|
||||||
|
# singular form → interest has plural
|
||||||
|
plural_key = key + "s"
|
||||||
|
if plural_key in interest_set and plural_key != key:
|
||||||
|
canon = next((t for t in interest_keywords if _term_key(t) == plural_key), None)
|
||||||
|
from_form = cf_map[key][0]
|
||||||
|
if canon:
|
||||||
|
_add(from_form, canon, "Singular variant")
|
||||||
|
|
||||||
|
# Rule 3: whitespace/hyphen normalization
|
||||||
|
for key in source_keys:
|
||||||
|
if key in interest_set:
|
||||||
|
continue
|
||||||
|
normalized = key.replace("-", "").replace("_", "").replace(" ", "")
|
||||||
|
if normalized in interest_set and normalized != key:
|
||||||
|
canon = next((t for t in interest_keywords if _term_key(t) == normalized), None)
|
||||||
|
from_form = cf_map[key][0]
|
||||||
|
if canon:
|
||||||
|
_add(from_form, canon, "Whitespace/punctuation variant")
|
||||||
|
|
||||||
|
suggestions.sort(key=lambda x: (x["from"].casefold(), x["to"].casefold()))
|
||||||
|
return suggestions
|
||||||
|
|
||||||
|
|
||||||
def _render_table(items: list[dict[str, Any]]) -> str:
|
def _render_table(items: list[dict[str, Any]]) -> str:
|
||||||
if not items:
|
if not items:
|
||||||
return "_None in this pass._\n"
|
return "_None in this pass._\n"
|
||||||
@@ -361,9 +476,19 @@ def main() -> None:
|
|||||||
markdown_output=args.markdown_output,
|
markdown_output=args.markdown_output,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# Load full term_stats for alias scanning (bundle only has top N)
|
||||||
|
stats_path = REPO_ROOT / "data" / "term_index" / "term_stats.json"
|
||||||
|
all_stats_terms: list[str] = []
|
||||||
|
if stats_path.exists():
|
||||||
|
stats_payload = _load_json(stats_path)
|
||||||
|
raw_terms = stats_payload.get("terms") if isinstance(stats_payload, dict) else []
|
||||||
|
if isinstance(raw_terms, list):
|
||||||
|
all_stats_terms = [str(t["term"]) for t in raw_terms if isinstance(t, dict) and isinstance(t.get("term"), str)]
|
||||||
|
|
||||||
interest_items = _prepare_interest_suggestions(bundle)
|
interest_items = _prepare_interest_suggestions(bundle)
|
||||||
reserved_terms = {_term_key(str(item.get("term") or "")) for item in interest_items}
|
reserved_terms = {_term_key(str(item.get("term") or "")) for item in interest_items}
|
||||||
watch_items = _prepare_watch_suggestions(bundle, reserved_terms=reserved_terms)
|
watch_items = _prepare_watch_suggestions(bundle, reserved_terms=reserved_terms)
|
||||||
|
alias_items = _prepare_alias_suggestions(bundle, all_terms=all_stats_terms)
|
||||||
|
|
||||||
suggestions = {
|
suggestions = {
|
||||||
"date": suggestion_date,
|
"date": suggestion_date,
|
||||||
@@ -373,10 +498,10 @@ def main() -> None:
|
|||||||
"summary": {
|
"summary": {
|
||||||
"interest_keyword_suggestions": len(interest_items),
|
"interest_keyword_suggestions": len(interest_items),
|
||||||
"watch_terms": len(watch_items),
|
"watch_terms": len(watch_items),
|
||||||
"alias_suggestions": 0,
|
"alias_suggestions": len(alias_items),
|
||||||
"stopword_suggestions": 0,
|
"stopword_suggestions": 0,
|
||||||
},
|
},
|
||||||
"alias_suggestions": [],
|
"alias_suggestions": alias_items,
|
||||||
"stopword_suggestions": [],
|
"stopword_suggestions": [],
|
||||||
"interest_keyword_suggestions": interest_items,
|
"interest_keyword_suggestions": interest_items,
|
||||||
"watch_terms": watch_items,
|
"watch_terms": watch_items,
|
||||||
@@ -400,7 +525,7 @@ def main() -> None:
|
|||||||
"markdown_output": str(markdown_output_path) if args.emit_markdown else None,
|
"markdown_output": str(markdown_output_path) if args.emit_markdown else None,
|
||||||
"interest_keyword_suggestions": len(interest_items),
|
"interest_keyword_suggestions": len(interest_items),
|
||||||
"watch_terms": len(watch_items),
|
"watch_terms": len(watch_items),
|
||||||
"alias_suggestions": 0,
|
"alias_suggestions": len(alias_items),
|
||||||
"stopword_suggestions": 0,
|
"stopword_suggestions": 0,
|
||||||
"emit_markdown": args.emit_markdown,
|
"emit_markdown": args.emit_markdown,
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -26,7 +26,7 @@ description: 生成 reader 项目的正式关键词 review 输入。当用户需
|
|||||||
|
|
||||||
## 工作流程
|
## 工作流程
|
||||||
|
|
||||||
1. 构建精简的审查数据包(临时工作文件):
|
### Phase 1:构建审查数据包
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
python skills/keyword-cleanup-review/scripts/build_review_bundle.py
|
python skills/keyword-cleanup-review/scripts/build_review_bundle.py
|
||||||
@@ -34,51 +34,92 @@ python skills/keyword-cleanup-review/scripts/build_review_bundle.py
|
|||||||
|
|
||||||
可选参数:
|
可选参数:
|
||||||
|
|
||||||
- `--days 7`
|
- `--days 7`(默认 7,建议传 365 覆盖全量)
|
||||||
- `--top 50`
|
- `--top 100`(考虑的词数)
|
||||||
- `--output outputs/term_index/review/keyword-cleanup-bundle.json`
|
- `--output outputs/term_index/review/keyword-cleanup-bundle.json`
|
||||||
|
|
||||||
2. 阅读建议模式:
|
#### 候选引擎策略
|
||||||
|
|
||||||
- `skills/keyword-cleanup-review/references/suggestion-schema.md`
|
根据 `configs/term_cleanup_policy.json` 的 `schema_version` 自动切换:
|
||||||
|
|
||||||
3. 运行 suggestions 生成脚本:
|
| 版本 | 策略 | 说明 |
|
||||||
|
|------|------|------|
|
||||||
|
| v1(旧) | 固定阈值(total≥3/days≥2 → interest) | 小数据集兼容 |
|
||||||
|
| v2(当前默认) | 百分位排名 + 增速因子 | 自适应数据量,不需要手工调阈值 |
|
||||||
|
|
||||||
|
v2 策略说明:
|
||||||
|
- **percentile**:total_count 在所有词里的排位占比。top 5% → interest 候选,5%-20% → watch 候选
|
||||||
|
- **growth**:recent_count / total_count,衡量近期活跃度。growth≥0.5 的排位外词也会主动推荐
|
||||||
|
|
||||||
|
### Phase 2:生成建议(规则层)
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
python scripts/generate_term_cleanup_suggestions.py ^
|
python scripts/generate_term_cleanup_suggestions.py \
|
||||||
--bundle outputs/term_index/review/keyword-cleanup-bundle.json
|
--bundle outputs/term_index/review/keyword-cleanup-bundle.json
|
||||||
```
|
```
|
||||||
|
|
||||||
默认生成:
|
如需人工审阅展示稿:
|
||||||
|
|
||||||
- 一份符合模式的 JSON 建议文件(正式建议产物,也是 review / apply 之间唯一正式输入)
|
|
||||||
|
|
||||||
如需人工审阅展示稿,再显式加:
|
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
python scripts/generate_term_cleanup_suggestions.py ^
|
python scripts/generate_term_cleanup_suggestions.py \
|
||||||
--bundle outputs/term_index/review/keyword-cleanup-bundle.json ^
|
--bundle outputs/term_index/review/keyword-cleanup-bundle.json \
|
||||||
--emit-markdown
|
--emit-markdown
|
||||||
```
|
```
|
||||||
|
|
||||||
这时才会额外生成:
|
#### 产出能力
|
||||||
|
|
||||||
- 一份简短的供人工审阅的 Markdown 报告(临时展示稿)
|
| 建议类型 | 状态 | 方法 |
|
||||||
|
|---------|------|------|
|
||||||
|
| interest 建议 | ✅ 已实现 | 百分位 top 5% + 增速促活 |
|
||||||
|
| watch 建议 | ✅ 已实现 | 百分位 5%-20% |
|
||||||
|
| alias 建议 | ✅ 已实现 | 规则层:大小写归一、单复数、去空格/连字符 |
|
||||||
|
| stopword 建议 | ❌ 规则层空缺 | 见 Phase 3(LLM 层) |
|
||||||
|
|
||||||
4. 严格保持边界:
|
默认生成:
|
||||||
|
- `term-cleanup-suggestions-YYYY-MM-DD.json`(正式建议产物)
|
||||||
|
|
||||||
|
显式加 `--emit-markdown` 额外生成:
|
||||||
|
- `term-cleanup-suggestions-YYYY-MM-DD.md`(临时展示稿)
|
||||||
|
|
||||||
|
### Phase 3:生成建议(LLM 层,可选)
|
||||||
|
|
||||||
|
规则层覆盖不了 alias(中英文对应、缩写展开、同义不同名)和 stopword 判断,需要 LLM 辅助:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python scripts/generate_term_cleanup_semantic_suggestions.py \
|
||||||
|
--bundle outputs/term_index/review/keyword-cleanup-bundle.json \
|
||||||
|
--suggestions outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json \
|
||||||
|
--output outputs/term_index/review/term-cleanup-semantic-suggestions-YYYY-MM-DD.json
|
||||||
|
```
|
||||||
|
|
||||||
|
从 `.env` 读取 LLM 配置(`LLM_API_URL` / `LLM_MODEL` / `LLM_API_KEY`),使用 DeepSeek API。
|
||||||
|
|
||||||
|
输出三部分:
|
||||||
|
|
||||||
|
| 输出 | 说明 |
|
||||||
|
|------|------|
|
||||||
|
| `semantic_alias` | 语义级别名(中英文、缩写、同义不同名) |
|
||||||
|
| `stopword` | 泛词过滤建议(规则层做不了的需要语义判断的) |
|
||||||
|
| `promote_to_interest` | 与用户关注方向一致的新词,建议加入 interest |
|
||||||
|
|
||||||
|
**注:LLM 层产物是候选,不应自动 apply,需要人工确认后由 OpenClaw 编排 apply。**
|
||||||
|
|
||||||
|
### Phase 4:输出给 OpenClaw 编排
|
||||||
|
|
||||||
|
- `suggestions JSON` = review / apply 之间唯一正式建议输入
|
||||||
|
- `semantic-suggestions JSON` = LLM 补充建议,需要人工筛选后合并到 suggestions JSON 再 apply
|
||||||
|
- Markdown = 临时展示层
|
||||||
|
- 后续汇报、确认、dry-run、apply、收尾清理由 OpenClaw 编排层执行
|
||||||
|
|
||||||
|
### Phase 5:严格保持边界
|
||||||
|
|
||||||
- 建议 `configs/term_aliases.json` 的修改
|
- 建议 `configs/term_aliases.json` 的修改
|
||||||
- 建议 `configs/term_stopwords.json` 的修改
|
- 建议 `configs/term_stopwords.json` 的修改
|
||||||
- 建议 `configs/filter_context.personal.json` 的新增
|
- 建议 `configs/filter_context.personal.json` 的新增
|
||||||
|
- **LLM 层产出(semantic-suggestions)不自动 apply**,需人工确认后由 OpenClaw 编排层执行
|
||||||
- 除非用户明确要求,否则不要直接编辑这些文件
|
- 除非用户明确要求,否则不要直接编辑这些文件
|
||||||
- 除非用户要求修改规则逻辑,否则不要建议直接编辑 `configs/filter_rules.json`
|
- 除非用户要求修改规则逻辑,否则不要建议直接编辑 `configs/filter_rules.json`
|
||||||
|
|
||||||
5. 输出交接口径:
|
|
||||||
|
|
||||||
- 将 JSON suggestions 视为正式 review 输入
|
|
||||||
- 将 Markdown 视为可选展示层
|
|
||||||
- 后续汇报、确认、dry-run apply、正式 apply、收尾清理应由 OpenClaw 编排层继续执行
|
|
||||||
|
|
||||||
## 审查启发式规则
|
## 审查启发式规则
|
||||||
|
|
||||||
优先考虑以下决策:
|
优先考虑以下决策:
|
||||||
@@ -141,6 +182,7 @@ JSON 输出应遵循:
|
|||||||
短期保留:
|
短期保留:
|
||||||
|
|
||||||
- `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json`
|
- `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json`
|
||||||
|
- `outputs/term_index/review/term-cleanup-semantic-suggestions-YYYY-MM-DD.json`
|
||||||
|
|
||||||
临时产物:
|
临时产物:
|
||||||
|
|
||||||
@@ -173,7 +215,9 @@ JSON 输出应遵循:
|
|||||||
## 资源
|
## 资源
|
||||||
|
|
||||||
- 脚本:
|
- 脚本:
|
||||||
- `scripts/build_review_bundle.py`
|
- `skills/keyword-cleanup-review/scripts/build_review_bundle.py`
|
||||||
- `scripts/generate_term_cleanup_suggestions.py`
|
- `scripts/generate_term_cleanup_suggestions.py`
|
||||||
|
- `scripts/generate_term_cleanup_semantic_suggestions.py`(LLM 层)
|
||||||
- 参考文档:
|
- 参考文档:
|
||||||
- `references/suggestion-schema.md`
|
- `references/suggestion-schema.md`
|
||||||
|
- `plans/keyword-cleanup-interest-watch-engine-improvement.md`(v2 引擎设计)
|
||||||
|
|||||||
@@ -9,16 +9,15 @@ from typing import Any
|
|||||||
|
|
||||||
|
|
||||||
DEFAULT_POLICY: dict[str, Any] = {
|
DEFAULT_POLICY: dict[str, Any] = {
|
||||||
"schema_version": "v1",
|
"schema_version": "v2",
|
||||||
"interest_keyword_review": {
|
"interest_keyword_review": {
|
||||||
"min_total_count": 3,
|
"percentile_min": 0.0,
|
||||||
"min_days_seen": 2,
|
"percentile_max": 0.05,
|
||||||
|
"growth_promotion": 0.5,
|
||||||
},
|
},
|
||||||
"watch_term_review": {
|
"watch_term_review": {
|
||||||
"min_total_count": 1,
|
"percentile_min": 0.05,
|
||||||
"min_days_seen": 1,
|
"percentile_max": 0.20,
|
||||||
"max_total_count": 2,
|
|
||||||
"max_days_seen": 2,
|
|
||||||
},
|
},
|
||||||
"alias_review": {
|
"alias_review": {
|
||||||
"min_total_count": 2,
|
"min_total_count": 2,
|
||||||
@@ -28,6 +27,11 @@ DEFAULT_POLICY: dict[str, Any] = {
|
|||||||
"max_total_count": 2,
|
"max_total_count": 2,
|
||||||
"max_days_seen": 2,
|
"max_days_seen": 2,
|
||||||
},
|
},
|
||||||
|
"notes": [
|
||||||
|
"v2: interest/watch 使用百分位排名 + 增速因子替代固定阈值",
|
||||||
|
"percentile 越小表示排名越高(top 5% = percentile 0.05)",
|
||||||
|
"growth = recent_count / total_count,衡量近期活跃度",
|
||||||
|
],
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -126,6 +130,37 @@ def _within_watch_thresholds(item: dict[str, Any], thresholds: dict[str, Any]) -
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _compute_percentile(value: int, sorted_values: list[int]) -> float:
|
||||||
|
"""
|
||||||
|
Return the percentile rank of `value` in `sorted_values` (ascending).
|
||||||
|
0.0 = highest frequency (top rank), 1.0 = lowest frequency (bottom rank).
|
||||||
|
"""
|
||||||
|
if not sorted_values:
|
||||||
|
return 1.0
|
||||||
|
# bisect_left — count of values strictly less than `value`
|
||||||
|
lo, hi = 0, len(sorted_values)
|
||||||
|
while lo < hi:
|
||||||
|
mid = (lo + hi) // 2
|
||||||
|
if sorted_values[mid] < value:
|
||||||
|
lo = mid + 1
|
||||||
|
else:
|
||||||
|
hi = mid
|
||||||
|
rank = lo
|
||||||
|
# invert: smallest value → rank=0 → 1.0 (bottom)
|
||||||
|
# largest value → rank=len → 0.0 (top)
|
||||||
|
return 1.0 - (rank / len(sorted_values))
|
||||||
|
|
||||||
|
|
||||||
|
def _compute_growth(recent_count: int, total_count: int) -> float:
|
||||||
|
"""
|
||||||
|
Return growth factor: recent_count / total_count.
|
||||||
|
Only meaningful when total_count >= 3; returns 0.0 for small counts.
|
||||||
|
"""
|
||||||
|
if total_count < 3:
|
||||||
|
return 0.0
|
||||||
|
return recent_count / total_count
|
||||||
|
|
||||||
|
|
||||||
def main() -> None:
|
def main() -> None:
|
||||||
parser = argparse.ArgumentParser(
|
parser = argparse.ArgumentParser(
|
||||||
description="Build a compact review bundle for the keyword-cleanup-review skill."
|
description="Build a compact review bundle for the keyword-cleanup-review skill."
|
||||||
@@ -235,6 +270,13 @@ def main() -> None:
|
|||||||
alias_values = _casefold_set(list(aliases.values()))
|
alias_values = _casefold_set(list(aliases.values()))
|
||||||
watch_set = _casefold_set([str(item.get("term", "")) for item in watchlist])
|
watch_set = _casefold_set([str(item.get("term", "")) for item in watchlist])
|
||||||
|
|
||||||
|
# Build a sorted list of all total_counts for percentile computation
|
||||||
|
all_total_counts = sorted(
|
||||||
|
int(item.get("total_count") or 0)
|
||||||
|
for item in stats_terms
|
||||||
|
if isinstance(item, dict) and isinstance(item.get("term"), str)
|
||||||
|
)
|
||||||
|
|
||||||
top_global_terms = []
|
top_global_terms = []
|
||||||
for item in stats_terms[: args.top]:
|
for item in stats_terms[: args.top]:
|
||||||
if not isinstance(item, dict):
|
if not isinstance(item, dict):
|
||||||
@@ -256,42 +298,108 @@ def main() -> None:
|
|||||||
"is_alias_target": folded in alias_values,
|
"is_alias_target": folded in alias_values,
|
||||||
"in_watchlist": folded in watch_set,
|
"in_watchlist": folded in watch_set,
|
||||||
"recent_count": recent_counter.get(term, 0),
|
"recent_count": recent_counter.get(term, 0),
|
||||||
|
"percentile": _compute_percentile(
|
||||||
|
int(item.get("total_count") or 0), all_total_counts
|
||||||
|
),
|
||||||
|
"growth": _compute_growth(
|
||||||
|
recent_counter.get(term, 0),
|
||||||
|
int(item.get("total_count") or 0),
|
||||||
|
),
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# Keep more uncovered terms for percentile-based selection
|
||||||
uncovered_terms = [
|
uncovered_terms = [
|
||||||
item for item in top_global_terms if not item["in_interest_keywords"] and not item["is_stopword"]
|
item for item in top_global_terms if not item["in_interest_keywords"] and not item["is_stopword"]
|
||||||
][:20]
|
][:100]
|
||||||
|
|
||||||
|
policy_version = (policy.get("schema_version") if isinstance(policy, dict) else None) or "v1"
|
||||||
interest_thresholds = policy.get("interest_keyword_review") if isinstance(policy, dict) else {}
|
interest_thresholds = policy.get("interest_keyword_review") if isinstance(policy, dict) else {}
|
||||||
watch_thresholds = policy.get("watch_term_review") if isinstance(policy, dict) else {}
|
watch_thresholds = policy.get("watch_term_review") if isinstance(policy, dict) else {}
|
||||||
|
|
||||||
|
if policy_version == "v2" or "percentile_max" in interest_thresholds:
|
||||||
|
# v2: percentile + growth based selection
|
||||||
|
pct_min_interest = float(interest_thresholds.get("percentile_min", 0.0))
|
||||||
|
pct_max_interest = float(interest_thresholds.get("percentile_max", 0.05))
|
||||||
|
growth_promo = float(interest_thresholds.get("growth_promotion", 0.5))
|
||||||
|
pct_min_watch = float(watch_thresholds.get("percentile_min", 0.05))
|
||||||
|
pct_max_watch = float(watch_thresholds.get("percentile_max", 0.20))
|
||||||
|
|
||||||
|
interest_candidates_raw = [
|
||||||
|
item for item in uncovered_terms
|
||||||
|
if not item["in_watchlist"]
|
||||||
|
and pct_min_interest <= item["percentile"] <= pct_max_interest
|
||||||
|
]
|
||||||
|
watch_candidates_raw = [
|
||||||
|
item for item in uncovered_terms
|
||||||
|
if not item["in_watchlist"]
|
||||||
|
and pct_min_watch < item["percentile"] <= pct_max_watch
|
||||||
|
]
|
||||||
|
# Growth boost: terms outside watch range but with strong growth signal
|
||||||
|
growth_boost_candidates = [
|
||||||
|
item for item in uncovered_terms
|
||||||
|
if not item["in_watchlist"]
|
||||||
|
and item["percentile"] > pct_max_watch
|
||||||
|
and item["growth"] >= growth_promo
|
||||||
|
]
|
||||||
|
else:
|
||||||
|
# v1 fallback: fixed thresholds
|
||||||
|
interest_candidates_raw = [
|
||||||
|
item for item in uncovered_terms
|
||||||
|
if not item["in_watchlist"] and _meets_min_thresholds(item, interest_thresholds)
|
||||||
|
]
|
||||||
|
watch_candidates_raw = [
|
||||||
|
item for item in uncovered_terms
|
||||||
|
if not item["in_watchlist"]
|
||||||
|
and not _meets_min_thresholds(item, interest_thresholds)
|
||||||
|
and _within_watch_thresholds(item, watch_thresholds)
|
||||||
|
]
|
||||||
|
growth_boost_candidates = []
|
||||||
|
|
||||||
interest_review_candidates = [
|
interest_review_candidates = [
|
||||||
{
|
{
|
||||||
"term": item["term"],
|
"term": item["term"],
|
||||||
"total_count": item["total_count"],
|
"total_count": item["total_count"],
|
||||||
"days_seen": item["days_seen"],
|
"days_seen": item["days_seen"],
|
||||||
|
"percentile": item["percentile"],
|
||||||
|
"growth": item["growth"],
|
||||||
"reason": (
|
"reason": (
|
||||||
"Meets the configured interest-keyword review threshold and is not yet covered "
|
f"top {item['percentile']:.1%} by frequency,"
|
||||||
"by interest keywords or stopwords."
|
f"growth={item['growth']:.0%},"
|
||||||
|
"not yet covered by interest keywords or stopwords."
|
||||||
),
|
),
|
||||||
}
|
}
|
||||||
for item in uncovered_terms
|
for item in interest_candidates_raw
|
||||||
if not item["in_watchlist"] and _meets_min_thresholds(item, interest_thresholds)
|
|
||||||
][:20]
|
][:20]
|
||||||
watch_review_candidates = [
|
watch_review_candidates = [
|
||||||
{
|
{
|
||||||
"term": item["term"],
|
"term": item["term"],
|
||||||
"total_count": item["total_count"],
|
"total_count": item["total_count"],
|
||||||
"days_seen": item["days_seen"],
|
"days_seen": item["days_seen"],
|
||||||
|
"percentile": item["percentile"],
|
||||||
|
"growth": item["growth"],
|
||||||
"reason": (
|
"reason": (
|
||||||
"Falls into the configured watch-term review range and should be observed "
|
f"top {item['percentile']:.1%} by frequency,"
|
||||||
"before promotion into interest keywords."
|
f"growth={item['growth']:.0%},"
|
||||||
|
"fell into watch-review range."
|
||||||
),
|
),
|
||||||
}
|
}
|
||||||
for item in uncovered_terms
|
for item in watch_candidates_raw
|
||||||
if not item["in_watchlist"]
|
|
||||||
and not _meets_min_thresholds(item, interest_thresholds)
|
|
||||||
and _within_watch_thresholds(item, watch_thresholds)
|
|
||||||
][:20]
|
][:20]
|
||||||
|
growth_boost_review_items = [
|
||||||
|
{
|
||||||
|
"term": item["term"],
|
||||||
|
"total_count": item["total_count"],
|
||||||
|
"days_seen": item["days_seen"],
|
||||||
|
"percentile": item["percentile"],
|
||||||
|
"growth": item["growth"],
|
||||||
|
"reason": (
|
||||||
|
f"growth spike: {item['growth']:.0%} of occurrences in recent window "
|
||||||
|
f"(total={item['total_count']}, days={item['days_seen']})."
|
||||||
|
),
|
||||||
|
}
|
||||||
|
for item in growth_boost_candidates
|
||||||
|
][:5]
|
||||||
recent_hot_terms = sorted(
|
recent_hot_terms = sorted(
|
||||||
({"term": term, "recent_count": count} for term, count in recent_counter.items()),
|
({"term": term, "recent_count": count} for term, count in recent_counter.items()),
|
||||||
key=lambda item: (-item["recent_count"], item["term"].casefold(), item["term"]),
|
key=lambda item: (-item["recent_count"], item["term"].casefold(), item["term"]),
|
||||||
@@ -330,6 +438,7 @@ def main() -> None:
|
|||||||
"governance_hints": {
|
"governance_hints": {
|
||||||
"interest_review_candidates": interest_review_candidates,
|
"interest_review_candidates": interest_review_candidates,
|
||||||
"watch_review_candidates": watch_review_candidates,
|
"watch_review_candidates": watch_review_candidates,
|
||||||
|
"growth_boost_review_items": growth_boost_review_items,
|
||||||
},
|
},
|
||||||
}
|
}
|
||||||
_save_json(args.output, bundle)
|
_save_json(args.output, bundle)
|
||||||
|
|||||||
@@ -6,8 +6,30 @@ from .freshrss_pipeline_jobs import (
|
|||||||
start_freshrss_pipeline_job,
|
start_freshrss_pipeline_job,
|
||||||
)
|
)
|
||||||
from .query_service import get_delivery_payload, get_run_report, get_run_status, list_run_artifacts, list_runs
|
from .query_service import get_delivery_payload, get_run_report, get_run_status, list_run_artifacts, list_runs
|
||||||
from .resume_jobs import get_resume_job_result, get_resume_job_status, start_resume_job
|
|
||||||
from .resume_service import inspect_resume_plan, resume_run
|
# NOTE: resume_jobs and resume_service are NOT eagerly imported here to avoid
|
||||||
|
# a circular import chain:
|
||||||
|
# workflows/freshrss_pipeline.py -> runtime -> resume_jobs -> resume_service
|
||||||
|
# -> workflows/freshrss_pipeline.py (circular!)
|
||||||
|
# They are lazy-loaded via __getattr__ when accessed as summary_mcp.runtime.*
|
||||||
|
|
||||||
|
|
||||||
|
def __getattr__(name):
|
||||||
|
import importlib
|
||||||
|
|
||||||
|
_LAZY = {
|
||||||
|
"get_resume_job_result": ("resume_jobs", "get_resume_job_result"),
|
||||||
|
"get_resume_job_status": ("resume_jobs", "get_resume_job_status"),
|
||||||
|
"start_resume_job": ("resume_jobs", "start_resume_job"),
|
||||||
|
"inspect_resume_plan": ("resume_service", "inspect_resume_plan"),
|
||||||
|
"resume_run": ("resume_service", "resume_run"),
|
||||||
|
}
|
||||||
|
if name in _LAZY:
|
||||||
|
mod_name, attr_name = _LAZY[name]
|
||||||
|
mod = importlib.import_module(f".{mod_name}", __package__)
|
||||||
|
return getattr(mod, attr_name)
|
||||||
|
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
||||||
|
|
||||||
|
|
||||||
__all__ = [
|
__all__ = [
|
||||||
"ArtifactRecord",
|
"ArtifactRecord",
|
||||||
|
|||||||
@@ -9,6 +9,8 @@ from pathlib import Path
|
|||||||
from typing import Any
|
from typing import Any
|
||||||
from uuid import uuid4
|
from uuid import uuid4
|
||||||
|
|
||||||
|
from dotenv import dotenv_values
|
||||||
|
|
||||||
from .run_store import RunStore
|
from .run_store import RunStore
|
||||||
from .query_service import _resolve_run_record
|
from .query_service import _resolve_run_record
|
||||||
|
|
||||||
@@ -29,6 +31,25 @@ DEFAULT_STAGES = [
|
|||||||
]
|
]
|
||||||
MIN_JOB_STALE_SECONDS = 30 * 60
|
MIN_JOB_STALE_SECONDS = 30 * 60
|
||||||
MAX_JOB_STALE_SECONDS = 6 * 60 * 60
|
MAX_JOB_STALE_SECONDS = 6 * 60 * 60
|
||||||
|
DEFAULT_DOTENV_PATH = REPO_ROOT / ".env"
|
||||||
|
|
||||||
|
|
||||||
|
def _build_subprocess_env() -> dict[str, str]:
|
||||||
|
"""Build an env dict for subprocess, merging parent env with .env values.
|
||||||
|
|
||||||
|
The subprocess inherits the Hermes MCP server's environment, but .env values
|
||||||
|
may not be in os.environ at the time the subprocess is spawned. This function
|
||||||
|
loads them from .env and merges them in so the child process sees all needed
|
||||||
|
variables (LLM_API_KEY, LLM_MODEL, LLM_API_URL, FRESHRSS_*, etc.) directly
|
||||||
|
in os.environ, avoiding any dotenv-loading timing issues inside the subprocess.
|
||||||
|
"""
|
||||||
|
env = os.environ.copy()
|
||||||
|
if DEFAULT_DOTENV_PATH.exists():
|
||||||
|
for key, value in dotenv_values(DEFAULT_DOTENV_PATH).items():
|
||||||
|
if isinstance(key, str) and isinstance(value, str) and value:
|
||||||
|
# Only set if not already present in parent env
|
||||||
|
env.setdefault(key, value)
|
||||||
|
return env
|
||||||
|
|
||||||
|
|
||||||
def _now() -> datetime:
|
def _now() -> datetime:
|
||||||
@@ -218,6 +239,7 @@ def start_freshrss_pipeline_job(
|
|||||||
stdout=subprocess.DEVNULL,
|
stdout=subprocess.DEVNULL,
|
||||||
stderr=subprocess.DEVNULL,
|
stderr=subprocess.DEVNULL,
|
||||||
start_new_session=True,
|
start_new_session=True,
|
||||||
|
env=_build_subprocess_env(),
|
||||||
)
|
)
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
report_file = _write_job_report(
|
report_file = _write_job_report(
|
||||||
|
|||||||
@@ -9,6 +9,8 @@ from pathlib import Path
|
|||||||
from typing import Any
|
from typing import Any
|
||||||
from uuid import uuid4
|
from uuid import uuid4
|
||||||
|
|
||||||
|
from dotenv import dotenv_values
|
||||||
|
|
||||||
from .query_service import _resolve_run_record
|
from .query_service import _resolve_run_record
|
||||||
from .resume_service import (
|
from .resume_service import (
|
||||||
SUPPORTED_RESUME_STAGES,
|
SUPPORTED_RESUME_STAGES,
|
||||||
@@ -38,6 +40,22 @@ DEFAULT_STAGES = [
|
|||||||
]
|
]
|
||||||
MIN_JOB_STALE_SECONDS = 30 * 60
|
MIN_JOB_STALE_SECONDS = 30 * 60
|
||||||
MAX_JOB_STALE_SECONDS = 6 * 60 * 60
|
MAX_JOB_STALE_SECONDS = 6 * 60 * 60
|
||||||
|
DEFAULT_DOTENV_PATH = REPO_ROOT / ".env"
|
||||||
|
|
||||||
|
|
||||||
|
def _build_subprocess_env() -> dict[str, str]:
|
||||||
|
"""Build an env dict for subprocess, merging parent env with .env values.
|
||||||
|
|
||||||
|
Ensures the subprocess sees all needed variables (LLM_API_KEY, LLM_MODEL,
|
||||||
|
LLM_API_URL, FRESHRSS_*, etc.) directly in os.environ, avoiding dotenv
|
||||||
|
timing issues in the child process.
|
||||||
|
"""
|
||||||
|
env = os.environ.copy()
|
||||||
|
if DEFAULT_DOTENV_PATH.exists():
|
||||||
|
for key, value in dotenv_values(DEFAULT_DOTENV_PATH).items():
|
||||||
|
if isinstance(key, str) and isinstance(value, str) and value:
|
||||||
|
env.setdefault(key, value)
|
||||||
|
return env
|
||||||
|
|
||||||
|
|
||||||
def _now() -> datetime:
|
def _now() -> datetime:
|
||||||
@@ -276,6 +294,7 @@ def start_resume_job(*, run_id: str) -> dict[str, Any]:
|
|||||||
stdout=subprocess.DEVNULL,
|
stdout=subprocess.DEVNULL,
|
||||||
stderr=subprocess.DEVNULL,
|
stderr=subprocess.DEVNULL,
|
||||||
start_new_session=True,
|
start_new_session=True,
|
||||||
|
env=_build_subprocess_env(),
|
||||||
)
|
)
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
report_file = _write_job_report(
|
report_file = _write_job_report(
|
||||||
|
|||||||
@@ -2,6 +2,7 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import json
|
import json
|
||||||
import os
|
import os
|
||||||
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||||
from datetime import date, datetime, timezone
|
from datetime import date, datetime, timezone
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any
|
from typing import Any
|
||||||
@@ -192,7 +193,7 @@ def _build_candidate_batch_payload(*, run_id: str, item_contexts: list[dict[str,
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
def _persist_summary_batch_artifact(*, run_store: RunStore, run_dir: Path, item_contexts: list[dict[str, Any]]) -> Path:
|
def _persist_summary_batch_artifact(*, run_store, run_dir: Path, item_contexts: list[dict[str, Any]]) -> Path:
|
||||||
output_path = _summary_batch_output(run_dir)
|
output_path = _summary_batch_output(run_dir)
|
||||||
_save_json(
|
_save_json(
|
||||||
output_path,
|
output_path,
|
||||||
@@ -202,7 +203,7 @@ def _persist_summary_batch_artifact(*, run_store: RunStore, run_dir: Path, item_
|
|||||||
return output_path
|
return output_path
|
||||||
|
|
||||||
|
|
||||||
def _persist_candidate_batch_artifact(*, run_store: RunStore, run_dir: Path, item_contexts: list[dict[str, Any]]) -> Path:
|
def _persist_candidate_batch_artifact(*, run_store, run_dir: Path, item_contexts: list[dict[str, Any]]) -> Path:
|
||||||
output_path = _candidate_batch_output(run_dir)
|
output_path = _candidate_batch_output(run_dir)
|
||||||
_save_json(
|
_save_json(
|
||||||
output_path,
|
output_path,
|
||||||
@@ -461,37 +462,50 @@ def run_freshrss_pipeline(
|
|||||||
summary_success_count = 0
|
summary_success_count = 0
|
||||||
summary_failed_count = 0
|
summary_failed_count = 0
|
||||||
summary_candidates = [ctx for ctx in item_contexts if ctx["extraction"] is not None and ctx["extraction"].success]
|
summary_candidates = [ctx for ctx in item_contexts if ctx["extraction"] is not None and ctx["extraction"].success]
|
||||||
for item_context in summary_candidates:
|
# Parallelize LLM summaries — I/O bound calls, independent per article
|
||||||
item_report = item_context["item_report"]
|
with ThreadPoolExecutor(max_workers=min(len(summary_candidates) or 1, 4)) as pool:
|
||||||
summary_exit_code, summary_payload, summary_report = run_loop_payload(
|
fut_map = {}
|
||||||
extracted_payload=item_context["extracted_payload"],
|
for item_context in summary_candidates:
|
||||||
prompt_path=resolved_prompt_path,
|
fut = pool.submit(
|
||||||
output_path=item_context["summary_output"],
|
run_loop_payload,
|
||||||
max_retries=max_retries,
|
extracted_payload=item_context["extracted_payload"],
|
||||||
timeout_seconds=timeout_seconds,
|
prompt_path=resolved_prompt_path,
|
||||||
api_key=resolved_llm_api_key,
|
output_path=item_context["summary_output"],
|
||||||
model=resolved_llm_model,
|
max_retries=max_retries,
|
||||||
api_url=resolved_llm_api_url,
|
timeout_seconds=timeout_seconds,
|
||||||
)
|
api_key=resolved_llm_api_key,
|
||||||
if summary_exit_code != 0 or summary_payload is None:
|
model=resolved_llm_model,
|
||||||
item_report["status"] = "summary_failed"
|
api_url=resolved_llm_api_url,
|
||||||
if summary_report is not None:
|
)
|
||||||
item_report["summary_errors"] = summary_report.errors
|
fut_map[fut] = item_context
|
||||||
summary_failed_count += 1
|
|
||||||
else:
|
|
||||||
item_context["summary_payload"] = summary_payload
|
|
||||||
item_report["status"] = "summarized"
|
|
||||||
summary_success_count += 1
|
|
||||||
|
|
||||||
run_store.update_stage(
|
for fut in as_completed(fut_map):
|
||||||
SUMMARY_STAGE,
|
item_context = fut_map[fut]
|
||||||
outputs={
|
item_report = item_context["item_report"]
|
||||||
"expected_items": extracted_success_count,
|
try:
|
||||||
"completed_items": summary_success_count + summary_failed_count,
|
summary_exit_code, summary_payload, summary_report = fut.result()
|
||||||
"success_count": summary_success_count,
|
except Exception as exc:
|
||||||
"failed_count": summary_failed_count,
|
summary_exit_code, summary_payload, summary_report = 1, None, None
|
||||||
},
|
|
||||||
)
|
if summary_exit_code != 0 or summary_payload is None:
|
||||||
|
item_report["status"] = "summary_failed"
|
||||||
|
if summary_report is not None:
|
||||||
|
item_report["summary_errors"] = summary_report.errors
|
||||||
|
summary_failed_count += 1
|
||||||
|
else:
|
||||||
|
item_context["summary_payload"] = summary_payload
|
||||||
|
item_report["status"] = "summarized"
|
||||||
|
summary_success_count += 1
|
||||||
|
|
||||||
|
run_store.update_stage(
|
||||||
|
SUMMARY_STAGE,
|
||||||
|
outputs={
|
||||||
|
"expected_items": extracted_success_count,
|
||||||
|
"completed_items": summary_success_count + summary_failed_count,
|
||||||
|
"success_count": summary_success_count,
|
||||||
|
"failed_count": summary_failed_count,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
summary_batch_output = _persist_summary_batch_artifact(
|
summary_batch_output = _persist_summary_batch_artifact(
|
||||||
run_store=run_store,
|
run_store=run_store,
|
||||||
|
|||||||
Reference in New Issue
Block a user