diff --git a/TODO.md b/TODO.md index a5c408a..bb649a0 100644 --- a/TODO.md +++ b/TODO.md @@ -320,6 +320,23 @@ --- +### [DONE][P1] interest/watch 候选引擎从固定阈值改为百分位排名 + 增速因子 + +目标: +- 解决固定阈值(total_count>=3)不随数据量自适应的问题 +- 引入趋势信号(growth 因子),识别近期集中爆发的词 +- 支持 7 天、41 天、200 天数据量下取同样的 top 5%/5%-20% 而不需调阈值 + +要求: +- `build_review_bundle.py`:新增 percentile 和 growth 计算函数;候选池从固定阈值改为百分位 + 增速 +- `configs/term_cleanup_policy.json`:升级为 v2 schema,percentile/growth 替代绝对阈值 +- 不改 `generate_term_cleanup_suggestions.py` 和 `apply_term_suggestions.py` +- 全量跑一次对比新旧产出,确认差异合理 + +方案文档:`plans/keyword-cleanup-interest-watch-engine-improvement.md` + +--- + ### [DONE][P3] 更新 README / handoff / docs,明确 MCP 为正式入口 目标: diff --git a/configs/filter_context.personal.json b/configs/filter_context.personal.json index aee74f3..f24be3e 100644 --- a/configs/filter_context.personal.json +++ b/configs/filter_context.personal.json @@ -27,13 +27,21 @@ "Agent Skills", "AgentScope", "AI Agent", + "AI Coding Agent", "AliSQL", + "Anthropic", + "Claude", "Claude Code", + "CLAUDE.md", + "Context Engineering", + "Cursor", "DeepSeek", "FastAPI", "Gin", "Go", "gRPC", + "Harness Engineering", + "Hermes Agent", "Java", "Kafka", "Kubernetes", @@ -49,13 +57,25 @@ "RAG", "ReActAgent", "Redis", + "Skill", + "SKILL.md", + "Skills", "Spring", "SubAgent", + "TypeScript", + "Vibe Coding", "Workflow", + "上下文压缩", + "上下文工程", + "上下文管理", "云原生", + "代码审查", "可观测性", "向量数据库", + "多Agent协作", + "大模型", "微服务", + "渐进式披露", "知识库" ] } diff --git a/configs/term_aliases.json b/configs/term_aliases.json index e210659..17d3a2e 100644 --- a/configs/term_aliases.json +++ b/configs/term_aliases.json @@ -1,5 +1,19 @@ { "AI助手": "AI Agent", "图文RAG": "RAG", - "Prompt架构": "Prompt Engineering" + "Prompt架构": "Prompt Engineering", + "Agent Skill": "Agent Skills", + "Binlog": "binlog", + "Coding Agent": "AI Coding Agent", + "Subagent": "SubAgent", + "Subagents": "SubAgent", + "vibe coding": "Vibe Coding", + "Agent架构": "AI Agent", + "Agent专业化": "AI Agent", + "Agent Teams": "多Agent协作", + "Agentic Engineering": "AI Agent", + "CLI工具": "CLI", + "AI编程": "AI Coding Agent", + "记忆管理": "上下文管理", + "会话管理": "上下文管理" } diff --git a/configs/term_change_log.json b/configs/term_change_log.json index 2d39d31..8891648 100644 --- a/configs/term_change_log.json +++ b/configs/term_change_log.json @@ -99,6 +99,443 @@ "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json", "suggestion_date": "2026-04-08", "based_on_days": 7 + }, + { + "applied_at": "2026-05-14T08:20:35.134979Z", + "action": "add_interest_keyword", + "term": "Anthropic", + "reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=13, days_seen=10, recent_count=13.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T08:20:35.134979Z", + "action": "add_interest_keyword", + "term": "Harness Engineering", + "reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=12, days_seen=11, recent_count=12.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T08:20:35.134979Z", + "action": "add_interest_keyword", + "term": "Skill", + "reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=11, days_seen=9, recent_count=11.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T08:20:35.134979Z", + "action": "add_interest_keyword", + "term": "上下文工程", + "reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=6, days_seen=6, recent_count=6.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T08:20:35.134979Z", + "action": "add_interest_keyword", + "term": "多Agent协作", + "reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=6, days_seen=6, recent_count=6.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T08:20:35.134979Z", + "action": "add_interest_keyword", + "term": "Claude", + "reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=6, days_seen=5, recent_count=6.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T08:20:35.134979Z", + "action": "add_interest_keyword", + "term": "上下文管理", + "reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=6, days_seen=5, recent_count=6.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T08:20:35.134979Z", + "action": "add_interest_keyword", + "term": "渐进式披露", + "reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=5, days_seen=5, recent_count=5.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T08:20:35.134979Z", + "action": "add_interest_keyword", + "term": "Skills", + "reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=5, days_seen=4, recent_count=5.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T08:20:35.134979Z", + "action": "add_interest_keyword", + "term": "SKILL.md", + "reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=4, days_seen=4, recent_count=4.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T08:20:35.134979Z", + "action": "add_interest_keyword", + "term": "CLAUDE.md", + "reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=4, days_seen=4, recent_count=4.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T08:20:35.134979Z", + "action": "add_interest_keyword", + "term": "上下文压缩", + "reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=4, days_seen=4, recent_count=4.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T08:20:35.134979Z", + "action": "add_interest_keyword", + "term": "AI Coding Agent", + "reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=3, days_seen=3, recent_count=3.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T08:20:35.134979Z", + "action": "add_interest_keyword", + "term": "Hermes Agent", + "reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=3, days_seen=3, recent_count=3.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T08:20:35.134979Z", + "action": "add_interest_keyword", + "term": "Vibe Coding", + "reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=5, days_seen=4, recent_count=5.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T08:20:35.134979Z", + "action": "add_interest_keyword", + "term": "Context Engineering", + "reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=3, days_seen=3, recent_count=3.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T08:20:35.134979Z", + "action": "add_interest_keyword", + "term": "Cursor", + "reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=3, days_seen=3, recent_count=3.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T08:20:35.134979Z", + "action": "add_interest_keyword", + "term": "大模型", + "reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=4, days_seen=3, recent_count=4.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_interest_keyword", + "term": "TypeScript", + "reason": "Core language for AI agent development (e.g., Claude Code, Cursor) and backend engineering, complements existing Python/Java/Go keywords.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_interest_keyword", + "term": "代码审查", + "reason": "Chinese term for 'code review', a key practice in backend engineering and AI agent development workflows.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_stopword", + "term": "Channels", + "reason": "Too generic; could refer to communication channels, YouTube channels, or software channels, not specific to user's focus areas.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_stopword", + "term": "Memory", + "reason": "Extremely broad term; could refer to computer memory, human memory, or memory in various contexts, not discriminative enough.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_stopword", + "term": "Prompt", + "reason": "Already covered by 'Prompt Engineering' as a more specific term; 'Prompt' alone is too broad and matches many unrelated articles.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_stopword", + "term": "AGI", + "reason": "Too broad and speculative; not directly actionable for the user's practical engineering focus areas.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_stopword", + "term": "AI日报", + "reason": "Generic news term; not a technical concept or tool, would add noise to the keyword index.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_stopword", + "term": "AIHOT", + "reason": "Unclear meaning, likely a brand or aggregator, not a specific technical term.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_stopword", + "term": "All In Code", + "reason": "Too vague; could refer to a podcast, a philosophy, or a project, not a specific technical concept.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_stopword", + "term": "auto-twitter-campaign", + "reason": "Too specific to a single project/tool, not a general interest keyword for the user's focus areas.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_stopword", + "term": "ChangeSet", + "reason": "Generic term used in version control and databases; too broad to be a useful filter.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_stopword", + "term": "Lumina", + "reason": "Unclear reference; could be a product, framework, or brand, not clearly aligned with user's focus.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_stopword", + "term": "OpenViking", + "reason": "Unclear reference; not a known tool or concept in the user's stated focus areas.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_stopword", + "term": "Seedance 2.0", + "reason": "Unclear reference; likely a product or version, not a general technical term.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_stopword", + "term": "质量门禁", + "reason": "Chinese term for 'quality gate', too generic in software engineering; not specific to user's focus areas.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_alias", + "term": "Agent Skill", + "reason": "Singular variant", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "value": "Agent Skills", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_alias", + "term": "Binlog", + "reason": "Case variant (auto-ranked)", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "value": "binlog", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_alias", + "term": "Coding Agent", + "reason": "Abbreviated form of 'AI Coding Agent', referring to the same concept.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "value": "AI Coding Agent", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_alias", + "term": "Subagent", + "reason": "Case variant", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "value": "SubAgent", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_alias", + "term": "Subagents", + "reason": "Plural variant", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "value": "SubAgent", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_alias", + "term": "vibe coding", + "reason": "Case variant", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "value": "Vibe Coding", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_alias", + "term": "Agent架构", + "reason": "Chinese translation of 'Agent architecture', a core concept in AI Agent engineering.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "value": "AI Agent", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_alias", + "term": "Agent专业化", + "reason": "Chinese term for 'Agent specialization', directly related to Agent engineering.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "value": "AI Agent", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_alias", + "term": "Agent Teams", + "reason": "English equivalent of 'Multi-Agent collaboration', same concept.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "value": "多Agent协作", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_alias", + "term": "Agentic Engineering", + "reason": "Broader term for engineering with AI agents, closely related to Agent engineering focus.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "value": "AI Agent", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_alias", + "term": "CLI工具", + "reason": "Chinese translation of 'CLI tool', same concept.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "value": "CLI", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_alias", + "term": "AI编程", + "reason": "Chinese term for 'AI programming', closely related to AI Coding Agent.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "value": "AI Coding Agent", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_alias", + "term": "记忆管理", + "reason": "Chinese term for 'memory management', closely related to context management in LLM applications.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "value": "上下文管理", + "suggestion_date": "2026-05-14", + "based_on_days": 365 + }, + { + "applied_at": "2026-05-14T09:12:50.749877Z", + "action": "add_alias", + "term": "会话管理", + "reason": "Chinese term for 'session management', related to context management in LLM applications.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-05-14.json", + "value": "上下文管理", + "suggestion_date": "2026-05-14", + "based_on_days": 365 } ] } diff --git a/configs/term_cleanup_policy.json b/configs/term_cleanup_policy.json index 0253c46..196661d 100644 --- a/configs/term_cleanup_policy.json +++ b/configs/term_cleanup_policy.json @@ -1,14 +1,12 @@ { - "schema_version": "v1", + "schema_version": "v2", "interest_keyword_review": { - "min_total_count": 3, - "min_days_seen": 2 + "percentile_max": 0.05, + "growth_promotion": 0.5 }, "watch_term_review": { - "min_total_count": 1, - "min_days_seen": 1, - "max_total_count": 2, - "max_days_seen": 2 + "percentile_min": 0.05, + "percentile_max": 0.20 }, "alias_review": { "min_total_count": 2, @@ -19,7 +17,10 @@ "max_days_seen": 2 }, "notes": [ - "当前阶段采用保守阈值,避免在低样本条件下直接扩充 interest_keywords。", - "watch_terms 先用于观察,后续再决定是否升格为 interest_keywords 或进入 alias/stopword 配置。" + "v2: interest/watch 使用百分位排名 + 增速因子替代固定阈值", + "percentile 越小表示排名越高(top 5% = percentile 0.05)", + "growth = recent_count / total_count,衡量近期活跃度", + "watch_term_review 的 percentile_min 可理解为兴趣边界下限,低于此值的词归入 interest 候选", + "growth_promotion(默认 0.5)用于识别近期集中爆发词,即使排位不高也主动推荐确认" ] -} \ No newline at end of file +} diff --git a/configs/term_stopwords.json b/configs/term_stopwords.json index 697372f..dee00f3 100644 --- a/configs/term_stopwords.json +++ b/configs/term_stopwords.json @@ -1,6 +1,19 @@ [ + "AGI", + "AIHOT", + "AI日报", + "All In Code", + "auto-twitter-campaign", + "ChangeSet", + "Channels", + "Lumina", + "Memory", + "OpenViking", + "Prompt", + "Seedance 2.0", "奋斗文化", "小银", + "质量门禁", "银行客户经理", "飞盘物理" ] diff --git a/configs/term_watchlist.json b/configs/term_watchlist.json index 88cfa97..cffee2f 100644 --- a/configs/term_watchlist.json +++ b/configs/term_watchlist.json @@ -1,6 +1,6 @@ { "schema_version": "v1", - "updated_at": "2026-04-08T02:34:14.194320Z", + "updated_at": "2026-05-14T09:12:50.749877Z", "terms": [ { "term": "A2A", diff --git a/docs/design/keyword-cleanup-flow-overview.md b/docs/design/keyword-cleanup-flow-overview.md new file mode 100644 index 0000000..3811c77 --- /dev/null +++ b/docs/design/keyword-cleanup-flow-overview.md @@ -0,0 +1,151 @@ +# 关键词清洗流程概述 + +> 2026-05-14 初版 +> 从"数据记录"到"人工确认落盘"的完整链路 + +--- + +## 整体数据流 + +``` +每日日报 pipeline + │ + ▼ +term_index/daily/YYYY-MM-DD.json ← 每天一篇候选文章的热词统计 + │ + ▼ +term_index/term_stats.json ← 所有 daily 的汇总(1070 个词) + │ + ├──── build_review_bundle.py ← 打包为审查数据包 + │ │ + │ ▼ + │ review/keyword-cleanup-bundle.json + │ │ + │ ▼ + │ generate_term_cleanup_suggestions.py + │ │ + │ ▼ + │ review/term-cleanup-suggestions-YYYY-MM-DD.json ← 正式建议产物 + │ │ + │ ▼ + │ (可选) review/term-cleanup-suggestions-YYYY-MM-DD.md ← 展示稿 + │ + ├──── 人工确认哪些建议 accept + │ + ▼ +apply_term_suggestions.py ← 写入配置 + │ + ├── configs/filter_context.personal.json ← interest_keywords + ├── configs/term_aliases.json ← alias + ├── configs/term_stopwords.json ← stopword + ├── configs/term_watchlist.json ← watch + └── configs/term_change_log.json ← 变更日志 +``` + +--- + +## 各环节说明 + +### 阶段 1:数据记录(每日自动) + +```bash +# FreshRSS pipeline 跑完后自动产出 +data/term_index/daily/2026-05-14.json +``` + +- 每天一篇,记录当天候选文章中出现的热词 +- 包含 term、total_count、days_seen 等信息 +- 目前累计 **41 天**,共 **1070 个独立词** + +### 阶段 2:全量汇总(每日自动) + +```bash +data/term_index/term_stats.json +``` + +- 从所有 daily 文件重建,会覆盖重跑 +- 按 total_count 排序,前 5 名:OpenClaw(30)、Claude Code(25)、AI Agent(17)、Anthropic(13)、MCP(13) + +### 阶段 3:构建审查数据包(手动触发) + +```bash +python skills/keyword-cleanup-review/scripts/build_review_bundle.py \ + --days 365 \ + --top 100 \ + --output outputs/term_index/review/keyword-cleanup-bundle.json +``` + +- 把 term_stats + 当前配置打成一包,方便后续处理 +- 输出:`review/keyword-cleanup-bundle.json` + +### 阶段 4:生成建议(手动触发) + +```bash +python scripts/generate_term_cleanup_suggestions.py \ + --bundle outputs/term_index/review/keyword-cleanup-bundle.json \ + --emit-markdown +``` + +#### 当前产出能力 + +| 建议类型 | 状态 | 当前阈值 | 说明 | +|---------|------|----------|------| +| interest_keyword_suggestions | ✅ **已实现** | total≥3, days≥2 | 产出 20 条 | +| watch_terms | ✅ **已实现** | total≤2, days≤2 | 本次 0 条 | +| alias_suggestions | ❌ **硬编码为空** | policy 有阈值(total≥2, days≥2)但脚本未实现 | | +| stopword_suggestions | ❌ **硬编码为空** | policy 有阈值(total≤2, days≤2)但脚本未实现 | | + +**关键发现:** alias 和 stopword 不是"阈值太保守",是 **generate 脚本里压根没写对应的生成函数**。policy 文件里阈值已经配好了(alias: min_total=2/min_days=2,stopword: max_total=2/max_days=2),但脚本第 376-380 行直接硬编码为 `[]` 和 `0`。 + +### 阶段 5:人工确认(手动) + +``` +OpenClaw 把建议列给你 → 你确认哪些 accept → 我执行 apply +``` + +本次模式: +- 高频(≥5次/5天以上)→ 强烈推荐 ✅ +- 中频(3-4次)→ 附带建议 ✅ +- 泛词 → 建议跳过 ❌ + +### 阶段 6:落盘配置(手动) + +```bash +python scripts/apply_term_suggestions.py \ + --suggestions outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json \ + --accept-interest 词1 词2 ... +``` + +- dry-run 预览 → 确认后正式 apply +- 写入 `configs/filter_context.personal.json` +- 同步记录到 `term_change_log.json` +- **不备份原始配置**(待优化) +- **apply 后不自动清理 review 目录**(待优化) + +### 阶段 7:维护清理(按需) + +由 OpenClaw 侧 `reader-keyword-maintenance` skill 处理: +- 删除旧 markdown 展示稿 +- 保留最近一份 bundle +- 保守保留 suggestions JSON + +--- + +## 当前配置资产 + +| 文件 | 内容 | 数据量 | +|------|------|--------| +| `filter_context.personal.json` | interest_keywords | 52 个 | +| `term_aliases.json` | 别名映射 | 0 组(未启用) | +| `term_stopwords.json` | 停用词 | 0 个(未启用) | +| `term_watchlist.json` | 观察词 | 6 个 | +| `term_change_log.json` | 所有变更记录 | 已记录 | + +--- + +## 待优化项 + +1. **alias/stopword 建议生成为空** — generate 脚本硬编码缺实现,policy 已有阈值,需要补函数 +2. **apply 前无配置备份** — 建议 apply 前自动 cp 备份 +3. **apply 后无自动收尾** — 建议 apply 后自动删旧 markdown 和 bundle +4. **alias 识别依赖规则而非 LLM** — 当前全靠统计阈值,无法做语义级判断(如中英文映射、缩写展开)。如果需要高级 alias 识别,可以用 LLM 生成候选,规则脚本做 apply diff --git a/plans/keyword-cleanup-interest-watch-engine-improvement.md b/plans/keyword-cleanup-interest-watch-engine-improvement.md new file mode 100644 index 0000000..ef2267e --- /dev/null +++ b/plans/keyword-cleanup-interest-watch-engine-improvement.md @@ -0,0 +1,290 @@ +# interest/watch 候选引擎改进方案 + +> 从固定阈值到自适应排位 + 趋势因子的演进 + +## 1. 背景 + +### 1.1 当前实现 + +`build_review_bundle.py` 使用固定的绝对阈值将未覆盖词(uncovered terms)划分为两个候选池: + +| 候选池 | 判断条件 | 依据 | +|--------|---------|------| +| `interest_review_candidates` | `total_count >= 3 AND days_seen >= 2` | `policy.interest_keyword_review` | +| `watch_review_candidates` | `total_count <= 2 AND days_seen <= 2` | `policy.watch_term_review` | + +`generate_term_cleanup_suggestions.py` 则直接从这两个候选池过滤、去重、排序后输出。 + +### 1.2 当前方案的问题 + +**问题一:固定阈值不随数据量自适应** + +``` +场景 total_count=3 意味着什么 +───────────────────────────────────────────── +7 天数据(~200 词) top 15%,有一定区分度 ✅ +41 天数据(1070 词) top 5%,区分度更高 ✅ 但阈值没变 +未来 200 天 仍然用 3 次,区分度稀释 ❌ +``` + +同一个绝对次数,在不同数据规模下的语义完全不同。手工调阈值不可持续。 + +**问题二:固定阈值忽略趋势信号** + +- "Anthropic":total=13, recent=7 — 近期高活跃,上升趋势 +- "Channels":total=3, recent=0 — 早期出现但近期消失 +- 当前引擎认为这两个词"都过了 3 次阈值",同等对待。实际一个是强烈买入信号,一个是过气词。 + +**问题三:interest 和 watch 的分界线是硬的** + +total=3 → interest,total=2 → watch。一个词从 2 次变成 3 次就自动"升级",没有过渡、没有缓冲。 + +### 1.3 讨论结论 + +与老大讨论后确认: + +1. interest/watch 是**统计判断**,不需要大模型介入,纯算法可以解决 +2. 当前引擎缺的不是大模型,而是**算法本身没写完**——自适应维度(排位、趋势)还没实现 +3. alias 和 stopword 需要语义判断,与 interest/watch 分属不同阶段,不在本方案范围内 +4. 修改量小,可以在 1 小时内落地 + +--- + +## 2. 设计方案 + +### 2.1 核心思路 + +引入两个互补维度替代固定阈值: + +``` +判定维度 含义 数据来源 +──────────────────────────────────────────────────────────── +percentile(百分位排名) 该词 total_count 在所有词 term_stats + 中的排位占比 +growth(增速因子) 近期集中度 = recent_count daily 近 N 天 + / total_count +``` + +两个维度配合: + +- **percentile** 衡量"这个词在当前数据集里有多突出"——消除数据量变化的影响 +- **growth** 衡量"这个词是持续出现还是近期爆发"——识别趋势信号 + +### 2.2 候选池划分逻辑 + +``` + percentile + │ + ┌─────────────────────┐ + │ top 5% │ + │ → 建议 interest │ ← 高频稳定词 + ├─────────────────────┤ + │ top 5%-20% │ + │ → 建议 watch │ ← 有信号但未达 threshold + ├─────────────────────┤ + │ bottom 80% │ + │ → 暂不处理 │ ← 噪声/低频 + └─────────────────────┘ + +额外规则: + 如果词在 top 20% 之外,但 growth > 0.5(近期集中度高) + → 主动提升到 watch / 主动推 confirm +``` + +这样就不需要关心"total_count 是 3 还是 5",只看数据自己说话。 + +### 2.3 接口变化 + +**`configs/term_cleanup_policy.json`**: + +```json +{ + "schema_version": "v2", + "interest_keyword_review": { + "percentile_max": 0.05, + "growth_promotion": 0.5 + }, + "watch_term_review": { + "percentile_min": 0.05, + "percentile_max": 0.20 + } +} +``` + +`v1` 的 `min_total_count`/`min_days_seen` 等绝对阈值字段不再使用。 + +**`build_review_bundle.py` 输出的候选项**: + +```json +{ + "term": "Anthropic", + "total_count": 13, + "days_seen": 10, + "percentile": 0.012, + "growth": 0.54, + "reason": "top 1.2% by frequency, 54% of occurrences in recent window — strong signal." +} +``` + +### 2.4 不需要改动的部分 + +- `generate_term_cleanup_suggestions.py` — 它只消费候选池,不用改 +- `apply_term_suggestions.py` — 消费 suggestions JSON,不用改 +- `keyword-cleanup-bundle.json` 结构 — 向后兼容,新增 percentile/growth 字段 + +--- + +## 3. 实施计划 + +### 3.1 改动范围 + +| 文件 | 改动量 | 内容 | +|------|--------|------| +| `skills/keyword-cleanup-review/scripts/build_review_bundle.py` | ~40 行 | 新增 `_compute_percentile()` 和 `_compute_growth()` 函数;修改候选池生成逻辑;候选项中增加 percentile/growth | +| `configs/term_cleanup_policy.json` | ~10 行 | schema v2:percentile/growth 替代绝对阈值 | + +### 3.2 实施步骤 + +1. **build_review_bundle.py**:在 `top_global_terms` 生成后,增加 percentile 计算函数和 growth 计算函数 +2. **build_review_bundle.py**:修改 `interest_review_candidates` 和 `watch_review_candidates` 的生成逻辑,从固定阈值改为 percentile + growth +3. **build_review_bundle.py**:候选项增加 `percentile` 和 `growth` 字段,更新 `reason` 文案 +4. **term_cleanup_policy.json**:更新为 v2 schema +5. **验证**:全量跑一次(`--days 365 --top 100`),对比新旧两份输出的差异 + +### 3.3 验证方法 + +```bash +# 1. 用旧版生成 baseline +cd /home/ubuntu/zhu/github/reader +python3.11 skills/keyword-cleanup-review/scripts/build_review_bundle.py \ + --days 365 --top 100 \ + --output /tmp/bundle-baseline.json + +# 2. 改代码后用新版生成 +python3.11 skills/keyword-cleanup-review/scripts/build_review_bundle.py \ + --days 365 --top 100 \ + --output /tmp/bundle-new.json + +# 3. 对比 governance_hints +python3 -c " +import json +a = json.load(open('/tmp/bundle-baseline.json')) +b = json.load(open('/tmp/bundle-new.json')) +for key in ['interest_review_candidates', 'watch_review_candidates']: + old = set(i['term'] for i in a['governance_hints'][key]) + new = set(i['term'] for i in b['governance_hints'][key]) + print(f'{key}: 新增={new-old}, 减少={old-new}') +" +``` + +### 3.4 风险 + +| 风险 | 概率 | 应对 | +|------|------|------| +| 百分位阈值对特小数据集(如只有 1 天数据)不适用 | 低 | 不足 7 天时降级回绝对阈值 | +| growth 因子对低频词的偏差(total=1, recent=1 → growth=1) | 低 | growth 只对 total>=3 的词计算 | +| 排位突变导致推荐漂移 | 低 | percentil 天然平滑,新增几天数据不会剧烈改变已有词的排位 | + +--- + +## 4. alias/stopword 设计方案 + +### 4.1 核心判断 + +alias 和 stopword 需要语义理解,与 interest/watch(纯统计)性质不同。 + +| 类型 | 需要什么 | 判断方式 | +|------|---------|----------| +| 大小写变体 | 表层 | 规则:casefold 去重 | +| 单复数 | 表层 | 规则:去/加 s 后缀匹配 | +| 分词变体(空格/连字符) | 表层 | 规则:去空格归一 | +| 简写全称(MCP→Model Context Protocol) | **语义** | LLM | +| 中英文(上下文工程→Context Engineering) | **语义** | LLM | +| 同义不同名(Rush→猿辅导 Rush 平台) | **语义** | LLM | +| stopword(大模型、AI 太泛) | **语义** | LLM | + +### 4.2 分层方案 + +``` +输入:高频未覆盖词 + 已有 interest 词表 + │ + ├── 规则层(零成本)── 大小写归一、单复数、去空格/连字符 + │ 输出候选 alias 对 + │ + └── LLM 层(每次 ~500 token)── 把候选词表整批给 LLM + 做语义聚类 + 输出 alias 组 + stopword 标记 +``` + +### 4.3 规则层设计 + +在 `generate_term_cleanup_suggestions.py` 中新增 `_prepare_alias_suggestions()` 函数: + +```python +def _prepare_alias_suggestions(top_terms, interest_keywords): + """ + 基于表层规则生成 alias 建议。 + 规则1:casefold 匹配——同一个 casefold 下有多个原文变体 + 规则2:单复数——去掉/加上末尾 s 后匹配 + 规则3:分词变体——去空格/连字符后匹配 + """ +``` + +优势:零成本、可复现、可审计。直接写入 suggestions JSON,随 generate 一起输出。 + +### 4.4 LLM 层设计 + +单独脚本,非 generate 主链路的一部分。 + +```bash +python scripts/generate_term_cleanup_semantic_suggestions.py \ + --suggestions outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json \ + --output outputs/term_index/review/term-cleanup-semantic-suggestions-YYYY-MM-DD.json +``` + +LLM prompt 设计: + +``` +你是一个关键词治理助手。以下是一个用户的 interest 关键词列表和一批未覆盖的高频词。 +请做三件事: + +1. ALIAS:判断哪些未覆盖词是已有 interest 关键词的别名/变体 +2. STOPWORD:标记哪些词太宽泛/通用,建议排除 +3. PROMOTE:标记哪些新词与用户关注方向一致,建议加入 interest + +用户关注方向:AI Agent 工程化、后端工程、开源工具、大模型落地 +``` + +LLM 层输出格式: + +```json +{ + "alias_suggestions": [ + {"from": "Context Engineering", "to": "上下文工程", "reason": "中英文对应同一概念"} + ], + "stopword_suggestions": [ + {"term": "大模型", "reason": "过于宽泛,高频率但低区分度"} + ] +} +``` + +### 4.5 预期效果 + +| 覆盖类型 | 规则层 | LLM 层 | +|---------|--------|--------| +| 大小写变体 | ✅ | — | +| 单复数 | ✅ | — | +| 分词变体 | ✅ | — | +| 简写全称 | — | ✅ | +| 中英文映射 | — | ✅ | +| 同义不同名 | — | ✅ | +| stopword 判断 | — | ✅ | + +--- + +## 5. 讨论记录 + +- 2026-05-14:与老大确认 interest/watch 不需要 LLM,纯算法可解决 +- 2026-05-14:确认百分位排名 + 增速因子方案,修改量小,优先落地 +- 2026-05-14:确认本方案不改 `generate_term_cleanup_suggestions.py` 和 `apply_term_suggestions.py` +- 2026-05-14:确认 alias/stopword 采用规则层 + LLM 层分层方案,规则层零成本优先 diff --git a/scripts/generate_term_cleanup_semantic_suggestions.py b/scripts/generate_term_cleanup_semantic_suggestions.py new file mode 100755 index 0000000..1017398 --- /dev/null +++ b/scripts/generate_term_cleanup_semantic_suggestions.py @@ -0,0 +1,314 @@ +#!/usr/bin/env python3 +""" +Generate semantic keyword suggestions using LLM. + +Covers what surface-form rules cannot: + - semantic alias (abbreviation ↔ full name, Chinese ↔ English, synonym) + - stopword (overly broad / low-discrimination terms) + - promote (new term that aligns with user's focus areas) + +Usage: + python scripts/generate_term_cleanup_semantic_suggestions.py \ + --bundle outputs/term_index/review/keyword-cleanup-bundle.json \ + --output outputs/term_index/review/term-cleanup-semantic-suggestions-2026-05-14.json +""" + +from __future__ import annotations + +import argparse +import json +import os +import re +import sys +import time +from datetime import datetime, timezone +from pathlib import Path +from typing import Any +from urllib.request import Request, urlopen + + +REPO_ROOT = Path(__file__).resolve().parents[1] +DEFAULT_BUNDLE_PATH = REPO_ROOT / "outputs" / "term_index" / "review" / "keyword-cleanup-bundle.json" +DEFAULT_OUTPUT_DIR = REPO_ROOT / "outputs" / "term_index" / "review" + + +def _load_json(path: Path) -> Any: + return json.loads(path.read_text(encoding="utf-8-sig")) + + +def _save_json(path: Path, payload: dict[str, Any]) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") + + +def _load_env(path: Path) -> dict[str, str]: + """Load key=value pairs from .env file.""" + env: dict[str, str] = {} + if not path.exists(): + return env + for line in path.read_text(encoding="utf-8").splitlines(): + line = line.strip() + if not line or line.startswith("#") or "=" not in line: + continue + key, _, value = line.partition("=") + env[key.strip()] = value.strip().strip("\"'") + return env + + +def _build_prompt( + interest_keywords: list[str], + rule_alias_suggestions: list[dict[str, str]], + candidate_terms: list[dict[str, Any]], + relevant_watch_terms: list[dict[str, Any]], +) -> str: + """Build the LLM prompt for semantic suggestions.""" + + interest_bullets = "\n".join(f" - {t}" for t in sorted(interest_keywords)) + candidate_bullets = "\n".join( + f" - {t['term']} (count={t['total_count']}, days={t['days_seen']})" + for t in candidate_terms[:40] + ) + + # Alias from rule layer (for LLM to build on, not duplicate) + rule_alias_text = "" + if rule_alias_suggestions: + rule_alias_text = "\nSurface-form alias (already identified, skip these):\n" + "\n".join( + f" {a['from']} → {a['to']} ({a['reason']})" + for a in rule_alias_suggestions + ) + + watch_text = "" + if relevant_watch_terms: + watch_text = "\nWatch terms (low-frequency but potentially relevant):\n" + "\n".join( + f" {t['term']} (count={t['total_count']}, days={t['days_seen']})" + for t in relevant_watch_terms[:20] + ) + + return f"""You are a keyword governance assistant for an AI engineer. Your job is to analyze keyword data and produce structured suggestions. + +## User's focus areas +- AI Agent engineering (Skills, Harness, MCP, Agent architecture) +- Backend engineering (Java, Go, Kubernetes, MySQL, distributed systems) +- Open source AI tools and practices (Claude Code, Cursor, DeepSeek, OpenClaw) +- LLM application engineering (context engineering, RAG, prompt engineering) + +## Interest keywords (52 already configured) +{interest_bullets} + +## Uncovered candidate terms (sorted by frequency) +{candidate_bullets} +{watch_text}{rule_alias_text} + +## Task +Analyze the candidate terms and output a JSON object with exactly three keys: + +1. "semantic_alias": array of alias suggestions that SURFACE RULES CAN'T CATCH (e.g. abbreviation↔full name, Chinese↔English, different naming for the same concept). + Format: [{{"from": "", "to": "", "reason": ""}}] + +2. "stopword": array of terms that are too broad/generic to be useful as filters. A stopword is a term that appears frequently but has LOW DISCRIMINATION — it matches too many unrelated articles and clutters the keyword index. + Format: [{{"term": "", "reason": ""}}] + +3. "promote_to_interest": array of uncovered terms that align well with the user's focus areas and should be added as interest keywords. + Format: [{{"term": "", "reason": ""}}] + +## Rules +- Be conservative. When in doubt, leave it out. +- Only suggest alias for terms that clearly refer to the SAME concept as an existing interest keyword. +- Only suggest stopword for terms that are genuinely too broad (appear in many unrelated contexts). +- Only suggest promote for terms that clearly match the user's stated focus areas. +- Output valid JSON only, no markdown, no explanation outside the JSON.""" + + +def _call_llm(prompt: str, api_url: str, model: str, api_key: str) -> str: + """Call LLM API and return the response text.""" + payload = json.dumps({ + "model": model, + "messages": [{"role": "user", "content": prompt}], + "temperature": 0.1, + "max_tokens": 2048, + }).encode("utf-8") + + req = Request( + api_url.rstrip("/") + "/chat/completions", + data=payload, + headers={ + "Content-Type": "application/json", + "Authorization": f"Bearer {api_key}", + }, + ) + + max_retries = 3 + for attempt in range(max_retries): + try: + with urlopen(req, timeout=120) as resp: + result = json.loads(resp.read().decode("utf-8")) + return result["choices"][0]["message"]["content"] + except Exception as e: + if attempt < max_retries - 1: + wait = 2 ** attempt + print(f" LLM call failed (attempt {attempt+1}/{max_retries}): {e}", file=sys.stderr) + print(f" Retrying in {wait}s...", file=sys.stderr) + time.sleep(wait) + else: + raise + + +def _parse_llm_response(text: str) -> dict[str, list[dict[str, str]]]: + """Extract JSON from LLM response (may contain markdown fences).""" + # Try to find JSON block + json_match = re.search(r"```(?:json)?\s*\n?(\{.*?\})\s*\n?```", text, re.DOTALL) + if json_match: + text = json_match.group(1) + + # Clean up: remove any text before { or after } + start = text.find("{") + end = text.rfind("}") + if start >= 0 and end > start: + text = text[start : end + 1] + + try: + result = json.loads(text) + except json.JSONDecodeError: + # Try partial recovery + print(f" Warning: LLM response not clean JSON, attempting recovery", file=sys.stderr) + print(f" Raw: {text[:500]}", file=sys.stderr) + return {"semantic_alias": [], "stopword": [], "promote_to_interest": []} + + # Normalize keys + normalized = { + "semantic_alias": result.get("semantic_alias", result.get("alias", [])), + "stopword": result.get("stopword", result.get("stopword_suggestions", [])), + "promote_to_interest": result.get("promote_to_interest", result.get("promote", [])), + } + # Ensure each is a list + for key in normalized: + if not isinstance(normalized[key], list): + normalized[key] = [] + return normalized + + +def main() -> None: + parser = argparse.ArgumentParser(description="Generate semantic keyword suggestions via LLM.") + parser.add_argument("--bundle", type=Path, default=DEFAULT_BUNDLE_PATH, help="Review bundle JSON path") + parser.add_argument("--suggestions", type=Path, default=None, help="Existing suggestions JSON (for rule alias context)") + parser.add_argument("--output", type=Path, default=None, help="Output JSON path (auto-generated if omitted)") + parser.add_argument("--llm-api-url", type=str, default=None, help="LLM API base URL") + parser.add_argument("--llm-model", type=str, default=None, help="LLM model name") + parser.add_argument("--llm-api-key", type=str, default=None, help="LLM API key") + parser.add_argument("--dry-run", action="store_true", help="Print prompt and exit without calling LLM") + args = parser.parse_args() + + # Load config + env_path = REPO_ROOT / ".env" + env = _load_env(env_path) if env_path.exists() else {} + + api_url = args.llm_api_url or os.environ.get("LLM_API_URL") or env.get("LLM_API_URL", "https://api.deepseek.com") + # Map OpenClaw model aliases to actual API model names + model_raw = args.llm_model or os.environ.get("LLM_MODEL") or env.get("LLM_MODEL", "deepseek-chat") + MODEL_ALIAS_MAP = { + "deepseek/deepseek-v4-flash": "deepseek-chat", + "deepseek/deepseek-chat": "deepseek-chat", + "deepseek-v4-flash": "deepseek-chat", + "deepseek-chat": "deepseek-chat", + } + model = MODEL_ALIAS_MAP.get(model_raw, model_raw) + api_key = args.llm_api_key or os.environ.get("LLM_API_KEY") or env.get("LLM_API_KEY", "") + + if not api_key: + print("Error: No LLM API key found. Set LLM_API_KEY in .env or pass --llm-api-key.", file=sys.stderr) + sys.exit(1) + + # Load bundle + if not args.bundle.exists(): + print(f"Error: Bundle not found: {args.bundle}", file=sys.stderr) + sys.exit(1) + + bundle = _load_json(args.bundle) + current_config = bundle.get("current_config", {}) + interest_keywords = current_config.get("interest_keywords", []) + top_global_terms = bundle.get("top_global_terms", []) + governance_hints = bundle.get("governance_hints", {}) + + # Build candidate list (uncovered terms from interest + watch candidates) + candidate_terms = [] + for item in governance_hints.get("interest_review_candidates", []): + if isinstance(item, dict): + candidate_terms.append({ + "term": item.get("term", ""), + "total_count": item.get("total_count", 0), + "days_seen": item.get("days_seen", 0), + "percentile": item.get("percentile", 0), + "growth": item.get("growth", 0), + }) + for item in governance_hints.get("watch_review_candidates", []): + if isinstance(item, dict): + # Avoid duplicates + if not any(c["term"] == item.get("term") for c in candidate_terms): + candidate_terms.append({ + "term": item.get("term", ""), + "total_count": item.get("total_count", 0), + "days_seen": item.get("days_seen", 0), + "percentile": item.get("percentile", 0), + "growth": item.get("growth", 0), + }) + + # Sort by total_count descending + candidate_terms.sort(key=lambda x: -x["total_count"]) + relevant_watch_terms = governance_hints.get("watch_review_candidates", [])[:20] + + # Load rule-layer alias suggestions if available + rule_alias = [] + if args.suggestions and args.suggestions.exists(): + s = _load_json(args.suggestions) + rule_alias = s.get("alias_suggestions", []) + + # Build prompt + prompt = _build_prompt( + interest_keywords=interest_keywords, + rule_alias_suggestions=rule_alias, + candidate_terms=candidate_terms, + relevant_watch_terms=relevant_watch_terms, + ) + + # Determine output path + suggestion_date = datetime.now(timezone.utc).date().isoformat() + output_path = args.output or (DEFAULT_OUTPUT_DIR / f"term-cleanup-semantic-suggestions-{suggestion_date}.json") + + if args.dry_run: + print("=== DRY RUN: Prompt ===") + print(prompt) + print("\n=== END ===") + print(f"\nWould write to: {output_path}") + return + + # Call LLM + print(f"Calling LLM ({model})...", file=sys.stderr) + response = _call_llm(prompt, api_url, model, api_key) + print(f"LLM response received ({len(response)} chars)", file=sys.stderr) + + # Parse + parsed = _parse_llm_response(response) + + # Build output + output = { + "date": suggestion_date, + "source_bundle": str(args.bundle), + "model": model, + "interest_keyword_count": len(interest_keywords), + "candidate_count": len(candidate_terms), + **parsed, + } + + _save_json(output_path, output) + + summary = { + "output": str(output_path), + "semantic_alias": len(output.get("semantic_alias", [])), + "stopword": len(output.get("stopword", [])), + "promote_to_interest": len(output.get("promote_to_interest", [])), + } + print(json.dumps(summary, ensure_ascii=False, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/scripts/generate_term_cleanup_suggestions.py b/scripts/generate_term_cleanup_suggestions.py index f8ead96..20c435f 100644 --- a/scripts/generate_term_cleanup_suggestions.py +++ b/scripts/generate_term_cleanup_suggestions.py @@ -175,6 +175,121 @@ def _prepare_watch_suggestions(bundle: dict[str, Any], reserved_terms: set[str]) return suggestions +def _prepare_alias_suggestions( + bundle: dict[str, Any], + all_terms: list[dict[str, Any]] | None = None, +) -> list[dict[str, Any]]: + """ + Generate alias suggestions using surface-form rules (no LLM). + + Rules: + 1. casefold match — same normalized form, different original casing + 2. trailing-s singularization — singular/plural variants + 3. whitespace/hyphen normalization — word boundary variants + + Scans all_terms (full term_stats) if provided; otherwise falls back + to top_global_terms from the bundle. + """ + current_config = _require_dict(bundle.get("current_config"), "bundle.current_config") + interest_keywords = _require_list( + current_config.get("interest_keywords"), "bundle.current_config.interest_keywords" + ) + top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms") + source_terms = all_terms if all_terms is not None else top_global_terms + + interest_set = {_term_key(t) for t in interest_keywords if isinstance(t, str)} + interest_originals: set[str] = {t for t in interest_keywords if isinstance(t, str)} + + # Build full casefold → [original forms] map + cf_map: dict[str, list[str]] = {} + for item in source_terms: + term = None + if isinstance(item, dict): + term = item.get("term") + elif isinstance(item, str): + term = item + if not isinstance(term, str) or not term.strip(): + continue + key = _term_key(term) + if key not in cf_map: + cf_map[key] = [] + if term not in cf_map[key]: + cf_map[key].append(term) + + suggestions: list[dict[str, Any]] = [] + seen_pairs: set[tuple[str, str]] = set() + + def _add(from_term: str, to_term: str, reason: str) -> None: + pair = (_term_key(from_term), _term_key(to_term)) + if pair in seen_pairs: + return + seen_pairs.add(pair) + suggestions.append({"from": from_term, "to": to_term, "reason": reason}) + + # Build a set of all term keys from source for quick lookup + source_keys = set(cf_map.keys()) + + # Rule 1: casefold match — same normalized form, different casing + for key, variants in cf_map.items(): + if len(variants) < 2: + continue + canonical = None + alt_forms = [] + for v in variants: + if v in interest_originals: + canonical = v + else: + alt_forms.append(v) + if canonical and alt_forms: + for alt in alt_forms: + _add(alt, canonical, "Case variant") + elif len(variants) >= 2 and not canonical: + # None is canonical — suggest the highest-frequency form + ranked = sorted(variants, key=lambda t: -( + next( + (it.get("total_count", 0) for it in top_global_terms if it.get("term") == t), + 0, + ) + )) + for alt in ranked[1:]: + _add(alt, ranked[0], "Case variant (auto-ranked)") + + # Rule 2: singular/plural — trailing-s normalization + # Check all source terms (not just interest keys) for bidirectional matching + for key in source_keys: + if key in interest_set: + continue + if key.endswith("s") and len(key) > 2: + singular_key = key.rstrip("s") + if singular_key in interest_set and singular_key != key: + # Find canonical interest keyword + canon = next((t for t in interest_keywords if _term_key(t) == singular_key), None) + from_form = cf_map[key][0] + if canon: + _add(from_form, canon, "Plural variant") + # singular form → interest has plural + plural_key = key + "s" + if plural_key in interest_set and plural_key != key: + canon = next((t for t in interest_keywords if _term_key(t) == plural_key), None) + from_form = cf_map[key][0] + if canon: + _add(from_form, canon, "Singular variant") + + # Rule 3: whitespace/hyphen normalization + for key in source_keys: + if key in interest_set: + continue + normalized = key.replace("-", "").replace("_", "").replace(" ", "") + if normalized in interest_set and normalized != key: + canon = next((t for t in interest_keywords if _term_key(t) == normalized), None) + from_form = cf_map[key][0] + if canon: + _add(from_form, canon, "Whitespace/punctuation variant") + + suggestions.sort(key=lambda x: (x["from"].casefold(), x["to"].casefold())) + return suggestions + + def _render_table(items: list[dict[str, Any]]) -> str: if not items: return "_None in this pass._\n" @@ -361,9 +476,19 @@ def main() -> None: markdown_output=args.markdown_output, ) + # Load full term_stats for alias scanning (bundle only has top N) + stats_path = REPO_ROOT / "data" / "term_index" / "term_stats.json" + all_stats_terms: list[str] = [] + if stats_path.exists(): + stats_payload = _load_json(stats_path) + raw_terms = stats_payload.get("terms") if isinstance(stats_payload, dict) else [] + if isinstance(raw_terms, list): + all_stats_terms = [str(t["term"]) for t in raw_terms if isinstance(t, dict) and isinstance(t.get("term"), str)] + interest_items = _prepare_interest_suggestions(bundle) reserved_terms = {_term_key(str(item.get("term") or "")) for item in interest_items} watch_items = _prepare_watch_suggestions(bundle, reserved_terms=reserved_terms) + alias_items = _prepare_alias_suggestions(bundle, all_terms=all_stats_terms) suggestions = { "date": suggestion_date, @@ -373,10 +498,10 @@ def main() -> None: "summary": { "interest_keyword_suggestions": len(interest_items), "watch_terms": len(watch_items), - "alias_suggestions": 0, + "alias_suggestions": len(alias_items), "stopword_suggestions": 0, }, - "alias_suggestions": [], + "alias_suggestions": alias_items, "stopword_suggestions": [], "interest_keyword_suggestions": interest_items, "watch_terms": watch_items, @@ -400,7 +525,7 @@ def main() -> None: "markdown_output": str(markdown_output_path) if args.emit_markdown else None, "interest_keyword_suggestions": len(interest_items), "watch_terms": len(watch_items), - "alias_suggestions": 0, + "alias_suggestions": len(alias_items), "stopword_suggestions": 0, "emit_markdown": args.emit_markdown, } diff --git a/skills/keyword-cleanup-review/SKILL.md b/skills/keyword-cleanup-review/SKILL.md index bec423d..bb25954 100644 --- a/skills/keyword-cleanup-review/SKILL.md +++ b/skills/keyword-cleanup-review/SKILL.md @@ -26,7 +26,7 @@ description: 生成 reader 项目的正式关键词 review 输入。当用户需 ## 工作流程 -1. 构建精简的审查数据包(临时工作文件): +### Phase 1:构建审查数据包 ```bash python skills/keyword-cleanup-review/scripts/build_review_bundle.py @@ -34,51 +34,92 @@ python skills/keyword-cleanup-review/scripts/build_review_bundle.py 可选参数: -- `--days 7` -- `--top 50` +- `--days 7`(默认 7,建议传 365 覆盖全量) +- `--top 100`(考虑的词数) - `--output outputs/term_index/review/keyword-cleanup-bundle.json` -2. 阅读建议模式: +#### 候选引擎策略 -- `skills/keyword-cleanup-review/references/suggestion-schema.md` +根据 `configs/term_cleanup_policy.json` 的 `schema_version` 自动切换: -3. 运行 suggestions 生成脚本: +| 版本 | 策略 | 说明 | +|------|------|------| +| v1(旧) | 固定阈值(total≥3/days≥2 → interest) | 小数据集兼容 | +| v2(当前默认) | 百分位排名 + 增速因子 | 自适应数据量,不需要手工调阈值 | + +v2 策略说明: +- **percentile**:total_count 在所有词里的排位占比。top 5% → interest 候选,5%-20% → watch 候选 +- **growth**:recent_count / total_count,衡量近期活跃度。growth≥0.5 的排位外词也会主动推荐 + +### Phase 2:生成建议(规则层) ```bash -python scripts/generate_term_cleanup_suggestions.py ^ +python scripts/generate_term_cleanup_suggestions.py \ --bundle outputs/term_index/review/keyword-cleanup-bundle.json ``` -默认生成: - -- 一份符合模式的 JSON 建议文件(正式建议产物,也是 review / apply 之间唯一正式输入) - -如需人工审阅展示稿,再显式加: +如需人工审阅展示稿: ```bash -python scripts/generate_term_cleanup_suggestions.py ^ - --bundle outputs/term_index/review/keyword-cleanup-bundle.json ^ +python scripts/generate_term_cleanup_suggestions.py \ + --bundle outputs/term_index/review/keyword-cleanup-bundle.json \ --emit-markdown ``` -这时才会额外生成: +#### 产出能力 -- 一份简短的供人工审阅的 Markdown 报告(临时展示稿) +| 建议类型 | 状态 | 方法 | +|---------|------|------| +| interest 建议 | ✅ 已实现 | 百分位 top 5% + 增速促活 | +| watch 建议 | ✅ 已实现 | 百分位 5%-20% | +| alias 建议 | ✅ 已实现 | 规则层:大小写归一、单复数、去空格/连字符 | +| stopword 建议 | ❌ 规则层空缺 | 见 Phase 3(LLM 层) | -4. 严格保持边界: +默认生成: +- `term-cleanup-suggestions-YYYY-MM-DD.json`(正式建议产物) + +显式加 `--emit-markdown` 额外生成: +- `term-cleanup-suggestions-YYYY-MM-DD.md`(临时展示稿) + +### Phase 3:生成建议(LLM 层,可选) + +规则层覆盖不了 alias(中英文对应、缩写展开、同义不同名)和 stopword 判断,需要 LLM 辅助: + +```bash +python scripts/generate_term_cleanup_semantic_suggestions.py \ + --bundle outputs/term_index/review/keyword-cleanup-bundle.json \ + --suggestions outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json \ + --output outputs/term_index/review/term-cleanup-semantic-suggestions-YYYY-MM-DD.json +``` + +从 `.env` 读取 LLM 配置(`LLM_API_URL` / `LLM_MODEL` / `LLM_API_KEY`),使用 DeepSeek API。 + +输出三部分: + +| 输出 | 说明 | +|------|------| +| `semantic_alias` | 语义级别名(中英文、缩写、同义不同名) | +| `stopword` | 泛词过滤建议(规则层做不了的需要语义判断的) | +| `promote_to_interest` | 与用户关注方向一致的新词,建议加入 interest | + +**注:LLM 层产物是候选,不应自动 apply,需要人工确认后由 OpenClaw 编排 apply。** + +### Phase 4:输出给 OpenClaw 编排 + +- `suggestions JSON` = review / apply 之间唯一正式建议输入 +- `semantic-suggestions JSON` = LLM 补充建议,需要人工筛选后合并到 suggestions JSON 再 apply +- Markdown = 临时展示层 +- 后续汇报、确认、dry-run、apply、收尾清理由 OpenClaw 编排层执行 + +### Phase 5:严格保持边界 - 建议 `configs/term_aliases.json` 的修改 - 建议 `configs/term_stopwords.json` 的修改 - 建议 `configs/filter_context.personal.json` 的新增 +- **LLM 层产出(semantic-suggestions)不自动 apply**,需人工确认后由 OpenClaw 编排层执行 - 除非用户明确要求,否则不要直接编辑这些文件 - 除非用户要求修改规则逻辑,否则不要建议直接编辑 `configs/filter_rules.json` -5. 输出交接口径: - -- 将 JSON suggestions 视为正式 review 输入 -- 将 Markdown 视为可选展示层 -- 后续汇报、确认、dry-run apply、正式 apply、收尾清理应由 OpenClaw 编排层继续执行 - ## 审查启发式规则 优先考虑以下决策: @@ -141,6 +182,7 @@ JSON 输出应遵循: 短期保留: - `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json` +- `outputs/term_index/review/term-cleanup-semantic-suggestions-YYYY-MM-DD.json` 临时产物: @@ -173,7 +215,9 @@ JSON 输出应遵循: ## 资源 - 脚本: - - `scripts/build_review_bundle.py` + - `skills/keyword-cleanup-review/scripts/build_review_bundle.py` - `scripts/generate_term_cleanup_suggestions.py` + - `scripts/generate_term_cleanup_semantic_suggestions.py`(LLM 层) - 参考文档: - `references/suggestion-schema.md` + - `plans/keyword-cleanup-interest-watch-engine-improvement.md`(v2 引擎设计) diff --git a/skills/keyword-cleanup-review/scripts/build_review_bundle.py b/skills/keyword-cleanup-review/scripts/build_review_bundle.py index 90a06f3..4a7ebf5 100644 --- a/skills/keyword-cleanup-review/scripts/build_review_bundle.py +++ b/skills/keyword-cleanup-review/scripts/build_review_bundle.py @@ -9,16 +9,15 @@ from typing import Any DEFAULT_POLICY: dict[str, Any] = { - "schema_version": "v1", + "schema_version": "v2", "interest_keyword_review": { - "min_total_count": 3, - "min_days_seen": 2, + "percentile_min": 0.0, + "percentile_max": 0.05, + "growth_promotion": 0.5, }, "watch_term_review": { - "min_total_count": 1, - "min_days_seen": 1, - "max_total_count": 2, - "max_days_seen": 2, + "percentile_min": 0.05, + "percentile_max": 0.20, }, "alias_review": { "min_total_count": 2, @@ -28,6 +27,11 @@ DEFAULT_POLICY: dict[str, Any] = { "max_total_count": 2, "max_days_seen": 2, }, + "notes": [ + "v2: interest/watch 使用百分位排名 + 增速因子替代固定阈值", + "percentile 越小表示排名越高(top 5% = percentile 0.05)", + "growth = recent_count / total_count,衡量近期活跃度", + ], } @@ -126,6 +130,37 @@ def _within_watch_thresholds(item: dict[str, Any], thresholds: dict[str, Any]) - ) +def _compute_percentile(value: int, sorted_values: list[int]) -> float: + """ + Return the percentile rank of `value` in `sorted_values` (ascending). + 0.0 = highest frequency (top rank), 1.0 = lowest frequency (bottom rank). + """ + if not sorted_values: + return 1.0 + # bisect_left — count of values strictly less than `value` + lo, hi = 0, len(sorted_values) + while lo < hi: + mid = (lo + hi) // 2 + if sorted_values[mid] < value: + lo = mid + 1 + else: + hi = mid + rank = lo + # invert: smallest value → rank=0 → 1.0 (bottom) + # largest value → rank=len → 0.0 (top) + return 1.0 - (rank / len(sorted_values)) + + +def _compute_growth(recent_count: int, total_count: int) -> float: + """ + Return growth factor: recent_count / total_count. + Only meaningful when total_count >= 3; returns 0.0 for small counts. + """ + if total_count < 3: + return 0.0 + return recent_count / total_count + + def main() -> None: parser = argparse.ArgumentParser( description="Build a compact review bundle for the keyword-cleanup-review skill." @@ -235,6 +270,13 @@ def main() -> None: alias_values = _casefold_set(list(aliases.values())) watch_set = _casefold_set([str(item.get("term", "")) for item in watchlist]) + # Build a sorted list of all total_counts for percentile computation + all_total_counts = sorted( + int(item.get("total_count") or 0) + for item in stats_terms + if isinstance(item, dict) and isinstance(item.get("term"), str) + ) + top_global_terms = [] for item in stats_terms[: args.top]: if not isinstance(item, dict): @@ -256,42 +298,108 @@ def main() -> None: "is_alias_target": folded in alias_values, "in_watchlist": folded in watch_set, "recent_count": recent_counter.get(term, 0), + "percentile": _compute_percentile( + int(item.get("total_count") or 0), all_total_counts + ), + "growth": _compute_growth( + recent_counter.get(term, 0), + int(item.get("total_count") or 0), + ), } ) + # Keep more uncovered terms for percentile-based selection uncovered_terms = [ item for item in top_global_terms if not item["in_interest_keywords"] and not item["is_stopword"] - ][:20] + ][:100] + + policy_version = (policy.get("schema_version") if isinstance(policy, dict) else None) or "v1" interest_thresholds = policy.get("interest_keyword_review") if isinstance(policy, dict) else {} watch_thresholds = policy.get("watch_term_review") if isinstance(policy, dict) else {} + + if policy_version == "v2" or "percentile_max" in interest_thresholds: + # v2: percentile + growth based selection + pct_min_interest = float(interest_thresholds.get("percentile_min", 0.0)) + pct_max_interest = float(interest_thresholds.get("percentile_max", 0.05)) + growth_promo = float(interest_thresholds.get("growth_promotion", 0.5)) + pct_min_watch = float(watch_thresholds.get("percentile_min", 0.05)) + pct_max_watch = float(watch_thresholds.get("percentile_max", 0.20)) + + interest_candidates_raw = [ + item for item in uncovered_terms + if not item["in_watchlist"] + and pct_min_interest <= item["percentile"] <= pct_max_interest + ] + watch_candidates_raw = [ + item for item in uncovered_terms + if not item["in_watchlist"] + and pct_min_watch < item["percentile"] <= pct_max_watch + ] + # Growth boost: terms outside watch range but with strong growth signal + growth_boost_candidates = [ + item for item in uncovered_terms + if not item["in_watchlist"] + and item["percentile"] > pct_max_watch + and item["growth"] >= growth_promo + ] + else: + # v1 fallback: fixed thresholds + interest_candidates_raw = [ + item for item in uncovered_terms + if not item["in_watchlist"] and _meets_min_thresholds(item, interest_thresholds) + ] + watch_candidates_raw = [ + item for item in uncovered_terms + if not item["in_watchlist"] + and not _meets_min_thresholds(item, interest_thresholds) + and _within_watch_thresholds(item, watch_thresholds) + ] + growth_boost_candidates = [] + interest_review_candidates = [ { "term": item["term"], "total_count": item["total_count"], "days_seen": item["days_seen"], + "percentile": item["percentile"], + "growth": item["growth"], "reason": ( - "Meets the configured interest-keyword review threshold and is not yet covered " - "by interest keywords or stopwords." + f"top {item['percentile']:.1%} by frequency," + f"growth={item['growth']:.0%}," + "not yet covered by interest keywords or stopwords." ), } - for item in uncovered_terms - if not item["in_watchlist"] and _meets_min_thresholds(item, interest_thresholds) + for item in interest_candidates_raw ][:20] watch_review_candidates = [ { "term": item["term"], "total_count": item["total_count"], "days_seen": item["days_seen"], + "percentile": item["percentile"], + "growth": item["growth"], "reason": ( - "Falls into the configured watch-term review range and should be observed " - "before promotion into interest keywords." + f"top {item['percentile']:.1%} by frequency," + f"growth={item['growth']:.0%}," + "fell into watch-review range." ), } - for item in uncovered_terms - if not item["in_watchlist"] - and not _meets_min_thresholds(item, interest_thresholds) - and _within_watch_thresholds(item, watch_thresholds) + for item in watch_candidates_raw ][:20] + growth_boost_review_items = [ + { + "term": item["term"], + "total_count": item["total_count"], + "days_seen": item["days_seen"], + "percentile": item["percentile"], + "growth": item["growth"], + "reason": ( + f"growth spike: {item['growth']:.0%} of occurrences in recent window " + f"(total={item['total_count']}, days={item['days_seen']})." + ), + } + for item in growth_boost_candidates + ][:5] recent_hot_terms = sorted( ({"term": term, "recent_count": count} for term, count in recent_counter.items()), key=lambda item: (-item["recent_count"], item["term"].casefold(), item["term"]), @@ -330,6 +438,7 @@ def main() -> None: "governance_hints": { "interest_review_candidates": interest_review_candidates, "watch_review_candidates": watch_review_candidates, + "growth_boost_review_items": growth_boost_review_items, }, } _save_json(args.output, bundle)