From 32ad5843483c67b8214fb466739ccdc8802a0999 Mon Sep 17 00:00:00 2001 From: zhuyongxin Date: Fri, 27 Mar 2026 16:59:52 +0800 Subject: [PATCH] Add keyword cleanup governance workflow --- .gitignore | 2 + README.md | 54 ++ TODO.md | 15 +- configs/filter_context.personal.json | 9 +- configs/term_aliases.json | 5 + configs/term_change_log.json | 4 + configs/term_cleanup_policy.json | 25 + configs/term_stopwords.json | 6 + configs/term_watchlist.json | 5 + docs/README.md | 22 +- docs/current/context-reset-brief.md | 45 +- docs/design/daily-keyword-index-design.md | 461 ++++++++++++++++++ outputs/README.md | 16 +- scripts/apply_term_suggestions.py | 402 +++++++++++++++ scripts/build_keyword_index.py | 76 +++ scripts/run_freshrss_pipeline.py | 4 + skills/keyword-cleanup-review/SKILL.md | 102 ++++ .../keyword-cleanup-review/agents/openai.yaml | 3 + .../references/suggestion-schema.md | 47 ++ .../references/trigger-notes.md | 1 + .../scripts/build_review_bundle.py | 340 +++++++++++++ src/summary_mcp/core/keyword_index.py | 228 +++++++++ src/summary_mcp/models/__init__.py | 7 +- src/summary_mcp/models/keyword_index.py | 35 ++ .../workflows/freshrss_pipeline.py | 19 + 25 files changed, 1915 insertions(+), 18 deletions(-) create mode 100644 configs/term_aliases.json create mode 100644 configs/term_change_log.json create mode 100644 configs/term_cleanup_policy.json create mode 100644 configs/term_stopwords.json create mode 100644 configs/term_watchlist.json create mode 100644 docs/design/daily-keyword-index-design.md create mode 100644 scripts/apply_term_suggestions.py create mode 100644 scripts/build_keyword_index.py create mode 100644 skills/keyword-cleanup-review/SKILL.md create mode 100644 skills/keyword-cleanup-review/agents/openai.yaml create mode 100644 skills/keyword-cleanup-review/references/suggestion-schema.md create mode 100644 skills/keyword-cleanup-review/references/trigger-notes.md create mode 100644 skills/keyword-cleanup-review/scripts/build_review_bundle.py create mode 100644 src/summary_mcp/core/keyword_index.py create mode 100644 src/summary_mcp/models/keyword_index.py diff --git a/.gitignore b/.gitignore index 16c6ba2..9f2b3b0 100644 --- a/.gitignore +++ b/.gitignore @@ -8,3 +8,5 @@ findings.md progress.md task_plan.md outputs/freshrss/ +data/term_index/ +outputs/term_index/ diff --git a/README.md b/README.md index 468ee5c..bd54e8b 100644 --- a/README.md +++ b/README.md @@ -92,6 +92,11 @@ This is the recommended production entrypoint. By default it writes only: - `outputs/freshrss/rerun//candidates/openclaw-delivery-payload.json` - `outputs/freshrss/rerun//run-report.json` +It also updates the daily keyword index runtime data: + +- `data/term_index/daily/YYYY-MM-DD.json` +- `data/term_index/term_stats.json` + If you need per-item intermediates, add `--debug-artifacts`. When OpenClaw is connected to the MCP server, it should call `run_freshrss_openclaw_pipeline` for the same behavior directly through MCP. The tool also supports `debug_artifacts=true` when deeper inspection is needed. @@ -160,3 +165,52 @@ The script writes by default: - `outputs/reference/candidates/openclaw-delivery-payload.json` Output layout details live in `outputs/README.md`. + +Keyword index defaults live in: + +- `configs/term_aliases.json` +- `configs/term_stopwords.json` +- `configs/term_cleanup_policy.json` +- `configs/term_watchlist.json` +- `configs/term_change_log.json` + +You can also rebuild the keyword index from an existing delivery payload: + +```bash +python scripts/build_keyword_index.py ^ + --input outputs/reference/candidates/openclaw-delivery-payload.json +``` + +Runtime keyword data is stored under `data/term_index/`. + +The keyword cleanup review skill lives in: + +- `skills/keyword-cleanup-review/` + +To build a review bundle for the LLM skill: + +```bash +python skills/keyword-cleanup-review/scripts/build_review_bundle.py ^ + --days 7 ^ + --top 50 ^ + --output outputs/term_index/review/keyword-cleanup-bundle.json +``` + +The review bundle now also carries cleanup governance context: + +- cleanup thresholds from `configs/term_cleanup_policy.json` +- the current watch list from `configs/term_watchlist.json` +- recent applied changes from `configs/term_change_log.json` + +The skill only produces review inputs and suggestions. It does not modify `term_aliases`, `term_stopwords`, or `filter_context.personal.json` automatically. + +To preview accepted suggestions before writing any config files: + +```bash +python scripts/apply_term_suggestions.py ^ + --suggestions outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json ^ + --accept-watch Cron Heartbeat Memory ^ + --dry-run +``` + +Remove `--dry-run` to write the accepted changes. The script can also apply accepted `alias`, `stopword`, and `interest keyword` suggestions through `--accept-alias`, `--accept-stopword`, and `--accept-interest`. Accepted watch terms are written into `configs/term_watchlist.json`, and every applied action is appended into `configs/term_change_log.json`. diff --git a/TODO.md b/TODO.md index 1b7c6ef..b49f133 100644 --- a/TODO.md +++ b/TODO.md @@ -20,6 +20,11 @@ - [x] 最终 payload 成功后才标记 FreshRSS 已读 - [x] MCP 工具 `run_freshrss_openclaw_pipeline` - [x] 默认精简输出模式 +- [x] 日报级 `keywords` 词元库与周期性词元清洗 skill 设计完成 +- [x] 日报级 `keywords` 词元库与全局词频统计实现完成 +- [x] `keyword-cleanup-review` skill 骨架与 review bundle 脚本实现完成 +- [x] 词元清洗低复杂治理层落地:`term_cleanup_policy` / `term_watchlist` / `term_change_log` +- [x] 已支持人工确认采纳建议并写入 `term_watchlist` / `term_change_log` --- @@ -38,6 +43,8 @@ - [ ] 设计“人工确认后再沉淀知识库”的状态流转 - [ ] 收敛 `paywall` 误判规则,降低中文文本误报 - [ ] 细化过滤规则并引入更多个性化上下文 +- [ ] 将 `keyword-cleanup-review` skill 接入周期性执行流程,产出别名/停用词/兴趣词建议 +- [ ] 增加清洗前后效果对比报告,验证配置调整是否真的改善过滤质量 - [ ] 增加批量 run 的保留策略与历史清理策略 - [ ] 为 OpenClaw 补一份更正式的 MCP 调用示例和接线说明 @@ -59,8 +66,8 @@ 优先做这三件事: -1. 明确 OpenClaw 侧的 MCP 接线方式 -2. 设计主动投递 / webhook 机制 +1. 将 `keyword-cleanup-review` skill 接入周期性执行流程 +2. 设计“人工确认后再沉淀知识库”的状态流转 3. 收敛规则误判,尤其是 `paywall` 相关启发式 --- @@ -71,4 +78,6 @@ - `docs/openclaw/openclaw-handoff.md` - `docs/openclaw/openclaw-candidate-input-field-spec.md` - `docs/openclaw/openclaw-delivery-payload-spec.md` -- `docs/current/context-reset-brief.md` +- `docs/design/daily-keyword-index-design.md` +- `skills/keyword-cleanup-review/SKILL.md` +- `docs/current/context-reset-brief.md` \ No newline at end of file diff --git a/configs/filter_context.personal.json b/configs/filter_context.personal.json index 0bc24a0..1167b8b 100644 --- a/configs/filter_context.personal.json +++ b/configs/filter_context.personal.json @@ -1,4 +1,4 @@ -{ +{ "source_tags": [ "backend", "ai-agent", @@ -48,6 +48,9 @@ "向量数据库", "知识库", "OpenAI", - "DeepSeek" + "DeepSeek", + "OpenClaw", + "AliSQL", + "MySQL复制延迟" ] -} +} \ No newline at end of file diff --git a/configs/term_aliases.json b/configs/term_aliases.json new file mode 100644 index 0000000..87e2055 --- /dev/null +++ b/configs/term_aliases.json @@ -0,0 +1,5 @@ +{ + "AI助手": "AI Agent", + "图文RAG": "RAG", + "Prompt架构": "Prompt Engineering" +} \ No newline at end of file diff --git a/configs/term_change_log.json b/configs/term_change_log.json new file mode 100644 index 0000000..e786764 --- /dev/null +++ b/configs/term_change_log.json @@ -0,0 +1,4 @@ +{ + "schema_version": "v1", + "entries": [] +} \ No newline at end of file diff --git a/configs/term_cleanup_policy.json b/configs/term_cleanup_policy.json new file mode 100644 index 0000000..0253c46 --- /dev/null +++ b/configs/term_cleanup_policy.json @@ -0,0 +1,25 @@ +{ + "schema_version": "v1", + "interest_keyword_review": { + "min_total_count": 3, + "min_days_seen": 2 + }, + "watch_term_review": { + "min_total_count": 1, + "min_days_seen": 1, + "max_total_count": 2, + "max_days_seen": 2 + }, + "alias_review": { + "min_total_count": 2, + "min_days_seen": 2 + }, + "stopword_review": { + "max_total_count": 2, + "max_days_seen": 2 + }, + "notes": [ + "当前阶段采用保守阈值,避免在低样本条件下直接扩充 interest_keywords。", + "watch_terms 先用于观察,后续再决定是否升格为 interest_keywords 或进入 alias/stopword 配置。" + ] +} \ No newline at end of file diff --git a/configs/term_stopwords.json b/configs/term_stopwords.json new file mode 100644 index 0000000..9ef77c0 --- /dev/null +++ b/configs/term_stopwords.json @@ -0,0 +1,6 @@ +[ + "小银", + "银行客户经理", + "飞盘物理", + "奋斗文化" +] \ No newline at end of file diff --git a/configs/term_watchlist.json b/configs/term_watchlist.json new file mode 100644 index 0000000..b5c6060 --- /dev/null +++ b/configs/term_watchlist.json @@ -0,0 +1,5 @@ +{ + "schema_version": "v1", + "updated_at": "2026-03-27T00:00:00Z", + "terms": [] +} \ No newline at end of file diff --git a/docs/README.md b/docs/README.md index d9bb0bd..d4fc7e4 100644 --- a/docs/README.md +++ b/docs/README.md @@ -1,4 +1,4 @@ -# 文档索引 +# 文档索引 ## 当前目录结构 @@ -31,17 +31,19 @@ - 过滤层的输入输出、规则结构与当前实现 7. `docs/design/filter-rule-engine-usage.md` - 规则怎么写、怎么跑、结果怎么解读的使用说明 -8. `docs/design/markdown-sink-design.md` +8. `docs/design/daily-keyword-index-design.md` + - 日报级词元库与周期性词元清洗 skill 设计 +9. `docs/design/markdown-sink-design.md` - 第一版 Markdown sink 的输入输出、目录结构与落地方式 -9. `docs/openclaw/openclaw-daily-digest-refactor.md` +10. `docs/openclaw/openclaw-daily-digest-refactor.md` - 为什么要从单篇入库改成 OpenClaw 日报聚合链路 -10. `docs/openclaw/article-candidate-daily-digest-schema.md` +11. `docs/openclaw/article-candidate-daily-digest-schema.md` - `ArticleCandidateRecord`、`OpenClawCandidateInput` 与 `DailyDigest` 的正式设计 -11. `docs/design/source-schema-design.md` +12. `docs/design/source-schema-design.md` - `source -> item -> document` 的对象设计 -12. `docs/notes/reading-pipeline-design-notes.md` +13. `docs/notes/reading-pipeline-design-notes.md` - 更上层的阅读流方案与阶段划分 -13. `docs/design/summary-loop-explained.md` +14. `docs/design/summary-loop-explained.md` - 当前 LLM 摘要校验闭环的解释 ## 当前文档分层 @@ -71,6 +73,10 @@ - 第一版规则过滤引擎设计与落地位置 - `docs/design/filter-rule-engine-usage.md` - 规则配置、调用方式与结果解读 +- `docs/design/daily-keyword-index-design.md` + - 日报级词元库与周期性词元清洗 skill 设计 +- `scripts/apply_term_suggestions.py` + - 人工确认后将建议写回 watchlist / change log 的脚本入口(见 `README.md` 用法) - `docs/design/markdown-sink-design.md` - 第一版 Markdown sink 设计与落地位置 @@ -103,4 +109,4 @@ - `TODO.md` 记录任务优先级与下一步 - `outputs/README.md` 记录当前输出目录约定 - `docs/archive/content-extract-mcp-mvp-archive.md` 只当历史快照,不再作为最新事实来源 -- 新增阶段性进展,优先更新 `README.md`、`TODO.md`、`docs/current/context-reset-brief.md` +- 新增阶段性进展,优先更新 `README.md`、`TODO.md`、`docs/current/context-reset-brief.md` \ No newline at end of file diff --git a/docs/current/context-reset-brief.md b/docs/current/context-reset-brief.md index 6060764..66d5600 100644 --- a/docs/current/context-reset-brief.md +++ b/docs/current/context-reset-brief.md @@ -23,6 +23,10 @@ OpenClaw 应通过 MCP 工具 `run_freshrss_openclaw_pipeline` 调用这条链 - 已完成“仅在最终 payload 成功写盘后再标记已读”的语义 - 已完成 MCP 工具 `run_freshrss_openclaw_pipeline` - 已完成默认精简输出模式,减少中间文件 +- 已完成日报级 `keywords` 词元库与全局词频统计 +- 已完成 `keyword-cleanup-review` skill 骨架与 review bundle 脚本 +- 已完成低复杂治理层:`term_cleanup_policy` / `term_watchlist` / `term_change_log` +- 已完成采纳建议写回脚本 `scripts/apply_term_suggestions.py` ## 当前 MCP 工具 @@ -47,6 +51,10 @@ OpenClaw 应通过 MCP 工具 `run_freshrss_openclaw_pipeline` 调用这条链 - `src/summary_mcp/server.py` - FreshRSS 统一工作流 - `src/summary_mcp/workflows/freshrss_pipeline.py` +- 词元统计核心 + - `src/summary_mcp/core/keyword_index.py` +- 词元统计模型 + - `src/summary_mcp/models/keyword_index.py` - 摘要循环 - `src/summary_mcp/core/summary_loop.py` - 提取主流程 @@ -63,6 +71,18 @@ OpenClaw 应通过 MCP 工具 `run_freshrss_openclaw_pipeline` 调用这条链 - `src/summary_mcp/models/openclaw_delivery.py` - 生产脚本入口 - `scripts/run_freshrss_pipeline.py` +- 词元统计重建脚本 + - `scripts/build_keyword_index.py` +- 词元清洗 skill + - `skills/keyword-cleanup-review/SKILL.md` +- skill review bundle 脚本 + - `skills/keyword-cleanup-review/scripts/build_review_bundle.py` +- 采纳建议写回脚本 + - `scripts/apply_term_suggestions.py` +- 清洗治理配置 + - `configs/term_cleanup_policy.json` + - `configs/term_watchlist.json` + - `configs/term_change_log.json` - OpenClaw 交接说明 - `docs/openclaw/openclaw-handoff.md` @@ -74,6 +94,19 @@ OpenClaw 应通过 MCP 工具 `run_freshrss_openclaw_pipeline` 调用这条链 - `candidates/openclaw-delivery-payload.json` - `run-report.json` +同时会更新本地运行数据: + +- `data/term_index/daily/YYYY-MM-DD.json` +- `data/term_index/term_stats.json` + +如果需要词元清洗审阅输入,可额外生成: + +- `outputs/term_index/review/keyword-cleanup-bundle.json` + +如果需要在人工确认后把建议正式写入 watchlist / change log,可使用: + +- `scripts/apply_term_suggestions.py` + 如果需要排障,可开启: - `debug_artifacts=true` @@ -90,6 +123,10 @@ OpenClaw 应通过 MCP 工具 `run_freshrss_openclaw_pipeline` 调用这条链 - MCP 工具入口可直接触发完整链路 - 微信公众号样本可直接使用 RSS 提供的 `summary` 内容提取,不再回源抓网页 - 精简输出模式已实际跑通 +- 日报级词元统计已通过离线样例验证,确认别名、停用词、非 `drop` 过滤和 rerun 覆盖逻辑正常 +- `keyword-cleanup-review` skill 已通过 `quick_validate.py` 结构校验 +- review bundle 脚本已实际跑通 +- `apply_term_suggestions.py` 已通过 dry-run 与临时副本写回验证 ## 当前已知限制 @@ -98,6 +135,8 @@ OpenClaw 应通过 MCP 工具 `run_freshrss_openclaw_pipeline` 调用这条链 - 某些规则仍偏保守,部分内容可能落到 `review` - `paywall` 相关启发式仍可能误判中文文本 - Webhook / 主动投递到 OpenClaw 外部接口尚未实现,当前是由 OpenClaw 通过 MCP 主动调用 +- 词元清洗 skill 当前已支持“bundle 构建 -> 建议审阅 -> 人工确认写回 watchlist/change_log”,但尚未接入周期性调度 +- 当前词元统计仍以前置 `OpenClawDeliveryPayload` 作为日报前代理输入,真实 `DailyDigest` 接入后还需切换上游 ## 当前最建议的交接阅读顺序 @@ -105,8 +144,10 @@ OpenClaw 应通过 MCP 工具 `run_freshrss_openclaw_pipeline` 调用这条链 2. `docs/openclaw/openclaw-handoff.md` 3. `docs/openclaw/openclaw-candidate-input-field-spec.md` 4. `docs/openclaw/openclaw-delivery-payload-spec.md` -5. `TODO.md` +5. `docs/design/daily-keyword-index-design.md` +6. `skills/keyword-cleanup-review/SKILL.md` +7. `TODO.md` ## 一句话结论 -当前仓库已经从“提取 MCP 原型”演进到“可供 OpenClaw 调用的 FreshRSS -> OpenClaw payload 上游处理器”,可以开始交接,但后续仍建议继续补 webhook / delivery 接线与规则收敛。 +当前仓库已经从“提取 MCP 原型”演进到“可供 OpenClaw 调用的 FreshRSS -> OpenClaw payload 上游处理器”,并已补上第一阶段的日报级词元统计能力和词元清洗 skill 骨架;后续重点转向 skill 周期调度、知识库状态流转和 webhook 接线。 \ No newline at end of file diff --git a/docs/design/daily-keyword-index-design.md b/docs/design/daily-keyword-index-design.md new file mode 100644 index 0000000..696ccaa --- /dev/null +++ b/docs/design/daily-keyword-index-design.md @@ -0,0 +1,461 @@ +# 日报级词元库与词元清洗 Skill 设计 + +## 1. 设计目标 + +当前项目已经具备: + +`FreshRSS -> extraction -> LLM summary -> rule engine -> OpenClaw payload` + +下一阶段希望新增“词元库”能力,目标不是做全文级检索索引,而是解决两件事: + +1. 为后续规则配置提供稳定、轻量、可读的词元来源 +2. 为周期性的词元清洗、合并和兴趣词补充提供数据基础 + +因此这套设计的核心原则是: + +- 词元来源轻量化 +- 数据粒度日报化 +- 主链路程序化维护 +- 清洗治理由独立 skill 周期性执行 +- LLM 不直接修改规则或兴趣词配置 + +## 2. 为什么不用逐篇词元库 + +逐篇记录每篇文章的词元事件,虽然可追溯,但当前阶段成本过高,收益不足: + +- 生成文件会很多 +- 存储和调试负担更大 +- 后续真正调规则时,用户更关心“最近日报里反复出现什么词”,而不是“某一篇文章具体抽到了什么词” +- 当前目标是服务日报与规则配置,不是做文章级分析平台 + +所以本项目不采用: + +`article -> term event -> term store` + +而采用: + +`daily digest -> keyword aggregate -> daily term index -> global term stats` + +## 3. 为什么只保留 `keywords` + +当前 LLM 摘要结果里已经有两个候选字段: + +- `keywords` +- `topics` + +本设计只使用 `keywords` 进入词元库,不使用 `topics` 作为主来源。 + +原因: + +- `keywords` 更具体,适合后续规则配置 + - 例如 `Java`、`Go`、`Python`、`MCP`、`RAG`、`Kafka` +- `topics` 更宽泛,适合摘要展示,不适合作为精确规则命中基础 + - 例如“后端工程”“AI Agent”“前沿科技”过于宽泛 +- 只保留一个字段可以显著控制词元库规模 +- 当前项目的真实需求是“词元可配置”,不是“主题聚类” + +结论: + +- `keywords` 进入词元库 +- `topics` 继续保留在单篇摘要结果中,但不纳入词元统计主流程 + +## 4. 词元库的上游边界 + +词元库不直接读取所有候选文章,而只读取“最终进入日报”的内容。 + +也就是说,真正的词元来源是: + +- `DailyDigest` +- 或等价的“已被日报选中”的 candidate 集合 + +不纳入词元库的内容: + +- 被 `drop` 的内容 +- 仅在中间候选层出现、但未进入日报的内容 +- 调试产物中的临时摘要结果 + +这样做的好处是: + +- 词元库只反映真正进入日级产物的内容 +- 高频词更接近长期兴趣,而不是临时噪声 +- 数据量更可控 + +## 5. 总体链路 + +目标链路调整为: + +`DailyDigest -> keyword normalization -> daily term index -> global term stats -> cleaning skill review -> human confirm -> config update` + +职责拆分如下: + +### 5.1 程序负责 + +- 从日报中提取 `keywords` +- 归一化词元 +- 应用别名映射 +- 应用停用词过滤 +- 生成每日词频 +- 更新全局累计统计 + +### 5.2 Skill 负责 + +- 周期性读取词频结果 +- 识别重复词、近义词、大小写变体 +- 识别泛词、噪声词、低价值词 +- 建议哪些词应合并、停用、加入兴趣词配置 + +### 5.3 人工负责 + +- 审核 skill 输出的建议 +- 决定是否更新: + - `term_aliases` + - `term_stopwords` + - `filter_context.personal.json` + +## 6. 数据文件设计 + +建议新增以下文件: + +- `data/term_index/daily/YYYY-MM-DD.json` + - 某一天日报的词元聚合结果 +- `data/term_index/term_stats.json` + - 全局累计词元统计 +- `configs/term_aliases.json` + - 词元别名归一配置 +- `configs/term_stopwords.json` + - 词元停用词配置 +- `configs/term_cleanup_policy.json` + - 清洗阈值与治理策略配置 +- `configs/term_watchlist.json` + - 当前处于观察状态的词元列表 +- `configs/term_change_log.json` + - 已确认生效的词元治理变更记录 + +说明: + +- 不新增逐篇 `term_events.jsonl` +- 不新增文章级明细文件 +- 默认只保留“日报聚合结果 + 全局统计结果” + +## 7. 每日词元文件结构 + +文件路径示例: + +- `data/term_index/daily/2026-03-26.json` + +建议结构: + +```json +{ + "date": "2026-03-26", + "source": "daily_digest", + "digest_id": "digest-2026-03-26", + "generated_at": "2026-03-26T21:30:00+08:00", + "candidate_count": 8, + "terms": [ + { + "term": "AI Agent", + "normalized_term": "AI Agent", + "count": 4 + }, + { + "term": "MCP", + "normalized_term": "MCP", + "count": 3 + }, + { + "term": "RAG", + "normalized_term": "RAG", + "count": 2 + } + ] +} +``` + +字段说明: + +- `date` + - 日报日期 +- `source` + - 固定标记为 `daily_digest` +- `digest_id` + - 日报对象唯一标识 +- `generated_at` + - 该词元文件生成时间 +- `candidate_count` + - 当天日报包含的条目数 +- `terms[]` + - 当天词元聚合结果 + +其中单条 `terms[]` 只保留: + +- `term` +- `normalized_term` +- `count` + +不保留: + +- 逐篇文章来源列表 +- 逐条命中明细 +- 本地文件路径 + +## 8. 全局统计文件结构 + +文件路径: + +- `data/term_index/term_stats.json` + +建议结构: + +```json +{ + "schema_version": "v1", + "generated_at": "2026-03-26T21:30:00+08:00", + "terms": [ + { + "term": "AI Agent", + "total_count": 18, + "days_seen": 6, + "first_seen": "2026-03-20", + "last_seen": "2026-03-26" + }, + { + "term": "MCP", + "total_count": 12, + "days_seen": 5, + "first_seen": "2026-03-21", + "last_seen": "2026-03-26" + } + ] +} +``` + +字段说明: + +- `term` + - 最终归一后的词元 +- `total_count` + - 累计出现次数 +- `days_seen` + - 出现过的天数 +- `first_seen` + - 首次出现日期 +- `last_seen` + - 最近出现日期 + +当前阶段不额外记录: + +- 各 category 分桶统计 +- keep/review/drop 分桶统计 +- 文章级来源列表 + +原因是:日报级词元库的第一目标是轻量稳定,不是分析平台。 + +## 9. 标准化与归一规则 + +程序在写入日报词元前,应先做标准化。 + +### 9.1 基础标准化 + +- 去除首尾空白 +- 保留中英文大小写风格中的稳定写法 +- 去重 +- 过滤空字符串 + +### 9.2 别名映射 + +通过 `configs/term_aliases.json` 做归一。 + +示例: + +```json +{ + "Agent": "AI Agent", + "智能体": "AI Agent", + "Postgres": "PostgreSQL", + "Model Context Protocol": "MCP" +} +``` + +### 9.3 停用词过滤 + +通过 `configs/term_stopwords.json` 过滤过泛词和噪声词。 + +示例: + +```json +[ + "技术", + "系统", + "方案", + "实践", + "文章" +] +``` + +## 10. 为什么不让 LLM 直接维护词元库 + +LLM 可以帮助做清洗建议,但不适合直接维护主词元库。 + +原因: + +- 主词元库更新应该稳定、低成本、可复现 +- 词频统计属于纯程序逻辑,没必要消耗模型调用 +- 如果让 LLM 直接写词元库,会引入不稳定和难审计问题 + +因此主流程固定为: + +- LLM 只负责在摘要结果里输出 `keywords` +- 程序负责归一、聚合、统计 + +## 11. 词元清洗 Skill 设计 + +新增一个周期性清洗 skill,定位是“治理器”,不是“实时生产者”。 + +### 11.1 Skill 输入 + +建议输入: + +- `data/term_index/term_stats.json` +- 最近 N 天的 `data/term_index/daily/*.json` +- `configs/term_aliases.json` +- `configs/term_stopwords.json` +- `configs/filter_context.personal.json` + +### 11.2 Skill 输出 + +skill 不直接修改配置文件,而是生成建议文件,例如: + +- `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.md` +- `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json` + +低复杂治理层建议补充三类输入: + +- `term_cleanup_policy` + - 用来定义 watch 和 interest 的最小证据阈值 +- `term_watchlist` + - 用来记录“先观察、暂不升级”的词元 +- `term_change_log` + - 用来记录已经确认落地的配置变更,避免后续遗忘上下文 + +建议项包括: + +- 建议合并的别名词 +- 建议新增的停用词 +- 建议加入 `interest_keywords` 的候选词 +- 建议降权观察的热点词 + +### 11.3 Skill 允许做什么 + +- 发现重复词 +- 发现大小写变体 +- 发现中英文混用的近义词 +- 发现持续高频但尚未进入兴趣配置的词 +- 发现明显过泛的词 + +### 11.4 Skill 不允许做什么 + +- 直接改 `filter_rules.json` +- 直接改 `filter_context.personal.json` +- 直接覆盖 `term_stats.json` +- 在无人工确认的情况下自动生效 + +## 12. 建议的清洗建议文件结构 + +建议 JSON 文件结构如下: + +```json +{ + "date": "2026-03-26", + "based_on_days": 7, + "alias_suggestions": [ + { + "from": "Agent", + "to": "AI Agent", + "reason": "和现有高频词语义一致,建议归并。" + } + ], + "stopword_suggestions": [ + { + "term": "系统", + "reason": "出现频繁但语义过泛,难以作为规则命中词。" + } + ], + "interest_keyword_suggestions": [ + { + "term": "MCP", + "reason": "最近多日持续高频,且符合当前 AI Agent 学习方向。" + } + ] +} +``` + +## 13. 周期与触发方式 + +建议触发周期: + +- 词元统计更新:每天一次,跟随日报生成 +- skill 清洗:每周一次,或人工手动触发 + +推荐流程: + +1. 当天日报生成完成 +2. 程序更新 `daily/YYYY-MM-DD.json` +3. 程序更新 `term_stats.json` +4. 每周或人工触发一次词元清洗 skill +5. skill 输出建议 +6. 人工确认后再更新配置文件 + +## 14. 与规则引擎的关系 + +这套词元库设计不是规则引擎的替代品,而是规则配置的辅助层。 + +关系如下: + +- `keywords` + - 是词元库来源 +- `term_stats` + - 是观察与调优依据 +- `filter_context.personal.json` + - 是真正给规则引擎使用的兴趣词配置 +- `filter_rules.json` + - 是最终裁决逻辑 + +也就是说: + +`日报词元统计 -> 清洗建议 -> 人工确认 -> 更新 interest_keywords -> 规则引擎命中` + +而不是: + +`日报词元统计 -> 自动改规则` + +## 15. 分阶段落地建议 + +### Phase 1 + +先做最小可用版本: + +- 只读取日报中的 `keywords` +- 生成每日词元文件 +- 生成全局累计词频文件 +- 支持 `term_aliases` 和 `term_stopwords` + +### Phase 2 + +再补治理层: + +- 增加词元清洗 skill +- 输出建议文件 +- 人工确认后更新配置 + +### Phase 3 + +最后再考虑增强: + +- 增加趋势分析 +- 增加最近 7 天热点词视图 +- 增加“建议加入兴趣词”的自动排序 + +## 16. 一句话结论 + +这套设计选择“只统计日报中的 `keywords`,由程序维护轻量词元库,再由独立 skill 周期性做清洗建议”,目的是在控制数据规模的前提下,为规则配置和长期兴趣演化提供稳定、可审计、可扩展的基础设施。 diff --git a/outputs/README.md b/outputs/README.md index ae530ee..a4eecf9 100644 --- a/outputs/README.md +++ b/outputs/README.md @@ -15,6 +15,17 @@ 这是当前推荐的生产模式。 +## 词元统计运行数据 + +词元统计不放在 `outputs/` 下,而放在单独的运行数据目录: + +- `data/term_index/daily/YYYY-MM-DD.json` + - 某一天的日报级 `keywords` 聚合结果 +- `data/term_index/term_stats.json` + - 全局累计词频统计 + +这两类文件属于可重建的本地运行数据,不作为仓库长期跟踪产物。 + ## 调试扩展输出 当启用 `debug_artifacts` 时,才会额外写出这些中间文件: @@ -57,6 +68,7 @@ - 默认优先保留最终产物,不把所有中间文件都当成长期资产 - 调试文件只在需要时生成 - `reference/` 只放可复用样例,不放真实生产批次数据 +- `data/term_index/` 只保存本地词元统计运行数据,不纳入 git 跟踪 ## 当前建议 @@ -65,5 +77,7 @@ - `openclaw-delivery-payload.json` - `run-report.json` - `freshrss.raw.json` +- `data/term_index/daily/YYYY-MM-DD.json` +- `data/term_index/term_stats.json` -其余文件默认都应视为调试辅助产物。 +其中前 3 个是本次批处理产物,后 2 个是持续累积的词元统计数据。 diff --git a/scripts/apply_term_suggestions.py b/scripts/apply_term_suggestions.py new file mode 100644 index 0000000..da4dfe6 --- /dev/null +++ b/scripts/apply_term_suggestions.py @@ -0,0 +1,402 @@ +from __future__ import annotations + +import argparse +import json +from datetime import UTC, datetime +from pathlib import Path +from typing import Any + + +REPO_ROOT = Path(__file__).resolve().parents[1] +DEFAULT_ALIASES_PATH = REPO_ROOT / "configs" / "term_aliases.json" +DEFAULT_STOPWORDS_PATH = REPO_ROOT / "configs" / "term_stopwords.json" +DEFAULT_CONTEXT_PATH = REPO_ROOT / "configs" / "filter_context.personal.json" +DEFAULT_WATCHLIST_PATH = REPO_ROOT / "configs" / "term_watchlist.json" +DEFAULT_CHANGE_LOG_PATH = REPO_ROOT / "configs" / "term_change_log.json" + + +def _load_json(path: Path, default: Any) -> Any: + if not path.exists(): + return default + return json.loads(path.read_text(encoding="utf-8-sig")) + + +def _save_json(path: Path, payload: dict[str, Any] | list[Any]) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") + + +def _term_key(value: str) -> str: + return value.strip().casefold() + + +def _utc_now() -> str: + return datetime.now(tz=UTC).isoformat().replace("+00:00", "Z") + + +def _load_suggestions(path: Path) -> dict[str, Any]: + payload = _load_json(path, {}) + if not isinstance(payload, dict): + raise RuntimeError("Suggestions file must contain a JSON object.") + return payload + + +def _load_aliases(path: Path) -> dict[str, str]: + payload = _load_json(path, {}) + if not isinstance(payload, dict): + raise RuntimeError("Aliases file must contain a JSON object.") + result: dict[str, str] = {} + for key, value in payload.items(): + if isinstance(key, str) and isinstance(value, str) and key.strip() and value.strip(): + result[key] = value + return result + + +def _load_stopwords(path: Path) -> list[str]: + payload = _load_json(path, []) + if not isinstance(payload, list): + raise RuntimeError("Stopwords file must contain a JSON array.") + return [item for item in payload if isinstance(item, str) and item.strip()] + + +def _load_context(path: Path) -> dict[str, Any]: + payload = _load_json(path, {}) + if not isinstance(payload, dict): + raise RuntimeError("Context file must contain a JSON object.") + payload.setdefault("interest_keywords", []) + if not isinstance(payload["interest_keywords"], list): + raise RuntimeError("filter_context.personal.json interest_keywords must be a JSON array.") + payload["interest_keywords"] = [item for item in payload["interest_keywords"] if isinstance(item, str) and item.strip()] + return payload + + +def _load_watchlist(path: Path) -> dict[str, Any]: + payload = _load_json( + path, + { + "schema_version": "v1", + "updated_at": _utc_now(), + "terms": [], + }, + ) + if not isinstance(payload, dict): + raise RuntimeError("Watchlist file must contain a JSON object.") + payload.setdefault("schema_version", "v1") + payload.setdefault("updated_at", _utc_now()) + terms = payload.get("terms", []) + if not isinstance(terms, list): + raise RuntimeError("Watchlist terms must be a JSON array.") + payload["terms"] = [item for item in terms if isinstance(item, dict) and isinstance(item.get("term"), str)] + return payload + + +def _load_change_log(path: Path) -> dict[str, Any]: + payload = _load_json( + path, + { + "schema_version": "v1", + "entries": [], + }, + ) + if not isinstance(payload, dict): + raise RuntimeError("Change log file must contain a JSON object.") + payload.setdefault("schema_version", "v1") + entries = payload.get("entries", []) + if not isinstance(entries, list): + raise RuntimeError("Change log entries must be a JSON array.") + payload["entries"] = [item for item in entries if isinstance(item, dict)] + return payload + + +def _build_index(entries: list[dict[str, Any]], field: str) -> dict[str, dict[str, Any]]: + index: dict[str, dict[str, Any]] = {} + for entry in entries: + value = entry.get(field) + if isinstance(value, str) and value.strip(): + index[_term_key(value)] = entry + return index + + +def _append_change( + change_log: dict[str, Any], + *, + applied_at: str, + action: str, + term: str, + value: str | None, + reason: str, + suggestions_path: Path, + suggestion_date: str | None, + based_on_days: int | None, +) -> None: + entry: dict[str, Any] = { + "applied_at": applied_at, + "action": action, + "term": term, + "reason": reason, + "suggestions_path": str(suggestions_path), + } + if value is not None: + entry["value"] = value + if suggestion_date is not None: + entry["suggestion_date"] = suggestion_date + if based_on_days is not None: + entry["based_on_days"] = based_on_days + change_log["entries"].append(entry) + + +def _remove_from_watchlist( + watchlist: dict[str, Any], + *, + term: str, +) -> bool: + folded = _term_key(term) + original_len = len(watchlist["terms"]) + watchlist["terms"] = [item for item in watchlist["terms"] if _term_key(str(item.get("term", ""))) != folded] + return len(watchlist["terms"]) != original_len + + +def main() -> None: + parser = argparse.ArgumentParser(description="Apply accepted term cleanup suggestions into repo config files.") + parser.add_argument("--suggestions", type=Path, required=True, help="Suggestion JSON file") + parser.add_argument("--aliases", type=Path, default=DEFAULT_ALIASES_PATH, help="Term aliases JSON file") + parser.add_argument("--stopwords", type=Path, default=DEFAULT_STOPWORDS_PATH, help="Term stopwords JSON file") + parser.add_argument("--context", type=Path, default=DEFAULT_CONTEXT_PATH, help="Personal filter context JSON file") + parser.add_argument("--watchlist", type=Path, default=DEFAULT_WATCHLIST_PATH, help="Watchlist JSON file") + parser.add_argument("--change-log", type=Path, default=DEFAULT_CHANGE_LOG_PATH, help="Change log JSON file") + parser.add_argument("--accept-watch", nargs="*", default=[], help="Accept watch_terms by term name") + parser.add_argument( + "--accept-interest", nargs="*", default=[], help="Accept interest_keyword_suggestions by term name" + ) + parser.add_argument("--accept-stopword", nargs="*", default=[], help="Accept stopword_suggestions by term name") + parser.add_argument("--accept-alias", nargs="*", default=[], help="Accept alias_suggestions by source term") + parser.add_argument("--dry-run", action="store_true", help="Preview changes without writing files") + args = parser.parse_args() + + suggestions = _load_suggestions(args.suggestions) + aliases = _load_aliases(args.aliases) + stopwords = _load_stopwords(args.stopwords) + context = _load_context(args.context) + watchlist = _load_watchlist(args.watchlist) + change_log = _load_change_log(args.change_log) + + alias_suggestions = suggestions.get("alias_suggestions", []) + stopword_suggestions = suggestions.get("stopword_suggestions", []) + interest_suggestions = suggestions.get("interest_keyword_suggestions", []) + watch_suggestions = suggestions.get("watch_terms", []) + if not all(isinstance(bucket, list) for bucket in [alias_suggestions, stopword_suggestions, interest_suggestions, watch_suggestions]): + raise RuntimeError("Suggestion JSON buckets must all be arrays.") + + alias_index = _build_index([item for item in alias_suggestions if isinstance(item, dict)], "from") + stopword_index = _build_index([item for item in stopword_suggestions if isinstance(item, dict)], "term") + interest_index = _build_index([item for item in interest_suggestions if isinstance(item, dict)], "term") + watch_index = _build_index([item for item in watch_suggestions if isinstance(item, dict)], "term") + + watch_map = {_term_key(str(item.get("term", ""))): item for item in watchlist["terms"]} + stopword_set = {_term_key(item) for item in stopwords} + interest_set = {_term_key(item) for item in context["interest_keywords"]} + alias_source_set = {_term_key(key) for key in aliases} + + applied_at = _utc_now() + suggestion_date = suggestions.get("date") if isinstance(suggestions.get("date"), str) else None + based_on_days = suggestions.get("based_on_days") if isinstance(suggestions.get("based_on_days"), int) else None + + applied: list[dict[str, Any]] = [] + skipped: list[dict[str, Any]] = [] + + def skip(action: str, term: str, reason: str) -> None: + skipped.append({"action": action, "term": term, "reason": reason}) + + def applied_entry(action: str, term: str, reason: str, value: str | None = None) -> None: + item: dict[str, Any] = {"action": action, "term": term, "reason": reason} + if value is not None: + item["value"] = value + applied.append(item) + + for raw_term in args.accept_watch: + term = raw_term.strip() + suggestion = watch_index.get(_term_key(term)) + if suggestion is None: + skip("add_watch_term", term, "Term not found in watch_terms suggestions.") + continue + if _term_key(term) in watch_map: + skip("add_watch_term", term, "Term already exists in watchlist.") + continue + reason = str(suggestion.get("reason") or "Accepted from watch_terms suggestion.") + watch_item = { + "term": str(suggestion.get("term") or term), + "added_at": applied_at, + "source": str(args.suggestions), + "reason": reason, + "status": "watching", + } + watchlist["terms"].append(watch_item) + watch_map[_term_key(watch_item["term"])] = watch_item + _append_change( + change_log, + applied_at=applied_at, + action="add_watch_term", + term=watch_item["term"], + value=None, + reason=reason, + suggestions_path=args.suggestions, + suggestion_date=suggestion_date, + based_on_days=based_on_days, + ) + applied_entry("add_watch_term", watch_item["term"], reason) + + for raw_term in args.accept_interest: + term = raw_term.strip() + suggestion = interest_index.get(_term_key(term)) + if suggestion is None: + skip("add_interest_keyword", term, "Term not found in interest_keyword_suggestions.") + continue + resolved_term = str(suggestion.get("term") or term) + if _term_key(resolved_term) in interest_set: + skip("add_interest_keyword", resolved_term, "Term already exists in interest_keywords.") + continue + reason = str(suggestion.get("reason") or "Accepted from interest_keyword_suggestions.") + context["interest_keywords"].append(resolved_term) + interest_set.add(_term_key(resolved_term)) + _append_change( + change_log, + applied_at=applied_at, + action="add_interest_keyword", + term=resolved_term, + value=None, + reason=reason, + suggestions_path=args.suggestions, + suggestion_date=suggestion_date, + based_on_days=based_on_days, + ) + if _remove_from_watchlist(watchlist, term=resolved_term): + _append_change( + change_log, + applied_at=applied_at, + action="remove_watch_term", + term=resolved_term, + value="promoted_to_interest_keyword", + reason="Removed from watchlist after promotion into interest_keywords.", + suggestions_path=args.suggestions, + suggestion_date=suggestion_date, + based_on_days=based_on_days, + ) + applied_entry("add_interest_keyword", resolved_term, reason) + + for raw_term in args.accept_stopword: + term = raw_term.strip() + suggestion = stopword_index.get(_term_key(term)) + if suggestion is None: + skip("add_stopword", term, "Term not found in stopword_suggestions.") + continue + resolved_term = str(suggestion.get("term") or term) + if _term_key(resolved_term) in stopword_set: + skip("add_stopword", resolved_term, "Term already exists in stopwords.") + continue + reason = str(suggestion.get("reason") or "Accepted from stopword_suggestions.") + stopwords.append(resolved_term) + stopword_set.add(_term_key(resolved_term)) + _append_change( + change_log, + applied_at=applied_at, + action="add_stopword", + term=resolved_term, + value=None, + reason=reason, + suggestions_path=args.suggestions, + suggestion_date=suggestion_date, + based_on_days=based_on_days, + ) + if _remove_from_watchlist(watchlist, term=resolved_term): + _append_change( + change_log, + applied_at=applied_at, + action="remove_watch_term", + term=resolved_term, + value="promoted_to_stopword", + reason="Removed from watchlist after being added to stopwords.", + suggestions_path=args.suggestions, + suggestion_date=suggestion_date, + based_on_days=based_on_days, + ) + applied_entry("add_stopword", resolved_term, reason) + + for raw_term in args.accept_alias: + term = raw_term.strip() + suggestion = alias_index.get(_term_key(term)) + if suggestion is None: + skip("add_alias", term, "Source term not found in alias_suggestions.") + continue + source_term = str(suggestion.get("from") or term) + target_term = str(suggestion.get("to") or "").strip() + if not target_term: + skip("add_alias", source_term, "Alias suggestion target is empty.") + continue + existing_target = aliases.get(source_term) + if existing_target == target_term: + skip("add_alias", source_term, "Alias already exists with the same target.") + continue + if _term_key(source_term) in alias_source_set and existing_target != target_term: + skip("add_alias", source_term, f"Alias source already exists with a different target: {existing_target}") + continue + reason = str(suggestion.get("reason") or "Accepted from alias_suggestions.") + aliases[source_term] = target_term + alias_source_set.add(_term_key(source_term)) + _append_change( + change_log, + applied_at=applied_at, + action="add_alias", + term=source_term, + value=target_term, + reason=reason, + suggestions_path=args.suggestions, + suggestion_date=suggestion_date, + based_on_days=based_on_days, + ) + if _remove_from_watchlist(watchlist, term=source_term): + _append_change( + change_log, + applied_at=applied_at, + action="remove_watch_term", + term=source_term, + value="resolved_as_alias_source", + reason="Removed from watchlist after alias mapping was accepted.", + suggestions_path=args.suggestions, + suggestion_date=suggestion_date, + based_on_days=based_on_days, + ) + applied_entry("add_alias", source_term, reason, value=target_term) + + watchlist["terms"] = sorted(watchlist["terms"], key=lambda item: (_term_key(str(item.get("term", ""))), str(item.get("term", "")))) + watchlist["updated_at"] = applied_at + stopwords = sorted(stopwords, key=lambda item: (_term_key(item), item)) + context["interest_keywords"] = sorted(context["interest_keywords"], key=lambda item: (_term_key(item), item)) + + summary = { + "dry_run": args.dry_run, + "suggestions": str(args.suggestions), + "applied_count": len(applied), + "skipped_count": len(skipped), + "applied": applied, + "skipped": skipped, + "output_paths": { + "aliases": str(args.aliases), + "stopwords": str(args.stopwords), + "context": str(args.context), + "watchlist": str(args.watchlist), + "change_log": str(args.change_log), + }, + } + + if not args.dry_run: + _save_json(args.aliases, aliases) + _save_json(args.stopwords, stopwords) + _save_json(args.context, context) + _save_json(args.watchlist, watchlist) + _save_json(args.change_log, change_log) + + print(json.dumps(summary, ensure_ascii=False, indent=2)) + + +if __name__ == "__main__": + main() \ No newline at end of file diff --git a/scripts/build_keyword_index.py b/scripts/build_keyword_index.py new file mode 100644 index 0000000..278fc05 --- /dev/null +++ b/scripts/build_keyword_index.py @@ -0,0 +1,76 @@ +from __future__ import annotations + +import argparse +import sys +from pathlib import Path + + +REPO_ROOT = Path(__file__).resolve().parents[1] +SRC_ROOT = REPO_ROOT / "src" + +if str(SRC_ROOT) not in sys.path: + sys.path.insert(0, str(SRC_ROOT)) + +from summary_mcp.core.keyword_index import persist_keyword_indexes +from summary_mcp.models.openclaw_delivery import OpenClawDeliveryPayload + + +def _load_payload(path: Path) -> OpenClawDeliveryPayload: + return OpenClawDeliveryPayload.model_validate_json(path.read_text(encoding="utf-8-sig")) + + +def main() -> None: + parser = argparse.ArgumentParser( + description="Build the daily keyword index and global keyword stats from an OpenClaw delivery payload." + ) + parser.add_argument( + "--input", + type=Path, + required=True, + help="OpenClaw delivery payload JSON file", + ) + parser.add_argument( + "--daily-dir", + type=Path, + default=REPO_ROOT / "data" / "term_index" / "daily", + help="Directory for daily keyword index files", + ) + parser.add_argument( + "--stats-output", + type=Path, + default=REPO_ROOT / "data" / "term_index" / "term_stats.json", + help="Global keyword stats output path", + ) + parser.add_argument( + "--aliases", + type=Path, + default=REPO_ROOT / "configs" / "term_aliases.json", + help="Keyword alias config JSON file", + ) + parser.add_argument( + "--stopwords", + type=Path, + default=REPO_ROOT / "configs" / "term_stopwords.json", + help="Keyword stopwords config JSON file", + ) + args = parser.parse_args() + + payload = _load_payload(args.input) + result = persist_keyword_indexes( + payload.candidates, + for_date=payload.date, + digest_id=payload.run_id, + source="openclaw_delivery_payload", + daily_dir=args.daily_dir, + stats_path=args.stats_output, + aliases_path=args.aliases, + stopwords_path=args.stopwords, + ) + + print(f"Saved daily keyword index to {result['daily_output']}") + print(f"Saved keyword stats to {result['stats_output']}") + print(f"Indexed {result['candidate_count']} candidates and {result['term_count']} keywords") + + +if __name__ == "__main__": + main() diff --git a/scripts/run_freshrss_pipeline.py b/scripts/run_freshrss_pipeline.py index 763c2ad..b588a61 100644 --- a/scripts/run_freshrss_pipeline.py +++ b/scripts/run_freshrss_pipeline.py @@ -88,6 +88,10 @@ def main() -> None: print(f"Saved FreshRSS pipeline run to {result['output_dir']}") print(f"Pulled {result['pulled_count']} items, delivered {result['delivered_count']} candidates") + print( + "Saved keyword index to " + f"{result['keyword_index']['daily_output']} and {result['keyword_index']['stats_output']}" + ) if args.mark_read: print(f"Marked {result['marked_read_count']} FreshRSS entries as read") diff --git a/skills/keyword-cleanup-review/SKILL.md b/skills/keyword-cleanup-review/SKILL.md new file mode 100644 index 0000000..d084ad1 --- /dev/null +++ b/skills/keyword-cleanup-review/SKILL.md @@ -0,0 +1,102 @@ +--- +name: keyword-cleanup-review +description: Review and curate this repository's daily keyword index and frequency stats. Use when the user wants to inspect `data/term_index/term_stats.json`, recent `data/term_index/daily/*.json`, `configs/term_aliases.json`, `configs/term_stopwords.json`, or `configs/filter_context.personal.json` to propose alias merges, stopwords, watch terms, or `interest_keywords` updates without directly modifying configs. +--- + +# Keyword Cleanup Review + +Use this skill to turn the repository's keyword statistics into reviewable cleanup suggestions. + +## Workflow + +1. Build a compact review bundle: + +```bash +python skills/keyword-cleanup-review/scripts/build_review_bundle.py +``` + +Optional knobs: + +- `--days 7` +- `--top 50` +- `--output outputs/term_index/review/keyword-cleanup-bundle.json` + +2. Read the generated bundle and the suggestion schema: + +- `outputs/term_index/review/keyword-cleanup-bundle.json` +- `skills/keyword-cleanup-review/references/suggestion-schema.md` + +3. Produce two outputs: + +- A short Markdown review for humans +- A JSON suggestion file matching the schema + +4. Keep the boundary strict: + +- Suggest changes to `configs/term_aliases.json` +- Suggest changes to `configs/term_stopwords.json` +- Suggest additions to `configs/filter_context.personal.json` +- Do not directly edit these files unless the user explicitly asks +- Do not suggest direct edits to `configs/filter_rules.json` unless the user asks for rule logic changes + +## Review Heuristics + +Prioritize these decisions: + +- Alias suggestion + - Same concept with different naming, casing, abbreviation, or Chinese/English variants +- Stopword suggestion + - Too generic, too broad, or too noisy to help filtering +- Interest keyword suggestion + - High-frequency and aligned with the user's backend engineering, AI-agent, and frontier-tech focus +- Watch term + - Recent and potentially important, but evidence is still weak + +Prefer conservative suggestions. If confidence is low, put the term into `watch_terms`. + +## Inputs + +Primary inputs: + +- `data/term_index/term_stats.json` +- `data/term_index/daily/*.json` +- `configs/term_aliases.json` +- `configs/term_stopwords.json` +- `configs/filter_context.personal.json` +- `configs/term_cleanup_policy.json` +- `configs/term_watchlist.json` +- `configs/term_change_log.json` + +The bundled script already compacts these into a single review bundle. + +## Output Expectations + +The Markdown output should: + +- Summarize the current state briefly +- List the top terms worth acting on +- Separate alias, stopword, interest-keyword, and watch-term recommendations +- Explain reasoning in short, concrete sentences + +The JSON output should follow: + +- `references/suggestion-schema.md` + +## Repository Notes + +Current repository behavior: + +- Keyword stats are program-maintained, not LLM-maintained +- Stats are built from `keywords`, not `topics` +- Stats only include non-`drop` candidates +- `data/term_index/term_stats.json` is rebuilt from daily files, so reruns overwrite the same day instead of double-counting +- cleanup policy, watchlist, and change log are repository-managed governance inputs and should be respected during review + +Keep suggestions aligned with that design. + +## Resources + +- Script: + - `scripts/build_review_bundle.py` +- Reference: + - `references/suggestion-schema.md` diff --git a/skills/keyword-cleanup-review/agents/openai.yaml b/skills/keyword-cleanup-review/agents/openai.yaml new file mode 100644 index 0000000..5da7df9 --- /dev/null +++ b/skills/keyword-cleanup-review/agents/openai.yaml @@ -0,0 +1,3 @@ +display_name: Keyword Cleanup Review +short_description: Review keyword stats and suggest cleanup updates. +default_prompt: Review the repository keyword index, identify duplicate or noisy terms, and produce alias, stopword, watch-term, and interest-keyword suggestions without modifying configs directly. diff --git a/skills/keyword-cleanup-review/references/suggestion-schema.md b/skills/keyword-cleanup-review/references/suggestion-schema.md new file mode 100644 index 0000000..4d49086 --- /dev/null +++ b/skills/keyword-cleanup-review/references/suggestion-schema.md @@ -0,0 +1,47 @@ +# Suggestion Schema + +When this skill produces review output, prefer two files: + +- Markdown summary for humans +- JSON suggestions for deterministic follow-up edits + +Recommended JSON shape: + +```json +{ + "date": "2026-03-27", + "based_on_days": 7, + "alias_suggestions": [ + { + "from": "Agent", + "to": "AI Agent", + "reason": "High overlap with existing repository terminology." + } + ], + "stopword_suggestions": [ + { + "term": "??", + "reason": "Too generic to be useful for filtering." + } + ], + "interest_keyword_suggestions": [ + { + "term": "MCP", + "reason": "Repeated high-frequency term aligned with current learning focus." + } + ], + "watch_terms": [ + { + "term": "OpenClaw", + "reason": "Trending recently but not enough history yet." + } + ] +} +``` + +Rules: + +- Only output suggestions, never claim they are already applied. +- Avoid suggesting changes that would directly edit `filter_rules.json`. +- Prefer small, reviewable batches over large refactors. +- If evidence is weak, put the term in `watch_terms` instead of alias or stopword suggestions. diff --git a/skills/keyword-cleanup-review/references/trigger-notes.md b/skills/keyword-cleanup-review/references/trigger-notes.md new file mode 100644 index 0000000..58e3fe1 --- /dev/null +++ b/skills/keyword-cleanup-review/references/trigger-notes.md @@ -0,0 +1 @@ +Provide keyword cleanup guidance for this repository's daily keyword index. Use this skill when the user wants to review, merge, clean, or curate `data/term_index/term_stats.json`, recent `data/term_index/daily/*.json`, `configs/term_aliases.json`, `configs/term_stopwords.json`, or `configs/filter_context.personal.json`, especially to propose alias merges, stopwords, or `interest_keywords` updates without directly modifying configs. diff --git a/skills/keyword-cleanup-review/scripts/build_review_bundle.py b/skills/keyword-cleanup-review/scripts/build_review_bundle.py new file mode 100644 index 0000000..90a06f3 --- /dev/null +++ b/skills/keyword-cleanup-review/scripts/build_review_bundle.py @@ -0,0 +1,340 @@ +from __future__ import annotations + +import argparse +import json +from collections import defaultdict +from datetime import UTC, datetime +from pathlib import Path +from typing import Any + + +DEFAULT_POLICY: dict[str, Any] = { + "schema_version": "v1", + "interest_keyword_review": { + "min_total_count": 3, + "min_days_seen": 2, + }, + "watch_term_review": { + "min_total_count": 1, + "min_days_seen": 1, + "max_total_count": 2, + "max_days_seen": 2, + }, + "alias_review": { + "min_total_count": 2, + "min_days_seen": 2, + }, + "stopword_review": { + "max_total_count": 2, + "max_days_seen": 2, + }, +} + + +def _load_json(path: Path) -> Any: + return json.loads(path.read_text(encoding="utf-8-sig")) + + +def _save_json(path: Path, payload: dict[str, Any]) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8") + + +def _load_mapping(path: Path) -> dict[str, str]: + if not path.exists(): + return {} + payload = _load_json(path) + return payload if isinstance(payload, dict) else {} + + +def _load_list(path: Path) -> list[str]: + if not path.exists(): + return [] + payload = _load_json(path) + if isinstance(payload, list): + return [item for item in payload if isinstance(item, str)] + return [] + + +def _load_interest_keywords(path: Path) -> list[str]: + if not path.exists(): + return [] + payload = _load_json(path) + if not isinstance(payload, dict): + return [] + values = payload.get("interest_keywords") + if not isinstance(values, list): + return [] + return [item for item in values if isinstance(item, str)] + + +def _casefold_set(values: list[str]) -> set[str]: + return {value.strip().casefold() for value in values if value.strip()} + + +def _load_policy(path: Path) -> dict[str, Any]: + if not path.exists(): + return DEFAULT_POLICY.copy() + payload = _load_json(path) + return payload if isinstance(payload, dict) else DEFAULT_POLICY.copy() + + +def _load_watchlist(path: Path) -> list[dict[str, Any]]: + if not path.exists(): + return [] + payload = _load_json(path) + if not isinstance(payload, dict): + return [] + terms = payload.get("terms") + if not isinstance(terms, list): + return [] + return [item for item in terms if isinstance(item, dict) and isinstance(item.get("term"), str)] + + +def _load_change_log(path: Path) -> list[dict[str, Any]]: + if not path.exists(): + return [] + payload = _load_json(path) + if not isinstance(payload, dict): + return [] + entries = payload.get("entries") + if not isinstance(entries, list): + return [] + return [item for item in entries if isinstance(item, dict)] + + +def _meets_min_thresholds(item: dict[str, Any], thresholds: dict[str, Any]) -> bool: + min_total_count = int(thresholds.get("min_total_count", 1)) + min_days_seen = int(thresholds.get("min_days_seen", 1)) + total_count = int(item.get("total_count") or 0) + days_seen = int(item.get("days_seen") or 0) + return total_count >= min_total_count and days_seen >= min_days_seen + + +def _within_watch_thresholds(item: dict[str, Any], thresholds: dict[str, Any]) -> bool: + min_total_count = int(thresholds.get("min_total_count", 1)) + min_days_seen = int(thresholds.get("min_days_seen", 1)) + max_total_count = int(thresholds.get("max_total_count", 999999)) + max_days_seen = int(thresholds.get("max_days_seen", 999999)) + total_count = int(item.get("total_count") or 0) + days_seen = int(item.get("days_seen") or 0) + return ( + total_count >= min_total_count + and days_seen >= min_days_seen + and total_count <= max_total_count + and days_seen <= max_days_seen + ) + + +def main() -> None: + parser = argparse.ArgumentParser( + description="Build a compact review bundle for the keyword-cleanup-review skill." + ) + repo_root = Path(__file__).resolve().parents[3] + parser.add_argument( + "--stats", + type=Path, + default=repo_root / "data" / "term_index" / "term_stats.json", + help="Global keyword stats file", + ) + parser.add_argument( + "--daily-dir", + type=Path, + default=repo_root / "data" / "term_index" / "daily", + help="Directory containing daily keyword index JSON files", + ) + parser.add_argument( + "--aliases", + type=Path, + default=repo_root / "configs" / "term_aliases.json", + help="Term aliases JSON file", + ) + parser.add_argument( + "--stopwords", + type=Path, + default=repo_root / "configs" / "term_stopwords.json", + help="Term stopwords JSON file", + ) + parser.add_argument( + "--context", + type=Path, + default=repo_root / "configs" / "filter_context.personal.json", + help="Personal filter context JSON file", + ) + parser.add_argument( + "--policy", + type=Path, + default=repo_root / "configs" / "term_cleanup_policy.json", + help="Keyword cleanup policy JSON file", + ) + parser.add_argument( + "--watchlist", + type=Path, + default=repo_root / "configs" / "term_watchlist.json", + help="Tracked watch terms JSON file", + ) + parser.add_argument( + "--change-log", + type=Path, + default=repo_root / "configs" / "term_change_log.json", + help="Applied keyword cleanup change log JSON file", + ) + parser.add_argument("--days", type=int, default=7, help="How many recent daily files to include") + parser.add_argument("--top", type=int, default=50, help="How many top global terms to include") + parser.add_argument( + "--output", + type=Path, + default=repo_root / "outputs" / "term_index" / "review" / "keyword-cleanup-bundle.json", + help="Output JSON file", + ) + args = parser.parse_args() + + stats_payload = _load_json(args.stats) if args.stats.exists() else {"terms": []} + stats_terms = stats_payload.get("terms") if isinstance(stats_payload, dict) else [] + if not isinstance(stats_terms, list): + stats_terms = [] + + aliases = _load_mapping(args.aliases) + stopwords = _load_list(args.stopwords) + interest_keywords = _load_interest_keywords(args.context) + policy = _load_policy(args.policy) + watchlist = _load_watchlist(args.watchlist) + change_log = _load_change_log(args.change_log) + + recent_daily_paths = sorted(args.daily_dir.glob("*.json"))[-args.days :] if args.daily_dir.exists() else [] + recent_daily: list[dict[str, Any]] = [] + recent_counter: defaultdict[str, int] = defaultdict(int) + for path in recent_daily_paths: + payload = _load_json(path) + if not isinstance(payload, dict): + continue + terms = payload.get("terms") + if not isinstance(terms, list): + terms = [] + compact_terms = [] + for term in terms: + if not isinstance(term, dict): + continue + normalized = term.get("normalized_term") + count = term.get("count") + if not isinstance(normalized, str) or not isinstance(count, int): + continue + compact_terms.append({"term": normalized, "count": count}) + recent_counter[normalized] += count + recent_daily.append( + { + "date": payload.get("date"), + "candidate_count": payload.get("candidate_count"), + "top_terms": compact_terms[:20], + } + ) + + interest_set = _casefold_set(interest_keywords) + stopword_set = _casefold_set(stopwords) + alias_keys = _casefold_set(list(aliases.keys())) + alias_values = _casefold_set(list(aliases.values())) + watch_set = _casefold_set([str(item.get("term", "")) for item in watchlist]) + + top_global_terms = [] + for item in stats_terms[: args.top]: + if not isinstance(item, dict): + continue + term = item.get("term") + if not isinstance(term, str): + continue + folded = term.strip().casefold() + top_global_terms.append( + { + "term": term, + "total_count": item.get("total_count"), + "days_seen": item.get("days_seen"), + "first_seen": item.get("first_seen"), + "last_seen": item.get("last_seen"), + "in_interest_keywords": folded in interest_set, + "is_stopword": folded in stopword_set, + "is_alias_source": folded in alias_keys, + "is_alias_target": folded in alias_values, + "in_watchlist": folded in watch_set, + "recent_count": recent_counter.get(term, 0), + } + ) + + uncovered_terms = [ + item for item in top_global_terms if not item["in_interest_keywords"] and not item["is_stopword"] + ][:20] + interest_thresholds = policy.get("interest_keyword_review") if isinstance(policy, dict) else {} + watch_thresholds = policy.get("watch_term_review") if isinstance(policy, dict) else {} + interest_review_candidates = [ + { + "term": item["term"], + "total_count": item["total_count"], + "days_seen": item["days_seen"], + "reason": ( + "Meets the configured interest-keyword review threshold and is not yet covered " + "by interest keywords or stopwords." + ), + } + for item in uncovered_terms + if not item["in_watchlist"] and _meets_min_thresholds(item, interest_thresholds) + ][:20] + watch_review_candidates = [ + { + "term": item["term"], + "total_count": item["total_count"], + "days_seen": item["days_seen"], + "reason": ( + "Falls into the configured watch-term review range and should be observed " + "before promotion into interest keywords." + ), + } + for item in uncovered_terms + if not item["in_watchlist"] + and not _meets_min_thresholds(item, interest_thresholds) + and _within_watch_thresholds(item, watch_thresholds) + ][:20] + recent_hot_terms = sorted( + ({"term": term, "recent_count": count} for term, count in recent_counter.items()), + key=lambda item: (-item["recent_count"], item["term"].casefold(), item["term"]), + )[:20] + + bundle = { + "generated_at": datetime.now(tz=UTC).isoformat(), + "days": args.days, + "top": args.top, + "sources": { + "stats": str(args.stats), + "daily_dir": str(args.daily_dir), + "aliases": str(args.aliases), + "stopwords": str(args.stopwords), + "context": str(args.context), + "policy": str(args.policy), + "watchlist": str(args.watchlist), + "change_log": str(args.change_log), + }, + "policy": policy, + "current_config": { + "alias_count": len(aliases), + "stopword_count": len(stopwords), + "interest_keyword_count": len(interest_keywords), + "watch_term_count": len(watchlist), + "aliases": aliases, + "stopwords": stopwords, + "interest_keywords": interest_keywords, + "watchlist": watchlist, + }, + "change_log_tail": change_log[-10:], + "top_global_terms": top_global_terms, + "recent_daily": recent_daily, + "recent_hot_terms": recent_hot_terms, + "uncovered_terms": uncovered_terms, + "governance_hints": { + "interest_review_candidates": interest_review_candidates, + "watch_review_candidates": watch_review_candidates, + }, + } + _save_json(args.output, bundle) + print(f"Saved keyword cleanup bundle to {args.output}") + + +if __name__ == "__main__": + main() \ No newline at end of file diff --git a/src/summary_mcp/core/keyword_index.py b/src/summary_mcp/core/keyword_index.py new file mode 100644 index 0000000..cfce533 --- /dev/null +++ b/src/summary_mcp/core/keyword_index.py @@ -0,0 +1,228 @@ +from __future__ import annotations + +import json +from collections import Counter +from datetime import UTC, date, datetime +from pathlib import Path +from typing import Iterable + +from summary_mcp.models.article_candidate import OpenClawCandidateInput +from summary_mcp.models.keyword_index import ( + DailyKeywordIndex, + DailyKeywordTerm, + KeywordStat, + KeywordStatsIndex, +) + + +REPO_ROOT = Path(__file__).resolve().parents[3] +DATA_ROOT = REPO_ROOT / "data" / "term_index" +DEFAULT_DAILY_DIR = DATA_ROOT / "daily" +DEFAULT_STATS_PATH = DATA_ROOT / "term_stats.json" +DEFAULT_ALIASES_PATH = REPO_ROOT / "configs" / "term_aliases.json" +DEFAULT_STOPWORDS_PATH = REPO_ROOT / "configs" / "term_stopwords.json" +DEFAULT_INCLUDE_DECISIONS = {"keep", "review"} + + +def _load_json(path: Path) -> object: + return json.loads(path.read_text(encoding="utf-8-sig")) + + +def _save_json(path: Path, payload: dict | list) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8") + + +def _term_key(value: str) -> str: + return value.strip().casefold() + + +def load_term_aliases(path: Path | None = None) -> dict[str, str]: + aliases_path = path or DEFAULT_ALIASES_PATH + if not aliases_path.exists(): + return {} + + payload = _load_json(aliases_path) + if not isinstance(payload, dict): + raise RuntimeError("Term aliases file must contain a JSON object.") + + aliases: dict[str, str] = {} + for raw_key, raw_value in payload.items(): + if not isinstance(raw_key, str) or not isinstance(raw_value, str): + continue + normalized_key = _term_key(raw_key) + normalized_value = raw_value.strip() + if not normalized_key or not normalized_value: + continue + aliases[normalized_key] = normalized_value + return aliases + + +def load_term_stopwords(path: Path | None = None) -> set[str]: + stopwords_path = path or DEFAULT_STOPWORDS_PATH + if not stopwords_path.exists(): + return set() + + payload = _load_json(stopwords_path) + if not isinstance(payload, list): + raise RuntimeError("Term stopwords file must contain a JSON array.") + + values: set[str] = set() + for item in payload: + if not isinstance(item, str): + continue + normalized = _term_key(item) + if normalized: + values.add(normalized) + return values + + +def normalize_keyword( + keyword: str, + *, + aliases: dict[str, str], + stopwords: set[str], +) -> str | None: + raw_value = keyword.strip() + if not raw_value: + return None + + aliased_value = aliases.get(_term_key(raw_value), raw_value).strip() + if not aliased_value: + return None + if _term_key(aliased_value) in stopwords: + return None + return aliased_value + + +def build_daily_keyword_index( + candidates: Iterable[OpenClawCandidateInput], + *, + for_date: date, + digest_id: str, + source: str = "openclaw_delivery_payload", + include_decisions: set[str] | None = None, + aliases: dict[str, str] | None = None, + stopwords: set[str] | None = None, +) -> DailyKeywordIndex: + allowed_decisions = include_decisions or DEFAULT_INCLUDE_DECISIONS + resolved_aliases = aliases or {} + resolved_stopwords = stopwords or set() + + term_counter: Counter[str] = Counter() + candidate_count = 0 + + for candidate in candidates: + if candidate.selection_decision not in allowed_decisions: + continue + + candidate_count += 1 + seen_for_candidate: set[str] = set() + for keyword in candidate.keywords: + normalized = normalize_keyword( + keyword, + aliases=resolved_aliases, + stopwords=resolved_stopwords, + ) + if normalized is None or normalized in seen_for_candidate: + continue + seen_for_candidate.add(normalized) + term_counter[normalized] += 1 + + terms = [ + DailyKeywordTerm(term=term, normalized_term=term, count=count) + for term, count in sorted(term_counter.items(), key=lambda item: (-item[1], item[0].casefold(), item[0])) + ] + + return DailyKeywordIndex( + schema_version="v1", + date=for_date, + source=source, + digest_id=digest_id, + generated_at=datetime.now(tz=UTC), + candidate_count=candidate_count, + terms=terms, + ) + + +def read_daily_keyword_index(path: Path) -> DailyKeywordIndex: + return DailyKeywordIndex.model_validate(_load_json(path)) + + +def rebuild_keyword_stats(*, daily_dir: Path = DEFAULT_DAILY_DIR) -> KeywordStatsIndex: + term_totals: dict[str, int] = {} + first_seen: dict[str, date] = {} + last_seen: dict[str, date] = {} + days_seen: dict[str, int] = {} + + if daily_dir.exists(): + for path in sorted(daily_dir.glob("*.json")): + daily_index = read_daily_keyword_index(path) + seen_today: set[str] = set() + for term in daily_index.terms: + normalized = term.normalized_term + term_totals[normalized] = term_totals.get(normalized, 0) + term.count + if normalized not in first_seen or daily_index.date < first_seen[normalized]: + first_seen[normalized] = daily_index.date + if normalized not in last_seen or daily_index.date > last_seen[normalized]: + last_seen[normalized] = daily_index.date + if normalized not in seen_today: + days_seen[normalized] = days_seen.get(normalized, 0) + 1 + seen_today.add(normalized) + + terms = [ + KeywordStat( + term=term, + total_count=term_totals[term], + days_seen=days_seen[term], + first_seen=first_seen[term], + last_seen=last_seen[term], + ) + for term in sorted(term_totals, key=lambda item: (-term_totals[item], item.casefold(), item)) + ] + + return KeywordStatsIndex( + schema_version="v1", + generated_at=datetime.now(tz=UTC), + terms=terms, + ) + + +def persist_keyword_indexes( + candidates: Iterable[OpenClawCandidateInput], + *, + for_date: date, + digest_id: str, + source: str = "openclaw_delivery_payload", + daily_dir: Path = DEFAULT_DAILY_DIR, + stats_path: Path = DEFAULT_STATS_PATH, + aliases_path: Path | None = None, + stopwords_path: Path | None = None, + include_decisions: set[str] | None = None, +) -> dict[str, object]: + aliases = load_term_aliases(aliases_path) + stopwords = load_term_stopwords(stopwords_path) + daily_index = build_daily_keyword_index( + candidates, + for_date=for_date, + digest_id=digest_id, + source=source, + include_decisions=include_decisions, + aliases=aliases, + stopwords=stopwords, + ) + + daily_path = daily_dir / f"{for_date.isoformat()}.json" + _save_json(daily_path, daily_index.model_dump(mode="json")) + + stats_index = rebuild_keyword_stats(daily_dir=daily_dir) + _save_json(stats_path, stats_index.model_dump(mode="json")) + + return { + "daily_output": str(daily_path), + "stats_output": str(stats_path), + "source": source, + "candidate_count": daily_index.candidate_count, + "term_count": len(daily_index.terms), + "top_terms": [term.model_dump(mode="json") for term in daily_index.terms[:10]], + } diff --git a/src/summary_mcp/models/__init__.py b/src/summary_mcp/models/__init__.py index 148460d..2da5477 100644 --- a/src/summary_mcp/models/__init__.py +++ b/src/summary_mcp/models/__init__.py @@ -22,6 +22,7 @@ from .daily_digest import ( DailyDigestSourceRef, DailyDigestStats, ) +from .keyword_index import DailyKeywordIndex, DailyKeywordTerm, KeywordStat, KeywordStatsIndex from .openclaw_delivery import ( OpenClawDeliveryPayload, OpenClawDeliveryStats, @@ -40,7 +41,11 @@ __all__ = [ "DailyDigestSection", "DailyDigestSourceRef", "DailyDigestStats", + "DailyKeywordIndex", + "DailyKeywordTerm", "DigestSectionHint", + "KeywordStat", + "KeywordStatsIndex", "KnowledgeDecision", "OpenClawCandidateInput", "OpenClawDeliveryPayload", @@ -52,4 +57,4 @@ __all__ = [ "build_openclaw_candidate_input", "candidate_id_for", "normalize_candidate_url", -] \ No newline at end of file +] diff --git a/src/summary_mcp/models/keyword_index.py b/src/summary_mcp/models/keyword_index.py new file mode 100644 index 0000000..0c57ccc --- /dev/null +++ b/src/summary_mcp/models/keyword_index.py @@ -0,0 +1,35 @@ +from __future__ import annotations + +from datetime import date, datetime + +from pydantic import BaseModel, Field + + +class DailyKeywordTerm(BaseModel): + term: str + normalized_term: str + count: int = Field(ge=1) + + +class DailyKeywordIndex(BaseModel): + schema_version: str = "v1" + date: date + source: str + digest_id: str + generated_at: datetime + candidate_count: int = Field(default=0, ge=0) + terms: list[DailyKeywordTerm] = Field(default_factory=list) + + +class KeywordStat(BaseModel): + term: str + total_count: int = Field(default=0, ge=0) + days_seen: int = Field(default=0, ge=0) + first_seen: date + last_seen: date + + +class KeywordStatsIndex(BaseModel): + schema_version: str = "v1" + generated_at: datetime + terms: list[KeywordStat] = Field(default_factory=list) diff --git a/src/summary_mcp/workflows/freshrss_pipeline.py b/src/summary_mcp/workflows/freshrss_pipeline.py index 85a5bae..b271eaa 100644 --- a/src/summary_mcp/workflows/freshrss_pipeline.py +++ b/src/summary_mcp/workflows/freshrss_pipeline.py @@ -6,6 +6,7 @@ from datetime import UTC, date, datetime from pathlib import Path from typing import Any +from summary_mcp.core.keyword_index import persist_keyword_indexes from summary_mcp.core.pipeline import extract_content from summary_mcp.core.summary_loop import resolve_llm_settings, run_loop_payload from summary_mcp.filters.engine import evaluate_filter_rules, load_filter_rules @@ -26,8 +27,13 @@ from summary_mcp.models.summary_io import ExtractionInput REPO_ROOT = Path(__file__).resolve().parents[3] OUTPUT_ROOT = REPO_ROOT / "outputs" FRESHRSS_OUTPUT_ROOT = OUTPUT_ROOT / "freshrss" +DATA_ROOT = REPO_ROOT / "data" / "term_index" DEFAULT_PROMPT_PATH = OUTPUT_ROOT / "prompts" / "llm-summary-prompt.txt" DEFAULT_RULES_PATH = REPO_ROOT / "configs" / "filter_rules.json" +DEFAULT_TERM_ALIASES_PATH = REPO_ROOT / "configs" / "term_aliases.json" +DEFAULT_TERM_STOPWORDS_PATH = REPO_ROOT / "configs" / "term_stopwords.json" +DEFAULT_TERM_DAILY_DIR = DATA_ROOT / "daily" +DEFAULT_TERM_STATS_PATH = DATA_ROOT / "term_stats.json" def _save_json(path: Path, payload: dict[str, Any] | list[Any]) -> None: @@ -237,6 +243,17 @@ def run_freshrss_pipeline( ) _save_json(delivery_output, delivery_payload.model_dump(mode="json")) + keyword_index_result = persist_keyword_indexes( + delivery_payload.candidates, + for_date=delivery_payload.date, + digest_id=delivery_payload.run_id, + source="openclaw_delivery_payload", + daily_dir=DEFAULT_TERM_DAILY_DIR, + stats_path=DEFAULT_TERM_STATS_PATH, + aliases_path=DEFAULT_TERM_ALIASES_PATH, + stopwords_path=DEFAULT_TERM_STOPWORDS_PATH, + ) + marked_count = 0 if mark_read and delivered_item_ids: client.mark_items_as_read(auth_token=auth_token, item_ids=delivered_item_ids) @@ -259,6 +276,7 @@ def run_freshrss_pipeline( "debug_artifacts": debug_artifacts, "raw_output": str(raw_output), "delivery_output": str(delivery_output), + "keyword_index": keyword_index_result, "status_counts": status_counts, "items": item_reports, } @@ -270,6 +288,7 @@ def run_freshrss_pipeline( "raw_output": str(raw_output), "delivery_output": str(delivery_output), "report_output": str(report_output), + "keyword_index": keyword_index_result, "pulled_count": len(items), "delivered_count": len(delivered_candidates), "marked_read_count": marked_count,