Add keyword cleanup governance workflow

This commit is contained in:
zhuyongxin
2026-03-27 16:59:52 +08:00
parent 100044e1f7
commit 32ad584348
25 changed files with 1915 additions and 18 deletions
+2
View File
@@ -8,3 +8,5 @@ findings.md
progress.md progress.md
task_plan.md task_plan.md
outputs/freshrss/ outputs/freshrss/
data/term_index/
outputs/term_index/
+54
View File
@@ -92,6 +92,11 @@ This is the recommended production entrypoint. By default it writes only:
- `outputs/freshrss/rerun/<timestamp>/candidates/openclaw-delivery-payload.json` - `outputs/freshrss/rerun/<timestamp>/candidates/openclaw-delivery-payload.json`
- `outputs/freshrss/rerun/<timestamp>/run-report.json` - `outputs/freshrss/rerun/<timestamp>/run-report.json`
It also updates the daily keyword index runtime data:
- `data/term_index/daily/YYYY-MM-DD.json`
- `data/term_index/term_stats.json`
If you need per-item intermediates, add `--debug-artifacts`. If you need per-item intermediates, add `--debug-artifacts`.
When OpenClaw is connected to the MCP server, it should call `run_freshrss_openclaw_pipeline` for the same behavior directly through MCP. The tool also supports `debug_artifacts=true` when deeper inspection is needed. When OpenClaw is connected to the MCP server, it should call `run_freshrss_openclaw_pipeline` for the same behavior directly through MCP. The tool also supports `debug_artifacts=true` when deeper inspection is needed.
@@ -160,3 +165,52 @@ The script writes by default:
- `outputs/reference/candidates/openclaw-delivery-payload.json` - `outputs/reference/candidates/openclaw-delivery-payload.json`
Output layout details live in `outputs/README.md`. Output layout details live in `outputs/README.md`.
Keyword index defaults live in:
- `configs/term_aliases.json`
- `configs/term_stopwords.json`
- `configs/term_cleanup_policy.json`
- `configs/term_watchlist.json`
- `configs/term_change_log.json`
You can also rebuild the keyword index from an existing delivery payload:
```bash
python scripts/build_keyword_index.py ^
--input outputs/reference/candidates/openclaw-delivery-payload.json
```
Runtime keyword data is stored under `data/term_index/`.
The keyword cleanup review skill lives in:
- `skills/keyword-cleanup-review/`
To build a review bundle for the LLM skill:
```bash
python skills/keyword-cleanup-review/scripts/build_review_bundle.py ^
--days 7 ^
--top 50 ^
--output outputs/term_index/review/keyword-cleanup-bundle.json
```
The review bundle now also carries cleanup governance context:
- cleanup thresholds from `configs/term_cleanup_policy.json`
- the current watch list from `configs/term_watchlist.json`
- recent applied changes from `configs/term_change_log.json`
The skill only produces review inputs and suggestions. It does not modify `term_aliases`, `term_stopwords`, or `filter_context.personal.json` automatically.
To preview accepted suggestions before writing any config files:
```bash
python scripts/apply_term_suggestions.py ^
--suggestions outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json ^
--accept-watch Cron Heartbeat Memory ^
--dry-run
```
Remove `--dry-run` to write the accepted changes. The script can also apply accepted `alias`, `stopword`, and `interest keyword` suggestions through `--accept-alias`, `--accept-stopword`, and `--accept-interest`. Accepted watch terms are written into `configs/term_watchlist.json`, and every applied action is appended into `configs/term_change_log.json`.
+11 -2
View File
@@ -20,6 +20,11 @@
- [x] 最终 payload 成功后才标记 FreshRSS 已读 - [x] 最终 payload 成功后才标记 FreshRSS 已读
- [x] MCP 工具 `run_freshrss_openclaw_pipeline` - [x] MCP 工具 `run_freshrss_openclaw_pipeline`
- [x] 默认精简输出模式 - [x] 默认精简输出模式
- [x] 日报级 `keywords` 词元库与周期性词元清洗 skill 设计完成
- [x] 日报级 `keywords` 词元库与全局词频统计实现完成
- [x] `keyword-cleanup-review` skill 骨架与 review bundle 脚本实现完成
- [x] 词元清洗低复杂治理层落地:`term_cleanup_policy` / `term_watchlist` / `term_change_log`
- [x] 已支持人工确认采纳建议并写入 `term_watchlist` / `term_change_log`
--- ---
@@ -38,6 +43,8 @@
- [ ] 设计“人工确认后再沉淀知识库”的状态流转 - [ ] 设计“人工确认后再沉淀知识库”的状态流转
- [ ] 收敛 `paywall` 误判规则,降低中文文本误报 - [ ] 收敛 `paywall` 误判规则,降低中文文本误报
- [ ] 细化过滤规则并引入更多个性化上下文 - [ ] 细化过滤规则并引入更多个性化上下文
- [ ] 将 `keyword-cleanup-review` skill 接入周期性执行流程,产出别名/停用词/兴趣词建议
- [ ] 增加清洗前后效果对比报告,验证配置调整是否真的改善过滤质量
- [ ] 增加批量 run 的保留策略与历史清理策略 - [ ] 增加批量 run 的保留策略与历史清理策略
- [ ] 为 OpenClaw 补一份更正式的 MCP 调用示例和接线说明 - [ ] 为 OpenClaw 补一份更正式的 MCP 调用示例和接线说明
@@ -59,8 +66,8 @@
优先做这三件事: 优先做这三件事:
1. 明确 OpenClaw 侧的 MCP 接线方式 1. 将 `keyword-cleanup-review` skill 接入周期性执行流程
2. 设计主动投递 / webhook 机制 2. 设计“人工确认后再沉淀知识库”的状态流转
3. 收敛规则误判,尤其是 `paywall` 相关启发式 3. 收敛规则误判,尤其是 `paywall` 相关启发式
--- ---
@@ -71,4 +78,6 @@
- `docs/openclaw/openclaw-handoff.md` - `docs/openclaw/openclaw-handoff.md`
- `docs/openclaw/openclaw-candidate-input-field-spec.md` - `docs/openclaw/openclaw-candidate-input-field-spec.md`
- `docs/openclaw/openclaw-delivery-payload-spec.md` - `docs/openclaw/openclaw-delivery-payload-spec.md`
- `docs/design/daily-keyword-index-design.md`
- `skills/keyword-cleanup-review/SKILL.md`
- `docs/current/context-reset-brief.md` - `docs/current/context-reset-brief.md`
+5 -2
View File
@@ -1,4 +1,4 @@
{ {
"source_tags": [ "source_tags": [
"backend", "backend",
"ai-agent", "ai-agent",
@@ -48,6 +48,9 @@
"向量数据库", "向量数据库",
"知识库", "知识库",
"OpenAI", "OpenAI",
"DeepSeek" "DeepSeek",
"OpenClaw",
"AliSQL",
"MySQL复制延迟"
] ]
} }
+5
View File
@@ -0,0 +1,5 @@
{
"AI助手": "AI Agent",
"图文RAG": "RAG",
"Prompt架构": "Prompt Engineering"
}
+4
View File
@@ -0,0 +1,4 @@
{
"schema_version": "v1",
"entries": []
}
+25
View File
@@ -0,0 +1,25 @@
{
"schema_version": "v1",
"interest_keyword_review": {
"min_total_count": 3,
"min_days_seen": 2
},
"watch_term_review": {
"min_total_count": 1,
"min_days_seen": 1,
"max_total_count": 2,
"max_days_seen": 2
},
"alias_review": {
"min_total_count": 2,
"min_days_seen": 2
},
"stopword_review": {
"max_total_count": 2,
"max_days_seen": 2
},
"notes": [
"当前阶段采用保守阈值,避免在低样本条件下直接扩充 interest_keywords。",
"watch_terms 先用于观察,后续再决定是否升格为 interest_keywords 或进入 alias/stopword 配置。"
]
}
+6
View File
@@ -0,0 +1,6 @@
[
"小银",
"银行客户经理",
"飞盘物理",
"奋斗文化"
]
+5
View File
@@ -0,0 +1,5 @@
{
"schema_version": "v1",
"updated_at": "2026-03-27T00:00:00Z",
"terms": []
}
+13 -7
View File
@@ -1,4 +1,4 @@
# 文档索引 # 文档索引
## 当前目录结构 ## 当前目录结构
@@ -31,17 +31,19 @@
- 过滤层的输入输出、规则结构与当前实现 - 过滤层的输入输出、规则结构与当前实现
7. `docs/design/filter-rule-engine-usage.md` 7. `docs/design/filter-rule-engine-usage.md`
- 规则怎么写、怎么跑、结果怎么解读的使用说明 - 规则怎么写、怎么跑、结果怎么解读的使用说明
8. `docs/design/markdown-sink-design.md` 8. `docs/design/daily-keyword-index-design.md`
- 日报级词元库与周期性词元清洗 skill 设计
9. `docs/design/markdown-sink-design.md`
- 第一版 Markdown sink 的输入输出、目录结构与落地方式 - 第一版 Markdown sink 的输入输出、目录结构与落地方式
9. `docs/openclaw/openclaw-daily-digest-refactor.md` 10. `docs/openclaw/openclaw-daily-digest-refactor.md`
- 为什么要从单篇入库改成 OpenClaw 日报聚合链路 - 为什么要从单篇入库改成 OpenClaw 日报聚合链路
10. `docs/openclaw/article-candidate-daily-digest-schema.md` 11. `docs/openclaw/article-candidate-daily-digest-schema.md`
- `ArticleCandidateRecord`、`OpenClawCandidateInput` 与 `DailyDigest` 的正式设计 - `ArticleCandidateRecord`、`OpenClawCandidateInput` 与 `DailyDigest` 的正式设计
11. `docs/design/source-schema-design.md` 12. `docs/design/source-schema-design.md`
- `source -> item -> document` 的对象设计 - `source -> item -> document` 的对象设计
12. `docs/notes/reading-pipeline-design-notes.md` 13. `docs/notes/reading-pipeline-design-notes.md`
- 更上层的阅读流方案与阶段划分 - 更上层的阅读流方案与阶段划分
13. `docs/design/summary-loop-explained.md` 14. `docs/design/summary-loop-explained.md`
- 当前 LLM 摘要校验闭环的解释 - 当前 LLM 摘要校验闭环的解释
## 当前文档分层 ## 当前文档分层
@@ -71,6 +73,10 @@
- 第一版规则过滤引擎设计与落地位置 - 第一版规则过滤引擎设计与落地位置
- `docs/design/filter-rule-engine-usage.md` - `docs/design/filter-rule-engine-usage.md`
- 规则配置、调用方式与结果解读 - 规则配置、调用方式与结果解读
- `docs/design/daily-keyword-index-design.md`
- 日报级词元库与周期性词元清洗 skill 设计
- `scripts/apply_term_suggestions.py`
- 人工确认后将建议写回 watchlist / change log 的脚本入口(见 `README.md` 用法)
- `docs/design/markdown-sink-design.md` - `docs/design/markdown-sink-design.md`
- 第一版 Markdown sink 设计与落地位置 - 第一版 Markdown sink 设计与落地位置
+43 -2
View File
@@ -23,6 +23,10 @@ OpenClaw 应通过 MCP 工具 `run_freshrss_openclaw_pipeline` 调用这条链
- 已完成“仅在最终 payload 成功写盘后再标记已读”的语义 - 已完成“仅在最终 payload 成功写盘后再标记已读”的语义
- 已完成 MCP 工具 `run_freshrss_openclaw_pipeline` - 已完成 MCP 工具 `run_freshrss_openclaw_pipeline`
- 已完成默认精简输出模式,减少中间文件 - 已完成默认精简输出模式,减少中间文件
- 已完成日报级 `keywords` 词元库与全局词频统计
- 已完成 `keyword-cleanup-review` skill 骨架与 review bundle 脚本
- 已完成低复杂治理层:`term_cleanup_policy` / `term_watchlist` / `term_change_log`
- 已完成采纳建议写回脚本 `scripts/apply_term_suggestions.py`
## 当前 MCP 工具 ## 当前 MCP 工具
@@ -47,6 +51,10 @@ OpenClaw 应通过 MCP 工具 `run_freshrss_openclaw_pipeline` 调用这条链
- `src/summary_mcp/server.py` - `src/summary_mcp/server.py`
- FreshRSS 统一工作流 - FreshRSS 统一工作流
- `src/summary_mcp/workflows/freshrss_pipeline.py` - `src/summary_mcp/workflows/freshrss_pipeline.py`
- 词元统计核心
- `src/summary_mcp/core/keyword_index.py`
- 词元统计模型
- `src/summary_mcp/models/keyword_index.py`
- 摘要循环 - 摘要循环
- `src/summary_mcp/core/summary_loop.py` - `src/summary_mcp/core/summary_loop.py`
- 提取主流程 - 提取主流程
@@ -63,6 +71,18 @@ OpenClaw 应通过 MCP 工具 `run_freshrss_openclaw_pipeline` 调用这条链
- `src/summary_mcp/models/openclaw_delivery.py` - `src/summary_mcp/models/openclaw_delivery.py`
- 生产脚本入口 - 生产脚本入口
- `scripts/run_freshrss_pipeline.py` - `scripts/run_freshrss_pipeline.py`
- 词元统计重建脚本
- `scripts/build_keyword_index.py`
- 词元清洗 skill
- `skills/keyword-cleanup-review/SKILL.md`
- skill review bundle 脚本
- `skills/keyword-cleanup-review/scripts/build_review_bundle.py`
- 采纳建议写回脚本
- `scripts/apply_term_suggestions.py`
- 清洗治理配置
- `configs/term_cleanup_policy.json`
- `configs/term_watchlist.json`
- `configs/term_change_log.json`
- OpenClaw 交接说明 - OpenClaw 交接说明
- `docs/openclaw/openclaw-handoff.md` - `docs/openclaw/openclaw-handoff.md`
@@ -74,6 +94,19 @@ OpenClaw 应通过 MCP 工具 `run_freshrss_openclaw_pipeline` 调用这条链
- `candidates/openclaw-delivery-payload.json` - `candidates/openclaw-delivery-payload.json`
- `run-report.json` - `run-report.json`
同时会更新本地运行数据:
- `data/term_index/daily/YYYY-MM-DD.json`
- `data/term_index/term_stats.json`
如果需要词元清洗审阅输入,可额外生成:
- `outputs/term_index/review/keyword-cleanup-bundle.json`
如果需要在人工确认后把建议正式写入 watchlist / change log,可使用:
- `scripts/apply_term_suggestions.py`
如果需要排障,可开启: 如果需要排障,可开启:
- `debug_artifacts=true` - `debug_artifacts=true`
@@ -90,6 +123,10 @@ OpenClaw 应通过 MCP 工具 `run_freshrss_openclaw_pipeline` 调用这条链
- MCP 工具入口可直接触发完整链路 - MCP 工具入口可直接触发完整链路
- 微信公众号样本可直接使用 RSS 提供的 `summary` 内容提取,不再回源抓网页 - 微信公众号样本可直接使用 RSS 提供的 `summary` 内容提取,不再回源抓网页
- 精简输出模式已实际跑通 - 精简输出模式已实际跑通
- 日报级词元统计已通过离线样例验证,确认别名、停用词、非 `drop` 过滤和 rerun 覆盖逻辑正常
- `keyword-cleanup-review` skill 已通过 `quick_validate.py` 结构校验
- review bundle 脚本已实际跑通
- `apply_term_suggestions.py` 已通过 dry-run 与临时副本写回验证
## 当前已知限制 ## 当前已知限制
@@ -98,6 +135,8 @@ OpenClaw 应通过 MCP 工具 `run_freshrss_openclaw_pipeline` 调用这条链
- 某些规则仍偏保守,部分内容可能落到 `review` - 某些规则仍偏保守,部分内容可能落到 `review`
- `paywall` 相关启发式仍可能误判中文文本 - `paywall` 相关启发式仍可能误判中文文本
- Webhook / 主动投递到 OpenClaw 外部接口尚未实现,当前是由 OpenClaw 通过 MCP 主动调用 - Webhook / 主动投递到 OpenClaw 外部接口尚未实现,当前是由 OpenClaw 通过 MCP 主动调用
- 词元清洗 skill 当前已支持“bundle 构建 -> 建议审阅 -> 人工确认写回 watchlist/change_log”,但尚未接入周期性调度
- 当前词元统计仍以前置 `OpenClawDeliveryPayload` 作为日报前代理输入,真实 `DailyDigest` 接入后还需切换上游
## 当前最建议的交接阅读顺序 ## 当前最建议的交接阅读顺序
@@ -105,8 +144,10 @@ OpenClaw 应通过 MCP 工具 `run_freshrss_openclaw_pipeline` 调用这条链
2. `docs/openclaw/openclaw-handoff.md` 2. `docs/openclaw/openclaw-handoff.md`
3. `docs/openclaw/openclaw-candidate-input-field-spec.md` 3. `docs/openclaw/openclaw-candidate-input-field-spec.md`
4. `docs/openclaw/openclaw-delivery-payload-spec.md` 4. `docs/openclaw/openclaw-delivery-payload-spec.md`
5. `TODO.md` 5. `docs/design/daily-keyword-index-design.md`
6. `skills/keyword-cleanup-review/SKILL.md`
7. `TODO.md`
## 一句话结论 ## 一句话结论
当前仓库已经从“提取 MCP 原型”演进到“可供 OpenClaw 调用的 FreshRSS -> OpenClaw payload 上游处理器”,可以开始交接,但后续仍建议继续补 webhook / delivery 接线与规则收敛。 当前仓库已经从“提取 MCP 原型”演进到“可供 OpenClaw 调用的 FreshRSS -> OpenClaw payload 上游处理器”,并已补上第一阶段的日报级词元统计能力和词元清洗 skill 骨架;后续重点转向 skill 周期调度、知识库状态流转和 webhook 接线。
+461
View File
@@ -0,0 +1,461 @@
# 日报级词元库与词元清洗 Skill 设计
## 1. 设计目标
当前项目已经具备:
`FreshRSS -> extraction -> LLM summary -> rule engine -> OpenClaw payload`
下一阶段希望新增“词元库”能力,目标不是做全文级检索索引,而是解决两件事:
1. 为后续规则配置提供稳定、轻量、可读的词元来源
2. 为周期性的词元清洗、合并和兴趣词补充提供数据基础
因此这套设计的核心原则是:
- 词元来源轻量化
- 数据粒度日报化
- 主链路程序化维护
- 清洗治理由独立 skill 周期性执行
- LLM 不直接修改规则或兴趣词配置
## 2. 为什么不用逐篇词元库
逐篇记录每篇文章的词元事件,虽然可追溯,但当前阶段成本过高,收益不足:
- 生成文件会很多
- 存储和调试负担更大
- 后续真正调规则时,用户更关心“最近日报里反复出现什么词”,而不是“某一篇文章具体抽到了什么词”
- 当前目标是服务日报与规则配置,不是做文章级分析平台
所以本项目不采用:
`article -> term event -> term store`
而采用:
`daily digest -> keyword aggregate -> daily term index -> global term stats`
## 3. 为什么只保留 `keywords`
当前 LLM 摘要结果里已经有两个候选字段:
- `keywords`
- `topics`
本设计只使用 `keywords` 进入词元库,不使用 `topics` 作为主来源。
原因:
- `keywords` 更具体,适合后续规则配置
- 例如 `Java`、`Go`、`Python`、`MCP`、`RAG`、`Kafka`
- `topics` 更宽泛,适合摘要展示,不适合作为精确规则命中基础
- 例如“后端工程”“AI Agent”“前沿科技”过于宽泛
- 只保留一个字段可以显著控制词元库规模
- 当前项目的真实需求是“词元可配置”,不是“主题聚类”
结论:
- `keywords` 进入词元库
- `topics` 继续保留在单篇摘要结果中,但不纳入词元统计主流程
## 4. 词元库的上游边界
词元库不直接读取所有候选文章,而只读取“最终进入日报”的内容。
也就是说,真正的词元来源是:
- `DailyDigest`
- 或等价的“已被日报选中”的 candidate 集合
不纳入词元库的内容:
- 被 `drop` 的内容
- 仅在中间候选层出现、但未进入日报的内容
- 调试产物中的临时摘要结果
这样做的好处是:
- 词元库只反映真正进入日级产物的内容
- 高频词更接近长期兴趣,而不是临时噪声
- 数据量更可控
## 5. 总体链路
目标链路调整为:
`DailyDigest -> keyword normalization -> daily term index -> global term stats -> cleaning skill review -> human confirm -> config update`
职责拆分如下:
### 5.1 程序负责
- 从日报中提取 `keywords`
- 归一化词元
- 应用别名映射
- 应用停用词过滤
- 生成每日词频
- 更新全局累计统计
### 5.2 Skill 负责
- 周期性读取词频结果
- 识别重复词、近义词、大小写变体
- 识别泛词、噪声词、低价值词
- 建议哪些词应合并、停用、加入兴趣词配置
### 5.3 人工负责
- 审核 skill 输出的建议
- 决定是否更新:
- `term_aliases`
- `term_stopwords`
- `filter_context.personal.json`
## 6. 数据文件设计
建议新增以下文件:
- `data/term_index/daily/YYYY-MM-DD.json`
- 某一天日报的词元聚合结果
- `data/term_index/term_stats.json`
- 全局累计词元统计
- `configs/term_aliases.json`
- 词元别名归一配置
- `configs/term_stopwords.json`
- 词元停用词配置
- `configs/term_cleanup_policy.json`
- 清洗阈值与治理策略配置
- `configs/term_watchlist.json`
- 当前处于观察状态的词元列表
- `configs/term_change_log.json`
- 已确认生效的词元治理变更记录
说明:
- 不新增逐篇 `term_events.jsonl`
- 不新增文章级明细文件
- 默认只保留“日报聚合结果 + 全局统计结果”
## 7. 每日词元文件结构
文件路径示例:
- `data/term_index/daily/2026-03-26.json`
建议结构:
```json
{
"date": "2026-03-26",
"source": "daily_digest",
"digest_id": "digest-2026-03-26",
"generated_at": "2026-03-26T21:30:00+08:00",
"candidate_count": 8,
"terms": [
{
"term": "AI Agent",
"normalized_term": "AI Agent",
"count": 4
},
{
"term": "MCP",
"normalized_term": "MCP",
"count": 3
},
{
"term": "RAG",
"normalized_term": "RAG",
"count": 2
}
]
}
```
字段说明:
- `date`
- 日报日期
- `source`
- 固定标记为 `daily_digest`
- `digest_id`
- 日报对象唯一标识
- `generated_at`
- 该词元文件生成时间
- `candidate_count`
- 当天日报包含的条目数
- `terms[]`
- 当天词元聚合结果
其中单条 `terms[]` 只保留:
- `term`
- `normalized_term`
- `count`
不保留:
- 逐篇文章来源列表
- 逐条命中明细
- 本地文件路径
## 8. 全局统计文件结构
文件路径:
- `data/term_index/term_stats.json`
建议结构:
```json
{
"schema_version": "v1",
"generated_at": "2026-03-26T21:30:00+08:00",
"terms": [
{
"term": "AI Agent",
"total_count": 18,
"days_seen": 6,
"first_seen": "2026-03-20",
"last_seen": "2026-03-26"
},
{
"term": "MCP",
"total_count": 12,
"days_seen": 5,
"first_seen": "2026-03-21",
"last_seen": "2026-03-26"
}
]
}
```
字段说明:
- `term`
- 最终归一后的词元
- `total_count`
- 累计出现次数
- `days_seen`
- 出现过的天数
- `first_seen`
- 首次出现日期
- `last_seen`
- 最近出现日期
当前阶段不额外记录:
- 各 category 分桶统计
- keep/review/drop 分桶统计
- 文章级来源列表
原因是:日报级词元库的第一目标是轻量稳定,不是分析平台。
## 9. 标准化与归一规则
程序在写入日报词元前,应先做标准化。
### 9.1 基础标准化
- 去除首尾空白
- 保留中英文大小写风格中的稳定写法
- 去重
- 过滤空字符串
### 9.2 别名映射
通过 `configs/term_aliases.json` 做归一。
示例:
```json
{
"Agent": "AI Agent",
"智能体": "AI Agent",
"Postgres": "PostgreSQL",
"Model Context Protocol": "MCP"
}
```
### 9.3 停用词过滤
通过 `configs/term_stopwords.json` 过滤过泛词和噪声词。
示例:
```json
[
"技术",
"系统",
"方案",
"实践",
"文章"
]
```
## 10. 为什么不让 LLM 直接维护词元库
LLM 可以帮助做清洗建议,但不适合直接维护主词元库。
原因:
- 主词元库更新应该稳定、低成本、可复现
- 词频统计属于纯程序逻辑,没必要消耗模型调用
- 如果让 LLM 直接写词元库,会引入不稳定和难审计问题
因此主流程固定为:
- LLM 只负责在摘要结果里输出 `keywords`
- 程序负责归一、聚合、统计
## 11. 词元清洗 Skill 设计
新增一个周期性清洗 skill,定位是“治理器”,不是“实时生产者”。
### 11.1 Skill 输入
建议输入:
- `data/term_index/term_stats.json`
- 最近 N 天的 `data/term_index/daily/*.json`
- `configs/term_aliases.json`
- `configs/term_stopwords.json`
- `configs/filter_context.personal.json`
### 11.2 Skill 输出
skill 不直接修改配置文件,而是生成建议文件,例如:
- `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.md`
- `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json`
低复杂治理层建议补充三类输入:
- `term_cleanup_policy`
- 用来定义 watch 和 interest 的最小证据阈值
- `term_watchlist`
- 用来记录“先观察、暂不升级”的词元
- `term_change_log`
- 用来记录已经确认落地的配置变更,避免后续遗忘上下文
建议项包括:
- 建议合并的别名词
- 建议新增的停用词
- 建议加入 `interest_keywords` 的候选词
- 建议降权观察的热点词
### 11.3 Skill 允许做什么
- 发现重复词
- 发现大小写变体
- 发现中英文混用的近义词
- 发现持续高频但尚未进入兴趣配置的词
- 发现明显过泛的词
### 11.4 Skill 不允许做什么
- 直接改 `filter_rules.json`
- 直接改 `filter_context.personal.json`
- 直接覆盖 `term_stats.json`
- 在无人工确认的情况下自动生效
## 12. 建议的清洗建议文件结构
建议 JSON 文件结构如下:
```json
{
"date": "2026-03-26",
"based_on_days": 7,
"alias_suggestions": [
{
"from": "Agent",
"to": "AI Agent",
"reason": "和现有高频词语义一致,建议归并。"
}
],
"stopword_suggestions": [
{
"term": "系统",
"reason": "出现频繁但语义过泛,难以作为规则命中词。"
}
],
"interest_keyword_suggestions": [
{
"term": "MCP",
"reason": "最近多日持续高频,且符合当前 AI Agent 学习方向。"
}
]
}
```
## 13. 周期与触发方式
建议触发周期:
- 词元统计更新:每天一次,跟随日报生成
- skill 清洗:每周一次,或人工手动触发
推荐流程:
1. 当天日报生成完成
2. 程序更新 `daily/YYYY-MM-DD.json`
3. 程序更新 `term_stats.json`
4. 每周或人工触发一次词元清洗 skill
5. skill 输出建议
6. 人工确认后再更新配置文件
## 14. 与规则引擎的关系
这套词元库设计不是规则引擎的替代品,而是规则配置的辅助层。
关系如下:
- `keywords`
- 是词元库来源
- `term_stats`
- 是观察与调优依据
- `filter_context.personal.json`
- 是真正给规则引擎使用的兴趣词配置
- `filter_rules.json`
- 是最终裁决逻辑
也就是说:
`日报词元统计 -> 清洗建议 -> 人工确认 -> 更新 interest_keywords -> 规则引擎命中`
而不是:
`日报词元统计 -> 自动改规则`
## 15. 分阶段落地建议
### Phase 1
先做最小可用版本:
- 只读取日报中的 `keywords`
- 生成每日词元文件
- 生成全局累计词频文件
- 支持 `term_aliases` 和 `term_stopwords`
### Phase 2
再补治理层:
- 增加词元清洗 skill
- 输出建议文件
- 人工确认后更新配置
### Phase 3
最后再考虑增强:
- 增加趋势分析
- 增加最近 7 天热点词视图
- 增加“建议加入兴趣词”的自动排序
## 16. 一句话结论
这套设计选择“只统计日报中的 `keywords`,由程序维护轻量词元库,再由独立 skill 周期性做清洗建议”,目的是在控制数据规模的前提下,为规则配置和长期兴趣演化提供稳定、可审计、可扩展的基础设施。
+15 -1
View File
@@ -15,6 +15,17 @@
这是当前推荐的生产模式。 这是当前推荐的生产模式。
## 词元统计运行数据
词元统计不放在 `outputs/` 下,而放在单独的运行数据目录:
- `data/term_index/daily/YYYY-MM-DD.json`
- 某一天的日报级 `keywords` 聚合结果
- `data/term_index/term_stats.json`
- 全局累计词频统计
这两类文件属于可重建的本地运行数据,不作为仓库长期跟踪产物。
## 调试扩展输出 ## 调试扩展输出
当启用 `debug_artifacts` 时,才会额外写出这些中间文件: 当启用 `debug_artifacts` 时,才会额外写出这些中间文件:
@@ -57,6 +68,7 @@
- 默认优先保留最终产物,不把所有中间文件都当成长期资产 - 默认优先保留最终产物,不把所有中间文件都当成长期资产
- 调试文件只在需要时生成 - 调试文件只在需要时生成
- `reference/` 只放可复用样例,不放真实生产批次数据 - `reference/` 只放可复用样例,不放真实生产批次数据
- `data/term_index/` 只保存本地词元统计运行数据,不纳入 git 跟踪
## 当前建议 ## 当前建议
@@ -65,5 +77,7 @@
- `openclaw-delivery-payload.json` - `openclaw-delivery-payload.json`
- `run-report.json` - `run-report.json`
- `freshrss.raw.json` - `freshrss.raw.json`
- `data/term_index/daily/YYYY-MM-DD.json`
- `data/term_index/term_stats.json`
其余文件默认都应视为调试辅助产物。 其中前 3 个是本次批处理产物,后 2 个是持续累积的词元统计数据。
+402
View File
@@ -0,0 +1,402 @@
from __future__ import annotations
import argparse
import json
from datetime import UTC, datetime
from pathlib import Path
from typing import Any
REPO_ROOT = Path(__file__).resolve().parents[1]
DEFAULT_ALIASES_PATH = REPO_ROOT / "configs" / "term_aliases.json"
DEFAULT_STOPWORDS_PATH = REPO_ROOT / "configs" / "term_stopwords.json"
DEFAULT_CONTEXT_PATH = REPO_ROOT / "configs" / "filter_context.personal.json"
DEFAULT_WATCHLIST_PATH = REPO_ROOT / "configs" / "term_watchlist.json"
DEFAULT_CHANGE_LOG_PATH = REPO_ROOT / "configs" / "term_change_log.json"
def _load_json(path: Path, default: Any) -> Any:
if not path.exists():
return default
return json.loads(path.read_text(encoding="utf-8-sig"))
def _save_json(path: Path, payload: dict[str, Any] | list[Any]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
def _term_key(value: str) -> str:
return value.strip().casefold()
def _utc_now() -> str:
return datetime.now(tz=UTC).isoformat().replace("+00:00", "Z")
def _load_suggestions(path: Path) -> dict[str, Any]:
payload = _load_json(path, {})
if not isinstance(payload, dict):
raise RuntimeError("Suggestions file must contain a JSON object.")
return payload
def _load_aliases(path: Path) -> dict[str, str]:
payload = _load_json(path, {})
if not isinstance(payload, dict):
raise RuntimeError("Aliases file must contain a JSON object.")
result: dict[str, str] = {}
for key, value in payload.items():
if isinstance(key, str) and isinstance(value, str) and key.strip() and value.strip():
result[key] = value
return result
def _load_stopwords(path: Path) -> list[str]:
payload = _load_json(path, [])
if not isinstance(payload, list):
raise RuntimeError("Stopwords file must contain a JSON array.")
return [item for item in payload if isinstance(item, str) and item.strip()]
def _load_context(path: Path) -> dict[str, Any]:
payload = _load_json(path, {})
if not isinstance(payload, dict):
raise RuntimeError("Context file must contain a JSON object.")
payload.setdefault("interest_keywords", [])
if not isinstance(payload["interest_keywords"], list):
raise RuntimeError("filter_context.personal.json interest_keywords must be a JSON array.")
payload["interest_keywords"] = [item for item in payload["interest_keywords"] if isinstance(item, str) and item.strip()]
return payload
def _load_watchlist(path: Path) -> dict[str, Any]:
payload = _load_json(
path,
{
"schema_version": "v1",
"updated_at": _utc_now(),
"terms": [],
},
)
if not isinstance(payload, dict):
raise RuntimeError("Watchlist file must contain a JSON object.")
payload.setdefault("schema_version", "v1")
payload.setdefault("updated_at", _utc_now())
terms = payload.get("terms", [])
if not isinstance(terms, list):
raise RuntimeError("Watchlist terms must be a JSON array.")
payload["terms"] = [item for item in terms if isinstance(item, dict) and isinstance(item.get("term"), str)]
return payload
def _load_change_log(path: Path) -> dict[str, Any]:
payload = _load_json(
path,
{
"schema_version": "v1",
"entries": [],
},
)
if not isinstance(payload, dict):
raise RuntimeError("Change log file must contain a JSON object.")
payload.setdefault("schema_version", "v1")
entries = payload.get("entries", [])
if not isinstance(entries, list):
raise RuntimeError("Change log entries must be a JSON array.")
payload["entries"] = [item for item in entries if isinstance(item, dict)]
return payload
def _build_index(entries: list[dict[str, Any]], field: str) -> dict[str, dict[str, Any]]:
index: dict[str, dict[str, Any]] = {}
for entry in entries:
value = entry.get(field)
if isinstance(value, str) and value.strip():
index[_term_key(value)] = entry
return index
def _append_change(
change_log: dict[str, Any],
*,
applied_at: str,
action: str,
term: str,
value: str | None,
reason: str,
suggestions_path: Path,
suggestion_date: str | None,
based_on_days: int | None,
) -> None:
entry: dict[str, Any] = {
"applied_at": applied_at,
"action": action,
"term": term,
"reason": reason,
"suggestions_path": str(suggestions_path),
}
if value is not None:
entry["value"] = value
if suggestion_date is not None:
entry["suggestion_date"] = suggestion_date
if based_on_days is not None:
entry["based_on_days"] = based_on_days
change_log["entries"].append(entry)
def _remove_from_watchlist(
watchlist: dict[str, Any],
*,
term: str,
) -> bool:
folded = _term_key(term)
original_len = len(watchlist["terms"])
watchlist["terms"] = [item for item in watchlist["terms"] if _term_key(str(item.get("term", ""))) != folded]
return len(watchlist["terms"]) != original_len
def main() -> None:
parser = argparse.ArgumentParser(description="Apply accepted term cleanup suggestions into repo config files.")
parser.add_argument("--suggestions", type=Path, required=True, help="Suggestion JSON file")
parser.add_argument("--aliases", type=Path, default=DEFAULT_ALIASES_PATH, help="Term aliases JSON file")
parser.add_argument("--stopwords", type=Path, default=DEFAULT_STOPWORDS_PATH, help="Term stopwords JSON file")
parser.add_argument("--context", type=Path, default=DEFAULT_CONTEXT_PATH, help="Personal filter context JSON file")
parser.add_argument("--watchlist", type=Path, default=DEFAULT_WATCHLIST_PATH, help="Watchlist JSON file")
parser.add_argument("--change-log", type=Path, default=DEFAULT_CHANGE_LOG_PATH, help="Change log JSON file")
parser.add_argument("--accept-watch", nargs="*", default=[], help="Accept watch_terms by term name")
parser.add_argument(
"--accept-interest", nargs="*", default=[], help="Accept interest_keyword_suggestions by term name"
)
parser.add_argument("--accept-stopword", nargs="*", default=[], help="Accept stopword_suggestions by term name")
parser.add_argument("--accept-alias", nargs="*", default=[], help="Accept alias_suggestions by source term")
parser.add_argument("--dry-run", action="store_true", help="Preview changes without writing files")
args = parser.parse_args()
suggestions = _load_suggestions(args.suggestions)
aliases = _load_aliases(args.aliases)
stopwords = _load_stopwords(args.stopwords)
context = _load_context(args.context)
watchlist = _load_watchlist(args.watchlist)
change_log = _load_change_log(args.change_log)
alias_suggestions = suggestions.get("alias_suggestions", [])
stopword_suggestions = suggestions.get("stopword_suggestions", [])
interest_suggestions = suggestions.get("interest_keyword_suggestions", [])
watch_suggestions = suggestions.get("watch_terms", [])
if not all(isinstance(bucket, list) for bucket in [alias_suggestions, stopword_suggestions, interest_suggestions, watch_suggestions]):
raise RuntimeError("Suggestion JSON buckets must all be arrays.")
alias_index = _build_index([item for item in alias_suggestions if isinstance(item, dict)], "from")
stopword_index = _build_index([item for item in stopword_suggestions if isinstance(item, dict)], "term")
interest_index = _build_index([item for item in interest_suggestions if isinstance(item, dict)], "term")
watch_index = _build_index([item for item in watch_suggestions if isinstance(item, dict)], "term")
watch_map = {_term_key(str(item.get("term", ""))): item for item in watchlist["terms"]}
stopword_set = {_term_key(item) for item in stopwords}
interest_set = {_term_key(item) for item in context["interest_keywords"]}
alias_source_set = {_term_key(key) for key in aliases}
applied_at = _utc_now()
suggestion_date = suggestions.get("date") if isinstance(suggestions.get("date"), str) else None
based_on_days = suggestions.get("based_on_days") if isinstance(suggestions.get("based_on_days"), int) else None
applied: list[dict[str, Any]] = []
skipped: list[dict[str, Any]] = []
def skip(action: str, term: str, reason: str) -> None:
skipped.append({"action": action, "term": term, "reason": reason})
def applied_entry(action: str, term: str, reason: str, value: str | None = None) -> None:
item: dict[str, Any] = {"action": action, "term": term, "reason": reason}
if value is not None:
item["value"] = value
applied.append(item)
for raw_term in args.accept_watch:
term = raw_term.strip()
suggestion = watch_index.get(_term_key(term))
if suggestion is None:
skip("add_watch_term", term, "Term not found in watch_terms suggestions.")
continue
if _term_key(term) in watch_map:
skip("add_watch_term", term, "Term already exists in watchlist.")
continue
reason = str(suggestion.get("reason") or "Accepted from watch_terms suggestion.")
watch_item = {
"term": str(suggestion.get("term") or term),
"added_at": applied_at,
"source": str(args.suggestions),
"reason": reason,
"status": "watching",
}
watchlist["terms"].append(watch_item)
watch_map[_term_key(watch_item["term"])] = watch_item
_append_change(
change_log,
applied_at=applied_at,
action="add_watch_term",
term=watch_item["term"],
value=None,
reason=reason,
suggestions_path=args.suggestions,
suggestion_date=suggestion_date,
based_on_days=based_on_days,
)
applied_entry("add_watch_term", watch_item["term"], reason)
for raw_term in args.accept_interest:
term = raw_term.strip()
suggestion = interest_index.get(_term_key(term))
if suggestion is None:
skip("add_interest_keyword", term, "Term not found in interest_keyword_suggestions.")
continue
resolved_term = str(suggestion.get("term") or term)
if _term_key(resolved_term) in interest_set:
skip("add_interest_keyword", resolved_term, "Term already exists in interest_keywords.")
continue
reason = str(suggestion.get("reason") or "Accepted from interest_keyword_suggestions.")
context["interest_keywords"].append(resolved_term)
interest_set.add(_term_key(resolved_term))
_append_change(
change_log,
applied_at=applied_at,
action="add_interest_keyword",
term=resolved_term,
value=None,
reason=reason,
suggestions_path=args.suggestions,
suggestion_date=suggestion_date,
based_on_days=based_on_days,
)
if _remove_from_watchlist(watchlist, term=resolved_term):
_append_change(
change_log,
applied_at=applied_at,
action="remove_watch_term",
term=resolved_term,
value="promoted_to_interest_keyword",
reason="Removed from watchlist after promotion into interest_keywords.",
suggestions_path=args.suggestions,
suggestion_date=suggestion_date,
based_on_days=based_on_days,
)
applied_entry("add_interest_keyword", resolved_term, reason)
for raw_term in args.accept_stopword:
term = raw_term.strip()
suggestion = stopword_index.get(_term_key(term))
if suggestion is None:
skip("add_stopword", term, "Term not found in stopword_suggestions.")
continue
resolved_term = str(suggestion.get("term") or term)
if _term_key(resolved_term) in stopword_set:
skip("add_stopword", resolved_term, "Term already exists in stopwords.")
continue
reason = str(suggestion.get("reason") or "Accepted from stopword_suggestions.")
stopwords.append(resolved_term)
stopword_set.add(_term_key(resolved_term))
_append_change(
change_log,
applied_at=applied_at,
action="add_stopword",
term=resolved_term,
value=None,
reason=reason,
suggestions_path=args.suggestions,
suggestion_date=suggestion_date,
based_on_days=based_on_days,
)
if _remove_from_watchlist(watchlist, term=resolved_term):
_append_change(
change_log,
applied_at=applied_at,
action="remove_watch_term",
term=resolved_term,
value="promoted_to_stopword",
reason="Removed from watchlist after being added to stopwords.",
suggestions_path=args.suggestions,
suggestion_date=suggestion_date,
based_on_days=based_on_days,
)
applied_entry("add_stopword", resolved_term, reason)
for raw_term in args.accept_alias:
term = raw_term.strip()
suggestion = alias_index.get(_term_key(term))
if suggestion is None:
skip("add_alias", term, "Source term not found in alias_suggestions.")
continue
source_term = str(suggestion.get("from") or term)
target_term = str(suggestion.get("to") or "").strip()
if not target_term:
skip("add_alias", source_term, "Alias suggestion target is empty.")
continue
existing_target = aliases.get(source_term)
if existing_target == target_term:
skip("add_alias", source_term, "Alias already exists with the same target.")
continue
if _term_key(source_term) in alias_source_set and existing_target != target_term:
skip("add_alias", source_term, f"Alias source already exists with a different target: {existing_target}")
continue
reason = str(suggestion.get("reason") or "Accepted from alias_suggestions.")
aliases[source_term] = target_term
alias_source_set.add(_term_key(source_term))
_append_change(
change_log,
applied_at=applied_at,
action="add_alias",
term=source_term,
value=target_term,
reason=reason,
suggestions_path=args.suggestions,
suggestion_date=suggestion_date,
based_on_days=based_on_days,
)
if _remove_from_watchlist(watchlist, term=source_term):
_append_change(
change_log,
applied_at=applied_at,
action="remove_watch_term",
term=source_term,
value="resolved_as_alias_source",
reason="Removed from watchlist after alias mapping was accepted.",
suggestions_path=args.suggestions,
suggestion_date=suggestion_date,
based_on_days=based_on_days,
)
applied_entry("add_alias", source_term, reason, value=target_term)
watchlist["terms"] = sorted(watchlist["terms"], key=lambda item: (_term_key(str(item.get("term", ""))), str(item.get("term", ""))))
watchlist["updated_at"] = applied_at
stopwords = sorted(stopwords, key=lambda item: (_term_key(item), item))
context["interest_keywords"] = sorted(context["interest_keywords"], key=lambda item: (_term_key(item), item))
summary = {
"dry_run": args.dry_run,
"suggestions": str(args.suggestions),
"applied_count": len(applied),
"skipped_count": len(skipped),
"applied": applied,
"skipped": skipped,
"output_paths": {
"aliases": str(args.aliases),
"stopwords": str(args.stopwords),
"context": str(args.context),
"watchlist": str(args.watchlist),
"change_log": str(args.change_log),
},
}
if not args.dry_run:
_save_json(args.aliases, aliases)
_save_json(args.stopwords, stopwords)
_save_json(args.context, context)
_save_json(args.watchlist, watchlist)
_save_json(args.change_log, change_log)
print(json.dumps(summary, ensure_ascii=False, indent=2))
if __name__ == "__main__":
main()
+76
View File
@@ -0,0 +1,76 @@
from __future__ import annotations
import argparse
import sys
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parents[1]
SRC_ROOT = REPO_ROOT / "src"
if str(SRC_ROOT) not in sys.path:
sys.path.insert(0, str(SRC_ROOT))
from summary_mcp.core.keyword_index import persist_keyword_indexes
from summary_mcp.models.openclaw_delivery import OpenClawDeliveryPayload
def _load_payload(path: Path) -> OpenClawDeliveryPayload:
return OpenClawDeliveryPayload.model_validate_json(path.read_text(encoding="utf-8-sig"))
def main() -> None:
parser = argparse.ArgumentParser(
description="Build the daily keyword index and global keyword stats from an OpenClaw delivery payload."
)
parser.add_argument(
"--input",
type=Path,
required=True,
help="OpenClaw delivery payload JSON file",
)
parser.add_argument(
"--daily-dir",
type=Path,
default=REPO_ROOT / "data" / "term_index" / "daily",
help="Directory for daily keyword index files",
)
parser.add_argument(
"--stats-output",
type=Path,
default=REPO_ROOT / "data" / "term_index" / "term_stats.json",
help="Global keyword stats output path",
)
parser.add_argument(
"--aliases",
type=Path,
default=REPO_ROOT / "configs" / "term_aliases.json",
help="Keyword alias config JSON file",
)
parser.add_argument(
"--stopwords",
type=Path,
default=REPO_ROOT / "configs" / "term_stopwords.json",
help="Keyword stopwords config JSON file",
)
args = parser.parse_args()
payload = _load_payload(args.input)
result = persist_keyword_indexes(
payload.candidates,
for_date=payload.date,
digest_id=payload.run_id,
source="openclaw_delivery_payload",
daily_dir=args.daily_dir,
stats_path=args.stats_output,
aliases_path=args.aliases,
stopwords_path=args.stopwords,
)
print(f"Saved daily keyword index to {result['daily_output']}")
print(f"Saved keyword stats to {result['stats_output']}")
print(f"Indexed {result['candidate_count']} candidates and {result['term_count']} keywords")
if __name__ == "__main__":
main()
+4
View File
@@ -88,6 +88,10 @@ def main() -> None:
print(f"Saved FreshRSS pipeline run to {result['output_dir']}") print(f"Saved FreshRSS pipeline run to {result['output_dir']}")
print(f"Pulled {result['pulled_count']} items, delivered {result['delivered_count']} candidates") print(f"Pulled {result['pulled_count']} items, delivered {result['delivered_count']} candidates")
print(
"Saved keyword index to "
f"{result['keyword_index']['daily_output']} and {result['keyword_index']['stats_output']}"
)
if args.mark_read: if args.mark_read:
print(f"Marked {result['marked_read_count']} FreshRSS entries as read") print(f"Marked {result['marked_read_count']} FreshRSS entries as read")
+102
View File
@@ -0,0 +1,102 @@
---
name: keyword-cleanup-review
description: Review and curate this repository's daily keyword index and frequency stats. Use when the user wants to inspect `data/term_index/term_stats.json`, recent `data/term_index/daily/*.json`, `configs/term_aliases.json`, `configs/term_stopwords.json`, or `configs/filter_context.personal.json` to propose alias merges, stopwords, watch terms, or `interest_keywords` updates without directly modifying configs.
---
# Keyword Cleanup Review
Use this skill to turn the repository's keyword statistics into reviewable cleanup suggestions.
## Workflow
1. Build a compact review bundle:
```bash
python skills/keyword-cleanup-review/scripts/build_review_bundle.py
```
Optional knobs:
- `--days 7`
- `--top 50`
- `--output outputs/term_index/review/keyword-cleanup-bundle.json`
2. Read the generated bundle and the suggestion schema:
- `outputs/term_index/review/keyword-cleanup-bundle.json`
- `skills/keyword-cleanup-review/references/suggestion-schema.md`
3. Produce two outputs:
- A short Markdown review for humans
- A JSON suggestion file matching the schema
4. Keep the boundary strict:
- Suggest changes to `configs/term_aliases.json`
- Suggest changes to `configs/term_stopwords.json`
- Suggest additions to `configs/filter_context.personal.json`
- Do not directly edit these files unless the user explicitly asks
- Do not suggest direct edits to `configs/filter_rules.json` unless the user asks for rule logic changes
## Review Heuristics
Prioritize these decisions:
- Alias suggestion
- Same concept with different naming, casing, abbreviation, or Chinese/English variants
- Stopword suggestion
- Too generic, too broad, or too noisy to help filtering
- Interest keyword suggestion
- High-frequency and aligned with the user's backend engineering, AI-agent, and frontier-tech focus
- Watch term
- Recent and potentially important, but evidence is still weak
Prefer conservative suggestions. If confidence is low, put the term into `watch_terms`.
## Inputs
Primary inputs:
- `data/term_index/term_stats.json`
- `data/term_index/daily/*.json`
- `configs/term_aliases.json`
- `configs/term_stopwords.json`
- `configs/filter_context.personal.json`
- `configs/term_cleanup_policy.json`
- `configs/term_watchlist.json`
- `configs/term_change_log.json`
The bundled script already compacts these into a single review bundle.
## Output Expectations
The Markdown output should:
- Summarize the current state briefly
- List the top terms worth acting on
- Separate alias, stopword, interest-keyword, and watch-term recommendations
- Explain reasoning in short, concrete sentences
The JSON output should follow:
- `references/suggestion-schema.md`
## Repository Notes
Current repository behavior:
- Keyword stats are program-maintained, not LLM-maintained
- Stats are built from `keywords`, not `topics`
- Stats only include non-`drop` candidates
- `data/term_index/term_stats.json` is rebuilt from daily files, so reruns overwrite the same day instead of double-counting
- cleanup policy, watchlist, and change log are repository-managed governance inputs and should be respected during review
Keep suggestions aligned with that design.
## Resources
- Script:
- `scripts/build_review_bundle.py`
- Reference:
- `references/suggestion-schema.md`
@@ -0,0 +1,3 @@
display_name: Keyword Cleanup Review
short_description: Review keyword stats and suggest cleanup updates.
default_prompt: Review the repository keyword index, identify duplicate or noisy terms, and produce alias, stopword, watch-term, and interest-keyword suggestions without modifying configs directly.
@@ -0,0 +1,47 @@
# Suggestion Schema
When this skill produces review output, prefer two files:
- Markdown summary for humans
- JSON suggestions for deterministic follow-up edits
Recommended JSON shape:
```json
{
"date": "2026-03-27",
"based_on_days": 7,
"alias_suggestions": [
{
"from": "Agent",
"to": "AI Agent",
"reason": "High overlap with existing repository terminology."
}
],
"stopword_suggestions": [
{
"term": "??",
"reason": "Too generic to be useful for filtering."
}
],
"interest_keyword_suggestions": [
{
"term": "MCP",
"reason": "Repeated high-frequency term aligned with current learning focus."
}
],
"watch_terms": [
{
"term": "OpenClaw",
"reason": "Trending recently but not enough history yet."
}
]
}
```
Rules:
- Only output suggestions, never claim they are already applied.
- Avoid suggesting changes that would directly edit `filter_rules.json`.
- Prefer small, reviewable batches over large refactors.
- If evidence is weak, put the term in `watch_terms` instead of alias or stopword suggestions.
@@ -0,0 +1 @@
Provide keyword cleanup guidance for this repository's daily keyword index. Use this skill when the user wants to review, merge, clean, or curate `data/term_index/term_stats.json`, recent `data/term_index/daily/*.json`, `configs/term_aliases.json`, `configs/term_stopwords.json`, or `configs/filter_context.personal.json`, especially to propose alias merges, stopwords, or `interest_keywords` updates without directly modifying configs.
@@ -0,0 +1,340 @@
from __future__ import annotations
import argparse
import json
from collections import defaultdict
from datetime import UTC, datetime
from pathlib import Path
from typing import Any
DEFAULT_POLICY: dict[str, Any] = {
"schema_version": "v1",
"interest_keyword_review": {
"min_total_count": 3,
"min_days_seen": 2,
},
"watch_term_review": {
"min_total_count": 1,
"min_days_seen": 1,
"max_total_count": 2,
"max_days_seen": 2,
},
"alias_review": {
"min_total_count": 2,
"min_days_seen": 2,
},
"stopword_review": {
"max_total_count": 2,
"max_days_seen": 2,
},
}
def _load_json(path: Path) -> Any:
return json.loads(path.read_text(encoding="utf-8-sig"))
def _save_json(path: Path, payload: dict[str, Any]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
def _load_mapping(path: Path) -> dict[str, str]:
if not path.exists():
return {}
payload = _load_json(path)
return payload if isinstance(payload, dict) else {}
def _load_list(path: Path) -> list[str]:
if not path.exists():
return []
payload = _load_json(path)
if isinstance(payload, list):
return [item for item in payload if isinstance(item, str)]
return []
def _load_interest_keywords(path: Path) -> list[str]:
if not path.exists():
return []
payload = _load_json(path)
if not isinstance(payload, dict):
return []
values = payload.get("interest_keywords")
if not isinstance(values, list):
return []
return [item for item in values if isinstance(item, str)]
def _casefold_set(values: list[str]) -> set[str]:
return {value.strip().casefold() for value in values if value.strip()}
def _load_policy(path: Path) -> dict[str, Any]:
if not path.exists():
return DEFAULT_POLICY.copy()
payload = _load_json(path)
return payload if isinstance(payload, dict) else DEFAULT_POLICY.copy()
def _load_watchlist(path: Path) -> list[dict[str, Any]]:
if not path.exists():
return []
payload = _load_json(path)
if not isinstance(payload, dict):
return []
terms = payload.get("terms")
if not isinstance(terms, list):
return []
return [item for item in terms if isinstance(item, dict) and isinstance(item.get("term"), str)]
def _load_change_log(path: Path) -> list[dict[str, Any]]:
if not path.exists():
return []
payload = _load_json(path)
if not isinstance(payload, dict):
return []
entries = payload.get("entries")
if not isinstance(entries, list):
return []
return [item for item in entries if isinstance(item, dict)]
def _meets_min_thresholds(item: dict[str, Any], thresholds: dict[str, Any]) -> bool:
min_total_count = int(thresholds.get("min_total_count", 1))
min_days_seen = int(thresholds.get("min_days_seen", 1))
total_count = int(item.get("total_count") or 0)
days_seen = int(item.get("days_seen") or 0)
return total_count >= min_total_count and days_seen >= min_days_seen
def _within_watch_thresholds(item: dict[str, Any], thresholds: dict[str, Any]) -> bool:
min_total_count = int(thresholds.get("min_total_count", 1))
min_days_seen = int(thresholds.get("min_days_seen", 1))
max_total_count = int(thresholds.get("max_total_count", 999999))
max_days_seen = int(thresholds.get("max_days_seen", 999999))
total_count = int(item.get("total_count") or 0)
days_seen = int(item.get("days_seen") or 0)
return (
total_count >= min_total_count
and days_seen >= min_days_seen
and total_count <= max_total_count
and days_seen <= max_days_seen
)
def main() -> None:
parser = argparse.ArgumentParser(
description="Build a compact review bundle for the keyword-cleanup-review skill."
)
repo_root = Path(__file__).resolve().parents[3]
parser.add_argument(
"--stats",
type=Path,
default=repo_root / "data" / "term_index" / "term_stats.json",
help="Global keyword stats file",
)
parser.add_argument(
"--daily-dir",
type=Path,
default=repo_root / "data" / "term_index" / "daily",
help="Directory containing daily keyword index JSON files",
)
parser.add_argument(
"--aliases",
type=Path,
default=repo_root / "configs" / "term_aliases.json",
help="Term aliases JSON file",
)
parser.add_argument(
"--stopwords",
type=Path,
default=repo_root / "configs" / "term_stopwords.json",
help="Term stopwords JSON file",
)
parser.add_argument(
"--context",
type=Path,
default=repo_root / "configs" / "filter_context.personal.json",
help="Personal filter context JSON file",
)
parser.add_argument(
"--policy",
type=Path,
default=repo_root / "configs" / "term_cleanup_policy.json",
help="Keyword cleanup policy JSON file",
)
parser.add_argument(
"--watchlist",
type=Path,
default=repo_root / "configs" / "term_watchlist.json",
help="Tracked watch terms JSON file",
)
parser.add_argument(
"--change-log",
type=Path,
default=repo_root / "configs" / "term_change_log.json",
help="Applied keyword cleanup change log JSON file",
)
parser.add_argument("--days", type=int, default=7, help="How many recent daily files to include")
parser.add_argument("--top", type=int, default=50, help="How many top global terms to include")
parser.add_argument(
"--output",
type=Path,
default=repo_root / "outputs" / "term_index" / "review" / "keyword-cleanup-bundle.json",
help="Output JSON file",
)
args = parser.parse_args()
stats_payload = _load_json(args.stats) if args.stats.exists() else {"terms": []}
stats_terms = stats_payload.get("terms") if isinstance(stats_payload, dict) else []
if not isinstance(stats_terms, list):
stats_terms = []
aliases = _load_mapping(args.aliases)
stopwords = _load_list(args.stopwords)
interest_keywords = _load_interest_keywords(args.context)
policy = _load_policy(args.policy)
watchlist = _load_watchlist(args.watchlist)
change_log = _load_change_log(args.change_log)
recent_daily_paths = sorted(args.daily_dir.glob("*.json"))[-args.days :] if args.daily_dir.exists() else []
recent_daily: list[dict[str, Any]] = []
recent_counter: defaultdict[str, int] = defaultdict(int)
for path in recent_daily_paths:
payload = _load_json(path)
if not isinstance(payload, dict):
continue
terms = payload.get("terms")
if not isinstance(terms, list):
terms = []
compact_terms = []
for term in terms:
if not isinstance(term, dict):
continue
normalized = term.get("normalized_term")
count = term.get("count")
if not isinstance(normalized, str) or not isinstance(count, int):
continue
compact_terms.append({"term": normalized, "count": count})
recent_counter[normalized] += count
recent_daily.append(
{
"date": payload.get("date"),
"candidate_count": payload.get("candidate_count"),
"top_terms": compact_terms[:20],
}
)
interest_set = _casefold_set(interest_keywords)
stopword_set = _casefold_set(stopwords)
alias_keys = _casefold_set(list(aliases.keys()))
alias_values = _casefold_set(list(aliases.values()))
watch_set = _casefold_set([str(item.get("term", "")) for item in watchlist])
top_global_terms = []
for item in stats_terms[: args.top]:
if not isinstance(item, dict):
continue
term = item.get("term")
if not isinstance(term, str):
continue
folded = term.strip().casefold()
top_global_terms.append(
{
"term": term,
"total_count": item.get("total_count"),
"days_seen": item.get("days_seen"),
"first_seen": item.get("first_seen"),
"last_seen": item.get("last_seen"),
"in_interest_keywords": folded in interest_set,
"is_stopword": folded in stopword_set,
"is_alias_source": folded in alias_keys,
"is_alias_target": folded in alias_values,
"in_watchlist": folded in watch_set,
"recent_count": recent_counter.get(term, 0),
}
)
uncovered_terms = [
item for item in top_global_terms if not item["in_interest_keywords"] and not item["is_stopword"]
][:20]
interest_thresholds = policy.get("interest_keyword_review") if isinstance(policy, dict) else {}
watch_thresholds = policy.get("watch_term_review") if isinstance(policy, dict) else {}
interest_review_candidates = [
{
"term": item["term"],
"total_count": item["total_count"],
"days_seen": item["days_seen"],
"reason": (
"Meets the configured interest-keyword review threshold and is not yet covered "
"by interest keywords or stopwords."
),
}
for item in uncovered_terms
if not item["in_watchlist"] and _meets_min_thresholds(item, interest_thresholds)
][:20]
watch_review_candidates = [
{
"term": item["term"],
"total_count": item["total_count"],
"days_seen": item["days_seen"],
"reason": (
"Falls into the configured watch-term review range and should be observed "
"before promotion into interest keywords."
),
}
for item in uncovered_terms
if not item["in_watchlist"]
and not _meets_min_thresholds(item, interest_thresholds)
and _within_watch_thresholds(item, watch_thresholds)
][:20]
recent_hot_terms = sorted(
({"term": term, "recent_count": count} for term, count in recent_counter.items()),
key=lambda item: (-item["recent_count"], item["term"].casefold(), item["term"]),
)[:20]
bundle = {
"generated_at": datetime.now(tz=UTC).isoformat(),
"days": args.days,
"top": args.top,
"sources": {
"stats": str(args.stats),
"daily_dir": str(args.daily_dir),
"aliases": str(args.aliases),
"stopwords": str(args.stopwords),
"context": str(args.context),
"policy": str(args.policy),
"watchlist": str(args.watchlist),
"change_log": str(args.change_log),
},
"policy": policy,
"current_config": {
"alias_count": len(aliases),
"stopword_count": len(stopwords),
"interest_keyword_count": len(interest_keywords),
"watch_term_count": len(watchlist),
"aliases": aliases,
"stopwords": stopwords,
"interest_keywords": interest_keywords,
"watchlist": watchlist,
},
"change_log_tail": change_log[-10:],
"top_global_terms": top_global_terms,
"recent_daily": recent_daily,
"recent_hot_terms": recent_hot_terms,
"uncovered_terms": uncovered_terms,
"governance_hints": {
"interest_review_candidates": interest_review_candidates,
"watch_review_candidates": watch_review_candidates,
},
}
_save_json(args.output, bundle)
print(f"Saved keyword cleanup bundle to {args.output}")
if __name__ == "__main__":
main()
+228
View File
@@ -0,0 +1,228 @@
from __future__ import annotations
import json
from collections import Counter
from datetime import UTC, date, datetime
from pathlib import Path
from typing import Iterable
from summary_mcp.models.article_candidate import OpenClawCandidateInput
from summary_mcp.models.keyword_index import (
DailyKeywordIndex,
DailyKeywordTerm,
KeywordStat,
KeywordStatsIndex,
)
REPO_ROOT = Path(__file__).resolve().parents[3]
DATA_ROOT = REPO_ROOT / "data" / "term_index"
DEFAULT_DAILY_DIR = DATA_ROOT / "daily"
DEFAULT_STATS_PATH = DATA_ROOT / "term_stats.json"
DEFAULT_ALIASES_PATH = REPO_ROOT / "configs" / "term_aliases.json"
DEFAULT_STOPWORDS_PATH = REPO_ROOT / "configs" / "term_stopwords.json"
DEFAULT_INCLUDE_DECISIONS = {"keep", "review"}
def _load_json(path: Path) -> object:
return json.loads(path.read_text(encoding="utf-8-sig"))
def _save_json(path: Path, payload: dict | list) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
def _term_key(value: str) -> str:
return value.strip().casefold()
def load_term_aliases(path: Path | None = None) -> dict[str, str]:
aliases_path = path or DEFAULT_ALIASES_PATH
if not aliases_path.exists():
return {}
payload = _load_json(aliases_path)
if not isinstance(payload, dict):
raise RuntimeError("Term aliases file must contain a JSON object.")
aliases: dict[str, str] = {}
for raw_key, raw_value in payload.items():
if not isinstance(raw_key, str) or not isinstance(raw_value, str):
continue
normalized_key = _term_key(raw_key)
normalized_value = raw_value.strip()
if not normalized_key or not normalized_value:
continue
aliases[normalized_key] = normalized_value
return aliases
def load_term_stopwords(path: Path | None = None) -> set[str]:
stopwords_path = path or DEFAULT_STOPWORDS_PATH
if not stopwords_path.exists():
return set()
payload = _load_json(stopwords_path)
if not isinstance(payload, list):
raise RuntimeError("Term stopwords file must contain a JSON array.")
values: set[str] = set()
for item in payload:
if not isinstance(item, str):
continue
normalized = _term_key(item)
if normalized:
values.add(normalized)
return values
def normalize_keyword(
keyword: str,
*,
aliases: dict[str, str],
stopwords: set[str],
) -> str | None:
raw_value = keyword.strip()
if not raw_value:
return None
aliased_value = aliases.get(_term_key(raw_value), raw_value).strip()
if not aliased_value:
return None
if _term_key(aliased_value) in stopwords:
return None
return aliased_value
def build_daily_keyword_index(
candidates: Iterable[OpenClawCandidateInput],
*,
for_date: date,
digest_id: str,
source: str = "openclaw_delivery_payload",
include_decisions: set[str] | None = None,
aliases: dict[str, str] | None = None,
stopwords: set[str] | None = None,
) -> DailyKeywordIndex:
allowed_decisions = include_decisions or DEFAULT_INCLUDE_DECISIONS
resolved_aliases = aliases or {}
resolved_stopwords = stopwords or set()
term_counter: Counter[str] = Counter()
candidate_count = 0
for candidate in candidates:
if candidate.selection_decision not in allowed_decisions:
continue
candidate_count += 1
seen_for_candidate: set[str] = set()
for keyword in candidate.keywords:
normalized = normalize_keyword(
keyword,
aliases=resolved_aliases,
stopwords=resolved_stopwords,
)
if normalized is None or normalized in seen_for_candidate:
continue
seen_for_candidate.add(normalized)
term_counter[normalized] += 1
terms = [
DailyKeywordTerm(term=term, normalized_term=term, count=count)
for term, count in sorted(term_counter.items(), key=lambda item: (-item[1], item[0].casefold(), item[0]))
]
return DailyKeywordIndex(
schema_version="v1",
date=for_date,
source=source,
digest_id=digest_id,
generated_at=datetime.now(tz=UTC),
candidate_count=candidate_count,
terms=terms,
)
def read_daily_keyword_index(path: Path) -> DailyKeywordIndex:
return DailyKeywordIndex.model_validate(_load_json(path))
def rebuild_keyword_stats(*, daily_dir: Path = DEFAULT_DAILY_DIR) -> KeywordStatsIndex:
term_totals: dict[str, int] = {}
first_seen: dict[str, date] = {}
last_seen: dict[str, date] = {}
days_seen: dict[str, int] = {}
if daily_dir.exists():
for path in sorted(daily_dir.glob("*.json")):
daily_index = read_daily_keyword_index(path)
seen_today: set[str] = set()
for term in daily_index.terms:
normalized = term.normalized_term
term_totals[normalized] = term_totals.get(normalized, 0) + term.count
if normalized not in first_seen or daily_index.date < first_seen[normalized]:
first_seen[normalized] = daily_index.date
if normalized not in last_seen or daily_index.date > last_seen[normalized]:
last_seen[normalized] = daily_index.date
if normalized not in seen_today:
days_seen[normalized] = days_seen.get(normalized, 0) + 1
seen_today.add(normalized)
terms = [
KeywordStat(
term=term,
total_count=term_totals[term],
days_seen=days_seen[term],
first_seen=first_seen[term],
last_seen=last_seen[term],
)
for term in sorted(term_totals, key=lambda item: (-term_totals[item], item.casefold(), item))
]
return KeywordStatsIndex(
schema_version="v1",
generated_at=datetime.now(tz=UTC),
terms=terms,
)
def persist_keyword_indexes(
candidates: Iterable[OpenClawCandidateInput],
*,
for_date: date,
digest_id: str,
source: str = "openclaw_delivery_payload",
daily_dir: Path = DEFAULT_DAILY_DIR,
stats_path: Path = DEFAULT_STATS_PATH,
aliases_path: Path | None = None,
stopwords_path: Path | None = None,
include_decisions: set[str] | None = None,
) -> dict[str, object]:
aliases = load_term_aliases(aliases_path)
stopwords = load_term_stopwords(stopwords_path)
daily_index = build_daily_keyword_index(
candidates,
for_date=for_date,
digest_id=digest_id,
source=source,
include_decisions=include_decisions,
aliases=aliases,
stopwords=stopwords,
)
daily_path = daily_dir / f"{for_date.isoformat()}.json"
_save_json(daily_path, daily_index.model_dump(mode="json"))
stats_index = rebuild_keyword_stats(daily_dir=daily_dir)
_save_json(stats_path, stats_index.model_dump(mode="json"))
return {
"daily_output": str(daily_path),
"stats_output": str(stats_path),
"source": source,
"candidate_count": daily_index.candidate_count,
"term_count": len(daily_index.terms),
"top_terms": [term.model_dump(mode="json") for term in daily_index.terms[:10]],
}
+5
View File
@@ -22,6 +22,7 @@ from .daily_digest import (
DailyDigestSourceRef, DailyDigestSourceRef,
DailyDigestStats, DailyDigestStats,
) )
from .keyword_index import DailyKeywordIndex, DailyKeywordTerm, KeywordStat, KeywordStatsIndex
from .openclaw_delivery import ( from .openclaw_delivery import (
OpenClawDeliveryPayload, OpenClawDeliveryPayload,
OpenClawDeliveryStats, OpenClawDeliveryStats,
@@ -40,7 +41,11 @@ __all__ = [
"DailyDigestSection", "DailyDigestSection",
"DailyDigestSourceRef", "DailyDigestSourceRef",
"DailyDigestStats", "DailyDigestStats",
"DailyKeywordIndex",
"DailyKeywordTerm",
"DigestSectionHint", "DigestSectionHint",
"KeywordStat",
"KeywordStatsIndex",
"KnowledgeDecision", "KnowledgeDecision",
"OpenClawCandidateInput", "OpenClawCandidateInput",
"OpenClawDeliveryPayload", "OpenClawDeliveryPayload",
+35
View File
@@ -0,0 +1,35 @@
from __future__ import annotations
from datetime import date, datetime
from pydantic import BaseModel, Field
class DailyKeywordTerm(BaseModel):
term: str
normalized_term: str
count: int = Field(ge=1)
class DailyKeywordIndex(BaseModel):
schema_version: str = "v1"
date: date
source: str
digest_id: str
generated_at: datetime
candidate_count: int = Field(default=0, ge=0)
terms: list[DailyKeywordTerm] = Field(default_factory=list)
class KeywordStat(BaseModel):
term: str
total_count: int = Field(default=0, ge=0)
days_seen: int = Field(default=0, ge=0)
first_seen: date
last_seen: date
class KeywordStatsIndex(BaseModel):
schema_version: str = "v1"
generated_at: datetime
terms: list[KeywordStat] = Field(default_factory=list)
@@ -6,6 +6,7 @@ from datetime import UTC, date, datetime
from pathlib import Path from pathlib import Path
from typing import Any from typing import Any
from summary_mcp.core.keyword_index import persist_keyword_indexes
from summary_mcp.core.pipeline import extract_content from summary_mcp.core.pipeline import extract_content
from summary_mcp.core.summary_loop import resolve_llm_settings, run_loop_payload from summary_mcp.core.summary_loop import resolve_llm_settings, run_loop_payload
from summary_mcp.filters.engine import evaluate_filter_rules, load_filter_rules from summary_mcp.filters.engine import evaluate_filter_rules, load_filter_rules
@@ -26,8 +27,13 @@ from summary_mcp.models.summary_io import ExtractionInput
REPO_ROOT = Path(__file__).resolve().parents[3] REPO_ROOT = Path(__file__).resolve().parents[3]
OUTPUT_ROOT = REPO_ROOT / "outputs" OUTPUT_ROOT = REPO_ROOT / "outputs"
FRESHRSS_OUTPUT_ROOT = OUTPUT_ROOT / "freshrss" FRESHRSS_OUTPUT_ROOT = OUTPUT_ROOT / "freshrss"
DATA_ROOT = REPO_ROOT / "data" / "term_index"
DEFAULT_PROMPT_PATH = OUTPUT_ROOT / "prompts" / "llm-summary-prompt.txt" DEFAULT_PROMPT_PATH = OUTPUT_ROOT / "prompts" / "llm-summary-prompt.txt"
DEFAULT_RULES_PATH = REPO_ROOT / "configs" / "filter_rules.json" DEFAULT_RULES_PATH = REPO_ROOT / "configs" / "filter_rules.json"
DEFAULT_TERM_ALIASES_PATH = REPO_ROOT / "configs" / "term_aliases.json"
DEFAULT_TERM_STOPWORDS_PATH = REPO_ROOT / "configs" / "term_stopwords.json"
DEFAULT_TERM_DAILY_DIR = DATA_ROOT / "daily"
DEFAULT_TERM_STATS_PATH = DATA_ROOT / "term_stats.json"
def _save_json(path: Path, payload: dict[str, Any] | list[Any]) -> None: def _save_json(path: Path, payload: dict[str, Any] | list[Any]) -> None:
@@ -237,6 +243,17 @@ def run_freshrss_pipeline(
) )
_save_json(delivery_output, delivery_payload.model_dump(mode="json")) _save_json(delivery_output, delivery_payload.model_dump(mode="json"))
keyword_index_result = persist_keyword_indexes(
delivery_payload.candidates,
for_date=delivery_payload.date,
digest_id=delivery_payload.run_id,
source="openclaw_delivery_payload",
daily_dir=DEFAULT_TERM_DAILY_DIR,
stats_path=DEFAULT_TERM_STATS_PATH,
aliases_path=DEFAULT_TERM_ALIASES_PATH,
stopwords_path=DEFAULT_TERM_STOPWORDS_PATH,
)
marked_count = 0 marked_count = 0
if mark_read and delivered_item_ids: if mark_read and delivered_item_ids:
client.mark_items_as_read(auth_token=auth_token, item_ids=delivered_item_ids) client.mark_items_as_read(auth_token=auth_token, item_ids=delivered_item_ids)
@@ -259,6 +276,7 @@ def run_freshrss_pipeline(
"debug_artifacts": debug_artifacts, "debug_artifacts": debug_artifacts,
"raw_output": str(raw_output), "raw_output": str(raw_output),
"delivery_output": str(delivery_output), "delivery_output": str(delivery_output),
"keyword_index": keyword_index_result,
"status_counts": status_counts, "status_counts": status_counts,
"items": item_reports, "items": item_reports,
} }
@@ -270,6 +288,7 @@ def run_freshrss_pipeline(
"raw_output": str(raw_output), "raw_output": str(raw_output),
"delivery_output": str(delivery_output), "delivery_output": str(delivery_output),
"report_output": str(report_output), "report_output": str(report_output),
"keyword_index": keyword_index_result,
"pulled_count": len(items), "pulled_count": len(items),
"delivered_count": len(delivered_candidates), "delivered_count": len(delivered_candidates),
"marked_read_count": marked_count, "marked_read_count": marked_count,