Compare commits

...
4 Commits
Author SHA1 Message Date
root 416414ae1d chore: normalize term stopwords ordering 2026-04-14 09:51:57 +08:00
root f3e7488fc8 docs: add reference templates for IMA notes and public digest 2026-04-14 09:51:57 +08:00
root e3663f681d feat: add keyword cleanup docs, skill updates, and delivery compatibility fix 2026-04-14 09:51:57 +08:00
root a06f2a1d08 chore: update term configs, skill docs and Python 3.10 compat
- term_change_log: record watch term add history
- term_watchlist: add initial watch terms from cleanup review
- term_aliases: minor update
- filter_context.personal: reorganize interest keywords
- keyword-cleanup-review skill: clarify JSON as formal artifact, markdown as temp review copy
- openclaw_delivery: add Python <3.11 UTC import compat
- daily-keyword-index-design: update suggestion artifact semantics
2026-04-14 09:51:57 +08:00
15 changed files with 1972 additions and 41 deletions
+26 -21
View File
@@ -23,34 +23,39 @@
"前沿科技" "前沿科技"
], ],
"interest_keywords": [ "interest_keywords": [
"Java", "Agent",
"Go", "Agent Skills",
"Python", "AgentScope",
"Spring", "AI Agent",
"AliSQL",
"Claude Code",
"DeepSeek",
"FastAPI", "FastAPI",
"Gin", "Gin",
"Go",
"gRPC", "gRPC",
"MySQL", "Java",
"PostgreSQL",
"Redis",
"Kafka", "Kafka",
"微服务",
"可观测性",
"Kubernetes", "Kubernetes",
"云原生",
"AI Agent",
"Agent",
"LLM", "LLM",
"RAG",
"MCP", "MCP",
"Prompt Engineering", "MySQL",
"Workflow", "MySQL复制延迟",
"向量数据库",
"知识库",
"OpenAI", "OpenAI",
"DeepSeek",
"OpenClaw", "OpenClaw",
"AliSQL", "PostgreSQL",
"MySQL复制延迟" "Prompt Engineering",
"Python",
"RAG",
"ReActAgent",
"Redis",
"Spring",
"SubAgent",
"Workflow",
"云原生",
"可观测性",
"向量数据库",
"微服务",
"知识库"
] ]
} }
+1 -1
View File
@@ -2,4 +2,4 @@
"AI助手": "AI Agent", "AI助手": "AI Agent",
"图文RAG": "RAG", "图文RAG": "RAG",
"Prompt架构": "Prompt Engineering" "Prompt架构": "Prompt Engineering"
} }
+102 -2
View File
@@ -1,4 +1,104 @@
{ {
"schema_version": "v1", "schema_version": "v1",
"entries": [] "entries": [
} {
"applied_at": "2026-04-08T02:34:14.194320Z",
"action": "add_watch_term",
"term": "A2A",
"reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=0.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json",
"suggestion_date": "2026-04-08",
"based_on_days": 7
},
{
"applied_at": "2026-04-08T02:34:14.194320Z",
"action": "add_watch_term",
"term": "Agentic Loop",
"reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=1.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json",
"suggestion_date": "2026-04-08",
"based_on_days": 7
},
{
"applied_at": "2026-04-08T02:34:14.194320Z",
"action": "add_watch_term",
"term": "AI Gateway",
"reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=1.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json",
"suggestion_date": "2026-04-08",
"based_on_days": 7
},
{
"applied_at": "2026-04-08T02:34:14.194320Z",
"action": "add_watch_term",
"term": "Claude Skills",
"reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=1.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json",
"suggestion_date": "2026-04-08",
"based_on_days": 7
},
{
"applied_at": "2026-04-08T02:34:14.194320Z",
"action": "add_watch_term",
"term": "Cron",
"reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=0.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json",
"suggestion_date": "2026-04-08",
"based_on_days": 7
},
{
"applied_at": "2026-04-08T02:34:14.194320Z",
"action": "add_watch_term",
"term": "CoPaw",
"reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=0.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json",
"suggestion_date": "2026-04-08",
"based_on_days": 7
},
{
"applied_at": "2026-04-08T02:34:14.194320Z",
"action": "add_interest_keyword",
"term": "Claude Code",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=7, days_seen=4, recent_count=4.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json",
"suggestion_date": "2026-04-08",
"based_on_days": 7
},
{
"applied_at": "2026-04-08T02:34:14.194320Z",
"action": "add_interest_keyword",
"term": "Agent Skills",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=6, days_seen=3, recent_count=0.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json",
"suggestion_date": "2026-04-08",
"based_on_days": 7
},
{
"applied_at": "2026-04-08T02:34:14.194320Z",
"action": "add_interest_keyword",
"term": "SubAgent",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=4, days_seen=4, recent_count=2.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json",
"suggestion_date": "2026-04-08",
"based_on_days": 7
},
{
"applied_at": "2026-04-08T02:34:14.194320Z",
"action": "add_interest_keyword",
"term": "AgentScope",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=4, days_seen=4, recent_count=1.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json",
"suggestion_date": "2026-04-08",
"based_on_days": 7
},
{
"applied_at": "2026-04-08T02:34:14.194320Z",
"action": "add_interest_keyword",
"term": "ReActAgent",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=3, days_seen=3, recent_count=1.",
"suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json",
"suggestion_date": "2026-04-08",
"based_on_days": 7
}
]
}
+3 -3
View File
@@ -1,6 +1,6 @@
[ [
"奋斗文化",
"小银", "小银",
"银行客户经理", "银行客户经理",
"飞盘物理", "飞盘物理"
"奋斗文化" ]
]
+46 -3
View File
@@ -1,5 +1,48 @@
{ {
"schema_version": "v1", "schema_version": "v1",
"updated_at": "2026-03-27T00:00:00Z", "updated_at": "2026-04-08T02:34:14.194320Z",
"terms": [] "terms": [
} {
"term": "A2A",
"added_at": "2026-04-08T02:34:14.194320Z",
"source": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json",
"reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=0.",
"status": "watching"
},
{
"term": "Agentic Loop",
"added_at": "2026-04-08T02:34:14.194320Z",
"source": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json",
"reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=1.",
"status": "watching"
},
{
"term": "AI Gateway",
"added_at": "2026-04-08T02:34:14.194320Z",
"source": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json",
"reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=1.",
"status": "watching"
},
{
"term": "Claude Skills",
"added_at": "2026-04-08T02:34:14.194320Z",
"source": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json",
"reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=1.",
"status": "watching"
},
{
"term": "CoPaw",
"added_at": "2026-04-08T02:34:14.194320Z",
"source": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json",
"reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=0.",
"status": "watching"
},
{
"term": "Cron",
"added_at": "2026-04-08T02:34:14.194320Z",
"source": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json",
"reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=0.",
"status": "watching"
}
]
}
+17 -4
View File
@@ -326,8 +326,13 @@ LLM 可以帮助做清洗建议,但不适合直接维护主词元库。
skill 不直接修改配置文件,而是生成建议文件,例如: skill 不直接修改配置文件,而是生成建议文件,例如:
- `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.md`
- `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json` - `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json`
- `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.md`
其中建议语义为:
- JSON 是 review / apply 之间的唯一正式建议产物
- Markdown 是人工临时审阅展示稿,不是长期真相来源
低复杂治理层建议补充三类输入: 低复杂治理层建议补充三类输入:
@@ -403,8 +408,9 @@ skill 不直接修改配置文件,而是生成建议文件,例如:
2. 程序更新 `daily/YYYY-MM-DD.json` 2. 程序更新 `daily/YYYY-MM-DD.json`
3. 程序更新 `term_stats.json` 3. 程序更新 `term_stats.json`
4. 每周或人工触发一次词元清洗 skill 4. 每周或人工触发一次词元清洗 skill
5. skill 输出建议 5. skill 生成 suggestions JSON(正式建议产物)
6. 人工确认后再更新配置文件 6. 如需要人工阅读,再临时生成 Markdown 展示稿
7. 人工确认后再更新配置文件
## 14. 与规则引擎的关系 ## 14. 与规则引擎的关系
@@ -445,7 +451,8 @@ skill 不直接修改配置文件,而是生成建议文件,例如:
再补治理层: 再补治理层:
- 增加词元清洗 skill - 增加词元清洗 skill
- 输出建议文件 - 输出建议文件(以 JSON 为正式产物)
- Markdown 仅作为按需生成的人工展示层
- 人工确认后更新配置 - 人工确认后更新配置
### Phase 3 ### Phase 3
@@ -459,3 +466,9 @@ skill 不直接修改配置文件,而是生成建议文件,例如:
## 16. 一句话结论 ## 16. 一句话结论
这套设计选择“只统计日报中的 `keywords`,由程序维护轻量词元库,再由独立 skill 周期性做清洗建议”,目的是在控制数据规模的前提下,为规则配置和长期兴趣演化提供稳定、可审计、可扩展的基础设施。 这套设计选择“只统计日报中的 `keywords`,由程序维护轻量词元库,再由独立 skill 周期性做清洗建议”,目的是在控制数据规模的前提下,为规则配置和长期兴趣演化提供稳定、可审计、可扩展的基础设施。
补充的产物策略是:
- facts/state 长期保留
- suggestions JSON 作为正式建议产物短期保留
- review bundle 与 Markdown 展示稿降级为临时工作文件 / 展示层
@@ -0,0 +1,163 @@
# reader MCP workflow service formalization summary (2026-04-07)
## Overview
On 2026-04-07, the reader project was formally advanced from a script-first integration model into a reader-centric MCP workflow service model.
The key shift is:
- before: OpenClaw primarily relied on long CLI / exec flows and direct output-path stitching
- now: reader exposes a formal workflow-oriented MCP surface with run-state, status queries, result reads, and minimal recovery
This document records the main outcomes and commits for the first formalization phase.
---
## Completed capability set
### 1. Run-state persistence
Commit:
- `72a6853` — `Add run-state persistence for FreshRSS pipeline`
Delivered:
- `run-state.json`
- `RunState / StageState / ArtifactRecord`
- stage-level state persistence for the FreshRSS pipeline
### 2. Architecture / implementation docs
Commit:
- `7563aa8` — `docs: add reader MCP architecture and implementation plan`
Delivered:
- architecture design
- implementation plan
- TODO-driven collaboration model
### 3. MCP run-status query tools
Commit:
- `d9173fb` — `feat: add MCP run status query tools`
Delivered:
- `get_run_status`
- `list_runs`
- `list_run_artifacts`
### 4. MCP result-read tools
Commit:
- `4a02894` — `Add MCP delivery payload and run report queries`
Delivered:
- `get_delivery_payload`
- `get_run_report`
### 5. Minimal resume design
Commit:
- `d91cdbc` — `docs: narrow resume_run minimal recovery design`
Delivered:
- narrowed design for `resume_run`
- explicit supported / unsupported recovery points
### 6. Minimal `resume_run`
Commit:
- `c622bc6` — `Implement minimal resume_run for freshrss runs`
Delivered:
- minimal `resume_run`
- supports only freshrss runs with `run-state.json`
- supports only recent resumable points
- explicitly rejects `fetch_feed` and `extract_articles`
### 7. Formal handoff / workflow docs
Commits:
- `4f219ef` — `docs: formalize reader MCP workflow service handoff`
- `2df0af5` — `docs: add openclaw orchestration flow for reader MCP`
Delivered:
- formal handoff aligned to actual implementation
- OpenClaw orchestration runbook
- explicit rule that OpenClaw should stop hand-stitching reader paths in the normal production flow
---
## Current formal MCP workflow surface
The current reader MCP workflow surface now includes:
- `run_freshrss_openclaw_pipeline`
- `get_run_status`
- `list_runs`
- `list_run_artifacts`
- `get_delivery_payload`
- `get_run_report`
- `resume_run` (minimal version)
---
## Current boundary
reader is now the upstream workflow engine for:
- FreshRSS pull
- extraction
- summary
- filter
- payload generation
- run-state persistence
- result read
- minimal recovery
OpenClaw / skill remains responsible for:
- digest markdown generation
- Hugo publishing
- chat reporting
- user confirmation
- IMA orchestration
---
## Current limitations
The first formalization phase is complete, but some constraints remain:
- `resume_run` is still minimal and does not support arbitrary stage re-entry
- historical runs without `run-state.json` are not formally recoverable
- some very old runs may still require conservative artifact/path discovery
- `rerun_stage` is not implemented
- deeper runtime consolidation of `run_freshrss_openclaw_pipeline` can still be improved later
---
## Practical conclusion
The reader project should now be treated as a formal MCP workflow service rather than as a long-running CLI-first integration point.
For normal production orchestration:
- start via MCP
- observe via MCP status tools
- read results via MCP result tools
- use `resume_run` only within the documented minimal recovery range
- keep CLI for debug / fallback only
+277
View File
@@ -0,0 +1,277 @@
# reader MCP Docker 部署计划
## 1. 目标
将 reader 作为正式 MCP workflow service 以 Docker 方式部署,满足以下原则:
1. 服务运行在容器内
2. 运行态与产物必须外置挂载,不闷在容器内
3. 配置统一记录在 `.env`
4. 读写行为与当前仓库约定保持一致
5. OpenClaw 后续可将该服务作为正式上游 MCP 使用
---
## 2. 部署原则
### 2.1 容器职责
容器只负责:
- 提供 reader MCP 服务运行环境
- 加载 reader 代码与依赖
- 读取挂载进来的配置与状态目录
- 对外暴露 MCP 服务入口
### 2.2 宿主机职责
宿主机负责持久化:
- 配置文件
- 运行态
- outputs 产物
- 数据目录
- configs
### 2.3 配置收口原则
所有环境配置统一放在 `.env`,避免:
- 零散写在 compose 内
- 零散写在 shell 命令里
- 零散写在 OpenClaw skill 里
---
## 3. 建议部署目录
建议在 reader 仓库内准备标准部署结构:
```text
/home/ubuntu/zhu/github/reader/
Dockerfile
docker-compose.yml
.env
outputs/
data/
configs/
knowledge-base/
```
说明:
- `Dockerfile`:构建 reader MCP 服务镜像
- `docker-compose.yml`:单服务部署编排
- `.env`:统一环境变量
- `outputs/`:产物、run-state、digest、payload 等外置持久化
- `data/`:term index 等数据外置持久化
- `configs/`:reader 运行配置外置持久化
- `knowledge-base/`:如当前 reader/skill 仍会依赖本地知识目录,可继续挂载
---
## 4. 必须挂载的目录 / 文件
### 必须挂载
- `.env`
- `outputs/`
- `data/`
- `configs/`
### 建议挂载
- `knowledge-base/`
### 通常不必挂载
- `docs/`
- `plans/`
- `.git/`
---
## 5. `.env` 统一配置建议
至少应包含以下配置:
### FreshRSS
- `FRESHRSS_API_BASE_URL`
- `FRESHRSS_USERNAME`
- `FRESHRSS_API_PASSWORD`
### 主 LLM
- `LLM_API_URL`
- `LLM_API_KEY`
- `LLM_MODEL`
### 单篇总结专用 LLM(如已使用)
- `ARTICLE_SUMMARY_API_URL`
- `ARTICLE_SUMMARY_API_KEY`
- `ARTICLE_SUMMARY_MODEL`
### IMA(如 reader / skill 仍依赖这些配置约定)
- `IMA_DAILY_KNOWLEDGE_BASE_ID`
- `IMA_DAILY_KNOWLEDGE_BASE_NAME`
### 运行控制
- `PYTHONUNBUFFERED=1`
- 视需要增加日志级别等配置
原则:
- 所有会影响服务行为的环境项,都优先进入 `.env`
- compose 文件只引用 `.env`,不在 compose 里硬编码业务参数
---
## 6. Dockerfile 设计建议
### 目标
- 使用 Python 3.11
- 安装 reader 依赖
- 默认启动 MCP 服务入口
### 建议思路
1. 基于 `python:3.11-slim`
2. 设置工作目录到 `/app`
3. 复制仓库代码
4. 安装依赖(如 `pip install -e .`)
5. 默认启动 reader MCP 服务
### 启动入口
优先使用当前正式服务入口,例如:
- `summary-mcp`
如果后续 reader 明确切换到别的稳定入口,再同步更新。
---
## 7. docker-compose 设计建议
建议先保持单服务简单结构,例如:
- service 名称:`reader-mcp`
- `env_file: .env`
- 挂载:
- `./outputs:/app/outputs`
- `./data:/app/data`
- `./configs:/app/configs`
- `./knowledge-base:/app/knowledge-base`(如需要)
- `./.env:/app/.env:ro`(可选,若程序直接读取文件)
- `restart: unless-stopped`
如果当前 MCP 服务是 stdio 型而不是 HTTP 型,需要进一步明确:
- 它是由 OpenClaw 以本地进程方式拉起
- 还是以常驻 sidecar / gateway adapter 方式挂接
因此 compose 的最终 command 需要结合实际接入方式确认。
---
## 8. 部署前确认项
在正式执行前,需要先确认以下问题:
### 8.1 MCP 连接方式
必须确认 reader MCP 服务的正式接入方式是:
1. **stdio 型**:OpenClaw/调用方本地拉起进程
2. **HTTP/SSE 型**:服务常驻监听端口,OpenClaw 远程连接
这会直接影响:
- Docker command
- 是否需要端口映射
- OpenClaw 接入配置
### 8.2 当前 `summary-mcp` 的服务形态
需要确认:
- 现有 `summary-mcp` 是 FastMCP stdio 默认模式
- 还是已有可直接 HTTP 化的运行方式
在这点没确认前,不要盲目写死端口暴露方案。
### 8.3 OpenClaw 侧接入点
部署完成后,还需要明确 OpenClaw 将如何引用该 MCP 服务:
- 本机命令型 MCP
- Docker 内服务桥接
- 或其它现有 OpenClaw MCP 配置方式
---
## 9. 执行顺序(建议)
### Phase A:部署方案落地
1. 确认 MCP 服务连接方式(stdio / HTTP)
2. 确认最终 Dockerfile 启动命令
3. 确认 compose 结构与挂载目录
4. 整理 `.env` 字段
### Phase B:容器化实现
1. 新建/更新 `Dockerfile`
2. 新建/更新 `docker-compose.yml`
3. 检查 `.dockerignore`
4. 核对路径是否与仓库内当前代码一致
### Phase C:本地部署验证
1. `docker compose build`
2. `docker compose up -d`
3. 验证服务启动
4. 验证容器外 `outputs/`、`data/` 等是否正常落盘
### Phase D:OpenClaw 接入验证
1. 让 OpenClaw 通过正式 MCP 路径连接 reader
2. 真跑一轮:
- run
- status
- payload
- report
3. 如有需要,验证一次最小 `resume_run`
---
## 10. 当前不在本轮范围内的事
本轮部署计划不直接处理:
- `rerun_stage`
- 更复杂的后台任务系统
- 多实例部署
- 横向扩展
- 生产告警体系
本轮只做:
- 单实例
- Docker 化
- 配置收口
- 挂载持久化
- OpenClaw 可正式接入
---
## 11. 一句话结论
reader 的下一步不是继续堆内部接口,而是:
**以 Docker 正式部署成 MCP workflow service,配置进 `.env`,状态和产物目录挂载到宿主机,然后由 OpenClaw 按正式 MCP 编排路径真实接入和验证。**
@@ -0,0 +1,257 @@
# keyword-cleanup 产物精简方案 v1
## 1. 背景
当前 keyword-cleanup 治理链已经从“只有 review bundle”演进到:
- term stats / daily term index
- review bundle
- suggestions json
- suggestions markdown
- apply -> config / watchlist / change_log
这说明链路已经打通,但也带来一个新问题:
> 中间产物偏多,容易让治理系统本身比被治理对象更重。
本方案的目标不是回退功能,而是重新划分:
- 哪些产物是长期资产
- 哪些产物只是决策输入
- 哪些产物只是运行时工作文件
从而把 keyword-cleanup 收敛成一个更轻的治理辅助层,而不是继续长成一个复杂子系统。
---
## 2. 设计目标
本轮精简目标:
1. 保留真正有长期价值的事实层与状态层数据
2. 保留唯一正式建议产物,用于 review / apply
3. 将 review bundle 和 markdown 展示稿降级为临时产物
4. 让主链路收敛到:
`stats -> suggestions json -> apply -> config`
而不是长期依赖:
`stats -> bundle -> suggestions json + md -> review -> apply`
---
## 3. 产物分层建议
### 3.1 长期保留:事实层
这些文件是系统长期事实基础,应继续长期保留:
- `data/term_index/daily/YYYY-MM-DD.json`
- `data/term_index/term_stats.json`
原因:
- daily 文件代表每日聚合观察结果
- global stats 是治理决策的核心事实来源
- 二者共同构成 term governance 的历史依据
### 3.2 长期保留:状态层
这些文件代表治理系统当前状态,应继续长期保留:
- `configs/filter_context.personal.json`
- `configs/term_watchlist.json`
- `configs/term_aliases.json`
- `configs/term_stopwords.json`
- `configs/term_change_log.json`
原因:
- 它们是已确认生效的治理结果
- 后续 reader 行为依赖这些配置
- `change_log` 负责回溯治理动作
### 3.3 短期保留:正式建议层
建议保留:
- `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json`
定位:
- 这是 review / apply 之间的唯一正式建议产物
- 机器可消费
- 可作为某次治理决策的外部依据
建议策略:
- 默认仅保留最近少量几份
- 或仅保留已经 apply 过的 suggestions JSON
- 避免无限累积所有历史 suggestions 文件
### 3.4 降级为临时产物:review bundle
建议降级:
- `outputs/term_index/review/keyword-cleanup-bundle.json`
定位:
- review 输入打包文件
- 只服务于 suggestions 生成过程
- 不属于长期治理资产
建议策略:
- 默认只保留当前最新一份
- 或迁移到更明确的 working/tmp 目录语义
- 不按日期长期积累
### 3.5 降级为临时产物:Markdown 展示稿
建议降级:
- `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.md`
定位:
- 纯人工审阅展示层
- 不是唯一真相
- 不参与 apply 逻辑
建议策略:
- 默认不长期持久化
- 需要人工审阅时临时生成
- 优先在聊天/界面中直接展示,而不是默认写成长期文件
---
## 4. 精简后的主链路
建议主链路口径收敛为:
1. 更新 daily term index
2. 更新 global term stats
3. 生成 suggestions JSON
4. 人工确认
5. apply 到 config
6. 记录 change log
即:
`stats -> suggestions json -> apply -> config`
其中:
- bundle = 内部工作层
- markdown = 展示层
- suggestions JSON = 唯一正式建议输入
---
## 5. 为什么这样收敛
### 5.1 避免中间层过多
如果 bundle / md / suggestions 都被长期持久化,就容易出现:
- 多份文件语义重叠
- 不知道谁是“准的”
- 哪些只是试跑产物,哪些是正式治理决策不清晰
### 5.2 保持系统重心正确
keyword-cleanup 的最终目的不是维护一个漂亮的 review 文件集合,而是:
- 持续积累稳定的关键词事实数据
- 让 interest/watch/alias/stopword 演化有据可依
- 让 reader 的长期偏好配置从真实日报里长出来
### 5.3 降低治理系统自身复杂度
治理系统应该比主系统更轻,而不是更重。
如果 review 产物越积越多,最终会反过来增加维护和理解成本。
---
## 6. 对现有实现的影响
本轮不要求删除已有能力,而是重新定义口径。
### 6.1 保留
- `build_review_bundle.py`
- `generate_term_cleanup_suggestions.py`
- `apply_term_suggestions.py`
### 6.2 调整口径
- `keyword-cleanup-bundle.json` 从“默认产物”降级为“临时工作文件”
- `term-cleanup-suggestions-YYYY-MM-DD.md` 从“正式产物”降级为“临时展示稿”
- `term-cleanup-suggestions-YYYY-MM-DD.json` 作为唯一正式建议产物保留
### 6.3 后续可选实现动作
- 覆盖式写入 bundle,而不是长期累积
- Markdown 按需生成,而不是默认总是落盘
- 增加清理策略,只保留最近 N 个 suggestions JSON
---
## 7. SOP 调整建议
### 7.1 review 阶段
默认步骤:
1. 生成或更新 term stats
2. 生成最新 bundle(临时)
3. 生成 suggestions JSON(正式)
4. 如需要人工阅读,再临时生成 Markdown 或直接在聊天展示
### 7.2 apply 阶段
apply 后以以下内容作为最终真相:
- config 文件当前值
- `term_change_log.json`
- 如需要,保留对应 suggestions JSON 作为决策依据
### 7.3 清理策略
建议:
- bundle:默认仅保留最新
- markdown:默认不归档
- suggestions JSON:保留少量最近记录或已应用记录
---
## 8. 非目标
本轮不做:
- 删除现有脚本
- 重写治理链路
- 一次性重构所有 review 文档
- 自动 apply 所有 suggestions
- 引入更复杂的存储系统
本轮只做一件事:
> 把 keyword-cleanup 的产物语义分清,长期保留该留的,临时化该临时的。
---
## 9. 一句话结论
keyword-cleanup 应该收敛为:
- **事实层长期保留**:daily / term_stats
- **状态层长期保留**:interest / watch / alias / stopword / change_log
- **正式建议层轻量保留**:suggestions JSON
- **中间输入层与展示层临时化**:bundle / markdown
最终目标是让 reader 的关键词治理成为一个轻量、可持续、可回溯的偏好演化机制,而不是一个不断膨胀的中间文件系统。
@@ -0,0 +1,534 @@
# keyword-cleanup-review 建议产物补齐设计
## 1. 背景与目标
当前仓库已经具备 keyword cleanup review 的大部分基础设施:
- 已有 review bundle 构建脚本 `skills/keyword-cleanup-review/scripts/build_review_bundle.py`
- 已有 term stats 与 daily term index 数据源
- 已有治理输入:`configs/term_cleanup_policy.json`、`configs/term_watchlist.json`、`configs/term_change_log.json`
- 已有建议落地脚本 `scripts/apply_term_suggestions.py`
当前缺口是:
> 缺少一层“把 `keyword-cleanup-bundle.json` 转成正式建议产物”的实现层。
也就是说,仓库现在能生成 review bundle,也能消费 suggestions JSON,但中间缺少稳定、可复用、可落盘的 suggestions 生成器。
本轮目标是补齐最小闭环,让仓库能够从 review bundle 稳定生成两份正式建议产物:
- `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json`
- `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.md`
并保证 JSON 与 `skills/keyword-cleanup-review/references/suggestion-schema.md` 对齐,且能直接衔接 `scripts/apply_term_suggestions.py`。
补充口径:本方案中的 JSON 是 review / apply 之间的唯一正式建议产物;bundle 与 Markdown 主要作为运行时工作文件和临时展示层,不建议与 facts/configs 一样长期沉淀。
---
## 2. 当前现状
### 2.1 已有输入层
`build_review_bundle.py` 已经把以下输入聚合成单个 bundle:
- `data/term_index/term_stats.json`
- `data/term_index/daily/*.json`
- `configs/term_aliases.json`
- `configs/term_stopwords.json`
- `configs/filter_context.personal.json`
- `configs/term_cleanup_policy.json`
- `configs/term_watchlist.json`
- `configs/term_change_log.json`
bundle 中已经包含:
- 当前配置快照
- 最近 N 天热点词
- uncovered terms
- interest review candidates
- watch review candidates
这些信息已经足够支撑“保守的、可审查的” suggestions 生成。
### 2.2 已有输出消费层
`scripts/apply_term_suggestions.py` 已经能消费 suggestions JSON,并将接受的建议写回:
- `configs/term_aliases.json`
- `configs/term_stopwords.json`
- `configs/filter_context.personal.json`
- `configs/term_watchlist.json`
- `configs/term_change_log.json`
这说明落地层已存在,缺的是中间的正式建议产物生成层。
---
## 3. 当前缺口
当前流程停在:
`build_review_bundle.py` -> `keyword-cleanup-bundle.json`
但缺少:
`keyword-cleanup-bundle.json` -> `term-cleanup-suggestions-YYYY-MM-DD.json/.md`
因此出现几个问题:
- README / 设计文档里已经引用 suggestions 产物,但仓库内没有稳定生成脚本
- 人工审阅与后续 apply 之间没有统一的正式交付格式
- 同一份 bundle 无法稳定、幂等地重放为同名 suggestions 产物
- alias / stopword / interest / watch 四类建议缺少统一出入口
---
## 4. 推荐最小闭环架构
推荐新增一层独立脚本:
- `scripts/generate_term_cleanup_suggestions.py`
职责:
- 输入:`outputs/term_index/review/keyword-cleanup-bundle.json`
- 输出:
- `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json`
- `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.md`
推荐最小数据流:
1. `build_review_bundle.py` 生成 bundle
2. `generate_term_cleanup_suggestions.py` 读取 bundle
3. 脚本基于治理 hints 生成 suggestions JSON
4. 同时渲染人类可审阅的 Markdown
5. 审阅后可用 `apply_term_suggestions.py` 选择性写回配置
本轮不引入默认 LLM 路径:
- 默认实现采用确定性规则生成
- 如果未来需要 LLM 参与,应作为显式可选增强,而不是默认路径
---
## 5. 产物设计
### 5.1 JSON 产物
JSON 必须与 `suggestion-schema.md` 对齐,至少包含:
```json
{
"date": "2026-04-08",
"based_on_days": 7,
"alias_suggestions": [],
"stopword_suggestions": [],
"interest_keyword_suggestions": [
{
"term": "Claude Code",
"reason": "Meets the configured interest-keyword review threshold and is not yet covered."
}
],
"watch_terms": [
{
"term": "A2A",
"reason": "Falls into the configured watch-term review range and should be observed first."
}
]
}
```
在不破坏兼容性的前提下,可以补充少量元数据字段,建议仅限:
- `source_bundle`
- `policy_schema_version`
- `summary`
建议项字段口径:
- `alias_suggestions[]`
- `from`
- `to`
- `reason`
- `stopword_suggestions[]`
- `term`
- `reason`
- `interest_keyword_suggestions[]`
- `term`
- `reason`
- 可选:`total_count`、`days_seen`、`recent_count`
- `watch_terms[]`
- `term`
- `reason`
- 可选:`total_count`、`days_seen`、`recent_count`
兼容性要求:
- `apply_term_suggestions.py` 只依赖分类 bucket 与关键字段名
- 因此额外证据字段只能追加,不能替换现有字段名
### 5.2 Markdown 产物
Markdown 推荐结构:
1. 标题与日期
2. 输入 bundle 与策略摘要
3. 当前现状摘要
- top/global 观察
- uncovered terms 概览
- 当前 watchlist / interest 覆盖情况
4. 建议摘要
- interest 建议数量
- watch 建议数量
- alias 建议数量
- stopword 建议数量
5. `interest_keyword_suggestions`
6. `watch_terms`
7. `alias_suggestions`
8. `stopword_suggestions`
9. 应用方式
- 指向生成的 JSON
- 给出 `apply_term_suggestions.py` 的调用示例
这样可以保证:
- 人可以直接审阅
- 机器可以直接消费同名 JSON
- Markdown 与 JSON 始终一一对应
---
## 6. 建议生成策略
### 6.1 本轮主链路:interest / watch
本轮先实现最小可用主链路:
- `interest_keyword_suggestions`
- `watch_terms`
直接复用 bundle 中已有的:
- `governance_hints.interest_review_candidates`
- `governance_hints.watch_review_candidates`
原因:
- 这些候选已经与 policy 对齐
- 这些候选已经排除了大部分已覆盖项
- 能直接与现有 apply 脚本形成闭环
### 6.2 alias / stopword 保守处理
本轮边界明确如下:
- `alias_suggestions` 先保持保守,默认可为空
- `stopword_suggestions` 先保持保守,默认可为空
- 后续如果补充更强证据或人工审查规则,再逐步增强
这样可以避免在证据不足时误伤配置。
### 6.3 Phase 2 新方向:alias review 交给 LLM 整理
对于 alias,不再优先走程序规则匹配。
Phase 2 建议改为:
- 程序继续负责准备 review 输入(term stats / daily / current aliases / stopwords / interest / watchlist)
- LLM 负责整理 alias 候选
- 默认先输出人工审阅汇报,而不是直接 apply
- 人工确认后,再决定是否写入 `term_aliases.json`
这样做的原因:
- alias 更偏语义整理,而不是简单趋势筛选
- 与 watch / interest 相比,alias 一旦错误归并,代价更高
- 对当前低频治理场景来说,LLM + 人工确认更轻,也比在程序里持续堆复杂规则更合适
---
## 7. 错误处理
脚本应做显式校验,并在失败时给出明确错误:
### 7.1 输入错误
- bundle 文件不存在 -> 直接失败
- bundle 不是 JSON object -> 直接失败
- 缺少关键字段(如 `days`、`governance_hints`)-> 直接失败
- 候选 bucket 结构错误 -> 直接失败
### 7.2 输出错误
- 输出目录不存在时自动创建
- JSON / Markdown 写入失败时直接退出非 0
### 7.3 数据去重与冲突
- 同一 term 不能同时出现在 interest 与 watch 中
- 优先级:`interest_keyword_suggestions` > `watch_terms`
- 已在 bundle 当前配置中覆盖的 term 不重复输出
---
## 8. 幂等性
本轮要求具备基础幂等性:
- 同一份 bundle 多次运行,默认生成同名产物
- 同一份 bundle 多次运行,JSON 内容顺序稳定
- Markdown 内容顺序稳定
建议做法:
- 优先使用 bundle 的 `generated_at` 日期作为 suggestions 文件日期
- term 排序按证据强度与 term 名稳定排序
- 不在默认输出中写入“每次运行变化”的当前时间戳
这样可以让生成器作为可重放步骤存在于 review 流程中。
---
## 9. 与 apply_term_suggestions.py 的衔接
正式链路应变成:
1. `build_review_bundle.py`
2. `generate_term_cleanup_suggestions.py`
3. 人工审阅 Markdown
4. `apply_term_suggestions.py --suggestions ...`
衔接要求:
- JSON bucket 名必须与 `apply_term_suggestions.py` 读取逻辑一致
- `date` 与 `based_on_days` 字段保留,用于 change log 回写
- 建议项中的 `reason` 直接沿用到 apply 后的 change log
这保证建议生成层不会成为孤立产物,而是正式进入 repo 治理闭环。
---
## 10. Python 3.11 依赖处理
### 10.1 当前现状
`build_review_bundle.py` 当前使用 `from datetime import UTC`,这要求 Python 3.11。
仓库整体 `pyproject.toml` 当前也声明 `requires-python = ">=3.11"`,因此短期内使用 `/usr/bin/python3.11` 运行是符合仓库现状的。
### 10.2 短期建议
短期先在文档与验证命令中明确:
- bundle 构建使用 `/usr/bin/python3.11`
- suggestions 生成脚本也按仓库当前 3.11 基线运行
### 10.3 中期建议
如果后续希望把 keyword cleanup 工具链下探到 Python 3.10,可做兼容改造:
- 把 `datetime.UTC` 替换为 `datetime.timezone.utc`
- 重新检查相关脚本是否还有其他 3.11-only 语法或库依赖
本轮不做超范围兼容重构,只在文档中把此约束说清楚。
---
## 11. 边界与非目标
本轮明确边界:
- 先实现 interest/watch 主链路
- alias/stopword 先保持保守或留待后续增强
- 不做超范围重构
- 不把 LLM 作为默认生成路径
- 不直接改 `filter_rules.json`
- 不自动 apply 建议到配置
因此,本轮交付定义为:
- 补齐 bundle -> suggestions 的正式实现层
- 让 review 流程可运行、可落盘、可审阅、可应用
而不是一次性做完所有高级治理逻辑。
---
## 12. 实施建议
建议按以下小步落地:
### Phase 1(已完成)
1. 新增 `scripts/generate_term_cleanup_suggestions.py`
2. 读取 bundle 并做结构校验
3. 生成稳定排序的 interest/watch suggestions JSON
4. Markdown 改成按需生成
5. README / skill 文档补一条生成命令
6. 用 `/usr/bin/python3.11` 完整跑通 bundle -> suggestions
完成后,keyword cleanup review 的最小正式链路变为:
`review bundle` -> `suggestions json` -> `apply accepted suggestions`
其中 Markdown 只是按需生成的展示层。
### Phase 2(下一步)
1. 保持程序继续准备 review 输入
2. 引入 LLM 做 alias 候选整理
3. 默认先生成 alias review 汇报,而不是直接 apply
4. 由人工确认后再决定是否写入 `term_aliases.json`
这样 alias review 会成为一个低频治理动作,而不是主链路里的自动归一步骤。
### Phase 3(下一步)
1. 保持程序继续准备 review 输入
2. 引入 LLM 做 stopword 候选整理
3. 默认先生成 stopword review 汇报,而不是直接 apply
4. 由人工确认后再决定是否写入 `term_stopwords.json`
这样 stopword review 会成为一个低频减噪动作,而不是主链路里的自动过滤步骤。
#### Phase 3 输入建议
建议给 LLM 的输入包括:
- `data/term_index/term_stats.json` 中的高频词与 recent evidence
- 最近 N 天 `data/term_index/daily/*.json` 的热点词上下文
- 当前 `configs/term_stopwords.json`
- 当前 `configs/term_aliases.json`
- 当前 `configs/filter_context.personal.json` 中的 `interest_keywords`
- 当前 `configs/term_watchlist.json`
程序层只负责把这些输入整理成紧凑 review context,不负责直接做 stopword 决策。
#### Phase 3 输出建议
建议 LLM 默认输出一份人工审阅汇报,而不是直接写配置:
- `建议加入 stopword`
- 词
- 简短理由
- 证据(如 total_count / days_seen / recent_count)
- `暂不建议加入 stopword`
- 词
- 为什么虽然偏泛,但当前还不能杀
- `需要人工判断`
- 词
- 风险点:可能是噪声,也可能仍保留有价值信号
如需结构化输出,可额外补一份 `stopword_review_candidates.json`,但该文件只作为 review 输入,不直接作为自动 apply 指令。
#### Phase 3 审阅原则
- 宁可少删,不乱杀
- 优先处理过泛、低辨识度、持续污染统计的词
- 对可能仍承载有效技术语义的词保持保守
- 默认先汇报,确认后再执行
#### Phase 3 汇报模板建议
建议 stopword review 默认按以下结构汇报给用户:
1. `建议加入 stopword`
- 词
- 理由:为什么这个词对治理帮助低、噪声高
- 证据:`total_count` / `days_seen` / `recent_count` 或最近出现上下文
2. `暂不建议加入 stopword`
- 词
- 理由:为什么当前不建议删掉
3. `需要人工判断`
- 词
- 风险点:泛词与有效主题词之间边界不清等
推荐汇报风格:
- 简短、保守、可审阅
- 先给判断,再给证据
- 不输出机器式原始 dump
- 不默认承诺“已应用”,只汇报“建议”
#### Phase 2 输入建议
建议给 LLM 的输入包括:
- `data/term_index/term_stats.json` 中的高频词与 recent evidence
- 最近 N 天 `data/term_index/daily/*.json` 的热点词上下文
- 当前 `configs/term_aliases.json`
- 当前 `configs/term_stopwords.json`
- 当前 `configs/filter_context.personal.json` 中的 `interest_keywords`
- 当前 `configs/term_watchlist.json`
程序层只负责把这些输入整理成紧凑 review context,不负责直接做 alias 决策。
#### Phase 2 输出建议
建议 LLM 默认输出一份人工审阅汇报,而不是直接写配置:
- `建议合并`
- `from -> to`
- 简短理由
- 证据(如 total_count / days_seen / recent_count)
- `暂不建议合并`
- 为什么不建议并掉
- `需要人工判断`
- 语义相近但风险较高的项
如需结构化输出,可额外补一份 `alias_review_candidates.json`,但该文件只作为 review 输入,不直接作为自动 apply 指令。
#### Phase 2 审阅原则
- 宁可少提,不乱提
- 优先整理明显同义 / 同概念 / 词形差异
- 不把公司名、产品名、泛概念词强行混并
- 默认先汇报,确认后再执行
#### Phase 2 汇报模板建议
建议 alias review 默认按以下结构汇报给用户:
1. `建议合并`
- `from -> to`
- 理由:为什么判断为同一概念或更合适的标准词
- 证据:`total_count` / `days_seen` / `recent_count` 或最近出现上下文
2. `暂不建议合并`
- 候选对
- 理由:为什么虽然相近,但当前不建议并
3. `需要人工判断`
- 候选对
- 风险点:歧义、范围差异、产品名/公司名混淆等
推荐汇报风格:
- 简短、保守、可审阅
- 先给判断,再给证据
- 不输出机器式原始 dump
- 不默认承诺“已应用”,只汇报“建议”
#### Phase 2 示例输出
建议合并:
- `Claude code -> Claude Code`
- 理由:明显属于同一产品名,仅是大小写写法不一致。
- 证据:`Claude Code` 在最近多日持续出现,而小写写法只是在少量上下文中作为变体出现。
- `Sub-Agent -> SubAgent`
- 理由:更像词形差异,不构成新的独立概念。
- 证据:两者都围绕同一 agent 架构语境出现,且没有稳定区分语义。
暂不建议合并:
- `Skills ↔ Agent Skills`
- 理由:前者过泛,后者更具体,当前强行归并会损失粒度。
- `Anthropic ↔ Claude`
- 理由:公司名与产品名并不等价,不应直接视为一个关键词。
需要人工判断:
- `AI助手 ↔ AI Agent`
- 风险点:语义可能接近,但中文表述范围更宽,是否并入需要结合你的使用语境判断。
+34
View File
@@ -0,0 +1,34 @@
# 示例文章标题
原文链接:
https://example.com/article
## 核心结论
这里先用一段短句概括最重要的判断。
如果结论较长,继续拆成第二个短段,而不是塞成一个大长段。
## 主要论点
先交代文章的核心主张。
再单独起一段解释支撑这个主张的关键论据。
如果还有补充判断,继续拆段,保证在 IMA 中阅读时不会挤成一整坨。
## 关键方法 / 机制
- 要点一
- 要点二
- 要点三
## 重要细节
- 细节一
- 细节二
## 可复用启发
- 启发一
- 启发二
+40
View File
@@ -0,0 +1,40 @@
+++
title = "AI 日报 · 示例"
date = 2026-04-01T16:55:00+08:00
summary = "围绕 Agent 架构分层、Skills 标准化与桌面 Agent 工程实践的当日观察。"
+++
> ⚠️ 格式规范(生成 Hugo 时必须遵守):
> - 文章编号:`1.` `2.` `3.`(阿拉伯数字 + 点),禁止 `① ② ③` / `一、二、三` 等变体
> - 四个 section 缺一不可:`今日概览` → `今日重点` → `趋势观察` → `延伸阅读`
> - 每篇文章结构:标题来源 → 摘要段 → "值得关注:"三点 → "这篇更值得关注的理由"段
# 今日概览
今天的公开候选主要集中在 AI Agent 的架构演进、工具化落地与工程化实践三条线索上。相比早期偏概念展示的讨论,这一批内容更强调模块化能力栈、真实部署路径与系统可维护性,说明行业关注点正在从“模型能做什么”转向“系统如何稳定落地并持续复用”。
## 今日重点
### 1. 学习笔记:从 Agent 到 Skills — AI 智能体架构的范式转变
来源:阿里云开发者
文章分析了 AI 智能体架构从单体 Agent 向模块化 Skills 的范式转变。Anthropic 先后推出 MCP 和 Agent Skills 开放标准,构建了知识、工具、协作和运行分层架构。文章通过一个自动化美化相册的真实项目,对比了 Claude Code 与 OpenClaw 两种实现方案,验证了新架构的可复用性与灵活性。
值得关注:
- Anthropic 在 14 个月内先后推出 MCP 和 Agent Skills 两个开放标准,推动 AI 智能体架构分层化。
- 新范式核心是构建薄 Agent 引擎与可组合的 Skills 库,取代为每个用例定制单体 Agent。
- 文章通过自动化美化相册项目,实操演示了 Skills、MCP、OpenClaw 和 A2A 协议如何协同工作。
这篇内容更值得关注的原因在于,它不只是提出了“Agent 要模块化”这个判断,而是把开放标准、分层架构和真实项目案例串成了一条完整论证链,能直接支撑今天日报的主线。
## 趋势观察
1. Agent 正在从单体能力转向可组合的模块化体系。无论是 Skills、MCP、记忆还是运行时编排,这批内容都在强调解耦与复用,而不是把智能体继续当成一个不可拆分的黑箱。
2. 工程化正在变成 AI 应用竞争的主战场。桌面 Agent、企业级架构和部署实践类内容增多,说明真正的差异化开始落在接入现有流程、控制风险和提升可维护性上。
3. AI 能力的竞争点正在上移。模型本身仍重要,但真正可持续的优势越来越来自系统设计、工作流整合和对业务场景的理解。
## 延伸阅读
- [学习笔记:从 Agent 到 Skills — AI 智能体架构的范式转变](https://example.com/a)|阿里云开发者
- [Agent Skills:打通可复用专业领域知识的最后一公里](https://example.com/b)|阿里云开发者
- [CoPaw深度解析:源码架构和功能实践](https://example.com/c)|阿里云开发者
@@ -0,0 +1,411 @@
from __future__ import annotations
import argparse
import json
from datetime import datetime, timezone
from pathlib import Path
from typing import Any
REPO_ROOT = Path(__file__).resolve().parents[1]
DEFAULT_BUNDLE_PATH = REPO_ROOT / "outputs" / "term_index" / "review" / "keyword-cleanup-bundle.json"
DEFAULT_OUTPUT_DIR = REPO_ROOT / "outputs" / "term_index" / "review"
def _load_json(path: Path) -> Any:
return json.loads(path.read_text(encoding="utf-8-sig"))
def _save_json(path: Path, payload: dict[str, Any]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
def _save_text(path: Path, content: str) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(content, encoding="utf-8")
def _term_key(value: str) -> str:
return value.strip().casefold()
def _utc_today() -> str:
return datetime.now(timezone.utc).date().isoformat()
def _require_dict(payload: Any, name: str) -> dict[str, Any]:
if not isinstance(payload, dict):
raise RuntimeError(f"{name} must be a JSON object.")
return payload
def _require_list(payload: Any, name: str) -> list[Any]:
if not isinstance(payload, list):
raise RuntimeError(f"{name} must be a JSON array.")
return payload
def _bundle_date(bundle: dict[str, Any]) -> str:
generated_at = bundle.get("generated_at")
if isinstance(generated_at, str) and generated_at.strip():
normalized = generated_at.replace("Z", "+00:00")
try:
return datetime.fromisoformat(normalized).date().isoformat()
except ValueError:
pass
return _utc_today()
def _recent_count_map(top_global_terms: list[dict[str, Any]]) -> dict[str, int]:
counts: dict[str, int] = {}
for item in top_global_terms:
term = item.get("term")
recent_count = item.get("recent_count")
if isinstance(term, str) and isinstance(recent_count, int):
counts[term] = recent_count
return counts
def _covered_term_sets(bundle: dict[str, Any]) -> tuple[set[str], set[str], set[str]]:
current_config = _require_dict(bundle.get("current_config"), "bundle.current_config")
interest_keywords = _require_list(current_config.get("interest_keywords"), "bundle.current_config.interest_keywords")
stopwords = _require_list(current_config.get("stopwords"), "bundle.current_config.stopwords")
watchlist = _require_list(current_config.get("watchlist"), "bundle.current_config.watchlist")
interest_set = {_term_key(item) for item in interest_keywords if isinstance(item, str) and item.strip()}
stopword_set = {_term_key(item) for item in stopwords if isinstance(item, str) and item.strip()}
watch_set = {
_term_key(str(item.get("term", "")))
for item in watchlist
if isinstance(item, dict) and isinstance(item.get("term"), str) and str(item.get("term", "")).strip()
}
return interest_set, stopword_set, watch_set
def _sort_key(item: dict[str, Any]) -> tuple[int, int, int, str, str]:
total_count = int(item.get("total_count") or 0)
days_seen = int(item.get("days_seen") or 0)
recent_count = int(item.get("recent_count") or 0)
term = str(item.get("term") or "")
return (-total_count, -days_seen, -recent_count, term.casefold(), term)
def _prepare_interest_suggestions(bundle: dict[str, Any]) -> list[dict[str, Any]]:
governance_hints = _require_dict(bundle.get("governance_hints"), "bundle.governance_hints")
candidates = _require_list(
governance_hints.get("interest_review_candidates"),
"bundle.governance_hints.interest_review_candidates",
)
top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms")
recent_counts = _recent_count_map([item for item in top_global_terms if isinstance(item, dict)])
interest_set, stopword_set, watch_set = _covered_term_sets(bundle)
suggestions: list[dict[str, Any]] = []
seen: set[str] = set()
for item in candidates:
if not isinstance(item, dict):
continue
term = item.get("term")
if not isinstance(term, str) or not term.strip():
continue
term_key = _term_key(term)
if term_key in seen or term_key in interest_set or term_key in stopword_set:
continue
total_count = int(item.get("total_count") or 0)
days_seen = int(item.get("days_seen") or 0)
recent_count = recent_counts.get(term, 0)
base_reason = str(item.get("reason") or "Meets the configured interest-keyword review threshold.")
if term_key in watch_set:
base_reason += " It is currently in watchlist and is ready for promotion."
reason = f"{base_reason} Evidence: total_count={total_count}, days_seen={days_seen}, recent_count={recent_count}."
suggestions.append(
{
"term": term,
"reason": reason,
"total_count": total_count,
"days_seen": days_seen,
"recent_count": recent_count,
}
)
seen.add(term_key)
suggestions.sort(key=_sort_key)
return suggestions
def _prepare_watch_suggestions(bundle: dict[str, Any], reserved_terms: set[str]) -> list[dict[str, Any]]:
governance_hints = _require_dict(bundle.get("governance_hints"), "bundle.governance_hints")
candidates = _require_list(
governance_hints.get("watch_review_candidates"),
"bundle.governance_hints.watch_review_candidates",
)
top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms")
recent_counts = _recent_count_map([item for item in top_global_terms if isinstance(item, dict)])
interest_set, stopword_set, watch_set = _covered_term_sets(bundle)
suggestions: list[dict[str, Any]] = []
seen: set[str] = set(reserved_terms)
for item in candidates:
if not isinstance(item, dict):
continue
term = item.get("term")
if not isinstance(term, str) or not term.strip():
continue
term_key = _term_key(term)
if term_key in seen or term_key in interest_set or term_key in stopword_set or term_key in watch_set:
continue
total_count = int(item.get("total_count") or 0)
days_seen = int(item.get("days_seen") or 0)
recent_count = recent_counts.get(term, 0)
base_reason = str(item.get("reason") or "Falls into the configured watch-term review range.")
reason = f"{base_reason} Evidence: total_count={total_count}, days_seen={days_seen}, recent_count={recent_count}."
suggestions.append(
{
"term": term,
"reason": reason,
"total_count": total_count,
"days_seen": days_seen,
"recent_count": recent_count,
}
)
seen.add(term_key)
suggestions.sort(key=_sort_key)
return suggestions
def _render_table(items: list[dict[str, Any]]) -> str:
if not items:
return "_None in this pass._\n"
lines = [
"| Term | Total | Days | Recent | Reason |",
"| --- | ---: | ---: | ---: | --- |",
]
for item in items:
term = str(item.get("term") or "")
total_count = int(item.get("total_count") or 0)
days_seen = int(item.get("days_seen") or 0)
recent_count = int(item.get("recent_count") or 0)
reason = str(item.get("reason") or "").replace("|", "\\|")
lines.append(f"| {term} | {total_count} | {days_seen} | {recent_count} | {reason} |")
return "\n".join(lines) + "\n"
def _render_simple_table(items: list[dict[str, Any]], first_column: str) -> str:
if not items:
return "_None in this pass._\n"
lines = [
f"| {first_column} | Reason |",
"| --- | --- |",
]
for item in items:
value = str(item.get(first_column.casefold()) or item.get(first_column) or "")
reason = str(item.get("reason") or "").replace("|", "\\|")
lines.append(f"| {value} | {reason} |")
return "\n".join(lines) + "\n"
def _render_markdown(
*,
suggestion_date: str,
bundle_path: Path,
json_output_path: Path,
bundle: dict[str, Any],
suggestions: dict[str, Any],
) -> str:
policy = _require_dict(bundle.get("policy"), "bundle.policy")
current_config = _require_dict(bundle.get("current_config"), "bundle.current_config")
top_global_terms = _require_list(bundle.get("top_global_terms"), "bundle.top_global_terms")
uncovered_terms = _require_list(bundle.get("uncovered_terms"), "bundle.uncovered_terms")
top_preview = [item for item in top_global_terms if isinstance(item, dict)][:5]
uncovered_preview = [item for item in uncovered_terms if isinstance(item, dict)][:5]
interest_items = suggestions["interest_keyword_suggestions"]
watch_items = suggestions["watch_terms"]
alias_items = suggestions["alias_suggestions"]
stopword_items = suggestions["stopword_suggestions"]
lines = [
f"# Term Cleanup Suggestions - {suggestion_date}",
"",
"## Review Context",
"",
f"- Source bundle: `{bundle_path}`",
f"- Suggestions JSON: `{json_output_path}`",
f"- Bundle generated_at: `{bundle.get('generated_at', 'unknown')}`",
f"- Based on days: `{suggestions['based_on_days']}`",
f"- Policy schema version: `{policy.get('schema_version', 'unknown')}`",
"- Scope: implement `interest_keyword_suggestions` and `watch_terms` main path first; keep alias/stopword conservative in this pass.",
"",
"## Current State",
"",
f"- Interest keywords: `{current_config.get('interest_keyword_count', 0)}`",
f"- Watch terms: `{current_config.get('watch_term_count', 0)}`",
f"- Stopwords: `{current_config.get('stopword_count', 0)}`",
f"- Aliases: `{current_config.get('alias_count', 0)}`",
f"- Top global terms considered: `{len(top_global_terms)}`",
f"- Uncovered terms considered: `{len(uncovered_terms)}`",
"",
"### Top Terms Snapshot",
"",
]
if top_preview:
for item in top_preview:
lines.append(
f"- `{item.get('term', '')}`: total_count={item.get('total_count', 0)}, days_seen={item.get('days_seen', 0)}, recent_count={item.get('recent_count', 0)}"
)
else:
lines.append("- No top terms available.")
lines.extend([
"",
"### Uncovered Terms Snapshot",
"",
])
if uncovered_preview:
for item in uncovered_preview:
lines.append(
f"- `{item.get('term', '')}`: total_count={item.get('total_count', 0)}, days_seen={item.get('days_seen', 0)}, recent_count={item.get('recent_count', 0)}"
)
else:
lines.append("- No uncovered terms available.")
lines.extend([
"",
"## Suggestion Summary",
"",
f"- `interest_keyword_suggestions`: `{len(interest_items)}`",
f"- `watch_terms`: `{len(watch_items)}`",
f"- `alias_suggestions`: `{len(alias_items)}`",
f"- `stopword_suggestions`: `{len(stopword_items)}`",
"",
"## Interest Keyword Suggestions",
"",
_render_table(interest_items).rstrip(),
"",
"## Watch Terms",
"",
_render_table(watch_items).rstrip(),
"",
"## Alias Suggestions",
"",
"_Conservative by design in this minimal version; no automatic alias suggestions are emitted yet._" if not alias_items else _render_simple_table(alias_items, "from").rstrip(),
"",
"## Stopword Suggestions",
"",
"_Conservative by design in this minimal version; no automatic stopword suggestions are emitted yet._" if not stopword_items else _render_simple_table(stopword_items, "term").rstrip(),
"",
"## Apply",
"",
"Review the Markdown first, then selectively apply accepted suggestions with the JSON file.",
"",
"```bash",
f"python scripts/apply_term_suggestions.py \\",
f" --suggestions {json_output_path} \\",
" --accept-interest \"Claude Code\" \\",
" --accept-watch \"A2A\" \\",
" --dry-run",
"```",
"",
])
return "\n".join(lines)
def _build_output_paths(
*,
output_dir: Path,
suggestion_date: str,
json_output: Path | None,
markdown_output: Path | None,
) -> tuple[Path, Path]:
stem = f"term-cleanup-suggestions-{suggestion_date}"
resolved_json = json_output or (output_dir / f"{stem}.json")
resolved_markdown = markdown_output or (output_dir / f"{stem}.md")
return resolved_json, resolved_markdown
def main() -> None:
parser = argparse.ArgumentParser(description="Generate term cleanup suggestions JSON and Markdown from review bundle.")
parser.add_argument("--bundle", type=Path, default=DEFAULT_BUNDLE_PATH, help="Review bundle JSON file")
parser.add_argument(
"--output-dir",
type=Path,
default=DEFAULT_OUTPUT_DIR,
help="Directory for generated suggestions outputs when explicit output paths are not provided",
)
parser.add_argument("--date", type=str, default=None, help="Override suggestions date (YYYY-MM-DD)")
parser.add_argument("--json-output", type=Path, default=None, help="Explicit suggestions JSON output path")
parser.add_argument("--markdown-output", type=Path, default=None, help="Explicit suggestions Markdown output path")
parser.add_argument(
"--emit-markdown",
action="store_true",
help="Also write the human-readable Markdown review draft. JSON suggestions are always written.",
)
args = parser.parse_args()
if not args.bundle.exists():
raise RuntimeError(f"Bundle file not found: {args.bundle}")
bundle = _require_dict(_load_json(args.bundle), "bundle")
days = bundle.get("days")
if not isinstance(days, int):
raise RuntimeError("bundle.days must be an integer.")
suggestion_date = args.date or _bundle_date(bundle)
json_output_path, markdown_output_path = _build_output_paths(
output_dir=args.output_dir,
suggestion_date=suggestion_date,
json_output=args.json_output,
markdown_output=args.markdown_output,
)
interest_items = _prepare_interest_suggestions(bundle)
reserved_terms = {_term_key(str(item.get("term") or "")) for item in interest_items}
watch_items = _prepare_watch_suggestions(bundle, reserved_terms=reserved_terms)
suggestions = {
"date": suggestion_date,
"based_on_days": days,
"source_bundle": str(args.bundle),
"policy_schema_version": _require_dict(bundle.get("policy"), "bundle.policy").get("schema_version", "unknown"),
"summary": {
"interest_keyword_suggestions": len(interest_items),
"watch_terms": len(watch_items),
"alias_suggestions": 0,
"stopword_suggestions": 0,
},
"alias_suggestions": [],
"stopword_suggestions": [],
"interest_keyword_suggestions": interest_items,
"watch_terms": watch_items,
}
markdown = _render_markdown(
suggestion_date=suggestion_date,
bundle_path=args.bundle,
json_output_path=json_output_path,
bundle=bundle,
suggestions=suggestions,
)
_save_json(json_output_path, suggestions)
if args.emit_markdown:
_save_text(markdown_output_path, markdown)
summary = {
"bundle": str(args.bundle),
"date": suggestion_date,
"json_output": str(json_output_path),
"markdown_output": str(markdown_output_path) if args.emit_markdown else None,
"interest_keyword_suggestions": len(interest_items),
"watch_terms": len(watch_items),
"alias_suggestions": 0,
"stopword_suggestions": 0,
"emit_markdown": args.emit_markdown,
}
print(json.dumps(summary, ensure_ascii=False, indent=2))
if __name__ == "__main__":
main()
+55 -6
View File
@@ -9,7 +9,7 @@ description: 审查和整理本仓库的每日关键词索引和频率统计。
## 工作流程 ## 工作流程
1. 构建精简的审查数据包: 1. 构建精简的审查数据包(临时工作文件):
```bash ```bash
python skills/keyword-cleanup-review/scripts/build_review_bundle.py python skills/keyword-cleanup-review/scripts/build_review_bundle.py
@@ -21,15 +21,32 @@ python skills/keyword-cleanup-review/scripts/build_review_bundle.py
- `--top 50` - `--top 50`
- `--output outputs/term_index/review/keyword-cleanup-bundle.json` - `--output outputs/term_index/review/keyword-cleanup-bundle.json`
2. 阅读生成的数据包和建议模式: 2. 阅读建议模式:
- `outputs/term_index/review/keyword-cleanup-bundle.json`
- `skills/keyword-cleanup-review/references/suggestion-schema.md` - `skills/keyword-cleanup-review/references/suggestion-schema.md`
3. 生成两份输出: 3. 运行 suggestions 生成脚本:
- 一份简短的供人工审阅的 Markdown 报告 ```bash
- 一份符合模式的 JSON 建议文件 python scripts/generate_term_cleanup_suggestions.py ^
--bundle outputs/term_index/review/keyword-cleanup-bundle.json
```
默认生成:
- 一份符合模式的 JSON 建议文件(正式建议产物)
如需人工审阅展示稿,再显式加:
```bash
python scripts/generate_term_cleanup_suggestions.py ^
--bundle outputs/term_index/review/keyword-cleanup-bundle.json ^
--emit-markdown
```
这时才会额外生成:
- 一份简短的供人工审阅的 Markdown 报告(临时展示稿)
4. 严格保持边界: 4. 严格保持边界:
@@ -78,10 +95,41 @@ Markdown 输出应:
- 分类别名、停用词、兴趣关键词和关注词建议 - 分类别名、停用词、兴趣关键词和关注词建议
- 用简短、具体的句子解释理由 - 用简短、具体的句子解释理由
说明:Markdown 主要用于人工临时审阅,不必默认当作长期资产保留。
JSON 输出应遵循: JSON 输出应遵循:
- `references/suggestion-schema.md` - `references/suggestion-schema.md`
说明:JSON 是 review / apply 之间的唯一正式建议产物,应优先保留。
## 产物保留策略
长期保留:
- `data/term_index/daily/*.json`
- `data/term_index/term_stats.json`
- `configs/filter_context.personal.json`
- `configs/term_watchlist.json`
- `configs/term_aliases.json`
- `configs/term_stopwords.json`
- `configs/term_change_log.json`
短期保留:
- `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json`
临时产物:
- `outputs/term_index/review/keyword-cleanup-bundle.json`
- `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.md`
默认执行口径:
- bundle 只作为运行时工作文件,默认只保留当前最新一份
- Markdown 只作为人工展示层,优先按需生成,不默认长期归档
- JSON suggestions 是 review / apply 之间唯一正式建议输入
## 仓库说明 ## 仓库说明
当前仓库行为: 当前仓库行为:
@@ -98,5 +146,6 @@ JSON 输出应遵循:
- 脚本: - 脚本:
- `scripts/build_review_bundle.py` - `scripts/build_review_bundle.py`
- `scripts/generate_term_cleanup_suggestions.py`
- 参考文档: - 参考文档:
- `references/suggestion-schema.md` - `references/suggestion-schema.md`
+6 -1
View File
@@ -1,6 +1,11 @@
from __future__ import annotations from __future__ import annotations
from datetime import UTC, date, datetime try:
from datetime import UTC, date, datetime
except ImportError: # Python < 3.11 compatibility
from datetime import timezone, date, datetime
UTC = timezone.utc
from pydantic import BaseModel, Field from pydantic import BaseModel, Field