diff --git a/configs/filter_context.personal.json b/configs/filter_context.personal.json index 1167b8b..aee74f3 100644 --- a/configs/filter_context.personal.json +++ b/configs/filter_context.personal.json @@ -23,34 +23,39 @@ "前沿科技" ], "interest_keywords": [ - "Java", - "Go", - "Python", - "Spring", + "Agent", + "Agent Skills", + "AgentScope", + "AI Agent", + "AliSQL", + "Claude Code", + "DeepSeek", "FastAPI", "Gin", + "Go", "gRPC", - "MySQL", - "PostgreSQL", - "Redis", + "Java", "Kafka", - "微服务", - "可观测性", "Kubernetes", - "云原生", - "AI Agent", - "Agent", "LLM", - "RAG", "MCP", - "Prompt Engineering", - "Workflow", - "向量数据库", - "知识库", + "MySQL", + "MySQL复制延迟", "OpenAI", - "DeepSeek", "OpenClaw", - "AliSQL", - "MySQL复制延迟" + "PostgreSQL", + "Prompt Engineering", + "Python", + "RAG", + "ReActAgent", + "Redis", + "Spring", + "SubAgent", + "Workflow", + "云原生", + "可观测性", + "向量数据库", + "微服务", + "知识库" ] -} \ No newline at end of file +} diff --git a/configs/term_aliases.json b/configs/term_aliases.json index 87e2055..e210659 100644 --- a/configs/term_aliases.json +++ b/configs/term_aliases.json @@ -2,4 +2,4 @@ "AI助手": "AI Agent", "图文RAG": "RAG", "Prompt架构": "Prompt Engineering" -} \ No newline at end of file +} diff --git a/configs/term_change_log.json b/configs/term_change_log.json index e786764..2d39d31 100644 --- a/configs/term_change_log.json +++ b/configs/term_change_log.json @@ -1,4 +1,104 @@ { "schema_version": "v1", - "entries": [] -} \ No newline at end of file + "entries": [ + { + "applied_at": "2026-04-08T02:34:14.194320Z", + "action": "add_watch_term", + "term": "A2A", + "reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=0.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json", + "suggestion_date": "2026-04-08", + "based_on_days": 7 + }, + { + "applied_at": "2026-04-08T02:34:14.194320Z", + "action": "add_watch_term", + "term": "Agentic Loop", + "reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=1.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json", + "suggestion_date": "2026-04-08", + "based_on_days": 7 + }, + { + "applied_at": "2026-04-08T02:34:14.194320Z", + "action": "add_watch_term", + "term": "AI Gateway", + "reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=1.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json", + "suggestion_date": "2026-04-08", + "based_on_days": 7 + }, + { + "applied_at": "2026-04-08T02:34:14.194320Z", + "action": "add_watch_term", + "term": "Claude Skills", + "reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=1.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json", + "suggestion_date": "2026-04-08", + "based_on_days": 7 + }, + { + "applied_at": "2026-04-08T02:34:14.194320Z", + "action": "add_watch_term", + "term": "Cron", + "reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=0.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json", + "suggestion_date": "2026-04-08", + "based_on_days": 7 + }, + { + "applied_at": "2026-04-08T02:34:14.194320Z", + "action": "add_watch_term", + "term": "CoPaw", + "reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=0.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json", + "suggestion_date": "2026-04-08", + "based_on_days": 7 + }, + { + "applied_at": "2026-04-08T02:34:14.194320Z", + "action": "add_interest_keyword", + "term": "Claude Code", + "reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=7, days_seen=4, recent_count=4.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json", + "suggestion_date": "2026-04-08", + "based_on_days": 7 + }, + { + "applied_at": "2026-04-08T02:34:14.194320Z", + "action": "add_interest_keyword", + "term": "Agent Skills", + "reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=6, days_seen=3, recent_count=0.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json", + "suggestion_date": "2026-04-08", + "based_on_days": 7 + }, + { + "applied_at": "2026-04-08T02:34:14.194320Z", + "action": "add_interest_keyword", + "term": "SubAgent", + "reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=4, days_seen=4, recent_count=2.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json", + "suggestion_date": "2026-04-08", + "based_on_days": 7 + }, + { + "applied_at": "2026-04-08T02:34:14.194320Z", + "action": "add_interest_keyword", + "term": "AgentScope", + "reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=4, days_seen=4, recent_count=1.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json", + "suggestion_date": "2026-04-08", + "based_on_days": 7 + }, + { + "applied_at": "2026-04-08T02:34:14.194320Z", + "action": "add_interest_keyword", + "term": "ReActAgent", + "reason": "Meets the configured interest-keyword review threshold and is not yet covered by interest keywords or stopwords. Evidence: total_count=3, days_seen=3, recent_count=1.", + "suggestions_path": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json", + "suggestion_date": "2026-04-08", + "based_on_days": 7 + } + ] +} diff --git a/configs/term_watchlist.json b/configs/term_watchlist.json index b5c6060..88cfa97 100644 --- a/configs/term_watchlist.json +++ b/configs/term_watchlist.json @@ -1,5 +1,48 @@ { "schema_version": "v1", - "updated_at": "2026-03-27T00:00:00Z", - "terms": [] -} \ No newline at end of file + "updated_at": "2026-04-08T02:34:14.194320Z", + "terms": [ + { + "term": "A2A", + "added_at": "2026-04-08T02:34:14.194320Z", + "source": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json", + "reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=0.", + "status": "watching" + }, + { + "term": "Agentic Loop", + "added_at": "2026-04-08T02:34:14.194320Z", + "source": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json", + "reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=1.", + "status": "watching" + }, + { + "term": "AI Gateway", + "added_at": "2026-04-08T02:34:14.194320Z", + "source": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json", + "reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=1.", + "status": "watching" + }, + { + "term": "Claude Skills", + "added_at": "2026-04-08T02:34:14.194320Z", + "source": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json", + "reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=1.", + "status": "watching" + }, + { + "term": "CoPaw", + "added_at": "2026-04-08T02:34:14.194320Z", + "source": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json", + "reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=0.", + "status": "watching" + }, + { + "term": "Cron", + "added_at": "2026-04-08T02:34:14.194320Z", + "source": "outputs/term_index/review/term-cleanup-suggestions-2026-04-08.json", + "reason": "Falls into the configured watch-term review range and should be observed before promotion into interest keywords. Evidence: total_count=2, days_seen=2, recent_count=0.", + "status": "watching" + } + ] +} diff --git a/docs/design/daily-keyword-index-design.md b/docs/design/daily-keyword-index-design.md index 696ccaa..d9e889e 100644 --- a/docs/design/daily-keyword-index-design.md +++ b/docs/design/daily-keyword-index-design.md @@ -326,8 +326,13 @@ LLM 可以帮助做清洗建议,但不适合直接维护主词元库。 skill 不直接修改配置文件,而是生成建议文件,例如: -- `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.md` - `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json` +- `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.md` + +其中建议语义为: + +- JSON 是 review / apply 之间的唯一正式建议产物 +- Markdown 是人工临时审阅展示稿,不是长期真相来源 低复杂治理层建议补充三类输入: @@ -403,8 +408,9 @@ skill 不直接修改配置文件,而是生成建议文件,例如: 2. 程序更新 `daily/YYYY-MM-DD.json` 3. 程序更新 `term_stats.json` 4. 每周或人工触发一次词元清洗 skill -5. skill 输出建议 -6. 人工确认后再更新配置文件 +5. skill 生成 suggestions JSON(正式建议产物) +6. 如需要人工阅读,再临时生成 Markdown 展示稿 +7. 人工确认后再更新配置文件 ## 14. 与规则引擎的关系 @@ -445,7 +451,8 @@ skill 不直接修改配置文件,而是生成建议文件,例如: 再补治理层: - 增加词元清洗 skill -- 输出建议文件 +- 输出建议文件(以 JSON 为正式产物) +- Markdown 仅作为按需生成的人工展示层 - 人工确认后更新配置 ### Phase 3 @@ -459,3 +466,9 @@ skill 不直接修改配置文件,而是生成建议文件,例如: ## 16. 一句话结论 这套设计选择“只统计日报中的 `keywords`,由程序维护轻量词元库,再由独立 skill 周期性做清洗建议”,目的是在控制数据规模的前提下,为规则配置和长期兴趣演化提供稳定、可审计、可扩展的基础设施。 + +补充的产物策略是: + +- facts/state 长期保留 +- suggestions JSON 作为正式建议产物短期保留 +- review bundle 与 Markdown 展示稿降级为临时工作文件 / 展示层 diff --git a/skills/keyword-cleanup-review/SKILL.md b/skills/keyword-cleanup-review/SKILL.md index 856898b..0b20d5a 100644 --- a/skills/keyword-cleanup-review/SKILL.md +++ b/skills/keyword-cleanup-review/SKILL.md @@ -9,7 +9,7 @@ description: 审查和整理本仓库的每日关键词索引和频率统计。 ## 工作流程 -1. 构建精简的审查数据包: +1. 构建精简的审查数据包(临时工作文件): ```bash python skills/keyword-cleanup-review/scripts/build_review_bundle.py @@ -21,15 +21,32 @@ python skills/keyword-cleanup-review/scripts/build_review_bundle.py - `--top 50` - `--output outputs/term_index/review/keyword-cleanup-bundle.json` -2. 阅读生成的数据包和建议模式: +2. 阅读建议模式: -- `outputs/term_index/review/keyword-cleanup-bundle.json` - `skills/keyword-cleanup-review/references/suggestion-schema.md` -3. 生成两份输出: +3. 运行 suggestions 生成脚本: -- 一份简短的供人工审阅的 Markdown 报告 -- 一份符合模式的 JSON 建议文件 +```bash +python scripts/generate_term_cleanup_suggestions.py ^ + --bundle outputs/term_index/review/keyword-cleanup-bundle.json +``` + +默认生成: + +- 一份符合模式的 JSON 建议文件(正式建议产物) + +如需人工审阅展示稿,再显式加: + +```bash +python scripts/generate_term_cleanup_suggestions.py ^ + --bundle outputs/term_index/review/keyword-cleanup-bundle.json ^ + --emit-markdown +``` + +这时才会额外生成: + +- 一份简短的供人工审阅的 Markdown 报告(临时展示稿) 4. 严格保持边界: @@ -78,10 +95,41 @@ Markdown 输出应: - 分类别名、停用词、兴趣关键词和关注词建议 - 用简短、具体的句子解释理由 +说明:Markdown 主要用于人工临时审阅,不必默认当作长期资产保留。 + JSON 输出应遵循: - `references/suggestion-schema.md` +说明:JSON 是 review / apply 之间的唯一正式建议产物,应优先保留。 + +## 产物保留策略 + +长期保留: + +- `data/term_index/daily/*.json` +- `data/term_index/term_stats.json` +- `configs/filter_context.personal.json` +- `configs/term_watchlist.json` +- `configs/term_aliases.json` +- `configs/term_stopwords.json` +- `configs/term_change_log.json` + +短期保留: + +- `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.json` + +临时产物: + +- `outputs/term_index/review/keyword-cleanup-bundle.json` +- `outputs/term_index/review/term-cleanup-suggestions-YYYY-MM-DD.md` + +默认执行口径: + +- bundle 只作为运行时工作文件,默认只保留当前最新一份 +- Markdown 只作为人工展示层,优先按需生成,不默认长期归档 +- JSON suggestions 是 review / apply 之间唯一正式建议输入 + ## 仓库说明 当前仓库行为: @@ -98,5 +146,6 @@ JSON 输出应遵循: - 脚本: - `scripts/build_review_bundle.py` + - `scripts/generate_term_cleanup_suggestions.py` - 参考文档: - `references/suggestion-schema.md` diff --git a/src/summary_mcp/models/openclaw_delivery.py b/src/summary_mcp/models/openclaw_delivery.py index 92b3ba8..0e12eb2 100644 --- a/src/summary_mcp/models/openclaw_delivery.py +++ b/src/summary_mcp/models/openclaw_delivery.py @@ -1,6 +1,11 @@ from __future__ import annotations -from datetime import UTC, date, datetime +try: + from datetime import UTC, date, datetime +except ImportError: # Python < 3.11 compatibility + from datetime import timezone, date, datetime + + UTC = timezone.utc from pydantic import BaseModel, Field