commit 1dfae8ca19ea0e4662a1957e004da337eb5fb515 Author: zhuyongxin Date: Tue Mar 24 17:01:35 2026 +0800 first commit diff --git a/.claude/settings.local.json b/.claude/settings.local.json new file mode 100644 index 0000000..c62a267 --- /dev/null +++ b/.claude/settings.local.json @@ -0,0 +1,5 @@ +{ + "enabledPlugins": { + "skill-creator@claude-plugins-official": true + } +} diff --git a/README.md b/README.md new file mode 100644 index 0000000..a26fcf6 --- /dev/null +++ b/README.md @@ -0,0 +1,30 @@ +# Content Extract MCP + +Python MCP scaffold for article content extraction. + +## Run + +```bash +pip install -e . +summary-mcp +``` + +The server exposes two tools: + +- `extract_url_content` +- `extract_item_content` + +Validate an LLM summary result: + +```bash +validate-llm-result outputs/result.json --extracted outputs/read-flow-2026.extracted.json +``` + +Run the minimal extraction-to-summary loop: + +```bash +python scripts/run_summary_loop.py ^ + --extracted outputs/read-flow-2026.extracted.json ^ + --prompt outputs/llm-summary-prompt.txt ^ + --output outputs/result.json +``` diff --git a/TODO.md b/TODO.md new file mode 100644 index 0000000..75b510d --- /dev/null +++ b/TODO.md @@ -0,0 +1,73 @@ +# TODO + +## 当前状态 + +项目当前处于 `Content Extract MCP` 的 MVP 完成阶段。 + +已完成: + +- [x] 明确整体阅读流链路 +- [x] 明确 `source -> item -> document/article` 的数据抽象 +- [x] 明确 `MCP 负责提取,LLM 负责摘要` 的职责边界 +- [x] 搭建 Python MCP 服务骨架 +- [x] 实现 `extract_url_content` +- [x] 实现 `extract_item_content` +- [x] 完成标题与正文提取 +- [x] 完成质量标记与结构化错误输出 +- [x] 用参考文章完成真实提取测试 +- [x] 输出结构化提取 JSON 文件 +- [x] 设计并迭代 LLM 摘要 prompt +- [x] 验证 LLM 输出的 `result.json` 基本符合预期 +- [x] 已封装 LLM 摘要校验 workflow skill + +--- + +## P0 - 近期必须完成 + +- [x] 为 LLM 摘要结果定义正式 JSON Schema +- [x] 增加一个本地校验脚本,自动校验 `result.json` 是否符合 schema +- [x] 把“提取 JSON -> LLM 摘要 JSON”串成一个标准化本地流程 +- [x] 清理或移除当前已不再使用的 `src/summary_mcp/core/summarizer.py` +- [x] 更新旧设计文档中仍然残留的 `summary mcp` 描述,避免和当前实现冲突 + +--- + +## P1 - 下一阶段推进 + +- [ ] 把上游 RSS 聚合结果映射成标准化 `item` +- [ ] 补充 `item` 的实际样例文件 +- [ ] 定义规则过滤层的输入输出 schema +- [ ] 设计知识库入库格式 +- [ ] 设计 webhook / 推送格式 +- [ ] 给提取结果增加更多正文清洗策略,比如尾部噪音清理 + +--- + +## P2 - 后续增强项 + +- [ ] 增加批量提取能力 +- [ ] 引入 Playwright 作为动态页面兜底抓取方案 +- [ ] 支持更多 `content_kind`,例如 `release`、`thread`、`video` +- [ ] 增加提取缓存、重试和更细粒度日志 +- [ ] 重命名包和项目名,从 `summary_mcp` 调整为更符合当前职责的名称 +- [ ] 与 OpenClaw 做更正式的工作流编排整合 + +--- + +## 当前建议的下一步 + + +优先做这两件事: + +1. 把上游 RSS 聚合结果映射成标准化 `item` +2. 补充一个真实 `item` 样例文件并开始设计规则过滤 schema + +--- + +## 收束上下文后建议先看 + +- docs/context-reset-brief.md +- docs/content-extract-mcp-mvp-archive.md +- TODO.md + + diff --git a/docs/content-extract-mcp-mvp-archive.md b/docs/content-extract-mcp-mvp-archive.md new file mode 100644 index 0000000..f72a7e0 --- /dev/null +++ b/docs/content-extract-mcp-mvp-archive.md @@ -0,0 +1,275 @@ +# Content Extract MCP MVP 归档记录 + +## 1. 归档目的 + +本文档用于记录当前 MVP 阶段已经完成的能力、实现边界、验证结果和已知限制。 + +当前结论:MCP 这一层不再负责生成摘要,而是只负责将网页或标准化 `item` 提取为结构化文章内容,再把结果交给上层 LLM 做摘要、分类和知识判断。 + +--- + +## 2. 当前 MVP 的目标与边界 + +当前 MVP 实现的是阅读流中的第 2 步前半段: + +`RSS / 页面来源 -> Content Extract MCP -> 结构化文章 JSON -> LLM 摘要` + +### 2.1 当前 MCP 负责的事情 + +- 接收单个 URL +- 接收标准化 `item` +- 抓取网页 HTML +- 抽取标题 +- 抽取正文纯文本 +- 返回结构化文章对象 +- 返回质量标记和结构化错误 + +### 2.2 当前 MCP 不负责的事情 + +- RSS 聚合 +- 摘要生成 +- 主题分类 +- 价值判断 +- 规则过滤 +- 知识库入库 +- 推送通知 + +这意味着: + +当前实现已经把“内容提取”这一层从整条流水线中独立出来,成为一个可被 LLM 或自动化编排系统复用的能力模块。 + +--- + +## 3. 当前实现的服务定位 + +当前项目虽然目录名仍然沿用了 `summary_mcp`,但实际职责已经调整为: + +`Content Extraction MCP` + +也就是: + +- MCP 负责内容提取 +- LLM 负责总结摘要 +- 规则引擎负责过滤 +- sink 负责入库与推送 + +这比早期“让 MCP 直接产出摘要”的思路更合理,因为职责边界更清楚,也更利于后续替换模型和 prompt。 + +--- + +## 4. 当前工具接口 + +当前 MCP 暴露两个 tool: + +- `extract_url_content` +- `extract_item_content` + +### 4.1 `extract_url_content` + +用途: + +- 直接输入 URL +- 适合单条文章调试 +- 适合上层 Agent 在没有标准化 `item` 时直接调用 + +### 4.2 `extract_item_content` + +用途: + +- 输入标准化 `item` +- 适合与上游 RSS 聚合层对接 +- 保留 `item_id`、`source_id` 等追踪信息 + +--- + +## 5. 当前输出结构 + +当前 MCP 输出的核心对象是结构化文章,而不是摘要结果。 + +关键字段包括: + +- `extract_id` +- `item_id` +- `source_id` +- `url` +- `title` +- `author` +- `published_at` +- `language` +- `content_kind` +- `plain_text` +- `quality_flags` +- `metadata` +- `pipeline_state` + +其中: + +- `plain_text` 是给上层 LLM 的核心输入 +- `quality_flags` 用于后续过滤参考 +- `metadata` 里包含当前提取来源和抽取器信息 + +--- + +## 6. 当前代码结构 + +当前目录结构如下: + +```text +src/summary_mcp/ + server.py + core/ + content_loader.py + errors.py + extractor.py + mapper.py + normalizer.py + pipeline.py + quality_checker.py + summarizer.py + models/ + document.py + item.py + summary_io.py +``` + +### 6.1 当前核心模块职责 + +- `server.py` + - MCP 服务入口 + - 注册 `extract_url_content` / `extract_item_content` + +- `normalizer.py` + - 统一 `url` 输入和 `item` 输入 + +- `content_loader.py` + - 负责正文来源选择和回源抓取 + +- `extractor.py` + - 负责标题提取和正文抽取 + +- `quality_checker.py` + - 负责生成质量标记 + +- `mapper.py` + - 负责将结果映射为结构化文章对象 + +- `pipeline.py` + - 负责串联整条提取流程 + +### 6.2 当前遗留项 + +- `core/summarizer.py` 仍保留在仓库中,但已经不再属于当前 MCP 的主流程 +- 包名 `summary_mcp` 和项目名 `summary-mcp` 仍然沿用了早期命名,后续建议重命名为更符合职责的名称 + +--- + +## 7. 当前技术选型 + +当前 MVP 使用: + +- Python +- MCP Python SDK (`FastMCP`) +- `httpx` 进行网络抓取 +- `trafilatura` 进行正文抽取 +- `beautifulsoup4` 作为 HTML 解析和兜底手段 +- `pydantic` 定义输入输出模型 + +设计原则是: + +- 外层 MCP +- 内层 extraction core +- 上层 LLM 单独负责总结和判断 + +--- + +## 8. 当前验证结果 + +### 8.1 语法与结构验证 + +已执行: + +- `python -m compileall src` + +结果: + +- 当前 Python 源码可以通过语法编译检查 + +### 8.2 真实 URL 提取验证 + +已使用以下参考文章进行了真实提取测试: + +- + +验证结果: + +- URL 抓取成功 +- 标题提取成功 +- 正文抽取成功 +- 输出返回结构化 JSON +- `quality_flags` 正常生成 +- `metadata` 中记录了 `content_source=fetched_html` 和 `extractor=trafilatura` + +输出文件: + +- `outputs/read-flow-2026.extracted.json` + +### 8.3 LLM 摘要链路验证 + +已基于提取结果生成: + +- `outputs/llm-summary-prompt.txt` +- `outputs/result.json` + +验证结果: + +- LLM 可基于提取结果生成结构化摘要 JSON +- 在 prompt 收紧后,`summary` 长度和 `keywords/topics` 区分已达到预期 + +--- + +## 9. 当前 MVP 已经达成的结论 + +当前 MVP 已经证明以下链路是成立的: + +1. 参考文章 URL 可以被 MCP 成功提取为结构化内容 +2. 提取结果可以独立保存为 JSON 文件 +3. JSON 文件可以被上层 LLM 消费 +4. LLM 可以输出稳定的摘要结构 +5. “MCP 负责提取,LLM 负责摘要” 这一职责拆分是可行的 + +这意味着: + +当前 MVP 的核心目标已经完成。 + +--- + +## 10. 已知限制 + +当前实现仍存在以下限制: + +- 还没有正式的 JSON Schema 校验器去验证 LLM 输出 +- 还没有批量处理能力 +- 还没有与 RSS 聚合层正式对接 +- 还没有接规则过滤器 +- 还没有接知识库 sink +- 还没有移除遗留的 `summarizer.py` +- 项目命名和职责命名仍存在历史包袱 + +--- + +## 11. 下一阶段建议 + +后续建议按以下顺序推进: + +1. 为 LLM 输出增加 schema 校验 +2. 将提取结果与摘要结果串成标准化流水线 +3. 接入 RSS 聚合层 +4. 引入规则过滤 +5. 接入知识库 sink 或 webhook +6. 最后再考虑与 OpenClaw 的进一步编排整合 + +--- + +## 12. 当前阶段一句话归档结论 + +当前 MVP 已经完成“内容提取 MCP”这一独立能力层:能够将文章 URL 或标准化 item 提取为结构化 JSON,并稳定交给上层 LLM 做摘要与后续处理。 diff --git a/docs/context-reset-brief.md b/docs/context-reset-brief.md new file mode 100644 index 0000000..772c441 --- /dev/null +++ b/docs/context-reset-brief.md @@ -0,0 +1,65 @@ +# 项目当前状态简报 + +## 当前已完成 + +- 已明确整体链路:`来源 -> 聚合池 -> 内容提取 MCP -> LLM 摘要 -> 校验 -> 过滤 -> 入库/推送` +- 已确定当前 MCP 的职责边界: + - 只负责内容提取 + - 不负责摘要、分类、价值判断 +- 已完成 Python MCP 骨架 +- 已实现两个 tool: + - `extract_url_content` + - `extract_item_content` +- 已完成真实 URL 提取验证 +- 已完成 LLM 摘要 prompt +- 已完成 LLM 摘要结果 schema 校验器 +- 已完成“提取 JSON -> LLM 摘要 JSON -> 校验”的最小闭环脚本 +- 已完成 LLM 摘要校验 skill 封装 +- 已清理旧的启发式 `summarizer.py` +- 已将旧的 MCP 设计文档更新为当前“Content Extract MCP”语义 + +## 当前关键文件 + +- MCP 入口: + - `src/summary_mcp/server.py` +- 提取主流程: + - `src/summary_mcp/core/pipeline.py` +- LLM 结果模型: + - `src/summary_mcp/models/llm_result.py` +- LLM 校验器: + - `src/summary_mcp/validators/llm_result.py` +- 校验 CLI: + - `src/summary_mcp/validate_llm_result.py` +- 最小闭环脚本: + - `scripts/run_summary_loop.py` +- 当前提示词: + - `outputs/llm-summary-prompt.txt` +- MVP 归档: + - `docs/content-extract-mcp-mvp-archive.md` +- 当前 TODO: + - `TODO.md` + +## 当前已经验证通过 + +- 参考文章 URL 可提取为结构化 JSON +- LLM 可根据提取结果生成摘要 JSON +- validator 可校验摘要 JSON +- 最小闭环脚本可直接调用 LLM 接口并产出通过校验的结果 + +## 当前未开始的下一阶段 + +- 让 FreshRSS 作为主聚合池 +- 设计 FreshRSS entry -> `item` 的映射 +- 准备真实 `item` 样例 +- 开始定义规则过滤层 schema + +## 收束后建议从这里继续 + +优先从这两个问题继续: + +1. FreshRSS 的 entry 字段如何映射成 `item` +2. 先做一个真实 `item` 样例文件,再讨论规则过滤 + +## 一句话结论 + +当前 MVP 已完成,下一阶段不再是继续打磨 MCP,而是开始接入 FreshRSS 上游并建立 `item` 标准化入口。 diff --git a/docs/reading-pipeline-design-notes.md b/docs/reading-pipeline-design-notes.md new file mode 100644 index 0000000..6d1e7a8 --- /dev/null +++ b/docs/reading-pipeline-design-notes.md @@ -0,0 +1,476 @@ +# 阅读流方案讨论纪要 + +## 参考来源 + +- Shawn Xie, 《Read Flow 2026》: + +## 1. 当前目标 + +本项目的目标不是直接复刻文章里的整套实现,而是先按下面这条链路,逐步做出一套接近文章效果的阅读流系统: + +1. 通过一层 RSS 聚合获取文章 +2. 通过一个 `skill` 或 `mcp` 获取文章摘要 +3. 基于规则对摘要结果进行过滤 +4. 将过滤后的内容推送到知识库或其他文档组件 +5. 完成消息推送 +6. 最后再考虑与 OpenClaw 结合 + +这里的 `skill / mcp` 只是整条链路中的一个组件,当前讨论的重点是第 2 步“页面摘要能力”的设计,而不是一次性把整套系统全部落地。 + +--- +## 2. 内容来源分类 + +为了便于后续实现,内容来源建议按两个维度分类: + +- 按来源类型分类:这个内容来自什么渠道 +- 按接入方式分类:系统用什么技术手段把它纳入统一流水线 + +这样做的原因是: + +- 同一种来源类型,可能有不同接入方式 +- 同一种接入方式,可能服务多种来源 +- 后面的摘要、过滤、入库更适合依赖标准化后的 source schema,而不是依赖渠道名称 + +### 2.1 按来源类型分类 + +建议定义以下几类: + +- `feed` + - 原生 RSS / Atom 源 + - 典型例子:博客、新闻站、技术周刊、文档更新 + +- `converted_feed` + - 原本不是 RSS,但通过第三方或自建转换后变成 feed + - 典型例子:微信公众号、部分社区栏目、社交平台账号流 + +- `event_feed` + - 更偏事件流而不是传统文章流 + - 典型例子:GitHub Releases、Commits、Discussions、Changelog + +- `page_watch` + - 没有现成 feed,需要定时轮询某个页面的更新 + - 典型例子:专题页、导航页、榜单页、产品更新页 + +- `custom_source` + - 无法直接归类,需要为特定站点写专门抓取逻辑 + - 典型例子:结构特殊的网站、内部页面、非标准内容源 + +### 2.2 按接入方式分类 + +建议定义以下几类: + +- `feed_direct` + - 直接消费 RSS / Atom + +- `feed_converted` + - 通过 RSSHub、wewe-rss、RSS-Bridge 等方式先转成 feed,再统一消费 + +- `api_backed` + - 通过平台 API 获取内容,再标准化成内部 item + +- `html_scraped` + - 直接抓取 HTML 页面,自己解析列表页和正文页 + +- `hybrid` + - 先消费 feed 元信息,再按需回源抓正文 + +### 2.3 你当前关心的几类内容如何归档 + +- RSS 文章 + - 来源类型:`feed` + - 接入方式:`feed_direct` + - 优先级:最高 + +- 微信文章 + - 来源类型:`converted_feed` + - 接入方式:通常是 `feed_converted` + - 优先级:高 + +- 其他文章 + - 如果是社区栏目、论坛专题:通常归到 `converted_feed` 或 `page_watch` + - 如果是普通网站更新页:通常归到 `page_watch` + - 如果是 GitHub 项目更新:通常归到 `event_feed` + - 如果是特殊站点:归到 `custom_source` + +结论: + +“其他文章”不应作为最终系统里的正式分类,因为它过于宽泛,后续规则过滤会很难维护。 + +### 2.4 建议的实现级分类 + +如果进入实现阶段,建议系统内部只保留下面这 5 类 source type: + +- `feed` +- `converted_feed` +- `event_feed` +- `page_watch` +- `custom_source` + +同时再给每个 source item 增加一个 `content_kind` 字段,用来表达内容形态: + +- `article` +- `thread` +- `release` +- `changelog` +- `video` +- `mixed` + +这样可以把“来源渠道”和“内容形态”拆开,后面的摘要与过滤规则会更稳定。 + +### 2.5 当前阶段建议先支持的来源 + +为了降低复杂度,第一阶段建议只支持两类: + +- `feed` +- `converted_feed` + +原因: + +- 它们最接近当前目标链路 +- 接入成本最低 +- 最容易验证第 2 步摘要和第 3 步过滤是否成立 +- 暂时不需要引入复杂的页面轮询和定制抓取逻辑 + +--- + +## 3. 系统边界的核心结论 + +当前最重要的不是先决定做 `Skill` 还是 `MCP`,而是先定义一个可复用的“摘要内核”。 + +建议的能力分层: + +`RSS 聚合 -> summary-core -> rule-engine -> sink -> push -> OpenClaw 编排` + +其中: + +- `summary-core`:负责抓取页面、抽取正文、生成结构化摘要 +- `rule-engine`:负责根据规则和评分过滤内容 +- `sink`:负责将结果写入知识库或文档系统 +- `push`:负责消息通知 +- `OpenClaw`:最后作为编排层或自动化入口接入 + +结论:先做“能力内核”,再决定用 `MCP` 还是 `Skill` 进行封装。 + +--- + +## 4. Skill 和 MCP 的定位 + +### Skill + +适合: + +- 人工触发 +- 代理式工作流 +- 提示词编排 +- 让 Agent 在上下文里决定何时调用摘要流程 + +优点: + +- 实现快 +- 适合探索和半自动流程 + +缺点: + +- 不适合做稳定的批处理基础设施 +- 不适合承载缓存、重试、队列、状态管理 + +结论: + +`Skill` 更像“编排层”,不适合作为底层基础能力的唯一实现。 + +### MCP + +适合: + +- 将摘要能力标准化成可调用工具 +- 给 OpenClaw、ChatGPT、Claude 等 Agent 统一接入 +- 为后续自动化编排预留稳定接口 + +优点: + +- 接口标准化 +- 易于被 Agent 调用 +- 后续接 OpenClaw 更自然 + +缺点: + +- 仍然需要自己实现抓取、抽取、摘要、缓存等能力 + +结论: + +`MCP` 适合作为“能力接口层”。 + +### 当前建议 + +推荐顺序: + +1. 先做 `summary-core` +2. 再暴露成 `MCP` +3. 最后根据需要加 `Skill` + +不建议一开始只做 `Skill`。 + +--- + +## 5. 页面摘要能力应输出什么 + +页面摘要不要只输出一段自然语言摘要,而应该输出结构化结果,方便后面的过滤、入库和推送。 + +建议输出结构: + +```json +{ + "url": "https://example.com/article", + "title": "文章标题", + "source": "站点名", + "author": "作者", + "published_at": "2026-03-23T08:00:00Z", + "language": "zh", + "summary": "3-5句摘要", + "highlights": ["要点1", "要点2", "要点3"], + "keywords": ["rss", "mcp", "knowledge-base"], + "topics": ["AI tools", "workflow"], + "quality_flags": { + "is_paywalled": false, + "is_truncated": false, + "is_low_content": false + }, + "scores": { + "readability": 0.84, + "novelty": 0.72, + "relevance": 0.91 + }, + "content_hash": "xxx" +} +``` + +这样做的价值: + +- 规则过滤直接使用结构化字段 +- 知识库入库更稳定 +- 推送内容可按场景裁剪 +- 后续可以追踪质量、去重和打分 + +--- + +## 6. 第 2 步“页面摘要能力”的技术拆分 + +页面摘要不是单一步骤,而是 3 个子能力: + +1. 获取页面内容 +2. 抽取正文 +3. 生成摘要 + +### 5.1 获取页面内容 + +有两条常见路径: + +- 直接使用 RSS item 里的 `content:encoded` 或 `summary` +- 根据 RSS item 的 `link` 回源抓取页面 + +建议策略: + +- 先使用 RSS 已提供的内容 +- 如果内容过短、质量不够,再回源抓页面正文 + +这样能平衡速度和完整度。 + +### 5.2 正文抽取 + +这是稳定性的关键环节。 + +候选技术: + +- `trafilatura` + - 对博客、新闻、文档类页面较稳 + - Python 生态成熟 + +- `readability-lxml` + - 经典轻量方案 + - 覆盖面广 + +- `Playwright + Readability` + - 适合 JS 动态页面 + - 成本最高,但兜底能力最强 + +建议组合: + +- 第一层:`httpx/requests + trafilatura` +- 第二层:`Playwright` 作为兜底方案 + +不建议一开始就默认使用 Playwright。 + +### 5.3 摘要生成 + +有两类方向: + +- 抽取式摘要 + - 成本低 + - 稳定 + - 可读性通常一般 + +- 生成式摘要 + - 更接近最终想要的“知识流”效果 + - 可读性更好 + - 需要更严格的结构化约束 + +建议: + +- 采用生成式摘要 +- 强制模型输出 JSON Schema +- 不允许自由散文式输出 + +这样后续流程会稳定很多。 + +--- + +## 7. 过滤层设计 + +过滤层不要完全依赖 LLM 的自由判断,建议采用“两层过滤”。 + +### 第一层:确定性规则 + +例如: + +- 来源白名单 / 黑名单 +- 关键词规则 +- 主题规则 +- 语言过滤 +- 最低字数要求 +- 重复检测 +- 发布时间窗口 +- relevance 阈值 + +### 第二层:可选的 LLM 判断 + +例如: + +- 是否值得进入知识库 +- 更偏资讯还是方法论 +- 是否符合个人关注主题 +- 是否值得推送 + +结论: + +先规则,后智能,不要反过来。 + +--- + +## 8. 知识库与推送层设计 + +第 4、5 步可以统一抽象为 `sink` 和 `push`。 + +### 第一阶段建议优先做的 sink + +- `Markdown sink` + - 输出到本地目录或 Git 仓库 + - 适合第一版 + - 可审计、易迁移 + +- `Webhook sink` + - 推送到 Telegram、企业微信、飞书、OpenClaw 或其他自动化入口 + - 扩展性最好 + +### 后续可扩展的 sink + +- Notion +- Obsidian Vault +- Logseq +- 数据库 + +结论: + +第一版不建议直接绑死在单一平台上,优先本地 Markdown + Webhook。 + +--- + +## 9. 现在真正该先定义的接口 + +摘要内核建议先定义统一接口,而不是先定义 UI 或 Agent 提示词。 + +建议输入: + +- `url` +- 可选 `raw_html` +- 可选 `rss_content` +- 可选 `user_profile` +- 可选 `summary_style` + +建议输出: + +- 结构化摘要 JSON +- 原始正文文本 +- 元信息 +- 错误状态 + +只要这个接口稳定: + +- CLI 可以调用 +- HTTP 服务可以调用 +- MCP 可以调用 +- OpenClaw Skill 也可以调用 + +这一步是整个系统里最关键的抽象。 + +--- + +## 10. 推荐的演进路线 + +建议按下面顺序推进: + +1. `RSS 聚合` + - 先通过 FreshRSS 或现有聚合器获取文章链接 +2. `summary-core` + - 输入 URL,输出结构化摘要 +3. `rule-engine` + - 对摘要 JSON 执行过滤规则 +4. `sink` + - 先落 Markdown,再做 webhook 推送 +5. `MCP server` + - 暴露 `summarize_url`、`filter_summary`、`store_note` 等工具 +6. `OpenClaw Skill` + - 让 OpenClaw 负责编排、人工确认或附加动作 +7. `OpenClaw 集成` + - 最后再接入完整自动化流程 + +这样每一层都可以独立测试,不会过早被某个平台或代理框架绑定。 + +--- + +## 11. 当前阶段的明确判断 + +### 不建议的方向 + +- 一开始只做 Skill +- 一开始把 OpenClaw 当成底层依赖 +- 一开始就直接绑死 Notion 一类单一知识库 +- 让过滤完全依赖 LLM 自由发挥 + +### 建议优先做的方向 + +- 先定义 `summary-core` 的输入输出 +- 摘要结果必须结构化 +- 过滤规则优先基于确定性逻辑 +- 入库优先选择 Markdown +- 对外优先暴露成 MCP,而不是先做纯 Skill + +--- + +## 12. 后续讨论建议 + +后面可以基于这份文档继续讨论以下问题: + +1. `summary-core` 的最小可用接口如何定义 +2. `MCP` 版和 `Skill` 版分别暴露哪些能力 +3. 正文抽取链路是否需要分层回退 +4. 规则引擎使用配置文件还是代码实现 +5. Markdown sink 的目录结构怎么设计 +6. Webhook / 推送层先接哪一个目标 +7. OpenClaw 在整条链路里承担“调度器”还是“人工审核入口” + +--- + +## 13. 当前阶段的一句话结论 + +当前阶段最合理的方向是:先把“页面摘要”做成一个独立、结构化、可复用的能力内核,再优先封装为 `MCP`,最后再用 `Skill` 和 OpenClaw 做编排与集成。 + diff --git a/docs/source-schema-design.md b/docs/source-schema-design.md new file mode 100644 index 0000000..a196a33 --- /dev/null +++ b/docs/source-schema-design.md @@ -0,0 +1,544 @@ +# Source Schema 设计草案 + +## 1. 文档目的 + +本文档用于定义阅读流系统中的来源与内容对象模型,目标是把“来源分类”的讨论收敛成一套可执行的数据结构,供后续的抓取、摘要、过滤、入库和推送流程统一使用。 + +这份文档关注的是数据抽象,而不是具体实现语言、数据库模型或 API 细节。 + +--- + +## 2. 设计目标 + +这套 schema 需要解决以下问题: + +- 不同渠道的内容如何进入同一条流水线 +- RSS、微信文章、普通网页、GitHub 更新如何被统一表示 +- 摘要器应该接收什么输入对象 +- 过滤规则应该依赖哪些标准字段 +- 入库与推送如何复用统一元信息 + +核心目标是建立稳定的中间层,避免后续各模块直接依赖渠道差异。 + +--- + +## 3. 设计原则 + +### 3.1 分层而不是混合 + +建议将系统中的对象分成三层: + +- `source` +- `item` +- `document` + +每一层职责不同: + +- `source`:描述一个来源渠道本身 +- `item`:描述一次抓取到的候选内容 +- `document`:描述经过正文抽取和摘要后的标准内容对象 + +### 3.2 先标准化,再智能化 + +在进入摘要、过滤、入库前,应该先把不同渠道的内容统一映射成相同的对象结构,而不是在每个处理阶段临时兼容各种来源格式。 + +### 3.3 渠道类型与内容形态分离 + +建议明确区分: + +- 来源渠道是什么 +- 内容本身是什么形态 + +例如: + +- 微信文章和 RSS 博文可能都属于 `article` +- GitHub Release 和 changelog 可能来自不同渠道,但内容形态更接近事件流 + +--- + +## 4. 对象分层 + +### 4.1 `source` + +`source` 是来源配置对象,用来描述某个渠道本身是什么、如何接入、是否启用。 + +特点: + +- 偏静态 +- 由用户或系统配置维护 +- 不代表具体某一篇内容 + +### 4.2 `item` + +`item` 是采集层产物,用来表示“这次抓回来的一个候选条目”。 + +特点: + +- 来自 RSS、API、HTML 抓取或转换 feed +- 可能只有标题、链接和摘要 +- 不保证已经拿到完整正文 + +### 4.3 `document` + +`document` 是标准内容对象,用来表示“已经过正文抽取和摘要处理,可以进入过滤、入库和推送阶段”。 + +特点: + +- 已经完成正文抽取 +- 已经有结构化摘要 +- 是后续规则引擎的主输入 + +--- + +## 5. 枚举设计 + +### 5.1 `source_type` + +表示来源类型。 + +可选值: + +- `feed` +- `converted_feed` +- `event_feed` +- `page_watch` +- `custom_source` + +说明: + +- `feed`:原生 RSS / Atom +- `converted_feed`:由 RSSHub、wewe-rss、RSS-Bridge 等转换而来 +- `event_feed`:更像事件流,例如 GitHub Releases、Commits、Discussions +- `page_watch`:轮询普通网页的更新情况 +- `custom_source`:为特殊站点定制的抓取源 + +### 5.2 `ingest_mode` + +表示接入方式。 + +可选值: + +- `feed_direct` +- `feed_converted` +- `api_backed` +- `html_scraped` +- `hybrid` + +说明: + +- `feed_direct`:直接消费 RSS / Atom +- `feed_converted`:先转换成 feed,再统一消费 +- `api_backed`:通过 API 获取数据 +- `html_scraped`:直接抓 HTML 页面并解析 +- `hybrid`:先消费 feed 元数据,再按需抓正文 + +### 5.3 `content_kind` + +表示内容形态。 + +可选值: + +- `article` +- `thread` +- `release` +- `changelog` +- `video` +- `mixed` + +说明: + +- `article`:标准文章 +- `thread`:串联式内容,例如帖子串、社交媒体长线程 +- `release`:版本发布说明 +- `changelog`:更新日志 +- `video`:视频内容 +- `mixed`:无法明确归入单一形态 + +### 5.4 `fetch_state` + +表示候选内容的正文抓取状态。 + +可选值: + +- `pending` +- `fetched` +- `failed` +- `skipped` + +### 5.5 `pipeline_state` + +表示内容在整条流水线中的处理状态。 + +可选值: + +- `ingested` +- `summarized` +- `filtered` +- `stored` +- `pushed` +- `dropped` + +--- + +## 6. `source` 对象设计 + +### 6.1 角色定义 + +`source` 用于描述一个来源本身,而不是某条具体内容。 + +它应该回答这些问题: + +- 这个来源是什么 +- 它属于哪一类来源 +- 用什么方式接入 +- 是否启用 +- 默认语言和默认内容形态是什么 + +### 6.2 建议字段 + +```json +{ + "source_id": "wechat-aiweekly", + "name": "AI Weekly 微信源", + "source_type": "converted_feed", + "ingest_mode": "feed_converted", + "content_kind": "article", + "base_url": "https://example.com", + "feed_url": "https://example.com/feed.xml", + "language": "zh", + "enabled": true, + "tags": ["ai", "newsletter"], + "priority": 80 +} +``` + +### 6.3 字段说明 + +- `source_id` + - 系统内部唯一标识 + +- `name` + - 人类可读名称 + +- `source_type` + - 来源类型 + +- `ingest_mode` + - 接入方式 + +- `content_kind` + - 默认内容形态 + +- `base_url` + - 来源站点主域名 + +- `feed_url` + - feed 地址;没有则可为空 + +- `language` + - 默认语言 + +- `enabled` + - 是否启用该来源 + +- `tags` + - 来源级标签,供过滤和统计使用 + +- `priority` + - 用于抓取调度、结果排序或推送优先级控制 + +### 6.4 最小可用版本 + +```json +{ + "source_id": "my-blog", + "source_type": "feed", + "ingest_mode": "feed_direct", + "feed_url": "https://example.com/rss.xml", + "enabled": true +} +``` + +--- + +## 7. `item` 对象设计 + +### 7.1 角色定义 + +`item` 是采集层的标准对象,用来统一表示从不同渠道抓回来的候选内容。 + +它应该回答这些问题: + +- 这条候选内容来自哪个 source +- 它的标题、链接和发布时间是什么 +- RSS 是否已经提供摘要或正文 +- 当前是否已抓到完整正文 + +### 7.2 建议字段 + +```json +{ + "item_id": "sha256:xxx", + "source_id": "wechat-aiweekly", + "external_id": "feed-entry-id-or-url", + "title": "文章标题", + "url": "https://example.com/post/123", + "author": "作者", + "published_at": "2026-03-23T08:00:00Z", + "discovered_at": "2026-03-23T10:00:00Z", + "content_kind": "article", + "language": "zh", + "raw_summary": "RSS 提供的摘要", + "raw_content": "RSS 提供的正文或片段", + "metadata": { + "feed_title": "源标题", + "categories": ["AI", "Tools"] + }, + "fetch_state": "pending" +} +``` + +### 7.3 字段说明 + +- `item_id` + - 内部唯一标识 + - 建议由 `source_id + canonical_url + published_at` 做 hash 生成 + +- `source_id` + - 指向来源配置对象 + +- `external_id` + - 来源平台本身的 id,例如 RSS entry id、原始 URL 或平台内容 id + +- `title` + - 候选内容标题 + +- `url` + - 原始内容链接 + +- `author` + - 作者信息;如果无法获取可为空 + +- `published_at` + - 原始发布时间 + +- `discovered_at` + - 被系统发现的时间 + +- `content_kind` + - 当前条目的内容形态 + +- `language` + - 当前条目的语言 + +- `raw_summary` + - 来源提供的摘要内容 + +- `raw_content` + - 来源直接提供的正文、正文片段或富文本转纯文本结果 + +- `metadata` + - 保留来源特有元信息,例如 feed 分类、栏目名、标签 + +- `fetch_state` + - 正文抓取状态 + +### 7.4 最小可用版本 + +```json +{ + "item_id": "sha256:xxx", + "source_id": "my-blog", + "title": "文章标题", + "url": "https://example.com/post/1", + "published_at": "2026-03-23T08:00:00Z", + "raw_summary": "摘要", + "raw_content": "", + "fetch_state": "pending" +} +``` + +--- + +## 8. `document` 对象设计 + +### 8.1 角色定义 + +`document` 是经过正文抽取和摘要处理后的标准内容对象。 + +它应该回答这些问题: + +- 最终正文是什么 +- 摘要和高亮点是什么 +- 这个内容值不值得进入下一步过滤或入库 +- 当前它处于流水线哪个阶段 + +### 8.2 建议字段 + +```json +{ + "document_id": "sha256:yyy", + "item_id": "sha256:xxx", + "source_id": "wechat-aiweekly", + "url": "https://example.com/post/123", + "title": "文章标题", + "author": "作者", + "published_at": "2026-03-23T08:00:00Z", + "language": "zh", + "content_kind": "article", + "plain_text": "抽取后的正文", + "summary": "3-5句摘要", + "highlights": ["要点1", "要点2", "要点3"], + "keywords": ["rss", "mcp"], + "topics": ["workflow", "knowledge-base"], + "scores": { + "relevance": 0.91, + "novelty": 0.72, + "readability": 0.84 + }, + "quality_flags": { + "is_paywalled": false, + "is_truncated": false, + "is_low_content": false + }, + "pipeline_state": "summarized" +} +``` + +### 8.3 字段说明 + +- `document_id` + - `document` 的唯一标识 + +- `item_id` + - 回溯到原始候选条目 + +- `source_id` + - 回溯到来源配置 + +- `url` + - 原始链接 + +- `title` + - 标题 + +- `author` + - 作者信息 + +- `published_at` + - 发布时间 + +- `language` + - 语言 + +- `content_kind` + - 内容形态 + +- `plain_text` + - 抽取后的正文纯文本 + +- `summary` + - 结构化摘要中的主摘要 + +- `highlights` + - 关键要点 + +- `keywords` + - 关键词 + +- `topics` + - 主题标签 + +- `scores` + - 打分结果,例如相关性、新颖度、可读性 + +- `quality_flags` + - 内容质量标志,例如是否付费墙、是否截断、是否低信息密度 + +- `pipeline_state` + - 当前流水线状态 + +### 8.4 最小可用版本 + +```json +{ + "document_id": "sha256:yyy", + "item_id": "sha256:xxx", + "title": "文章标题", + "url": "https://example.com/post/1", + "plain_text": "正文", + "summary": "3-5句摘要", + "highlights": ["点1", "点2"], + "scores": { + "relevance": 0.9 + }, + "pipeline_state": "summarized" +} +``` + +--- + +## 9. 三层对象之间的关系 + +建议关系如下: + +- 一个 `source` 可以产生多个 `item` +- 一个 `item` 在多数情况下对应一个 `document` +- `document` 是 `item` 经过正文抽取和摘要处理后的结果 + +简化理解: + +- `source` 解决“从哪里来” +- `item` 解决“抓到了什么” +- `document` 解决“它值不值得留下” + +--- + +## 10. 为什么必须分层 + +如果不区分 `source`、`item` 和 `document`,后面会出现这些问题: + +- 来源配置和运行时数据混在一起 +- RSS 字段、网页抓取字段和摘要字段混在一起 +- 过滤规则既要判断来源,又要判断内容质量,还要判断摘要分数 +- 不同来源的差异会不断渗透到下游模块 + +分层后的收益是: + +- 采集层可以独立演进 +- 摘要器输入更稳定 +- 过滤器只依赖统一对象,不依赖来源细节 +- 入库和推送更容易复用 + +--- + +## 11. 当前阶段的最小实现建议 + +为了降低复杂度,第一阶段建议: + +- 只支持 `feed` 和 `converted_feed` +- 只实现 `source`、`item`、`document` 这三层基础对象 +- `item` 只要求能容纳 RSS 拉回来的标题、链接、摘要、正文片段 +- `document` 只要求能容纳正文、摘要、高亮点和基础打分 + +这样做的好处是: + +- 结构足够稳定 +- 成本足够低 +- 可以直接支撑下一步 `summary-core` 设计 + +--- + +## 12. 后续可继续讨论的问题 + +1. 是否需要给 `source` 增加抓取周期、超时、重试策略字段 +2. `item_id` 和 `document_id` 的生成策略是否要统一 +3. `metadata` 是否需要按来源类型拆出专用字段 +4. `document` 的 `scores` 和 `quality_flags` 是否要定义更严格的 schema +5. 这套 schema 最终是落成 JSON 文件、数据库表,还是 Pydantic 模型 + +--- + +## 13. 当前阶段的一句话结论 + +当前最合适的做法是先固定 `source -> item -> document` 这三层对象模型,再围绕它设计 `summary-core`、过滤器和入库流程。 diff --git a/docs/summary-core-interface-design.md b/docs/summary-core-interface-design.md new file mode 100644 index 0000000..ec0048c --- /dev/null +++ b/docs/summary-core-interface-design.md @@ -0,0 +1,503 @@ +# Summary Core Interface 设计草案 + +## 1. 文档目的 + +本文档用于定义 `summary-core` 的输入输出接口,目标是把“页面摘要能力”从概念讨论收敛成一套稳定、可复用、可封装的数据接口。 + +这份文档不限定实现语言,也不限定它最终通过 CLI、HTTP、MCP 还是 Skill 暴露,只关注能力边界和数据契约。 + +--- + +## 2. 角色定位 + +`summary-core` 是阅读流系统中的摘要内核,负责把候选内容转换成可进入过滤、入库和推送阶段的标准化内容对象。 + +它在整条链路中的位置是: + +`RSS 聚合 / 页面抓取 -> summary-core -> rule-engine -> sink -> push` + +它的职责是: + +- 接收一个候选内容对象或与之等价的输入 +- 获取正文或补全正文 +- 抽取正文纯文本 +- 生成结构化摘要 +- 输出标准化的 `document` 对象或与之兼容的结果 + +它不负责: + +- 维护 RSS 订阅源 +- 执行规则过滤 +- 写入知识库 +- 执行消息推送 +- 负责编排整个自动化流程 + +--- + +## 3. 设计原则 + +### 3.1 输入以 `item` 为主,补充字段为辅 + +`summary-core` 的标准输入应优先基于 `item`,这样可以和上游采集层稳定衔接。 + +但为了适配不同调用方式,也允许附带额外输入: + +- `raw_html` +- `rss_content` +- `user_profile` +- `summary_style` + +### 3.2 输出以 `document` 为中心 + +`summary-core` 的输出应该尽量直接映射到 `document`,而不是返回一段松散文本。 + +这样后续的规则引擎、知识库入库和推送层都可以稳定消费。 + +### 3.3 失败也要结构化表达 + +正文抓取失败、内容过短、抽取结果异常、疑似付费墙等情况,不能只返回报错字符串,应该返回结构化错误或状态标志。 + +### 3.4 同一套数据契约服务多种封装方式 + +不管未来是: + +- 本地 CLI +- HTTP API +- MCP tool +- OpenClaw Skill + +都应尽量复用同一套输入输出 schema,而不是每个入口都定义一套不同的数据格式。 + +--- + +## 4. 处理阶段拆分 + +`summary-core` 内部建议拆成以下几个阶段: + +1. 输入标准化 +2. 正文获取 +3. 正文抽取 +4. 摘要生成 +5. 质量标记 +6. 输出映射 + +### 4.1 输入标准化 + +目标: + +- 统一不同调用来源的数据结构 +- 优先使用 `item` +- 补齐缺失字段 + +### 4.2 正文获取 + +目标: + +- 优先使用上游已提供的 `raw_content` +- 若内容不足,再根据 `url` 回源抓取页面 +- 如有 `raw_html`,优先直接使用 + +### 4.3 正文抽取 + +目标: + +- 从 HTML 或富文本中抽取正文 +- 生成稳定的纯文本内容 +- 识别是否正文过短、被截断或包含大量噪音 + +### 4.4 摘要生成 + +目标: + +- 基于正文生成结构化摘要 +- 输出摘要、高亮点、关键词、主题和评分 +- 保证结构化输出稳定,不允许自由散文式结果 + +### 4.5 质量标记 + +目标: + +- 判断是否存在付费墙、截断、低信息密度等问题 +- 为后续过滤器提供可用标志位 + +### 4.6 输出映射 + +目标: + +- 将摘要结果映射为标准 `document` +- 保持与上游 `item` 的可追溯关系 + +--- + +## 5. 输入模型 + +### 5.1 标准输入结构 + +建议标准输入命名为 `SummaryInput`。 + +```json +{ + "item": { + "item_id": "sha256:xxx", + "source_id": "my-blog", + "title": "文章标题", + "url": "https://example.com/post/1", + "published_at": "2026-03-23T08:00:00Z", + "raw_summary": "RSS 摘要", + "raw_content": "RSS 正文片段", + "content_kind": "article", + "language": "zh" + }, + "raw_html": "...", + "rss_content": "RSS 提供的正文内容", + "user_profile": { + "interests": ["AI", "workflow"], + "languages": ["zh", "en"] + }, + "summary_style": "default" +} +``` + +### 5.2 输入字段说明 + +- `item` + - 标准主输入 + - 推荐必填 + +- `raw_html` + - 可选 + - 当调用方已经抓到网页 HTML 时可直接传入,避免重复抓取 + +- `rss_content` + - 可选 + - 当调用方希望优先使用 RSS 正文时可单独传入 + +- `user_profile` + - 可选 + - 用于后续个性化摘要、主题提取或兴趣打分 + - 第一阶段可以不启用,但接口层建议预留 + +- `summary_style` + - 可选 + - 用于控制摘要策略,例如 `default`、`brief`、`detailed`、`bullet` + +### 5.3 最小可用输入 + +第一阶段建议只要求: + +```json +{ + "item": { + "item_id": "sha256:xxx", + "title": "文章标题", + "url": "https://example.com/post/1", + "raw_summary": "RSS 摘要", + "raw_content": "" + } +} +``` + +--- + +## 6. 输出模型 + +### 6.1 标准输出结构 + +建议标准输出命名为 `SummaryOutput`。 + +```json +{ + "success": true, + "document": { + "document_id": "sha256:yyy", + "item_id": "sha256:xxx", + "source_id": "my-blog", + "url": "https://example.com/post/1", + "title": "文章标题", + "author": "作者", + "published_at": "2026-03-23T08:00:00Z", + "language": "zh", + "content_kind": "article", + "plain_text": "抽取后的正文", + "summary": "3-5句摘要", + "highlights": ["要点1", "要点2", "要点3"], + "keywords": ["rss", "mcp"], + "topics": ["workflow", "knowledge-base"], + "scores": { + "relevance": 0.91, + "novelty": 0.72, + "readability": 0.84 + }, + "quality_flags": { + "is_paywalled": false, + "is_truncated": false, + "is_low_content": false + }, + "pipeline_state": "summarized" + }, + "debug": { + "content_source": "fetched_html", + "extractor": "trafilatura", + "model": "summary-model-name" + }, + "warnings": [] +} +``` + +### 6.2 输出字段说明 + +- `success` + - 是否成功完成摘要流程 + +- `document` + - 标准输出主体 + - 结构应尽量与 `document` schema 对齐 + +- `debug` + - 可选 + - 记录本次实际用了什么正文来源、抽取器和模型 + - 便于后续排查问题 + +- `warnings` + - 可选 + - 用于记录非致命异常,例如正文过短、内容疑似截断、元信息缺失 + +### 6.3 最小可用输出 + +```json +{ + "success": true, + "document": { + "document_id": "sha256:yyy", + "item_id": "sha256:xxx", + "title": "文章标题", + "url": "https://example.com/post/1", + "plain_text": "正文", + "summary": "3-5句摘要", + "highlights": ["点1", "点2"], + "scores": { + "relevance": 0.9 + }, + "pipeline_state": "summarized" + }, + "warnings": [] +} +``` + +--- + +## 7. 错误模型 + +### 7.1 为什么要有独立错误模型 + +如果 `summary-core` 只在失败时返回字符串错误,会导致: + +- 上层流程难以判断是否应该重试 +- 过滤器和编排层难以做自动决策 +- MCP 或 Skill 封装时难以标准化输出 + +因此建议引入结构化错误模型。 + +### 7.2 建议错误结构 + +```json +{ + "success": false, + "error": { + "code": "CONTENT_FETCH_FAILED", + "message": "Failed to fetch article content", + "retryable": true, + "stage": "fetch", + "details": { + "url": "https://example.com/post/1" + } + }, + "warnings": [] +} +``` + +### 7.3 建议错误码 + +- `INVALID_INPUT` +- `CONTENT_FETCH_FAILED` +- `CONTENT_EXTRACTION_FAILED` +- `CONTENT_TOO_SHORT` +- `PAYWALL_DETECTED` +- `SUMMARY_GENERATION_FAILED` +- `UNSUPPORTED_CONTENT_KIND` +- `UNKNOWN_ERROR` + +### 7.4 建议错误字段 + +- `code` + - 机器可识别错误码 + +- `message` + - 人类可读错误信息 + +- `retryable` + - 是否建议重试 + +- `stage` + - 出错阶段,例如 `normalize`、`fetch`、`extract`、`summarize` + +- `details` + - 补充调试信息 + +--- + +## 8. 内容来源优先级策略 + +为了保证 `summary-core` 在不同输入条件下行为稳定,建议明确正文来源优先级。 + +推荐顺序: + +1. `raw_html` +2. `item.raw_content` +3. `rss_content` +4. 根据 `item.url` 回源抓取 + +解释: + +- 如果调用方已经传入 `raw_html`,优先直接用,避免重复请求 +- 如果 RSS 已经提供了高质量正文,可优先利用 +- 如果 RSS 内容不足,再回源抓页面 + +这一策略需要在实现阶段进一步细化,例如通过长度阈值判断是否需要回源抓取。 + +--- + +## 9. 与 `document` schema 的关系 + +`summary-core` 的输出不应另起炉灶,而应尽量直接映射到 `document`。 + +建议遵守以下原则: + +- `SummaryOutput.document` 与 `document` schema 保持高度一致 +- `summary-core` 只负责生成 `document`,不负责直接写入知识库 +- 上游保留 `item_id` 和 `source_id`,确保可追溯 + +这样做的好处是: + +- 下游过滤器只需要消费 `document` +- 存储层无需理解摘要实现细节 +- MCP 封装和 CLI 调用都可以复用同一结果结构 + +--- + +## 10. 调用方式兼容性 + +同一套接口建议兼容以下几种封装方式: + +### 10.1 CLI + +适合: + +- 本地验证 +- 单条 URL 调试 +- 批处理脚本调用 + +建议形式: + +- `summary-core input.json` +- `summary-core --url https://example.com/post/1` + +### 10.2 HTTP API + +适合: + +- 被其他服务调用 +- 作为统一后端能力暴露 + +建议形式: + +- `POST /summarize` + +### 10.3 MCP Tool + +适合: + +- 给 OpenClaw、ChatGPT、Claude 等 Agent 调用 + +建议形式: + +- `summarize_item` +- `summarize_url` + +### 10.4 Skill + +适合: + +- 让 Agent 在上下文中决定是否调用摘要流程 +- 与其他工具组合编排 + +结论: + +接口 schema 应保持统一,不因为调用方式不同而拆成不同语义模型。 + +--- + +## 11. 当前阶段的最小可用接口 + +第一阶段建议只实现一个最小接口: + +输入: + +```json +{ + "item": { + "item_id": "sha256:xxx", + "title": "文章标题", + "url": "https://example.com/post/1", + "raw_summary": "RSS 摘要", + "raw_content": "" + } +} +``` + +输出: + +```json +{ + "success": true, + "document": { + "document_id": "sha256:yyy", + "item_id": "sha256:xxx", + "title": "文章标题", + "url": "https://example.com/post/1", + "plain_text": "正文", + "summary": "3-5句摘要", + "highlights": ["点1", "点2"], + "scores": { + "relevance": 0.9 + }, + "pipeline_state": "summarized" + }, + "warnings": [] +} +``` + +这个阶段先不要求: + +- 个性化摘要 +- 多风格摘要模板 +- 批量接口 +- 复杂质量评分 +- 高级缓存策略 + +--- + +## 12. 后续可继续讨论的问题 + +1. `SummaryInput` 是否必须包含完整 `item`,还是允许只传 URL +2. `summary_style` 是否应该在第一阶段就进入 schema +3. `debug` 字段是否应面向开发环境可选开启 +4. 错误码是否需要更细分成抓取类、抽取类、模型类 +5. `plain_text` 是否需要附带 token 数、字数等统计信息 +6. `SummaryOutput` 是否需要保留原始抽取文本和最终摘要之间的映射信息 + +--- + +## 13. 当前阶段的一句话结论 + +当前阶段最合理的做法是先固定 `summary-core` 的统一输入输出契约,让它稳定接收 `item`,稳定产出 `document`,再在此基础上封装 CLI、MCP 和 Skill。 diff --git a/docs/summary-loop-explained.md b/docs/summary-loop-explained.md new file mode 100644 index 0000000..a23940c --- /dev/null +++ b/docs/summary-loop-explained.md @@ -0,0 +1,229 @@ +# Summary Loop 工作说明 + +## 1. 目的 + +本文档说明 `scripts/run_summary_loop.py` 如何把以下流程串成一个最小闭环: + +`提取 JSON -> LLM 生成摘要 JSON -> validator 校验 -> 失败则修复重试` + +这个脚本的目标不是单次调用 LLM,而是让 LLM 输出进入一个“生成后验收”的自动反馈回路。 + +--- + +## 2. 核心思路 + +这个闭环由三部分组成: + +- 提取结果 + - 来自 `outputs/*.extracted.json` + - 提供标题、链接、正文、质量标记 + +- LLM 摘要 + - 基于提取结果和 prompt 生成 `result.json` + +- validator + - 检查 `result.json` 是否符合结构和业务规则 + - 不通过则返回错误列表 + +因此,LLM 不是“自己知道哪里错了”,而是脚本在每轮生成后用 validator 明确指出错误,再要求它修复。 + +--- + +## 3. 运行顺序 + +脚本入口: + +- `scripts/run_summary_loop.py` + +执行顺序如下: + +1. 读取提取结果 JSON +2. 读取摘要 prompt 模板 +3. 从提取结果中裁剪出摘要真正需要的输入字段 +4. 调用 LLM 生成摘要 JSON +5. 解析模型输出中的 JSON +6. 将结果写入输出文件 +7. 调用 validator 校验结果 +8. 如果通过,结束 +9. 如果失败,构造修复 prompt,再次调用 LLM + +--- + +## 4. 输入来源 + +脚本主要使用两个输入文件: + +- 提取结果:`outputs/read-flow-2026.extracted.json` +- 摘要 prompt:`outputs/llm-summary-prompt.txt` + +提取结果不会整包无差别塞给 LLM,而是先裁剪成更小的摘要输入: + +- `article.title` +- `article.url` +- `article.plain_text` +- `article.quality_flags` +- `warnings` + +这样做是为了减少 payload,降低模型超时概率,也让 prompt 更聚焦。 + +--- + +## 5. 首次生成 + +第一次生成时,脚本会把: + +- prompt 模板 +- 结构化文章输入 + +拼成一个完整请求,发给 LLM。 + +对应函数: + +- `build_initial_prompt(...)` +- `call_llm(...)` + +模型返回后,脚本会尝试提取一个 JSON 对象。 + +如果模型输出不是合法 JSON: + +- 这一轮直接视为失败 +- 会生成一份失败校验结果 +- 然后进入修复重试 + +--- + +## 6. validator 如何加入流程 + +validator 不是附加步骤,而是主流程里的硬门槛。 + +对应函数: + +- `validate_llm_result(...)` + +校验内容包括两层: + +### 6.1 结构校验 + +- 必填字段是否存在 +- 字段类型是否正确 +- `category` 是否属于允许枚举 +- `summary` 长度是否合规 +- `highlights` / `keywords` / `topics` 数量是否合规 + +### 6.2 业务校验 + +- `keywords` 和 `topics` 不能重复 +- 列表内部不能重复 +- `title` 和 `url` 必须与提取结果一致 + +只有 validator 返回 `valid: true`,这一轮结果才会被接受。 + +--- + +## 7. LLM 如何知道自己生成错了 + +LLM 本身不会主动知道哪些字段不合规。 + +脚本会在校验失败后,把以下信息重新发给 LLM: + +- validator 错误列表 +- 原始提取结果 +- 当前错误的摘要 JSON + +然后要求它: + +- 只修复错误字段 +- 保留已经正确的字段 +- 继续只输出合法 JSON + +对应函数: + +- `build_repair_prompt(...)` + +也就是说: + +- prompt 负责提出目标 +- validator 负责判错 +- repair prompt 负责把错误喂回模型 + +这三者一起形成自动修复回路。 + +--- + +## 8. 时序图 + +```text +Extracted JSON + | + v +run_summary_loop.py + | + |-- 读取 prompt 模板 + |-- 构造摘要输入 + | + v +LLM + | + v +摘要 JSON + | + v +validator + | + |-- valid = true ------> 结束 + | + |-- valid = false + | | + | |-- errors + | v + | repair prompt + | | + +------> LLM 重试 +``` + +--- + +## 9. 生成的中间文件 + +每次执行脚本都会在输出目录下留下调试痕迹: + +- `result.loop.attempt-1.raw.txt` + - 模型原始输出 + +- `result.loop.attempt-1.json` + - 提取出的 JSON 结果 + +- `result.loop.attempt-1.validation.json` + - 这一轮的校验报告 + +- `result.loop.json` + - 当前最终结果 + +这些文件的价值是: + +- 可以回看模型原始输出 +- 可以定位是 JSON 解析失败还是 schema 校验失败 +- 可以看每轮修复到底修了什么 + +--- + +## 10. 当前脚本参数 + +当前脚本支持: + +- `--extracted` +- `--prompt` +- `--output` +- `--max-retries` +- `--timeout` +- `--api-key` +- `--model` +- `--api-url` + +这意味着它既可以走环境变量,也可以直接用命令行传入模型配置。 + +--- + +## 11. 一句话结论 + +`run_summary_loop.py` 的本质不是“调一次 LLM”,而是“让 LLM 生成结果后必须经过 validator 验收,失败就拿着错误清单继续修”,直到结果通过或达到重试上限。 diff --git a/docs/summary-mcp-service-design.md b/docs/summary-mcp-service-design.md new file mode 100644 index 0000000..68f67f0 --- /dev/null +++ b/docs/summary-mcp-service-design.md @@ -0,0 +1,378 @@ +# Content Extract MCP Service 设计草案 + +## 1. 文档目的 + +本文档用于定义当前仓库中已经落地的 MCP 服务设计,即“内容提取 MCP”。 + +目标是明确: + +- 这个 MCP 服务当前真正负责什么 +- 它对外暴露哪些 tool +- 每个 tool 的输入输出结构是什么 +- validator 与 LLM 摘要如何接在 MCP 之后 +- 当前 MVP 已完成到哪一层 + +这份文档描述的是当前真实实现,而不是早期“摘要 MCP”设想。 + +--- + +## 2. 服务定位 + +这个 MCP 服务的角色是:为上层 Agent、OpenClaw 或其他自动化流程提供统一的“文章内容提取”能力。 + +它在整条链路中的位置是: + +`RSS 聚合 -> Content Extract MCP -> LLM 摘要 -> 校验 -> 规则过滤 -> 入库 -> 推送` + +它不是完整阅读流系统,也不是摘要服务本身,而是阅读流中的“结构化正文提取层”。 + +### 2.1 服务负责的事情 + +- 接收 URL 或标准化 `item` +- 获取正文或补全正文 +- 抽取文章标题 +- 抽取正文纯文本 +- 返回结构化文章对象 +- 返回质量标记和结构化错误 + +### 2.2 服务不负责的事情 + +- 管理 RSS 订阅源 +- 生成摘要 +- 主题分类 +- 价值判断 +- 规则过滤 +- 写入知识库 +- 发送通知 +- 执行全流程编排 + +结论: + +当前 MCP 服务边界应保持干净,只负责把网页或 item 转成结构化文章数据。 + +--- + +## 3. 总体架构 + +建议并且当前实现采用的是: + +```text +MCP Server Layer + -> Tool Handlers + -> extraction core + -> normalizer + -> content_loader + -> extractor + -> quality_checker + -> mapper +``` + +### 3.1 外层 MCP 层职责 + +- 注册 tools +- 接收和校验 tool 输入 +- 调用内部 extraction core +- 将结果包装成 MCP tool 输出 + +### 3.2 内层 extraction core 职责 + +- 处理正文获取、标题提取、正文抽取、质量检查、结果映射 +- 与具体 MCP SDK 解耦 + +### 3.3 为什么必须分层 + +如果把内容提取逻辑直接写死在 MCP handler 里,后续会出现这些问题: + +- 业务逻辑难测试 +- 协议层与抽取逻辑耦合 +- 以后想补 CLI、批处理或其他入口时需要重复实现 + +因此当前原则是: + +- MCP 是外壳 +- extraction core 是内核 + +--- + +## 4. 当前暴露的 Tools + +当前实现只暴露两个 tool: + +- `extract_url_content` +- `extract_item_content` + +### 4.1 `extract_url_content` + +角色: + +- 直接输入 URL +- 适合单条文章测试 +- 适合手动调试或上层 Agent 直接调用 + +建议输入: + +```json +{ + "url": "https://example.com/post/1", + "language_hint": "zh" +} +``` + +建议输出: + +```json +{ + "success": true, + "article": { + "extract_id": "sha256:yyy", + "item_id": null, + "source_id": null, + "url": "https://example.com/post/1", + "title": "文章标题", + "language": "zh", + "content_kind": "article", + "plain_text": "抽取后的正文", + "quality_flags": { + "is_paywalled": false, + "is_truncated": false, + "is_low_content": false + }, + "metadata": { + "content_source": "fetched_html", + "extractor": "trafilatura", + "char_count": 1234 + }, + "pipeline_state": "extracted" + }, + "warnings": [], + "debug": { + "content_source": "fetched_html", + "extractor": "trafilatura" + } +} +``` + +### 4.2 `extract_item_content` + +角色: + +- 输入标准化 `item` +- 与上游 RSS 聚合层对接 +- 保留 `item_id`、`source_id` 等追踪信息 + +建议输入: + +```json +{ + "item": { + "item_id": "sha256:xxx", + "source_id": "my-blog", + "title": "文章标题", + "url": "https://example.com/post/1", + "published_at": "2026-03-23T08:00:00Z", + "raw_summary": "RSS 摘要", + "raw_content": "RSS 正文片段", + "content_kind": "article", + "language": "zh" + } +} +``` + +建议输出: + +```json +{ + "success": true, + "article": { + "extract_id": "sha256:yyy", + "item_id": "sha256:xxx", + "source_id": "my-blog", + "url": "https://example.com/post/1", + "title": "文章标题", + "published_at": "2026-03-23T08:00:00Z", + "language": "zh", + "content_kind": "article", + "plain_text": "抽取后的正文", + "quality_flags": { + "is_paywalled": false, + "is_truncated": false, + "is_low_content": false + }, + "metadata": { + "content_source": "item.raw_content", + "extractor": "inline", + "char_count": 1234 + }, + "pipeline_state": "extracted" + }, + "warnings": [], + "debug": { + "content_source": "item.raw_content", + "extractor": "inline" + } +} +``` + +--- + +## 5. 统一输出规范 + +当前两个 tool 都复用同一个输出结构,即 `ExtractionOutput`。 + +成功时: + +- `success = true` +- 返回 `article` +- 可附带 `warnings` +- 可附带 `debug` + +失败时: + +- `success = false` +- 返回结构化 `error` +- 不返回不完整的 `article` + +这样做的好处是: + +- 上层流程只需要消费一个稳定 schema +- 后续 LLM 摘要和 validator 可以独立接上 +- OpenClaw 或其他 Agent 更容易编排 + +--- + +## 6. 错误返回规范 + +建议错误结构如下: + +```json +{ + "success": false, + "error": { + "code": "CONTENT_FETCH_FAILED", + "message": "Failed to fetch article content", + "retryable": true, + "stage": "fetch", + "details": { + "url": "https://example.com/post/1" + } + }, + "warnings": [] +} +``` + +当前错误码包括: + +- `INVALID_INPUT` +- `CONTENT_FETCH_FAILED` +- `CONTENT_EXTRACTION_FAILED` +- `CONTENT_TOO_SHORT` +- `UNKNOWN_ERROR` + +使用原则: + +- 抓取失败不等于程序崩溃 +- 可以重试的错误应标记 `retryable = true` +- 错误应指出具体阶段 + +--- + +## 7. 当前技术选型 + +当前 MVP 使用: + +- Python +- MCP Python SDK (`FastMCP`) +- `httpx` 进行网页抓取 +- `trafilatura` 进行正文抽取 +- `beautifulsoup4` 作为 HTML 兜底解析 +- `pydantic` 进行输入输出约束 + +设计原则是: + +- 外层 MCP +- 内层 extraction core +- LLM 摘要和校验作为 MCP 之后的独立环节 + +--- + +## 8. 当前目录结构建议 + +当前实现大致如下: + +```text +summary_mcp/ + server.py + core/ + normalizer.py + content_loader.py + extractor.py + quality_checker.py + mapper.py + pipeline.py + models/ + item.py + document.py + summary_io.py + llm_result.py + validators/ + llm_result.py +``` + +说明: + +- `server.py`:MCP 服务入口 +- `core/`:提取内核 +- `models/`:提取和校验相关模型 +- `validators/`:LLM 输出校验逻辑 + +--- + +## 9. 与 LLM 摘要层的关系 + +当前 MCP 不负责摘要,但它与摘要层的接口已经明确: + +1. MCP 产出结构化 `article` +2. 上层 LLM 根据 `article.title`、`article.url`、`article.plain_text` 等生成摘要 JSON +3. validator 对摘要 JSON 做 schema 校验与业务校验 +4. 若不合格,进入修复重试 + +这条闭环当前已经通过本地脚本验证: + +- `scripts/run_summary_loop.py` + +因此,当前正确的职责拆分是: + +- MCP 负责提取 +- LLM 负责摘要 +- validator 负责验收 + +--- + +## 10. 当前阶段验收结论 + +当前 MVP 已完成以下验证: + +- 真实 URL 可以提取为结构化文章 JSON +- 提取结果可以保存为文件 +- 提取结果可以喂给 LLM 生成摘要 JSON +- 摘要 JSON 可以通过 validator 校验 +- 整条“提取 -> 摘要 -> 校验”最小闭环已经跑通 + +--- + +## 11. 后续扩展方向 + +后续建议按下面顺序推进: + +- 接入真实 RSS 聚合结果并映射为 `item` +- 定义规则过滤层 schema +- 设计知识库入库格式 +- 增加批量处理能力 +- 引入 Playwright 作为动态页面兜底方案 +- 重命名包和项目名,使其与当前职责一致 + +--- + +## 12. 当前阶段一句话结论 + +当前仓库中的 MCP 已经从早期“摘要 MCP”演进为“内容提取 MCP”,它的职责是稳定产出结构化文章 JSON,并把摘要与验收环节留给后续的 LLM 和 validator 流程。 diff --git a/infra/freshrss/README.md b/infra/freshrss/README.md new file mode 100644 index 0000000..f5ee26c --- /dev/null +++ b/infra/freshrss/README.md @@ -0,0 +1,37 @@ +# FreshRSS Local Setup + +This is the minimal local Docker setup for FreshRSS. + +## Start + +```bash +docker compose up -d +``` + +Then open: + +```text +http://localhost:8080 +``` + +## Notes + +- This setup is intended for local MVP use. +- FreshRSS can use SQLite during initial setup, so no separate database container is required. +- Persistent data is stored in: + - `./data` + - `./extensions` + +## First-Time Setup + +On the web installer: + +1. Choose `SQLite` +2. Create the admin account +3. Finish installation + +## Stop + +```bash +docker compose down +``` diff --git a/infra/freshrss/docker-compose.yml b/infra/freshrss/docker-compose.yml new file mode 100644 index 0000000..e3841e5 --- /dev/null +++ b/infra/freshrss/docker-compose.yml @@ -0,0 +1,13 @@ +services: + freshrss: + image: freshrss/freshrss:latest + container_name: freshrss + restart: unless-stopped + ports: + - "8080:80" + environment: + TZ: Asia/Shanghai + CRON_MIN: "13,43" + volumes: + - ./data:/var/www/FreshRSS/data + - ./extensions:/var/www/FreshRSS/extensions diff --git a/outputs/llm-summary-prompt.txt b/outputs/llm-summary-prompt.txt new file mode 100644 index 0000000..cafa1cb --- /dev/null +++ b/outputs/llm-summary-prompt.txt @@ -0,0 +1,63 @@ +你是一个高质量中文内容摘要助手。你的任务是根据用户提供的结构化文章提取结果,生成稳定、可复用的摘要结果。 + +输入说明: +- 输入是一个 JSON 对象 +- `article.title` 是文章标题 +- `article.url` 是原文链接 +- `article.plain_text` 是已抽取的正文纯文本 +- `article.quality_flags` 和 `warnings` 可作为质量参考 +- 不要重复输出原始 JSON + +请完成以下任务: +1. 生成 `summary` +2. 生成 `highlights` +3. 生成 `keywords` +4. 生成 `topics` +5. 判断 `category` +6. 判断 `worth_keeping` 并给出 `reason` + +输出要求: +- 只输出合法 JSON +- 不要输出 Markdown 代码块 +- 不要补充解释性文字 +- 如果正文信息不足,要如实反映,不要编造 + +字段约束: +- `summary` + - 用 2 到 3 句话 + - 总长度不超过 140 个中文字符 + - 只保留文章最核心的目标、方法和结论 + - 不要展开细节,不要重复 `highlights` + +- `highlights` + - 输出 3 到 5 条 + - 每条一句话 + - 每条只表达一个独立要点 + +- `keywords` + - 输出 5 到 8 个 + - 必须是具体的实体、工具名、方法名、关键概念 + - 优先名词短语 + - 不要使用过于抽象的分类词 + +- `topics` + - 输出 3 到 5 个 + - 必须是比 `keywords` 更高层的主题标签 + - 用于分类归档 + - 不要与 `keywords` 重复 + +- `category` + - 只能是:`资讯`、`方法论`、`工具实践`、`观点评论` 之一 + +输出 JSON 结构如下: +{ + "title": "文章标题", + "url": "原文链接", + "summary": "2-3句摘要", + "highlights": ["要点1", "要点2", "要点3"], + "keywords": ["关键词1", "关键词2", "关键词3"], + "topics": ["主题1", "主题2", "主题3"], + "category": "资讯 | 方法论 | 工具实践 | 观点评论", + "worth_keeping": true, + "reason": "一句话理由" +} diff --git a/outputs/read-flow-2026.extracted.json b/outputs/read-flow-2026.extracted.json new file mode 100644 index 0000000..b378e16 --- /dev/null +++ b/outputs/read-flow-2026.extracted.json @@ -0,0 +1,32 @@ +{ + "success": true, + "article": { + "extract_id": "sha256:1ee177246bce46534c81718bee693fea43a4b491f545e61b3c41f22e20994057", + "item_id": null, + "source_id": null, + "url": "https://shawnxie.top/blogs/tools/read-flow-2026.html", + "title": "信息过载时代,我的漏斗式阅读工作流", + "author": null, + "published_at": null, + "language": null, + "content_kind": "article", + "plain_text": "信息过载时代,我的漏斗式阅读工作流\n这两年我越来越强烈地感觉到,信息问题早就不是获取不到,而是处理不过来。\n真正让我疲惫的,不是没东西看,而是每天都有太多东西值得看:公众号文章、技术博客、GitHub Release、AI 新闻、社区讨论、长文、短讯、碎片化观点等等,全都在争夺我的注意力。\n如果不做点什么,一个人的信息生活很容易退化成这样:\n- 收藏夹里躺满了文章,真正读完的却寥寥无几\n- 每天浏览大量内容,能留在脑子里的却屈指可数\n- 输入看似充实,输出却异常薄弱\n- 以为自己一直在吸收信息,实际上只是在不同窗口间疲于切换\n2026年,我开始认真整理自己的一套信息处理工作流。我不需要大而全的平台,也不想要那种号称能“一键替我读完互联网”的 AI 产品。\n我只需要一套围绕自己运转的信息吸收漏斗:既能尽可能广泛地捕捉信息,又能提前过滤掉噪音和重复内容,只把真正值得投入时间的内容呈现在我面前。更重要的是,这套系统还要能将我精读后的高价值内容沉淀下来,反过来优化下一轮的信息筛选。\n信息处理不追求看得更多,而是在于更稳定地吸收、判断和沉淀。\n这篇文章,我将把这套工作流思路完整写下来,包括:\n- 为什么我要这样做\n- 它是怎么搭起来的\n- 每一层分别承担什么职责\n- 未来优化方向\n如果你也在被信息过载折磨,或者你已经有不少输入工具,却还是觉得每天看了很多,脑子里没留下什么,那也许这篇文章会有点参考价值。\n为什么要做信息吸收漏斗?\n我以前也尝试过很多方法:\n- 订阅很多 RSS,然后每天刷\n- 把值得看的东西先扔进稍后阅读\n- 用收藏、标签、笔记软件把文章存起来\n- 靠搜索和回忆去找以前读过的内容\n这些工具各有用处,但通常只能解决信息处理中的某一个环节。而个人信息处理真正的难点,从来都不在于单点突破,而在于如何将这些分散的环节串联起来。信息系统如果不能同时解决这两个问题,就很容易失控:\n输入太多,内容太杂\n几乎每一个你关心的领域,都在持续产生新内容。技术突破、产品迭代、AI进展、工具更新、商业动态、研究成果、行业变化……\n而且在未经过滤之前,重要新闻、工具发布、经验复盘、广告软文、标题党、重复报道等往往都混在一起,让人难以分辨。\n如果每一条内容都需要你从零开始判断其价值,那日积月累下来的认知负担会非常高。很多时候,人真正感到疲惫的不是读一篇长文,而是反复做这些微小但无意义的判断:\n- 这篇文章值不值得点开?\n- 这条消息是不是和刚才那条重复?\n- 这个标题是不是在夸大其词?\n- 这条内容到底在说什么?\n- 这篇文章我只是需要了解一下,还是值得保存?\n缺少精读回路\n很多内容看完就过去了,既没有标记,也没有沉淀,更没有真正融入自己的知识体系。\n如果信息系统永远只知道最新内容是什么,却不知道什么内容真正对你有帮助,那最多只能算是一个信息输送管道,而不是一个能越来越懂你的个性化系统。\n所以我要的不是做一个更强的信息池,而是做一个更稳的信息漏斗。\n桶的思路是往里装更多,漏斗的思路是让信息在往下走的过程中不断收窄,最后留下真正值得进入大脑和长期记忆的部分。\n从实践效果来看,这套漏斗式信息处理工作流至少给我带来了几个显著的变化:\n- 我不再需要每天从海量标题中艰难地筛选重点内容\n- 低质量和重复的信息大部分在上游就被过滤掉了\n- 真正值得精读的内容,会在更靠后的环节中呈现在我面前\n- 我读过并认为有价值的内容,终于开始能够反过来优化后续的筛选逻辑\n不是追求更快地刷完信息,而是实现更稳定地吸收知识。\n为什么基于 OpenClaw 搭建?\n由于流程尚处探索阶段,选择先用 OpenClaw 将整条链路串联起来,让系统先运行起来,验证可行性,发现问题后再逐步优化和完善。\nOpenClaw 主要扮演编排层的角色,负责以下工作:\n- 定时触发各类任务\n- 串联不同的外部工具和系统\n- 与飞书文档进行交互\n- 与 Lumina 知识库进行联动\n- 在需要的环节引入 AI 进行结构化内容生成\n这种方式起步快、迭代快,非常适合边运行边优化,不需要一开始就投入大量精力构建完整的前后端系统。\n如果你也想搭建一套类似的信息处理流程,需要做这些准备:\n1. 可持续维护的信息源\n这是整个系统的上游基础,需要先想清楚几个问题:\n- 自己真正长期关注的主题领域是什么\n- 这些领域的主要信息来源有哪些\n- 哪些信息源值得长期订阅,哪些只是偶尔查看即可\n这一步不需要追求大而全,但要尽量保持稳定和高质量。\n如果你还没有构建自己的信息源,可以先参考我收集的 RSS 订阅源。\n2. 统一的聚合池\n我选择的聚合池是 FreshRSS,一个开源的RSS阅读器,支持多平台,支持自定义订阅,内容分类管理和API,非常适合用来进行内容存储和同步。\n它并非最终阅读工具,而是信息汇聚的中转站,让所有信息源先汇聚到同一个地方,形成可持续消费的内容候选池。\n3. 自动化编排层\n这是 OpenClaw 最能发挥价值的地方,极大地降低了自动化流程实现的难度,通过自然语言描述诉求,在对话中逐步落地想法。我主要用它来完成以下任务:\n- 定时执行 digest(自定义的预处理Skill)\n- 定时执行 daily-review(自定义的内容精选Skill)\n- 发布飞书文档\n- 调用 Lumina(个人开源的知识库项目)\n- 串联各个技能和脚本\n4. 长期知识沉淀层\n在我的工作流中,这一层使用 Lumina——个人开发的信息管理工作台。当然也可以使用别的笔记工具,如Notion、Obsidian等。\nLumina 只承接我经过筛选后确认值得长期保留的优质内容,这一步直接决定了后续反馈机制的质量。\n一个系统最终能学到什么,很大程度上取决于你为它提供了什么样的正反馈样本。\n信息处理工作流\n一句话描述:用 RSS 尽可能广泛地捕捉信息,用 FreshRSS 进行稳定聚合,用 Digest 完成预处理,用 Daily Review 实现每日精选,通过人工精读判断内容的长期价值,用 Lumina 进行知识沉淀,最后将这些沉淀的价值反向转化为轻量的个性化信号。\n这套流程的核心优势不是追求全自动,而是在于分层处理的设计理念。这意味着:\n- 并非所有信息都值得投入时间精读\n- 并非所有信息都值得长期留存\n- 并非所有信息都需要进行个性化加权\n- 并非所有信息都适合直接作为输出内容\n一旦层次划分清晰,系统就不会退化为单一维度的推荐流,而会演变成一个真正的认知加工流程。漏斗的真正价值不在于让信息越来越少,而在于实现了这几个关键目标:\n- 上游宽广:确保不会错过真正重要的信息变化\n- 中游稳定:有效过滤噪音,避免其直接干扰注意力系统\n- 下游精准:让我不必在不值得精读的内容上浪费时间\n- 回流轻柔:通过轻量反馈机制,避免系统演变成封闭的信息茧房\n个人信息系统核心能力不是如何接入更多信息源,而是要明确:哪些信息值得进入系统,哪些信息值得投入时间精读,哪些内容值得长期留存,哪些知识最终真正融入了你的思考和行动?\n当能够清晰地回答这些问题时,就不再是互联网信息的被动接收者,你将拥有一套真正属于自己的认知处理系统。接下来将按顺序详细介绍每一层的工作原理。\n信息源:以 RSS 为主,但不局限于原生 RSS\n整个系统的最上游是信息源,主要依赖 RSS。\nRSS 是最被低估的个人信息基础设施。其天然符合个人信息系统最核心的几个要求:\n- 订阅权完全掌握在自己手中\n- 信息来源清晰明确\n- 更新内容结构化\n- 不受平台推荐算法直接支配\n- 便于程序接入和自动化处理\n然而在现实中,有一个不可避免的问题:并非所有值得关注的信息源都提供 RSS 订阅。例如:部分公众号内容、某些社区的特定栏目、垂直网站的更新页面和社交平台上的账号动态。\n因此,需要在信息源进入流程之前,尽可能将其统一转换为 RSS 或类似 feed 的格式。常见的方案有:\n- RSSHub / RSS-Bridge 等转换工具:能够将大量原本不提供 RSS 订阅的内容源,转换为可以被订阅和程序自动化消费的 feed 格式;\n- GitHub feed:项目的 Releases、Commits、Discussions 等,都天然提供了结构化的 feed 接口;\n- wewe-rss:能够将公众号文章转换成RSS订阅源;\n- nitter:将 Twitter/X 动态转换为RSS源;\n- 自定义抓取脚本:对于没有现成 RSS 解决方案的页面,也可以自行编写轻量级的抓取脚本,将其转换为内部可用的 feed 格式。\n这一步的原则很简单:上游信息源可以多种多样,但在进入系统之前,格式必须尽可能统一。只有这样,下游的预处理和筛选环节才能稳定可靠地运行。\n更多信息源归一处理方案可参考之前文章碎片时间刷文章!懒人阅读方案分享。\n聚合池:用 FreshRSS 打造稳定的“中间水库”\n所有订阅源最终都会汇聚到 FreshRSS 中,它并非\"我每天真正坐下来阅读的地方\",而是流程的缓冲层,主要有以下作用:\n实现信息来源的统一管理\n无论内容来自博客、社区、GitHub 还是转换后的 feed,最终都以统一的格式呈现,成为可被消费的标准化对象。\n让下游环节无需直接对接互联网\ndigest 模块无需再逐个访问各个网站抓取今日更新内容,只需从 FreshRSS 这个统一的内容池中获取未读候选即可。大大降低了系统各环节之间的耦合度。\n确保了信息处理的时间连续性\n信息系统最忌讳的是“今天临时查看一下、明天就忘了、后天又重新开始”的碎片化处理方式。FreshRSS 提供了一个稳定的时间窗口,使后续任务能够按照固定节奏有序运行。\n我的目标不是追求无边界的信息获取,而是实现有边界的信息处理。\n预处理:Digest 将海量候选内容转化为可判断对象\n如果把 FreshRSS 比作蓄水池,那么 Digest 就是这套系统中的第一道加工厂,专注于完成内容预处理任务。\n让人每天真正感到疲惫的,往往不是精读一篇高质量文章,而是反复判断大量低质量内容是否值得投入时间。\n核心处理任务\n- URL 精确去重\n- 相似内容去重\n- 正文抓取\n- 质量检查\n- 噪音过滤\n- 摘要生成\n- 初步排序与文档输出\n在真正的“阅读”开始之前,先将互联网上天然混乱的信息整理成一批更具可读性的候选内容。\n为什么重要?\n如果没有预处理环节,后续的所有精选工作都将建立在一堆未经整理的原始标题之上。而经过 Digest 处理后,情况会改善:\n- 标题党内容大幅减少\n- 重复报道得到有效过滤\n- 无法获取正文的无效数据明显减少\n- 每条候选内容至少附带一个可供快速判断的摘要\n这将显著降低后续阅读的认知成本。Digest 的职责并非编辑终稿,而是将候选内容池整理干净、结构化,为后续的精选环节奠定基础。\nAI 精选:Daily Review 为我呈现重点关注内容\n如果说 Digest 的作用是将原始候选内容处理得更具可读性,那么 Daily Review 则是从这些经过预处理的候选中,进一步提炼出当天真正值得关注的精华内容。\n这一步让流程从预处理迈向编辑阶段。不再追求内容的广度,而是致力于打造重点更突出、结构更清晰的阅读体验,让最终产物更像一份真正意义上的日报,而非简单堆砌的摘要集合。\n目前,我将 Daily Review 设计为几个固定栏目,将不同类型的信息分配到不同的认知槽位中,包含:\n- 今日大事:聚焦具有公共重要性的事件\n- 变更与实践:关注对个人有直接操作价值的内容\n- 安全与风险:警惕潜在的风险因素\n- 开源与工具:追踪工具生态的发展变化\n- 洞察与数据点:把握行业趋势和关键数据\n- 主题深挖:将单一新闻事件提升到趋势层面进行分析\nLLM 赋能\n从 Digest 输出的候选内容,到 Daily Review 最终的成稿层,中间有很多工作适合 AI 来完成:\n- 合并同一事件的多个信息来源\n- 提炼核心主题\n- 为内容分类并分配到对应栏目\n- 识别值得深入探讨的话题\n- 将零散的候选内容重组为人类可以快速阅读的日报结构\n但我对 AI 在这一环节的应用始终保持克制。不是让 AI 全自动生成日报,而是扮演结构化整理者的角色,从候选内容中筛选出当天和我相关的精华部分。\n精读留存:将 Human in the loop 置于系统核心\n前面所有流程,都是为将信息筛选到值得精读的阶段。真正让系统不至于退化为另一种自动化信息流的,正是这一关键环节:Human in the loop(人在回路中)。\n我坚信,在个人知识系统中,最不能完全外包的是长期价值判断能力。\n系统可以协助我完成许多任务,如信息收集、去重、摘要、聚类、排序和精选等。但它无法完全替我做出关键决策:\n- 哪些内容真正值得纳入长期知识库\n- 哪些内容在未来会持续发挥价值\n- 哪些内容会对我的写作和判断产生深远影响\n因此,在我的工作流中,Lumina 前面始终设有一道人工筛选门槛。我会从之前各个环节产生的内容中,挑选出真正值得精读和长期留存的内容。\n这是整套系统中最关键的价值确认环节。能够进入 Lumina 的内容,不仅代表我看过,更意味着我认为这篇内容值得在未来持续为我所用。\n个人画像:让系统学会识别对我真正有价值的内容\n如果流程到 Lumina 就戛然而止,那么它仍然只是一个单纯的过滤和沉淀系统。只有当沉淀的内容开始反过来影响上游的信息选择时,整个系统才真正形成了闭环。\n为此我加入了一层设计较为克制的兴趣画像逻辑。\n刻意控制了影响力,不希望整个系统演变成另一个猜你喜欢的推荐引擎。我只需要它能稍微更懂我,但又不过度迎合我的偏好,保留对公共重要性内容的关注和探索未知领域的空间。\n这层画像目前主要承担轻量 rerank 的功能,作为辅助信号,轻微影响 Digest 和 Daily Review 环节中候选内容的排序。主要从以下几个维度逐步学习我的偏好:\n- 长期精读的主题领域\n- 能稳定提供价值的信息源\n- 偏好的内容格式\n- 最终存入 Lumina 知识库的内容类型\n轻量引导的设计,旨在减少无效信息对注意力的浪费,同时避免构建封闭的信息茧房。个性化推荐应帮助我们减少无意义的判断,而非让我们躲进舒适区。\n沉淀:将信息流转化为长期内容资产\n信息漏斗的终点不是读完,而是沉淀。目前从两个方向拓展沉淀的价值:\n1. 周刊生成\n当系统积累了一周的高质量内容后,就不应再局限于每天生成一份日报。一周的时间跨度非常适合进行复盘总结。此时,系统已经拥有了丰富的素材:\n- 一周的 Digest 预处理结果\n- 一周的 Daily Review 精选内容\n- 若干经过价值确认的 Lumina 知识库内容\n- 一些开始反复出现的热门主题\n此时生成周刊,比单纯从网页上抓取热点更有价值。因为这些内容已经经过了个人筛选和沉淀,带有明显的个性化痕迹。周刊不应仅仅是本周发生了什么的简单罗列,更应该具备深度和价值:\n- 本周有哪些主题值得重点关注和记忆\n- 哪些变化只是短期噪音,哪些是值得重视的长期信号\n- 哪些内容具有长期参考价值,值得反复回看\n周刊合集👉🏻:肖恩技术周刊\n2. 主题文章生成\n另一个方向是让系统能够逐渐识别哪些主题已经积累了足够的素材,值得写成长篇文章。\n虽然信息流中的内容看起来是离散的,但如果拉长时间维度观察,就会发现很多内容其实都在指向同一个核心主题,例如:\n- AI Agent 工程的发展趋势\n- 开源工具链的演变\n- 内容平台分发机制的变革\n- 隐私保护、合规要求与数据治理\n一旦某个主题在一段时间内反复出现,并且我多次对相关内容进行精读、收藏和沉淀,那么它就不应再仅仅是多条零散的新闻,而应该逐渐发展成为一个可以深入挖掘和输出的长期主题。\n内容合集👉🏻:今日观察\n未来迭代方向\n尽管工作流目前已经能够稳定运行,但它远未达到最终完成的状态。可以继续完善的方面有:\n细化反馈机制\n目前系统中最强的反馈信号是这篇内容是否被存入 Lumina。但在理想状态下,我希望系统能够逐渐识别更多层次的用户行为:\n- 点开内容但未读完\n- 读完内容但未收藏\n- 内容值得精读\n- 内容值得长期沉淀\n- 内容最终影响了写作、决策或实际实现\n- 不喜欢的内容(负反馈)\n一旦这些层次的反馈机制得以完善,兴趣画像将变得更加精准,不再只是一个粗粒度的偏好集合。\n进一步抽象流程\n目前,我更倾向于继续在现有的 OpenClaw 体系内迭代优化,这是探索阶段最适合的方式。\n但如果这套流程能够变得更加稳定,职责边界也更加清晰,那么抽象出其中的信息处理内核,会更有利于后续工程化迭代。\n不过比起将其产品化,还是先让这套漏斗系统持续稳定地运转,越来越懂我。\n结语\n信息处理的关键,从来不是看到更多,而是让真正重要的信息被自己接住。\n当信息经过筛选、精读、沉淀和反馈,真正融入知识结构时,我们不再是信息的被动消费者,将真正成为自己信息环境的主人。\n感谢阅读\n微信公众号「肖恩聊技术」\n如果这篇文章对你有帮助,欢迎扫码关注,获取原创文章推送。", + "quality_flags": { + "is_paywalled": false, + "is_truncated": false, + "is_low_content": false + }, + "metadata": { + "content_source": "fetched_html", + "extractor": "trafilatura", + "char_count": 7019 + }, + "pipeline_state": "extracted" + }, + "error": null, + "debug": { + "content_source": "fetched_html", + "extractor": "trafilatura" + }, + "warnings": [] +} \ No newline at end of file diff --git a/outputs/result.json b/outputs/result.json new file mode 100644 index 0000000..65875e6 --- /dev/null +++ b/outputs/result.json @@ -0,0 +1,30 @@ +{ +"title": "信息过载时代,我的漏斗式阅读工作流", +"url": "https://shawnxie.top/blogs/tools/read-flow-2026.html", +"summary": "作者为解决信息过载问题,构建了一套以RSS为上游、FreshRSS为聚合池、OpenClaw为编排层的漏斗式工作流,通过分层筛选、AI精选和人工精读,将信息逐步沉淀至Lumina知识库,形成稳定可控的个人信息处理闭环。", +"highlights": [ +"以RSS为主统一信息源,通过转换工具将公众号、社交动态等非RSS内容标准化接入。", +"Digest预处理负责去重、抓取正文和生成摘要,Daily Review结合AI将内容分类为日报栏目。", +"人工精读是长期价值判断的核心环节,只有经过筛选的内容才进入Lumina知识库沉淀。", +"通过轻量兴趣画像和反馈机制,系统能逐步优化筛选排序,同时避免形成信息茧房。", +"沉淀后的内容可进一步生成周刊或主题文章,将信息流转化为长期资产。" +], +"keywords": [ +"RSS", +"FreshRSS", +"OpenClaw", +"Digest", +"Daily Review", +"Lumina", +"信息漏斗", +"人在回路" +], +"topics": [ +"个人信息管理", +"阅读工作流", +"知识沉淀" +], +"category": "方法论", +"worth_keeping": true, +"reason": "系统性地阐述了个人信息处理的分层架构与工程实践,兼具理念清晰度和可操作性,对知识工作者有较高参考价值。" +} \ No newline at end of file diff --git a/outputs/result.loop.attempt-1.json b/outputs/result.loop.attempt-1.json new file mode 100644 index 0000000..53ceba6 --- /dev/null +++ b/outputs/result.loop.attempt-1.json @@ -0,0 +1,32 @@ +{ + "title": "信息过载时代,我的漏斗式阅读工作流", + "url": "https://shawnxie.top/blogs/tools/read-flow-2026.html", + "summary": "文章提出一套应对信息过载的“漏斗式阅读工作流”,以RSS采集、FreshRSS聚合、Digest预处理、Daily Review精选、人工精读和Lumina沉淀构成闭环。核心不是全自动读完互联网,而是分层过滤噪音、保留人工价值判断,并用轻量反馈持续优化筛选。", + "highlights": [ + "作者认为信息焦虑的核心已从“获取不到”转向“处理不过来”,疲惫主要来自大量低效判断与缺少沉淀回路。", + "整套流程以OpenClaw为编排层,串联RSS信息源、FreshRSS、飞书文档、Lumina及AI处理环节,便于快速迭代。", + "Digest负责去重、抓取正文、质量检查、噪音过滤、生成摘要和初步排序,把原始信息整理成可判断候选集。", + "Daily Review借助LLM做栏目化精选与结构化整理,但作者强调AI只做辅助,长期价值判断必须保留人为参与。", + "系统最终通过Lumina沉淀高价值内容,并用轻量兴趣画像反向影响排序,形成不过度迎合的个性化闭环。" + ], + "keywords": [ + "OpenClaw", + "FreshRSS", + "Lumina", + "Digest", + "Daily Review", + "RSSHub", + "RSS-Bridge", + "Human in the loop" + ], + "topics": [ + "个人知识管理", + "信息过载治理", + "自动化工作流", + "内容筛选与沉淀", + "个性化信息系统" + ], + "category": "工具实践", + "worth_keeping": true, + "reason": "文章给出了从信息采集、预处理到沉淀反馈的完整可执行链路,对搭建个人信息处理系统有较强参考价值。" +} \ No newline at end of file diff --git a/outputs/result.loop.attempt-1.raw.txt b/outputs/result.loop.attempt-1.raw.txt new file mode 100644 index 0000000..53ceba6 --- /dev/null +++ b/outputs/result.loop.attempt-1.raw.txt @@ -0,0 +1,32 @@ +{ + "title": "信息过载时代,我的漏斗式阅读工作流", + "url": "https://shawnxie.top/blogs/tools/read-flow-2026.html", + "summary": "文章提出一套应对信息过载的“漏斗式阅读工作流”,以RSS采集、FreshRSS聚合、Digest预处理、Daily Review精选、人工精读和Lumina沉淀构成闭环。核心不是全自动读完互联网,而是分层过滤噪音、保留人工价值判断,并用轻量反馈持续优化筛选。", + "highlights": [ + "作者认为信息焦虑的核心已从“获取不到”转向“处理不过来”,疲惫主要来自大量低效判断与缺少沉淀回路。", + "整套流程以OpenClaw为编排层,串联RSS信息源、FreshRSS、飞书文档、Lumina及AI处理环节,便于快速迭代。", + "Digest负责去重、抓取正文、质量检查、噪音过滤、生成摘要和初步排序,把原始信息整理成可判断候选集。", + "Daily Review借助LLM做栏目化精选与结构化整理,但作者强调AI只做辅助,长期价值判断必须保留人为参与。", + "系统最终通过Lumina沉淀高价值内容,并用轻量兴趣画像反向影响排序,形成不过度迎合的个性化闭环。" + ], + "keywords": [ + "OpenClaw", + "FreshRSS", + "Lumina", + "Digest", + "Daily Review", + "RSSHub", + "RSS-Bridge", + "Human in the loop" + ], + "topics": [ + "个人知识管理", + "信息过载治理", + "自动化工作流", + "内容筛选与沉淀", + "个性化信息系统" + ], + "category": "工具实践", + "worth_keeping": true, + "reason": "文章给出了从信息采集、预处理到沉淀反馈的完整可执行链路,对搭建个人信息处理系统有较强参考价值。" +} \ No newline at end of file diff --git a/outputs/result.loop.attempt-1.validation.json b/outputs/result.loop.attempt-1.validation.json new file mode 100644 index 0000000..90f8372 --- /dev/null +++ b/outputs/result.loop.attempt-1.validation.json @@ -0,0 +1,37 @@ +{ + "valid": true, + "errors": [], + "warnings": [], + "normalized_result": { + "title": "信息过载时代,我的漏斗式阅读工作流", + "url": "https://shawnxie.top/blogs/tools/read-flow-2026.html", + "summary": "文章提出一套应对信息过载的“漏斗式阅读工作流”,以RSS采集、FreshRSS聚合、Digest预处理、Daily Review精选、人工精读和Lumina沉淀构成闭环。核心不是全自动读完互联网,而是分层过滤噪音、保留人工价值判断,并用轻量反馈持续优化筛选。", + "highlights": [ + "作者认为信息焦虑的核心已从“获取不到”转向“处理不过来”,疲惫主要来自大量低效判断与缺少沉淀回路。", + "整套流程以OpenClaw为编排层,串联RSS信息源、FreshRSS、飞书文档、Lumina及AI处理环节,便于快速迭代。", + "Digest负责去重、抓取正文、质量检查、噪音过滤、生成摘要和初步排序,把原始信息整理成可判断候选集。", + "Daily Review借助LLM做栏目化精选与结构化整理,但作者强调AI只做辅助,长期价值判断必须保留人为参与。", + "系统最终通过Lumina沉淀高价值内容,并用轻量兴趣画像反向影响排序,形成不过度迎合的个性化闭环。" + ], + "keywords": [ + "OpenClaw", + "FreshRSS", + "Lumina", + "Digest", + "Daily Review", + "RSSHub", + "RSS-Bridge", + "Human in the loop" + ], + "topics": [ + "个人知识管理", + "信息过载治理", + "自动化工作流", + "内容筛选与沉淀", + "个性化信息系统" + ], + "category": "工具实践", + "worth_keeping": true, + "reason": "文章给出了从信息采集、预处理到沉淀反馈的完整可执行链路,对搭建个人信息处理系统有较强参考价值。" + } +} \ No newline at end of file diff --git a/outputs/result.loop.json b/outputs/result.loop.json new file mode 100644 index 0000000..53ceba6 --- /dev/null +++ b/outputs/result.loop.json @@ -0,0 +1,32 @@ +{ + "title": "信息过载时代,我的漏斗式阅读工作流", + "url": "https://shawnxie.top/blogs/tools/read-flow-2026.html", + "summary": "文章提出一套应对信息过载的“漏斗式阅读工作流”,以RSS采集、FreshRSS聚合、Digest预处理、Daily Review精选、人工精读和Lumina沉淀构成闭环。核心不是全自动读完互联网,而是分层过滤噪音、保留人工价值判断,并用轻量反馈持续优化筛选。", + "highlights": [ + "作者认为信息焦虑的核心已从“获取不到”转向“处理不过来”,疲惫主要来自大量低效判断与缺少沉淀回路。", + "整套流程以OpenClaw为编排层,串联RSS信息源、FreshRSS、飞书文档、Lumina及AI处理环节,便于快速迭代。", + "Digest负责去重、抓取正文、质量检查、噪音过滤、生成摘要和初步排序,把原始信息整理成可判断候选集。", + "Daily Review借助LLM做栏目化精选与结构化整理,但作者强调AI只做辅助,长期价值判断必须保留人为参与。", + "系统最终通过Lumina沉淀高价值内容,并用轻量兴趣画像反向影响排序,形成不过度迎合的个性化闭环。" + ], + "keywords": [ + "OpenClaw", + "FreshRSS", + "Lumina", + "Digest", + "Daily Review", + "RSSHub", + "RSS-Bridge", + "Human in the loop" + ], + "topics": [ + "个人知识管理", + "信息过载治理", + "自动化工作流", + "内容筛选与沉淀", + "个性化信息系统" + ], + "category": "工具实践", + "worth_keeping": true, + "reason": "文章给出了从信息采集、预处理到沉淀反馈的完整可执行链路,对搭建个人信息处理系统有较强参考价值。" +} \ No newline at end of file diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..d6f6505 --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,27 @@ +[project] +name = "summary-mcp" +version = "0.1.0" +description = "MCP service scaffold for article content extraction" +readme = "README.md" +requires-python = ">=3.11" +dependencies = [ + "beautifulsoup4>=4.12.0", + "httpx>=0.27.0", + "mcp>=1.17.0", + "pydantic>=2.9.0", + "trafilatura>=1.12.0", +] + +[project.scripts] +summary-mcp = "summary_mcp.server:main" +validate-llm-result = "summary_mcp.validate_llm_result:main" + +[build-system] +requires = ["setuptools>=68.0"] +build-backend = "setuptools.build_meta" + +[tool.setuptools] +package-dir = {"" = "src"} + +[tool.setuptools.packages.find] +where = ["src"] diff --git a/scripts/__pycache__/run_summary_loop.cpython-312.pyc b/scripts/__pycache__/run_summary_loop.cpython-312.pyc new file mode 100644 index 0000000..2b279ba Binary files /dev/null and b/scripts/__pycache__/run_summary_loop.cpython-312.pyc differ diff --git a/scripts/run_summary_loop.py b/scripts/run_summary_loop.py new file mode 100644 index 0000000..ecc2622 --- /dev/null +++ b/scripts/run_summary_loop.py @@ -0,0 +1,235 @@ +from __future__ import annotations + +import argparse +import json +import os +import re +import sys +from pathlib import Path +from typing import Any + +import httpx + + +REPO_ROOT = Path(__file__).resolve().parents[1] +SRC_ROOT = REPO_ROOT / "src" + +if str(SRC_ROOT) not in sys.path: + sys.path.insert(0, str(SRC_ROOT)) + +from summary_mcp.validators.llm_result import validate_llm_result + + +JSON_BLOCK_RE = re.compile(r"```(?:json)?\s*(\{.*\})\s*```", re.DOTALL) + + +def load_text(path: Path) -> str: + return path.read_text(encoding="utf-8") + + +def load_json(path: Path) -> dict[str, Any]: + return json.loads(load_text(path)) + + +def save_json(path: Path, payload: dict[str, Any]) -> None: + path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8") + + +def build_summary_input(extracted: dict[str, Any]) -> dict[str, Any]: + article = extracted.get('article') or {} + return { + 'article': { + 'title': article.get('title'), + 'url': article.get('url'), + 'plain_text': article.get('plain_text'), + 'quality_flags': article.get('quality_flags'), + }, + 'warnings': extracted.get('warnings', []), + } + + +def build_initial_prompt(prompt_template: str, extracted: dict[str, Any]) -> str: + summary_input = build_summary_input(extracted) + return ( + f"{prompt_template}\n\n" + "Below is the structured extracted article input. Generate the final summary JSON from it.\n\n" + f"{json.dumps(summary_input, ensure_ascii=False, indent=2)}" + ) + + +def build_repair_prompt( + errors: list[str], + extracted: dict[str, Any], + result_json: dict[str, Any], +) -> str: + summary_input = build_summary_input(extracted) + return ( + "Please repair the following invalid summary JSON.\n\n" + "Requirements:\n" + "- Output valid JSON only\n" + "- Keep fields that are already correct\n" + "- Fix only the validator-reported errors\n" + "- Do not add explanations\n\n" + f"validator errors:\n{json.dumps(errors, ensure_ascii=False, indent=2)}\n\n" + f"Extracted article input:\n{json.dumps(summary_input, ensure_ascii=False, indent=2)}\n\n" + f"Current summary JSON:\n{json.dumps(result_json, ensure_ascii=False, indent=2)}\n" + ) + + +def extract_json_text(raw_text: str) -> str: + fenced = JSON_BLOCK_RE.search(raw_text) + if fenced: + return fenced.group(1) + + stripped = raw_text.strip() + start = stripped.find("{") + end = stripped.rfind("}") + if start == -1 or end == -1 or end <= start: + raise ValueError("Model output does not contain a JSON object.") + return stripped[start : end + 1] + + +def call_llm( + prompt: str, + timeout_seconds: float, + api_key: str | None, + model: str | None, + api_url: str | None, +) -> str: + api_key = api_key or os.environ.get("LLM_API_KEY") or os.environ.get("OPENAI_API_KEY") + model = model or os.environ.get("LLM_MODEL") or os.environ.get("OPENAI_MODEL") + api_url = api_url or os.environ.get("LLM_API_URL", "https://api.openai.com/v1/chat/completions") + + if not api_key: + raise RuntimeError("Missing LLM_API_KEY or OPENAI_API_KEY, or pass --api-key.") + if not model: + raise RuntimeError("Missing LLM_MODEL or OPENAI_MODEL, or pass --model.") + + headers = { + "Authorization": f"Bearer {api_key}", + "Content-Type": "application/json", + } + payload = { + "model": model, + "messages": [ + { + "role": "system", + "content": "You are a precise JSON generator. Always output a single valid JSON object.", + }, + {"role": "user", "content": prompt}, + ], + "temperature": 0.2, + } + + with httpx.Client(timeout=timeout_seconds) as client: + response = client.post(api_url, headers=headers, json=payload) + response.raise_for_status() + data = response.json() + + try: + return data["choices"][0]["message"]["content"] + except (KeyError, IndexError, TypeError) as exc: + raise RuntimeError(f"Unexpected LLM response shape: {json.dumps(data, ensure_ascii=False)[:1000]}") from exc + + +def run_loop( + extracted_path: Path, + prompt_path: Path, + output_path: Path, + max_retries: int, + timeout_seconds: float, + api_key: str | None, + model: str | None, + api_url: str | None, +) -> int: + extracted = load_json(extracted_path) + prompt_template = load_text(prompt_path) + output_path.parent.mkdir(parents=True, exist_ok=True) + + last_errors: list[str] = [] + last_result: dict[str, Any] | None = None + + for attempt in range(1, max_retries + 2): + if attempt == 1: + prompt = build_initial_prompt(prompt_template, extracted) + else: + assert last_result is not None + prompt = build_repair_prompt(last_errors, extracted, last_result) + + raw_output = call_llm(prompt, timeout_seconds, api_key, model, api_url) + raw_path = output_path.with_name(f"{output_path.stem}.attempt-{attempt}.raw.txt") + raw_path.write_text(raw_output, encoding="utf-8") + + try: + result_payload = json.loads(extract_json_text(raw_output)) + except (json.JSONDecodeError, ValueError) as exc: + last_errors = [f"Model output is not valid JSON: {exc}"] + last_result = {"raw_output": raw_output} + validation_path = output_path.with_name(f"{output_path.stem}.attempt-{attempt}.validation.json") + save_json( + validation_path, + { + "valid": False, + "errors": last_errors, + "warnings": [], + "normalized_result": None, + }, + ) + if attempt > max_retries: + output_path.write_text(raw_output, encoding="utf-8") + return 1 + continue + + attempt_path = output_path.with_name(f"{output_path.stem}.attempt-{attempt}.json") + save_json(attempt_path, result_payload) + save_json(output_path, result_payload) + + report = validate_llm_result(output_path, extracted_path) + validation_path = output_path.with_name(f"{output_path.stem}.attempt-{attempt}.validation.json") + save_json( + validation_path, + { + "valid": report.valid, + "errors": report.errors, + "warnings": report.warnings, + "normalized_result": report.normalized_result, + }, + ) + + if report.valid: + return 0 + + last_errors = report.errors + last_result = result_payload + + return 1 + + +def main() -> None: + parser = argparse.ArgumentParser(description="Run the minimal extraction -> LLM summary -> validation loop.") + parser.add_argument("--extracted", type=Path, required=True, help="Extracted article JSON file") + parser.add_argument("--prompt", type=Path, required=True, help="LLM prompt template file") + parser.add_argument("--output", type=Path, required=True, help="Target path for the final summary JSON") + parser.add_argument("--max-retries", type=int, default=2, help="Number of repair retries after the initial attempt") + parser.add_argument("--timeout", type=float, default=60.0, help="LLM request timeout in seconds") + parser.add_argument("--api-key", type=str, default=None, help="LLM API key") + parser.add_argument("--model", type=str, default=None, help="LLM model name") + parser.add_argument("--api-url", type=str, default=None, help="LLM chat completions API URL") + args = parser.parse_args() + + raise SystemExit( + run_loop( + extracted_path=args.extracted, + prompt_path=args.prompt, + output_path=args.output, + max_retries=args.max_retries, + timeout_seconds=args.timeout, + api_key=args.api_key, + model=args.model, + api_url=args.api_url, + ) + ) + + +if __name__ == "__main__": + main() diff --git a/skills/llm-summary-review/SKILL.md b/skills/llm-summary-review/SKILL.md new file mode 100644 index 0000000..13b93c1 --- /dev/null +++ b/skills/llm-summary-review/SKILL.md @@ -0,0 +1,101 @@ +--- +name: llm-summary-review +description: Validate and refine LLM-generated summary JSON for extracted articles in this repository. Use when the user wants to review, validate, repair, or iterate on `outputs/result.json` or similar summary outputs produced from extracted article JSON. +--- + +# LLM Summary Review + +Use this skill when working on the repository's summary loop after article extraction is done. + +## What This Skill Does + +- Verifies that an LLM summary result matches the expected JSON contract +- Reuses the repository validator instead of re-checking fields manually +- Repairs invalid outputs by telling the LLM exactly what to fix +- Keeps the workflow aligned with the extraction JSON produced by this project + +## Inputs + +Typical files: + +- Extracted article JSON: `outputs/*.extracted.json` +- LLM summary result JSON: `outputs/result.json` +- Prompt template: `outputs/llm-summary-prompt.txt` + +## Workflow + +1. Validate the current summary result with the repository validator: + +```bash +python -m summary_mcp.validate_llm_result outputs/result.json --extracted outputs/read-flow-2026.extracted.json +``` + +2. If validation passes: + - Report that the result is structurally valid + - Briefly note any warnings + - Do not rewrite the result unless the user asks + +3. If validation fails: + - Read the validator errors carefully + - Ask the LLM to regenerate or repair only the failing parts + - Re-run the validator until it passes or a retry limit is hit + +## Repair Prompt Pattern + +When asking an LLM to repair a bad result, provide: + +- The original extracted article JSON +- The current invalid summary JSON +- The validator error list +- A strict instruction to preserve valid fields and fix only the failing ones + +Use this repair template: + +```text +请修复下面这份不符合要求的摘要 JSON。 + +要求: +- 只输出合法 JSON +- 保留已经正确的字段 +- 只修复 validator 报出的错误 +- 不要补充解释 + +validator errors: +{{errors}} + +原始提取结果: +{{extracted_json}} + +当前摘要结果: +{{result_json}} +``` + +## Validation Rules + +The validator currently enforces: + +- Required fields exist +- Field types are correct +- `category` is one of: `资讯` `方法论` `工具实践` `观点评论` +- `summary` length is within bounds +- `highlights`, `keywords`, and `topics` counts are within bounds +- `keywords` and `topics` do not overlap +- `title` and `url` match the extracted article when an extracted JSON file is provided + +## Repository Implementation + +Relevant code: + +- Validator model: `src/summary_mcp/models/llm_result.py` +- Validator logic: `src/summary_mcp/validators/llm_result.py` +- CLI entry: `src/summary_mcp/validate_llm_result.py` + +Prefer using the existing validator rather than recreating checks in free-form reasoning. + +## When To Stop + +Stop when one of these is true: + +- The validator returns `valid: true` +- The user asks to inspect the remaining failures manually +- Repeated retries fail and the user should decide how to proceed diff --git a/skills/llm-summary-review/agents/openai.yaml b/skills/llm-summary-review/agents/openai.yaml new file mode 100644 index 0000000..a68dd1b --- /dev/null +++ b/skills/llm-summary-review/agents/openai.yaml @@ -0,0 +1,3 @@ +display_name: LLM Summary Review +short_description: Validate and repair summary JSON outputs for this repository. +default_prompt: Validate an LLM summary JSON against the repository rules, explain any failures, and repair the output if needed. diff --git a/skills/llm-summary-review/scripts/validate_result.py b/skills/llm-summary-review/scripts/validate_result.py new file mode 100644 index 0000000..20c604f --- /dev/null +++ b/skills/llm-summary-review/scripts/validate_result.py @@ -0,0 +1,37 @@ +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path + + +def main() -> None: + parser = argparse.ArgumentParser(description="Validate an LLM summary JSON using the repository validator.") + parser.add_argument("result", type=Path, help="Path to the summary result JSON file") + parser.add_argument("--extracted", type=Path, default=None, help="Optional extracted article JSON file") + args = parser.parse_args() + + repo_root = Path(__file__).resolve().parents[3] + sys.path.insert(0, str(repo_root / "src")) + + from summary_mcp.validators.llm_result import validate_llm_result + + report = validate_llm_result(args.result, args.extracted) + print( + json.dumps( + { + "valid": report.valid, + "errors": report.errors, + "warnings": report.warnings, + "normalized_result": report.normalized_result, + }, + ensure_ascii=False, + indent=2, + ) + ) + raise SystemExit(0 if report.valid else 1) + + +if __name__ == "__main__": + main() diff --git a/src/summary_mcp.egg-info/PKG-INFO b/src/summary_mcp.egg-info/PKG-INFO new file mode 100644 index 0000000..aba4838 --- /dev/null +++ b/src/summary_mcp.egg-info/PKG-INFO @@ -0,0 +1,27 @@ +Metadata-Version: 2.4 +Name: summary-mcp +Version: 0.1.0 +Summary: MCP service scaffold for article summarization +Requires-Python: >=3.11 +Description-Content-Type: text/markdown +Requires-Dist: beautifulsoup4>=4.12.0 +Requires-Dist: httpx>=0.27.0 +Requires-Dist: mcp>=1.17.0 +Requires-Dist: pydantic>=2.9.0 +Requires-Dist: trafilatura>=1.12.0 + +# Summary MCP + +Python MCP scaffold for article summarization. + +## Run + +```bash +pip install -e . +summary-mcp +``` + +The server exposes two tools: + +- `summarize_url` +- `summarize_item` diff --git a/src/summary_mcp.egg-info/SOURCES.txt b/src/summary_mcp.egg-info/SOURCES.txt new file mode 100644 index 0000000..c3d37a9 --- /dev/null +++ b/src/summary_mcp.egg-info/SOURCES.txt @@ -0,0 +1,23 @@ +README.md +pyproject.toml +src/summary_mcp/__init__.py +src/summary_mcp/server.py +src/summary_mcp.egg-info/PKG-INFO +src/summary_mcp.egg-info/SOURCES.txt +src/summary_mcp.egg-info/dependency_links.txt +src/summary_mcp.egg-info/entry_points.txt +src/summary_mcp.egg-info/requires.txt +src/summary_mcp.egg-info/top_level.txt +src/summary_mcp/core/__init__.py +src/summary_mcp/core/content_loader.py +src/summary_mcp/core/errors.py +src/summary_mcp/core/extractor.py +src/summary_mcp/core/mapper.py +src/summary_mcp/core/normalizer.py +src/summary_mcp/core/pipeline.py +src/summary_mcp/core/quality_checker.py +src/summary_mcp/core/summarizer.py +src/summary_mcp/models/__init__.py +src/summary_mcp/models/document.py +src/summary_mcp/models/item.py +src/summary_mcp/models/summary_io.py \ No newline at end of file diff --git a/src/summary_mcp.egg-info/dependency_links.txt b/src/summary_mcp.egg-info/dependency_links.txt new file mode 100644 index 0000000..8b13789 --- /dev/null +++ b/src/summary_mcp.egg-info/dependency_links.txt @@ -0,0 +1 @@ + diff --git a/src/summary_mcp.egg-info/entry_points.txt b/src/summary_mcp.egg-info/entry_points.txt new file mode 100644 index 0000000..3d797a0 --- /dev/null +++ b/src/summary_mcp.egg-info/entry_points.txt @@ -0,0 +1,2 @@ +[console_scripts] +summary-mcp = summary_mcp.server:main diff --git a/src/summary_mcp.egg-info/requires.txt b/src/summary_mcp.egg-info/requires.txt new file mode 100644 index 0000000..02ada98 --- /dev/null +++ b/src/summary_mcp.egg-info/requires.txt @@ -0,0 +1,5 @@ +beautifulsoup4>=4.12.0 +httpx>=0.27.0 +mcp>=1.17.0 +pydantic>=2.9.0 +trafilatura>=1.12.0 diff --git a/src/summary_mcp.egg-info/top_level.txt b/src/summary_mcp.egg-info/top_level.txt new file mode 100644 index 0000000..842d052 --- /dev/null +++ b/src/summary_mcp.egg-info/top_level.txt @@ -0,0 +1 @@ +summary_mcp diff --git a/src/summary_mcp/__init__.py b/src/summary_mcp/__init__.py new file mode 100644 index 0000000..173c38d --- /dev/null +++ b/src/summary_mcp/__init__.py @@ -0,0 +1,2 @@ +"""Summary MCP service package.""" + diff --git a/src/summary_mcp/__pycache__/__init__.cpython-312.pyc b/src/summary_mcp/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000..8904a37 Binary files /dev/null and b/src/summary_mcp/__pycache__/__init__.cpython-312.pyc differ diff --git a/src/summary_mcp/__pycache__/server.cpython-312.pyc b/src/summary_mcp/__pycache__/server.cpython-312.pyc new file mode 100644 index 0000000..732d36f Binary files /dev/null and b/src/summary_mcp/__pycache__/server.cpython-312.pyc differ diff --git a/src/summary_mcp/__pycache__/validate_llm_result.cpython-312.pyc b/src/summary_mcp/__pycache__/validate_llm_result.cpython-312.pyc new file mode 100644 index 0000000..784095c Binary files /dev/null and b/src/summary_mcp/__pycache__/validate_llm_result.cpython-312.pyc differ diff --git a/src/summary_mcp/core/__init__.py b/src/summary_mcp/core/__init__.py new file mode 100644 index 0000000..a7965bd --- /dev/null +++ b/src/summary_mcp/core/__init__.py @@ -0,0 +1 @@ +"""Core pipeline for the summary MCP service.""" diff --git a/src/summary_mcp/core/__pycache__/__init__.cpython-312.pyc b/src/summary_mcp/core/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000..33d7c1a Binary files /dev/null and b/src/summary_mcp/core/__pycache__/__init__.cpython-312.pyc differ diff --git a/src/summary_mcp/core/__pycache__/content_loader.cpython-312.pyc b/src/summary_mcp/core/__pycache__/content_loader.cpython-312.pyc new file mode 100644 index 0000000..518a0a0 Binary files /dev/null and b/src/summary_mcp/core/__pycache__/content_loader.cpython-312.pyc differ diff --git a/src/summary_mcp/core/__pycache__/errors.cpython-312.pyc b/src/summary_mcp/core/__pycache__/errors.cpython-312.pyc new file mode 100644 index 0000000..a439389 Binary files /dev/null and b/src/summary_mcp/core/__pycache__/errors.cpython-312.pyc differ diff --git a/src/summary_mcp/core/__pycache__/extractor.cpython-312.pyc b/src/summary_mcp/core/__pycache__/extractor.cpython-312.pyc new file mode 100644 index 0000000..3bd5a3f Binary files /dev/null and b/src/summary_mcp/core/__pycache__/extractor.cpython-312.pyc differ diff --git a/src/summary_mcp/core/__pycache__/mapper.cpython-312.pyc b/src/summary_mcp/core/__pycache__/mapper.cpython-312.pyc new file mode 100644 index 0000000..13de24e Binary files /dev/null and b/src/summary_mcp/core/__pycache__/mapper.cpython-312.pyc differ diff --git a/src/summary_mcp/core/__pycache__/normalizer.cpython-312.pyc b/src/summary_mcp/core/__pycache__/normalizer.cpython-312.pyc new file mode 100644 index 0000000..d176113 Binary files /dev/null and b/src/summary_mcp/core/__pycache__/normalizer.cpython-312.pyc differ diff --git a/src/summary_mcp/core/__pycache__/pipeline.cpython-312.pyc b/src/summary_mcp/core/__pycache__/pipeline.cpython-312.pyc new file mode 100644 index 0000000..d840bbd Binary files /dev/null and b/src/summary_mcp/core/__pycache__/pipeline.cpython-312.pyc differ diff --git a/src/summary_mcp/core/__pycache__/quality_checker.cpython-312.pyc b/src/summary_mcp/core/__pycache__/quality_checker.cpython-312.pyc new file mode 100644 index 0000000..9a6b0fe Binary files /dev/null and b/src/summary_mcp/core/__pycache__/quality_checker.cpython-312.pyc differ diff --git a/src/summary_mcp/core/__pycache__/summarizer.cpython-312.pyc b/src/summary_mcp/core/__pycache__/summarizer.cpython-312.pyc new file mode 100644 index 0000000..e2efa22 Binary files /dev/null and b/src/summary_mcp/core/__pycache__/summarizer.cpython-312.pyc differ diff --git a/src/summary_mcp/core/content_loader.py b/src/summary_mcp/core/content_loader.py new file mode 100644 index 0000000..5e9c83c --- /dev/null +++ b/src/summary_mcp/core/content_loader.py @@ -0,0 +1,38 @@ +from __future__ import annotations + +import httpx + +from summary_mcp.core.errors import SummaryError +from summary_mcp.models.summary_io import ExtractionInput + + +def choose_inline_content(extraction_input: ExtractionInput) -> tuple[str | None, str]: + if extraction_input.raw_html: + return extraction_input.raw_html, "raw_html" + + if extraction_input.item and extraction_input.item.raw_content and len(extraction_input.item.raw_content.strip()) >= 500: + return extraction_input.item.raw_content, "item.raw_content" + + if extraction_input.rss_content and len(extraction_input.rss_content.strip()) >= 500: + return extraction_input.rss_content, "rss_content" + + return None, "none" + + +def fetch_html(url: str) -> str: + headers = { + "User-Agent": "summary-mcp/0.1 (+https://modelcontextprotocol.io/)", + } + try: + with httpx.Client(follow_redirects=True, timeout=15.0, headers=headers) as client: + response = client.get(url) + response.raise_for_status() + return response.text + except httpx.HTTPError as exc: + raise SummaryError( + code="CONTENT_FETCH_FAILED", + message="Failed to fetch article content", + retryable=True, + stage="fetch", + details={"url": url, "reason": str(exc)}, + ) from exc diff --git a/src/summary_mcp/core/errors.py b/src/summary_mcp/core/errors.py new file mode 100644 index 0000000..2167f51 --- /dev/null +++ b/src/summary_mcp/core/errors.py @@ -0,0 +1,24 @@ +from __future__ import annotations + +from dataclasses import dataclass, field +from typing import Any + +from summary_mcp.models.summary_io import ErrorInfo + + +@dataclass +class SummaryError(Exception): + code: str + message: str + retryable: bool + stage: str + details: dict[str, Any] = field(default_factory=dict) + + def to_error_info(self) -> ErrorInfo: + return ErrorInfo( + code=self.code, + message=self.message, + retryable=self.retryable, + stage=self.stage, + details=self.details, + ) diff --git a/src/summary_mcp/core/extractor.py b/src/summary_mcp/core/extractor.py new file mode 100644 index 0000000..7f527ca --- /dev/null +++ b/src/summary_mcp/core/extractor.py @@ -0,0 +1,63 @@ +from __future__ import annotations + +from bs4 import BeautifulSoup +import trafilatura + +from summary_mcp.core.errors import SummaryError + + +def extract_title(content: str) -> str | None: + soup = BeautifulSoup(content, "html.parser") + og_title = soup.find("meta", attrs={"property": "og:title"}) + if og_title and og_title.get("content"): + return og_title["content"].strip() + + if soup.title and soup.title.string: + raw_title = soup.title.string.strip() + return raw_title.split("|", 1)[0].strip() + + heading = soup.find(["h1", "h2"]) + if heading: + heading_text = heading.get_text(" ", strip=True) + if heading_text: + return heading_text + + return None + + +def _dedupe_leading_lines(text: str) -> str: + lines = [line.strip() for line in text.splitlines() if line.strip()] + if len(lines) >= 2 and lines[0] == lines[1]: + lines.pop(0) + return "\n".join(lines).strip() + + +def extract_plain_text(content: str, content_source: str) -> tuple[str, str]: + if content_source == "raw_html" or content.lstrip().startswith("<"): + extracted = trafilatura.extract(content, include_links=False, include_formatting=False) + if extracted and len(extracted.strip()) >= 200: + return _dedupe_leading_lines(extracted), "trafilatura" + + soup = BeautifulSoup(content, "html.parser") + fallback = " ".join(soup.stripped_strings) + if len(fallback.strip()) >= 200: + return fallback.strip(), "beautifulsoup" + + raise SummaryError( + code="CONTENT_EXTRACTION_FAILED", + message="Failed to extract article body from HTML", + retryable=False, + stage="extract", + ) + + if len(content.strip()) < 200: + raise SummaryError( + code="CONTENT_TOO_SHORT", + message="Content is too short to summarize reliably", + retryable=False, + stage="extract", + details={"length": len(content.strip())}, + ) + + return content.strip(), "inline" + diff --git a/src/summary_mcp/core/mapper.py b/src/summary_mcp/core/mapper.py new file mode 100644 index 0000000..02cf2bf --- /dev/null +++ b/src/summary_mcp/core/mapper.py @@ -0,0 +1,40 @@ +from __future__ import annotations + +import hashlib + +from summary_mcp.models.document import ExtractedArticle, QualityFlags +from summary_mcp.models.item import Item + + +def _extract_id(item: Item, plain_text: str) -> str: + seed = f"{item.url}|{item.title or ''}|{len(plain_text)}" + digest = hashlib.sha256(seed.encode("utf-8")).hexdigest() + return f"sha256:{digest}" + + +def build_article( + item: Item, + plain_text: str, + quality_flags: QualityFlags, + content_source: str, + extractor_name: str, +) -> ExtractedArticle: + return ExtractedArticle( + extract_id=_extract_id(item, plain_text), + item_id=item.item_id, + source_id=item.source_id, + url=item.url, + title=item.title, + author=item.author, + published_at=item.published_at, + language=item.language, + content_kind=item.content_kind, + plain_text=plain_text, + quality_flags=quality_flags, + metadata={ + "content_source": content_source, + "extractor": extractor_name, + "char_count": len(plain_text), + }, + pipeline_state="extracted", + ) diff --git a/src/summary_mcp/core/normalizer.py b/src/summary_mcp/core/normalizer.py new file mode 100644 index 0000000..796b3da --- /dev/null +++ b/src/summary_mcp/core/normalizer.py @@ -0,0 +1,25 @@ +from __future__ import annotations + +from summary_mcp.core.errors import SummaryError +from summary_mcp.models.item import Item +from summary_mcp.models.summary_io import ExtractionInput + + +def normalize_input(extraction_input: ExtractionInput) -> ExtractionInput: + if extraction_input.item is not None: + return extraction_input + + if extraction_input.url: + return ExtractionInput( + item=Item(url=extraction_input.url, title=None), + raw_html=extraction_input.raw_html, + rss_content=extraction_input.rss_content, + language_hint=extraction_input.language_hint, + ) + + raise SummaryError( + code="INVALID_INPUT", + message="Either item or url must be provided", + retryable=False, + stage="normalize", + ) diff --git a/src/summary_mcp/core/pipeline.py b/src/summary_mcp/core/pipeline.py new file mode 100644 index 0000000..562d6b7 --- /dev/null +++ b/src/summary_mcp/core/pipeline.py @@ -0,0 +1,46 @@ +from __future__ import annotations + +from summary_mcp.core.content_loader import choose_inline_content, fetch_html +from summary_mcp.core.errors import SummaryError +from summary_mcp.core.extractor import extract_plain_text, extract_title +from summary_mcp.core.mapper import build_article +from summary_mcp.core.normalizer import normalize_input +from summary_mcp.core.quality_checker import assess_quality +from summary_mcp.models.summary_io import DebugInfo, ExtractionInput, ExtractionOutput + + +def extract_content(extraction_input: ExtractionInput) -> ExtractionOutput: + try: + normalized = normalize_input(extraction_input) + assert normalized.item is not None + + inline_content, content_source = choose_inline_content(normalized) + if inline_content is None: + inline_content = fetch_html(str(normalized.item.url)) + content_source = "fetched_html" + + if normalized.item.title is None and ( + content_source in {"raw_html", "fetched_html"} or inline_content.lstrip().startswith("<") + ): + normalized.item.title = extract_title(inline_content) + + plain_text, extractor_name = extract_plain_text(inline_content, content_source) + quality_flags, warnings = assess_quality(plain_text) + article = build_article( + normalized.item, + plain_text, + quality_flags, + content_source, + extractor_name, + ) + return ExtractionOutput( + success=True, + article=article, + debug=DebugInfo( + content_source=content_source, + extractor=extractor_name, + ), + warnings=warnings, + ) + except SummaryError as exc: + return ExtractionOutput(success=False, error=exc.to_error_info(), warnings=[]) diff --git a/src/summary_mcp/core/quality_checker.py b/src/summary_mcp/core/quality_checker.py new file mode 100644 index 0000000..795cd2e --- /dev/null +++ b/src/summary_mcp/core/quality_checker.py @@ -0,0 +1,25 @@ +from __future__ import annotations + +from summary_mcp.models.document import QualityFlags + + +PAYWALL_HINTS = ("subscribe to read", "会员", "付费", "订阅后查看", "sign in to continue") + + +def assess_quality(text: str) -> tuple[QualityFlags, list[str]]: + lowered = text.lower() + flags = QualityFlags( + is_paywalled=any(hint in lowered for hint in PAYWALL_HINTS), + is_truncated=text.endswith("...") or text.endswith("……"), + is_low_content=len(text.strip()) < 500, + ) + + warnings: list[str] = [] + if flags.is_paywalled: + warnings.append("Potential paywall detected in content.") + if flags.is_truncated: + warnings.append("Content may be truncated.") + if flags.is_low_content: + warnings.append("Content has low information density.") + + return flags, warnings diff --git a/src/summary_mcp/models/__init__.py b/src/summary_mcp/models/__init__.py new file mode 100644 index 0000000..cd13687 --- /dev/null +++ b/src/summary_mcp/models/__init__.py @@ -0,0 +1 @@ +"""Shared models for the summary MCP service.""" diff --git a/src/summary_mcp/models/__pycache__/__init__.cpython-312.pyc b/src/summary_mcp/models/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000..69e78e5 Binary files /dev/null and b/src/summary_mcp/models/__pycache__/__init__.cpython-312.pyc differ diff --git a/src/summary_mcp/models/__pycache__/document.cpython-312.pyc b/src/summary_mcp/models/__pycache__/document.cpython-312.pyc new file mode 100644 index 0000000..185fd71 Binary files /dev/null and b/src/summary_mcp/models/__pycache__/document.cpython-312.pyc differ diff --git a/src/summary_mcp/models/__pycache__/item.cpython-312.pyc b/src/summary_mcp/models/__pycache__/item.cpython-312.pyc new file mode 100644 index 0000000..57207cd Binary files /dev/null and b/src/summary_mcp/models/__pycache__/item.cpython-312.pyc differ diff --git a/src/summary_mcp/models/__pycache__/llm_result.cpython-312.pyc b/src/summary_mcp/models/__pycache__/llm_result.cpython-312.pyc new file mode 100644 index 0000000..b32d7c4 Binary files /dev/null and b/src/summary_mcp/models/__pycache__/llm_result.cpython-312.pyc differ diff --git a/src/summary_mcp/models/__pycache__/summary_io.cpython-312.pyc b/src/summary_mcp/models/__pycache__/summary_io.cpython-312.pyc new file mode 100644 index 0000000..c2b88ad Binary files /dev/null and b/src/summary_mcp/models/__pycache__/summary_io.cpython-312.pyc differ diff --git a/src/summary_mcp/models/document.py b/src/summary_mcp/models/document.py new file mode 100644 index 0000000..8c9b19c --- /dev/null +++ b/src/summary_mcp/models/document.py @@ -0,0 +1,33 @@ +from __future__ import annotations + +from datetime import datetime +from typing import Any, Literal + +from pydantic import BaseModel, Field, HttpUrl + +from .item import ContentKind + + +PipelineState = Literal["ingested", "extracted", "filtered", "stored", "pushed", "dropped"] + + +class QualityFlags(BaseModel): + is_paywalled: bool = False + is_truncated: bool = False + is_low_content: bool = False + + +class ExtractedArticle(BaseModel): + extract_id: str + item_id: str | None = None + source_id: str | None = None + url: HttpUrl + title: str | None = None + author: str | None = None + published_at: datetime | None = None + language: str | None = None + content_kind: ContentKind = "article" + plain_text: str + quality_flags: QualityFlags = Field(default_factory=QualityFlags) + metadata: dict[str, Any] = Field(default_factory=dict) + pipeline_state: PipelineState = "extracted" diff --git a/src/summary_mcp/models/item.py b/src/summary_mcp/models/item.py new file mode 100644 index 0000000..14d9d16 --- /dev/null +++ b/src/summary_mcp/models/item.py @@ -0,0 +1,27 @@ +from __future__ import annotations + +from datetime import datetime +from typing import Any, Literal + +from pydantic import BaseModel, Field, HttpUrl + + +ContentKind = Literal["article", "thread", "release", "changelog", "video", "mixed"] +FetchState = Literal["pending", "fetched", "failed", "skipped"] + + +class Item(BaseModel): + item_id: str | None = None + source_id: str | None = None + external_id: str | None = None + title: str | None = None + url: HttpUrl + author: str | None = None + published_at: datetime | None = None + discovered_at: datetime | None = None + content_kind: ContentKind = "article" + language: str | None = None + raw_summary: str | None = None + raw_content: str | None = None + metadata: dict[str, Any] = Field(default_factory=dict) + fetch_state: FetchState = "pending" diff --git a/src/summary_mcp/models/llm_result.py b/src/summary_mcp/models/llm_result.py new file mode 100644 index 0000000..e9de07c --- /dev/null +++ b/src/summary_mcp/models/llm_result.py @@ -0,0 +1,30 @@ +from __future__ import annotations + +from typing import Literal + +from pydantic import BaseModel, Field, HttpUrl, field_validator + + +Category = Literal["资讯", "方法论", "工具实践", "观点评论"] + + +class LlmSummaryResult(BaseModel): + title: str = Field(min_length=1) + url: HttpUrl + summary: str = Field(min_length=20, max_length=140) + highlights: list[str] = Field(min_length=3, max_length=5) + keywords: list[str] = Field(min_length=5, max_length=8) + topics: list[str] = Field(min_length=3, max_length=5) + category: Category + worth_keeping: bool + reason: str = Field(min_length=1) + + @field_validator("title", "summary", "reason") + @classmethod + def normalize_text_fields(cls, value: str) -> str: + return value.strip() + + @field_validator("highlights", "keywords", "topics") + @classmethod + def normalize_list_fields(cls, values: list[str]) -> list[str]: + return [value.strip() for value in values if value.strip()] diff --git a/src/summary_mcp/models/summary_io.py b/src/summary_mcp/models/summary_io.py new file mode 100644 index 0000000..44e2195 --- /dev/null +++ b/src/summary_mcp/models/summary_io.py @@ -0,0 +1,37 @@ +from __future__ import annotations + +from typing import Any + +from pydantic import BaseModel, Field + +from .document import ExtractedArticle +from .item import Item + + +class ExtractionInput(BaseModel): + item: Item | None = None + raw_html: str | None = None + rss_content: str | None = None + language_hint: str | None = None + url: str | None = None + + +class ErrorInfo(BaseModel): + code: str + message: str + retryable: bool + stage: str + details: dict[str, Any] = Field(default_factory=dict) + + +class DebugInfo(BaseModel): + content_source: str | None = None + extractor: str | None = None + + +class ExtractionOutput(BaseModel): + success: bool + article: ExtractedArticle | None = None + error: ErrorInfo | None = None + debug: DebugInfo | None = None + warnings: list[str] = Field(default_factory=list) diff --git a/src/summary_mcp/server.py b/src/summary_mcp/server.py new file mode 100644 index 0000000..0d32706 --- /dev/null +++ b/src/summary_mcp/server.py @@ -0,0 +1,40 @@ +from __future__ import annotations + +from mcp.server.fastmcp import FastMCP + +from summary_mcp.core.pipeline import extract_content +from summary_mcp.models.item import Item +from summary_mcp.models.summary_io import ExtractionInput + + +mcp = FastMCP(name="content-extract-mcp") + + +@mcp.tool() +def extract_url_content(url: str, language_hint: str | None = None) -> dict: + """Extract structured article content from a single URL.""" + result = extract_content( + ExtractionInput( + url=url, + language_hint=language_hint, + ) + ) + return result.model_dump(mode="json") + + +@mcp.tool() +def extract_item_content(item: dict) -> dict: + """Extract structured article content from a normalized item object.""" + parsed_item = Item.model_validate(item) + result = extract_content( + ExtractionInput(item=parsed_item) + ) + return result.model_dump(mode="json") + + +def main() -> None: + mcp.run() + + +if __name__ == "__main__": + main() diff --git a/src/summary_mcp/validate_llm_result.py b/src/summary_mcp/validate_llm_result.py new file mode 100644 index 0000000..426711f --- /dev/null +++ b/src/summary_mcp/validate_llm_result.py @@ -0,0 +1,33 @@ +from __future__ import annotations + +import argparse +import json +from pathlib import Path + +from summary_mcp.validators.llm_result import validate_llm_result + + +def main() -> None: + parser = argparse.ArgumentParser(description="Validate an LLM summary JSON result.") + parser.add_argument("result", type=Path, help="Path to the LLM result JSON file") + parser.add_argument( + "--extracted", + type=Path, + default=None, + help="Optional extracted article JSON used for title/url consistency checks", + ) + args = parser.parse_args() + + report = validate_llm_result(args.result, args.extracted) + payload = { + "valid": report.valid, + "errors": report.errors, + "warnings": report.warnings, + "normalized_result": report.normalized_result, + } + print(json.dumps(payload, ensure_ascii=False, indent=2)) + raise SystemExit(0 if report.valid else 1) + + +if __name__ == "__main__": + main() diff --git a/src/summary_mcp/validators/__init__.py b/src/summary_mcp/validators/__init__.py new file mode 100644 index 0000000..b11e9b5 --- /dev/null +++ b/src/summary_mcp/validators/__init__.py @@ -0,0 +1 @@ +"""Validation helpers for LLM outputs.""" diff --git a/src/summary_mcp/validators/__pycache__/__init__.cpython-312.pyc b/src/summary_mcp/validators/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000..a7cdfcc Binary files /dev/null and b/src/summary_mcp/validators/__pycache__/__init__.cpython-312.pyc differ diff --git a/src/summary_mcp/validators/__pycache__/llm_result.cpython-312.pyc b/src/summary_mcp/validators/__pycache__/llm_result.cpython-312.pyc new file mode 100644 index 0000000..95d107a Binary files /dev/null and b/src/summary_mcp/validators/__pycache__/llm_result.cpython-312.pyc differ diff --git a/src/summary_mcp/validators/llm_result.py b/src/summary_mcp/validators/llm_result.py new file mode 100644 index 0000000..cd0862b --- /dev/null +++ b/src/summary_mcp/validators/llm_result.py @@ -0,0 +1,84 @@ +from __future__ import annotations + +import json +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any + +from pydantic import ValidationError + +from summary_mcp.models.llm_result import LlmSummaryResult + + +@dataclass +class ValidationReport: + valid: bool + errors: list[str] = field(default_factory=list) + warnings: list[str] = field(default_factory=list) + normalized_result: dict[str, Any] | None = None + + +def _load_json(path: Path) -> dict[str, Any]: + with path.open("r", encoding="utf-8") as handle: + return json.load(handle) + + +def _validate_business_rules(result: LlmSummaryResult, extracted: dict[str, Any] | None) -> tuple[list[str], list[str]]: + errors: list[str] = [] + warnings: list[str] = [] + + keyword_overlap = set(result.keywords) & set(result.topics) + if keyword_overlap: + errors.append(f"`keywords` and `topics` must not overlap: {sorted(keyword_overlap)}") + + if len(set(result.highlights)) != len(result.highlights): + errors.append("`highlights` contains duplicate entries") + if len(set(result.keywords)) != len(result.keywords): + errors.append("`keywords` contains duplicate entries") + if len(set(result.topics)) != len(result.topics): + errors.append("`topics` contains duplicate entries") + + if extracted: + article = extracted.get("article") or {} + extracted_title = article.get("title") + extracted_url = article.get("url") + if extracted_title and result.title != extracted_title: + errors.append("`title` does not match extracted article title") + if extracted_url and str(result.url) != extracted_url: + errors.append("`url` does not match extracted article url") + + if result.category == "\u8d44\u8baf" and result.worth_keeping: + warnings.append("`??` category marked as worth keeping; check if this is intentional.") + + return errors, warnings + + +def validate_llm_result( + result_path: Path, + extracted_path: Path | None = None, +) -> ValidationReport: + try: + raw_result = _load_json(result_path) + except json.JSONDecodeError as exc: + return ValidationReport(valid=False, errors=[f"Invalid JSON: {exc}"]) + + extracted: dict[str, Any] | None = None + if extracted_path is not None: + try: + extracted = _load_json(extracted_path) + except json.JSONDecodeError as exc: + return ValidationReport(valid=False, errors=[f"Invalid extracted JSON: {exc}"]) + + try: + parsed = LlmSummaryResult.model_validate(raw_result) + except ValidationError as exc: + errors = [f"{'.'.join(str(part) for part in error['loc'])}: {error['msg']}" for error in exc.errors()] + return ValidationReport(valid=False, errors=errors) + + errors, warnings = _validate_business_rules(parsed, extracted) + return ValidationReport( + valid=not errors, + errors=errors, + warnings=warnings, + normalized_result=parsed.model_dump(mode="json"), + )