Add FreshRSS ingestion and rule filtering

This commit is contained in:
zhuyongxin
2026-03-24 18:37:14 +08:00
parent c8d96705d1
commit ff3d15b8c4
21 changed files with 1460 additions and 21 deletions
+2
View File
@@ -1,2 +1,4 @@
.claude/ .claude/
.codex/ .codex/
__pycache__/
*.pyc
+52 -3
View File
@@ -1,6 +1,6 @@
# Content Extract MCP # Content Extract MCP
Python MCP scaffold for article content extraction. Python MCP scaffold for article content extraction, structured summary validation, and deterministic filtering.
## Run ## Run
@@ -9,10 +9,11 @@ pip install -e .
summary-mcp summary-mcp
``` ```
The server exposes two tools: The server exposes three tools:
- `extract_url_content` - `extract_url_content`
- `extract_item_content` - `extract_item_content`
- `filter_summary_result`
Validate an LLM summary result: Validate an LLM summary result:
@@ -28,3 +29,51 @@ python scripts/run_summary_loop.py ^
--prompt outputs/llm-summary-prompt.txt ^ --prompt outputs/llm-summary-prompt.txt ^
--output outputs/result.json --output outputs/result.json
``` ```
Pull FreshRSS entries and map them into normalized `item` objects:
```bash
set FRESHRSS_API_BASE_URL=http://127.0.0.1:8081/api/greader.php
set FRESHRSS_USERNAME=bot
set FRESHRSS_API_PASSWORD=your-api-password
python scripts/pull_freshrss_items.py --limit 5
```
The script writes:
- `outputs/freshrss.raw.json`
- `outputs/freshrss.items.json`
Pull FreshRSS entries and run content extraction for each mapped item:
```bash
set FRESHRSS_API_BASE_URL=http://127.0.0.1:8081/api/greader.php
set FRESHRSS_USERNAME=osiman
set FRESHRSS_API_PASSWORD=your-api-password
python scripts/run_freshrss_extract.py --limit 1
```
The script writes:
- `outputs/freshrss.raw.json`
- `outputs/freshrss.items.json`
- `outputs/freshrss.extracted.json`
Run deterministic filter rules against a structured summary result:
```bash
python scripts/run_filter_rules.py ^
--summary outputs/result.json ^
--extracted outputs/read-flow-2026.extracted.json ^
--output outputs/filter-decision.json
```
You can optionally pass a context file to inject interest topics or source tags:
```bash
python scripts/run_filter_rules.py ^
--summary outputs/result.json ^
--extracted outputs/read-flow-2026.extracted.json ^
--context outputs/filter-context.json ^
--output outputs/filter-decision.with-context.json
```
+13 -9
View File
@@ -2,7 +2,7 @@
## 当前状态 ## 当前状态
项目当前处于 `Content Extract MCP` 的 MVP 完成阶段。 项目当前处于 `Content Extract MCP` 的 MVP 完成并进入规则过滤阶段。
已完成: 已完成:
@@ -19,6 +19,10 @@
- [x] 设计并迭代 LLM 摘要 prompt - [x] 设计并迭代 LLM 摘要 prompt
- [x] 验证 LLM 输出的 `result.json` 基本符合预期 - [x] 验证 LLM 输出的 `result.json` 基本符合预期
- [x] 已封装 LLM 摘要校验 workflow skill - [x] 已封装 LLM 摘要校验 workflow skill
- [x] 接通 FreshRSS 上游并完成 `item` 映射
- [x] 跑通 `FreshRSS -> item -> content extraction` 单条链路
- [x] 定义规则过滤层 schema
- [x] 实现第一版规则引擎与过滤 MCP tool
--- ---
@@ -34,12 +38,14 @@
## P1 - 下一阶段推进 ## P1 - 下一阶段推进
- [ ] 把上游 RSS 聚合结果映射成标准化 `item` - [x] 把上游 RSS 聚合结果映射成标准化 `item`
- [ ] 补充 `item` 的实际样例文件 - [x] 补充 `item` 的实际样例文件
- [ ] 定义规则过滤层的输入输出 schema - [x] 跑通 `FreshRSS -> item -> content extraction` 单条链路
- [x] 定义规则过滤层的输入输出 schema
- [ ] 设计知识库入库格式 - [ ] 设计知识库入库格式
- [ ] 设计 webhook / 推送格式 - [ ] 设计 webhook / 推送格式
- [ ] 给提取结果增加更多正文清洗策略,比如尾部噪音清理 - [ ] 给提取结果增加更多正文清洗策略,比如尾部噪音清理
- [ ] 细化过滤规则并引入更多个性化上下文
--- ---
@@ -56,18 +62,16 @@
## 当前建议的下一步 ## 当前建议的下一步
优先做这两件事: 优先做这两件事:
1. 把上游 RSS 聚合结果映射成标准化 `item` 1. 设计知识库入库格式
2. 补充一个真实 `item` 样例文件并开始设计规则过滤 schema 2. 设计 webhook / 推送格式
--- ---
## 收束上下文后建议先看 ## 收束上下文后建议先看
- docs/context-reset-brief.md - docs/context-reset-brief.md
- docs/filter-rule-engine-design.md
- docs/content-extract-mcp-mvp-archive.md - docs/content-extract-mcp-mvp-archive.md
- TODO.md - TODO.md
+102
View File
@@ -0,0 +1,102 @@
[
{
"rule_id": "drop-low-content",
"enabled": true,
"stop_on_match": true,
"conditions_all": [
{ "field": "article.quality_flags.is_low_content", "op": "eq", "value": true }
],
"action": {
"decision": "drop",
"reason": "Extracted article quality is too low.",
"labels": ["quality", "low-content"],
"priority": 100
}
},
{
"rule_id": "review-truncated-content",
"enabled": true,
"stop_on_match": false,
"conditions_all": [
{ "field": "article.quality_flags.is_truncated", "op": "eq", "value": true }
],
"action": {
"decision": "review",
"reason": "Extracted article may be truncated.",
"labels": ["quality", "truncated"],
"priority": 95
}
},
{
"rule_id": "review-paywalled-content",
"enabled": true,
"stop_on_match": false,
"conditions_all": [
{ "field": "article.quality_flags.is_paywalled", "op": "eq", "value": true }
],
"action": {
"decision": "review",
"reason": "Article may be behind a paywall.",
"labels": ["quality", "paywall"],
"priority": 95
}
},
{
"rule_id": "drop-ephemeral-news",
"enabled": true,
"stop_on_match": true,
"conditions_all": [
{ "field": "summary.category", "op": "eq", "value": "资讯" },
{ "field": "summary.worth_keeping", "op": "eq", "value": false }
],
"action": {
"decision": "drop",
"reason": "News-like content marked as not worth keeping.",
"labels": ["summary", "ephemeral-news"],
"priority": 90
}
},
{
"rule_id": "keep-worth-keeping-method",
"enabled": true,
"stop_on_match": false,
"conditions_all": [
{ "field": "summary.worth_keeping", "op": "eq", "value": true },
{ "field": "summary.category", "op": "in", "value": ["方法论", "工具实践"] }
],
"action": {
"decision": "keep",
"reason": "Structured summary marked the content as worth keeping in a durable category.",
"labels": ["summary", "durable"],
"priority": 80
}
},
{
"rule_id": "keep-interest-topic",
"enabled": true,
"stop_on_match": false,
"conditions_all": [
{ "field": "summary.topics", "op": "overlap", "value": { "from_field": "context.interest_topics" } }
],
"action": {
"decision": "keep",
"reason": "Topics overlap with current interest profile.",
"labels": ["interest", "topic-match"],
"priority": 75
}
},
{
"rule_id": "review-worth-keeping-other",
"enabled": true,
"stop_on_match": false,
"conditions_all": [
{ "field": "summary.worth_keeping", "op": "eq", "value": true }
],
"action": {
"decision": "review",
"reason": "Worth-keeping signal is positive but no stronger keep rule matched.",
"labels": ["summary", "needs-review"],
"priority": 60
}
}
]
+59
View File
@@ -0,0 +1,59 @@
# 文档索引
## 当前推荐阅读顺序
1. `context-reset-brief.md`
- 当前真实进度与下一步入口
2. `summary-mcp-service-design.md`
- 当前 MCP 服务的职责、接口和边界
3. `filter-rule-engine-design.md`
- 过滤层的输入输出、规则结构与当前实现
4. `source-schema-design.md`
- `source -> item -> document` 的对象设计
5. `reading-pipeline-design-notes.md`
- 更上层的阅读流方案与阶段划分
6. `summary-loop-explained.md`
- 当前 LLM 摘要校验闭环的解释
## 当前文档分层
### 1. 当前状态与导航
- `README.md`
- 仓库入口与脚本运行方式
- `TODO.md`
- 当前优先级、已完成项、下一阶段任务
- `docs/context-reset-brief.md`
- 当前阶段状态的最短摘要
- `docs/README.md`
- 文档索引与阅读顺序
### 2. 当前实现设计
- `docs/summary-mcp-service-design.md`
- 当前内容提取 MCP 的真实设计
- `docs/summary-core-interface-design.md`
- 摘要/提取内核的接口抽象
- `docs/source-schema-design.md`
- `source`、`item`、`document` 三层 schema
- `docs/summary-loop-explained.md`
- 提取 JSON -> LLM 摘要 JSON -> 校验 的闭环说明
- `docs/filter-rule-engine-design.md`
- 第一版规则过滤引擎设计与落地位置
### 3. 上下游方案设计
- `docs/reading-pipeline-design-notes.md`
- 整体阅读流、规则、sink、push 的方案笔记
### 4. 历史归档
- `docs/content-extract-mcp-mvp-archive.md`
- MVP 阶段归档,部分状态已被后续进展覆盖
## 当前文档维护原则
- `context-reset-brief.md` 记录当前最新状态
- `TODO.md` 记录任务优先级与下一步
- `content-extract-mcp-mvp-archive.md` 只当历史快照,不再作为最新事实来源
- 新增阶段性进展,优先更新 `README.md`、`TODO.md`、`context-reset-brief.md`
+42 -8
View File
@@ -7,9 +7,10 @@
- 只负责内容提取 - 只负责内容提取
- 不负责摘要、分类、价值判断 - 不负责摘要、分类、价值判断
- 已完成 Python MCP 骨架 - 已完成 Python MCP 骨架
- 已实现两个 tool: - 已实现三个 tool:
- `extract_url_content` - `extract_url_content`
- `extract_item_content` - `extract_item_content`
- `filter_summary_result`
- 已完成真实 URL 提取验证 - 已完成真实 URL 提取验证
- 已完成 LLM 摘要 prompt - 已完成 LLM 摘要 prompt
- 已完成 LLM 摘要结果 schema 校验器 - 已完成 LLM 摘要结果 schema 校验器
@@ -17,6 +18,12 @@
- 已完成 LLM 摘要校验 skill 封装 - 已完成 LLM 摘要校验 skill 封装
- 已清理旧的启发式 `summarizer.py` - 已清理旧的启发式 `summarizer.py`
- 已将旧的 MCP 设计文档更新为当前“Content Extract MCP”语义 - 已将旧的 MCP 设计文档更新为当前“Content Extract MCP”语义
- 已完成 FreshRSS `greader` API 接入
- 已完成 FreshRSS entry -> `item` 映射
- 已产出真实 `item` 样例文件
- 已跑通 `FreshRSS -> item -> content extraction` 单条链路
- 已定义过滤层输入输出 schema
- 已完成第一版规则引擎、本地脚本和 MCP tool
## 当前关键文件 ## 当前关键文件
@@ -28,12 +35,34 @@
- `src/summary_mcp/models/llm_result.py` - `src/summary_mcp/models/llm_result.py`
- LLM 校验器: - LLM 校验器:
- `src/summary_mcp/validators/llm_result.py` - `src/summary_mcp/validators/llm_result.py`
- 过滤模型:
- `src/summary_mcp/models/filtering.py`
- 过滤引擎:
- `src/summary_mcp/filters/engine.py`
- 默认过滤规则:
- `configs/filter_rules.json`
- 校验 CLI: - 校验 CLI:
- `src/summary_mcp/validate_llm_result.py` - `src/summary_mcp/validate_llm_result.py`
- 最小闭环脚本: - 最小闭环脚本:
- `scripts/run_summary_loop.py` - `scripts/run_summary_loop.py`
- FreshRSS 拉取脚本:
- `scripts/pull_freshrss_items.py`
- FreshRSS 提取脚本:
- `scripts/run_freshrss_extract.py`
- 过滤脚本:
- `scripts/run_filter_rules.py`
- 当前提示词: - 当前提示词:
- `outputs/llm-summary-prompt.txt` - `outputs/llm-summary-prompt.txt`
- FreshRSS 原始响应样例:
- `outputs/freshrss.raw.json`
- FreshRSS item 样例:
- `outputs/freshrss.items.json`
- FreshRSS 提取结果:
- `outputs/freshrss.extracted.json`
- 过滤结果样例:
- `outputs/filter-decision.json`
- 文档索引:
- `docs/README.md`
- MVP 归档: - MVP 归档:
- `docs/content-extract-mcp-mvp-archive.md` - `docs/content-extract-mcp-mvp-archive.md`
- 当前 TODO: - 当前 TODO:
@@ -45,21 +74,26 @@
- LLM 可根据提取结果生成摘要 JSON - LLM 可根据提取结果生成摘要 JSON
- validator 可校验摘要 JSON - validator 可校验摘要 JSON
- 最小闭环脚本可直接调用 LLM 接口并产出通过校验的结果 - 最小闭环脚本可直接调用 LLM 接口并产出通过校验的结果
- FreshRSS API 可拉取真实 entry
- 真实 entry 可映射为标准化 `item`
- 标准化 `item` 可继续进入内容提取流程
- 规则引擎可对结构化摘要结果输出 `keep / drop / review` 决策
- MCP tool 可承接“由上层 LLM/Agent 调用过滤”的模式
## 当前未开始的下一阶段 ## 当前未开始的下一阶段
- 让 FreshRSS 作为主聚合池 - 设计知识库入库格式
- 设计 FreshRSS entry -> `item` 的映射 - 设计 webhook / 推送格式
- 准备真实 `item` 样例 - 把 FreshRSS 拉取与提取流程进一步批量化/调度化
- 开始定义规则过滤层 schema - 迭代更细的过滤规则与个性化上下文
## 收束后建议从这里继续 ## 收束后建议从这里继续
优先从这两个问题继续: 优先从这两个问题继续:
1. FreshRSS 的 entry 字段如何映射成 `item` 1. 先设计知识库 sink 的输入输出格式
2. 先做一个真实 `item` 样例文件,再讨论规则过滤 2. 再决定先接 webhook / 推送还是继续细化过滤规则
## 一句话结论 ## 一句话结论
当前 MVP 已完成,下一阶段不再是继续打磨 MCP,而是开始接入 FreshRSS 上游并建立 `item` 标准化入口。 当前 MVP 已完成,FreshRSS 上游和第一版规则过滤层都已接通,下一阶段应转向下游 sink 与推送设计。
+225
View File
@@ -0,0 +1,225 @@
# 规则过滤引擎设计
## 1. 目标
当前过滤层的定位是:
- 不让 LLM 直接做最终过滤决策
- 让 LLM 只产出结构化信号
- 由规则引擎输出最终 `keep / drop / review` 决策
当前链路是:
`item -> content extraction -> llm summary -> filter rule engine`
## 2. 输入输出
### 2.1 FilterInput
过滤层统一读取四类输入:
- `item`
- `article`
- `summary`
- `context`
其中:
- `item` 表示上游标准化候选条目
- `article` 表示正文提取结果
- `summary` 表示结构化 LLM 摘要结果
- `context` 表示额外的用户偏好或运行时上下文
### 2.2 FilterDecisionResult
过滤结果统一输出:
- `decision`
- `keep`
- `drop`
- `review`
- `matched_rules`
- `reasons`
- `labels`
- `priority`
- `matches`
这样后续的知识库 sink、推送层或人工审核都可以稳定消费。
## 3. 规则结构
当前规则文件位置:
- `configs/filter_rules.json`
单条规则结构为:
```json
{
"rule_id": "keep-worth-keeping-method",
"enabled": true,
"stop_on_match": false,
"conditions_all": [
{ "field": "summary.worth_keeping", "op": "eq", "value": true },
{ "field": "summary.category", "op": "in", "value": ["方法论", "工具实践"] }
],
"action": {
"decision": "keep",
"reason": "Structured summary marked the content as worth keeping in a durable category.",
"labels": ["summary", "durable"],
"priority": 80
}
}
```
## 4. 当前支持的操作符
- `eq`
- `ne`
- `in`
- `not_in`
- `contains`
- `overlap`
- `gte`
- `lte`
- `exists`
当前条件组合方式:
- `conditions_all`
- `conditions_any`
## 5. 动态上下文字段
规则支持从其他字段动态取值,例如:
```json
{ "field": "summary.topics", "op": "overlap", "value": { "from_field": "context.interest_topics" } }
```
这允许上层 Agent 在调用 `filter_summary_result` 时,把当前关注主题动态注入,而不必把偏好硬编码在规则文件里。
## 6. 当前默认规则思路
当前默认规则分为三类:
- 质量拦截
- 低质量正文直接 `drop`
- 疑似截断或付费墙进入 `review`
- 摘要价值判断
- `资讯 + worth_keeping=false` 直接 `drop`
- `方法论/工具实践 + worth_keeping=true` 直接 `keep`
- 个性化补充信号
- `summary.topics` 与 `context.interest_topics` 重叠时提升为 `keep`
- 其他 `worth_keeping=true` 的结果默认进入 `review`
## 7. 规则执行流程
当前过滤流程按以下顺序执行:
1. 上层先产出结构化 `summary`
2. 可选附带 `item`、`article` 与 `context`
3. 规则引擎按 `priority` 从高到低遍历规则
4. 每条规则根据 `conditions_all` / `conditions_any` 判断是否命中
5. 命中的规则被收集为 `matches`
6. 最终根据命中结果收敛为 `keep / drop / review`
当前决策收敛规则是:
- 只要命中任意 `drop`,最终结果就是 `drop`
- 否则只要命中任意 `keep`,最终结果就是 `keep`
- 否则如果命中任意 `review`,最终结果就是 `review`
- 如果没有任何规则命中,默认回落到 `review`
这样设计的原因是:
- `drop` 应该拥有最高约束力
- `keep` 只在没有硬性淘汰时生效
- 默认不自动放行未知内容
## 8. 为什么让 LLM 调用规则 tool,而不是直接裁决
当前架构故意拆成两层:
- LLM 负责生成结构化信号
- 规则引擎负责输出最终过滤决策
原因是:
- LLM 适合做语义理解、归类、摘要和价值信号提取
- 规则引擎适合做稳定、可复现、可审计的最终判断
因此推荐的调用方式是:
1. LLM 先调用提取 tool
2. LLM 或本地脚本产出 `summary_result`
3. LLM 再调用 `filter_summary_result`
4. 后续根据过滤结果决定是否入库、推送或人工审核
这意味着:
- LLM 是编排者
- rule engine 是裁决器
而不是让 LLM 在过滤阶段再次自由发挥。
## 9. 调用示例
### 9.1 本地脚本
```bash
python scripts/run_filter_rules.py ^
--summary outputs/result.json ^
--extracted outputs/read-flow-2026.extracted.json ^
--output outputs/filter-decision.json
```
### 9.2 带上下文的调用
```json
{
"interest_topics": ["个人信息管理", "阅读工作流"]
}
```
将这份上下文作为 `--context` 传入后,规则就可以根据当前关注主题做加权。
### 9.3 MCP tool
`filter_summary_result` 接收:
- `summary_result`
- 可选 `extracted_article`
- 可选 `item`
- 可选 `context`
返回:
- `decision`
- `matched_rules`
- `reasons`
- `labels`
- `priority`
- `matches`
## 10. 当前落地位置
- 过滤模型:
- `src/summary_mcp/models/filtering.py`
- 规则引擎:
- `src/summary_mcp/filters/engine.py`
- 默认规则:
- `configs/filter_rules.json`
- 本地运行脚本:
- `scripts/run_filter_rules.py`
- MCP tool:
- `filter_summary_result`
## 11. 当前阶段的设计结论
第一版过滤层采用“规则主导,LLM 提供信号”的架构:
- LLM 不直接做最终裁决
- 上层 Agent 可以决定是否调用过滤 tool
- 真正的过滤决策由规则引擎输出
- 灰区内容后续再考虑是否引入第二层 LLM 辅助判断
+39
View File
@@ -0,0 +1,39 @@
{
"decision": "keep",
"matched_rules": [
"keep-worth-keeping-method",
"review-worth-keeping-other"
],
"reasons": [
"Structured summary marked the content as worth keeping in a durable category.",
"Worth-keeping signal is positive but no stronger keep rule matched."
],
"labels": [
"durable",
"needs-review",
"summary"
],
"priority": 80,
"matches": [
{
"rule_id": "keep-worth-keeping-method",
"decision": "keep",
"reason": "Structured summary marked the content as worth keeping in a durable category.",
"labels": [
"summary",
"durable"
],
"priority": 80
},
{
"rule_id": "review-worth-keeping-other",
"decision": "review",
"reason": "Worth-keeping signal is positive but no stronger keep rule matched.",
"labels": [
"summary",
"needs-review"
],
"priority": 60
}
]
}
+53
View File
@@ -0,0 +1,53 @@
{
"decision": "keep",
"matched_rules": [
"keep-worth-keeping-method",
"keep-interest-topic",
"review-worth-keeping-other"
],
"reasons": [
"Structured summary marked the content as worth keeping in a durable category.",
"Topics overlap with current interest profile.",
"Worth-keeping signal is positive but no stronger keep rule matched."
],
"labels": [
"durable",
"interest",
"needs-review",
"summary",
"topic-match"
],
"priority": 80,
"matches": [
{
"rule_id": "keep-worth-keeping-method",
"decision": "keep",
"reason": "Structured summary marked the content as worth keeping in a durable category.",
"labels": [
"summary",
"durable"
],
"priority": 80
},
{
"rule_id": "keep-interest-topic",
"decision": "keep",
"reason": "Topics overlap with current interest profile.",
"labels": [
"interest",
"topic-match"
],
"priority": 75
},
{
"rule_id": "review-worth-keeping-other",
"decision": "review",
"reason": "Worth-keeping signal is positive but no stronger keep rule matched.",
"labels": [
"summary",
"needs-review"
],
"priority": 60
}
]
}
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
+92
View File
@@ -0,0 +1,92 @@
from __future__ import annotations
import argparse
import json
import os
import sys
from pathlib import Path
from typing import Any
REPO_ROOT = Path(__file__).resolve().parents[1]
SRC_ROOT = REPO_ROOT / "src"
if str(SRC_ROOT) not in sys.path:
sys.path.insert(0, str(SRC_ROOT))
from summary_mcp.integrations.freshrss import FreshRSSClient, map_entry_to_item
def _save_json(path: Path, payload: dict[str, Any] | list[Any]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
def _load_required(name: str, value: str | None) -> str:
resolved = value or os.environ.get(name)
if not resolved:
cli_name = name.lower().replace("_", "-")
raise RuntimeError(f"Missing required value: pass --{cli_name} or set {name}.")
return resolved
def main() -> None:
parser = argparse.ArgumentParser(description="Pull FreshRSS entries and map them into normalized item objects.")
parser.add_argument("--api-base-url", type=str, default=None, help="FreshRSS greader API base URL")
parser.add_argument("--username", type=str, default=None, help="FreshRSS API username")
parser.add_argument("--api-password", type=str, default=None, help="FreshRSS API password")
parser.add_argument(
"--stream-id",
type=str,
default="user/-/state/com.google/reading-list",
help="Google Reader API stream id",
)
parser.add_argument("--limit", type=int, default=10, help="Maximum number of entries to request")
parser.add_argument("--continuation", type=str, default=None, help="Continuation token for paging")
parser.add_argument(
"--raw-output",
type=Path,
default=REPO_ROOT / "outputs" / "freshrss.raw.json",
help="Where to save the raw FreshRSS response",
)
parser.add_argument(
"--items-output",
type=Path,
default=REPO_ROOT / "outputs" / "freshrss.items.json",
help="Where to save the mapped item list",
)
parser.add_argument("--timeout", type=float, default=20.0, help="Request timeout in seconds")
args = parser.parse_args()
api_base_url = _load_required("FRESHRSS_API_BASE_URL", args.api_base_url)
username = _load_required("FRESHRSS_USERNAME", args.username)
api_password = _load_required("FRESHRSS_API_PASSWORD", args.api_password)
client = FreshRSSClient(
api_base_url=api_base_url,
username=username,
api_password=api_password,
timeout_seconds=args.timeout,
)
auth_token = client.client_login()
payload = client.fetch_stream_contents(
auth_token=auth_token,
stream_id=args.stream_id,
limit=args.limit,
continuation=args.continuation,
)
entries = payload.get("items")
if not isinstance(entries, list):
raise RuntimeError("FreshRSS stream response does not contain an items array.")
mapped_items = [map_entry_to_item(entry).model_dump(mode="json") for entry in entries]
_save_json(args.raw_output, payload)
_save_json(args.items_output, mapped_items)
print(f"Saved {len(mapped_items)} mapped items to {args.items_output}")
if __name__ == "__main__":
main()
+77
View File
@@ -0,0 +1,77 @@
from __future__ import annotations
import argparse
import json
import sys
from pathlib import Path
from typing import Any
REPO_ROOT = Path(__file__).resolve().parents[1]
SRC_ROOT = REPO_ROOT / "src"
if str(SRC_ROOT) not in sys.path:
sys.path.insert(0, str(SRC_ROOT))
from summary_mcp.filters.engine import load_filter_rules, evaluate_filter_rules
from summary_mcp.models.document import ExtractedArticle
from summary_mcp.models.filtering import FilterContext, FilterInput
from summary_mcp.models.item import Item
from summary_mcp.models.llm_result import LlmSummaryResult
def _load_json(path: Path) -> dict[str, Any]:
return json.loads(path.read_text(encoding="utf-8-sig"))
def _save_json(path: Path, payload: dict[str, Any]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
def main() -> None:
parser = argparse.ArgumentParser(description="Run deterministic filter rules against a structured summary result.")
parser.add_argument("--summary", type=Path, required=True, help="Structured LLM summary JSON file")
parser.add_argument("--extracted", type=Path, default=None, help="Extracted article JSON file")
parser.add_argument("--item", type=Path, default=None, help="Normalized item JSON file")
parser.add_argument("--context", type=Path, default=None, help="Optional filter context JSON file")
parser.add_argument(
"--rules",
type=Path,
default=REPO_ROOT / "configs" / "filter_rules.json",
help="Filter rule config JSON file",
)
parser.add_argument(
"--output",
type=Path,
default=REPO_ROOT / "outputs" / "filter-decision.json",
help="Where to save the filter decision",
)
args = parser.parse_args()
summary = LlmSummaryResult.model_validate(_load_json(args.summary))
article = None
if args.extracted is not None:
extracted_payload = _load_json(args.extracted)
article_payload = extracted_payload.get("article", extracted_payload)
article = ExtractedArticle.model_validate(article_payload)
item = None
if args.item is not None:
item = Item.model_validate(_load_json(args.item))
context = FilterContext.model_validate(_load_json(args.context)) if args.context else FilterContext()
rules = load_filter_rules(args.rules)
decision = evaluate_filter_rules(
FilterInput(item=item, article=article, summary=summary, context=context),
rules,
)
_save_json(args.output, decision.model_dump(mode="json"))
print(f"Saved filter decision to {args.output}")
if __name__ == "__main__":
main()
+125
View File
@@ -0,0 +1,125 @@
from __future__ import annotations
import argparse
import json
import os
import sys
from pathlib import Path
from typing import Any
REPO_ROOT = Path(__file__).resolve().parents[1]
SRC_ROOT = REPO_ROOT / "src"
if str(SRC_ROOT) not in sys.path:
sys.path.insert(0, str(SRC_ROOT))
from summary_mcp.core.pipeline import extract_content
from summary_mcp.integrations.freshrss import FreshRSSClient, map_entry_to_item
from summary_mcp.models.summary_io import ExtractionInput
def _save_json(path: Path, payload: dict[str, Any] | list[Any]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
def _load_required(name: str, value: str | None) -> str:
resolved = value or os.environ.get(name)
if not resolved:
cli_name = name.lower().replace("_", "-")
raise RuntimeError(f"Missing required value: pass --{cli_name} or set {name}.")
return resolved
def main() -> None:
parser = argparse.ArgumentParser(description="Pull FreshRSS entries and run content extraction for each mapped item.")
parser.add_argument("--api-base-url", type=str, default=None, help="FreshRSS greader API base URL")
parser.add_argument("--username", type=str, default=None, help="FreshRSS API username")
parser.add_argument("--api-password", type=str, default=None, help="FreshRSS API password")
parser.add_argument(
"--stream-id",
type=str,
default="user/-/state/com.google/reading-list",
help="Google Reader API stream id",
)
parser.add_argument("--limit", type=int, default=5, help="Maximum number of entries to request")
parser.add_argument("--continuation", type=str, default=None, help="Continuation token for paging")
parser.add_argument(
"--raw-output",
type=Path,
default=REPO_ROOT / "outputs" / "freshrss.raw.json",
help="Where to save the raw FreshRSS response",
)
parser.add_argument(
"--items-output",
type=Path,
default=REPO_ROOT / "outputs" / "freshrss.items.json",
help="Where to save the mapped item list",
)
parser.add_argument(
"--extracted-output",
type=Path,
default=REPO_ROOT / "outputs" / "freshrss.extracted.json",
help="Where to save the extraction results",
)
parser.add_argument("--timeout", type=float, default=20.0, help="Request timeout in seconds")
args = parser.parse_args()
api_base_url = _load_required("FRESHRSS_API_BASE_URL", args.api_base_url)
username = _load_required("FRESHRSS_USERNAME", args.username)
api_password = _load_required("FRESHRSS_API_PASSWORD", args.api_password)
client = FreshRSSClient(
api_base_url=api_base_url,
username=username,
api_password=api_password,
timeout_seconds=args.timeout,
)
auth_token = client.client_login()
payload = client.fetch_stream_contents(
auth_token=auth_token,
stream_id=args.stream_id,
limit=args.limit,
continuation=args.continuation,
)
entries = payload.get("items")
if not isinstance(entries, list):
raise RuntimeError("FreshRSS stream response does not contain an items array.")
mapped_items = [map_entry_to_item(entry) for entry in entries]
extraction_results: list[dict[str, Any]] = []
success_count = 0
for item in mapped_items:
extraction = extract_content(ExtractionInput(item=item))
extraction_results.append(
{
"item": item.model_dump(mode="json"),
"extraction": extraction.model_dump(mode="json"),
}
)
if extraction.success:
success_count += 1
_save_json(args.raw_output, payload)
_save_json(args.items_output, [item.model_dump(mode="json") for item in mapped_items])
_save_json(
args.extracted_output,
{
"stream_id": args.stream_id,
"requested_limit": args.limit,
"entry_count": len(entries),
"extracted_success_count": success_count,
"results": extraction_results,
},
)
print(
f"Saved {len(mapped_items)} mapped items and {success_count} successful extractions to {args.extracted_output}"
)
if __name__ == "__main__":
main()
+3
View File
@@ -0,0 +1,3 @@
from .engine import DEFAULT_RULES_PATH, evaluate_filter_rules, load_filter_rules
__all__ = ["DEFAULT_RULES_PATH", "evaluate_filter_rules", "load_filter_rules"]
+136
View File
@@ -0,0 +1,136 @@
from __future__ import annotations
import json
from pathlib import Path
from typing import Any
from summary_mcp.models.filtering import (
FieldCondition,
FilterDecisionResult,
FilterInput,
FilterRule,
MatchedRule,
)
REPO_ROOT = Path(__file__).resolve().parents[3]
DEFAULT_RULES_PATH = REPO_ROOT / "configs" / "filter_rules.json"
def load_filter_rules(path: Path | None = None) -> list[FilterRule]:
rules_path = path or DEFAULT_RULES_PATH
payload = json.loads(rules_path.read_text(encoding="utf-8-sig"))
if not isinstance(payload, list):
raise RuntimeError("Filter rules file must contain a JSON array.")
return [FilterRule.model_validate(item) for item in payload]
def _normalize_value(value: Any) -> Any:
if hasattr(value, "model_dump"):
return value.model_dump(mode="json")
return value
def _resolve_field(filter_input: FilterInput, field_path: str) -> Any:
current: Any = filter_input
for part in field_path.split("."):
current = _normalize_value(current)
if isinstance(current, dict):
if part not in current:
return None
current = current[part]
continue
return None
return _normalize_value(current)
def _expected_value(filter_input: FilterInput, value: Any) -> Any:
if isinstance(value, dict) and "from_field" in value:
field_name = value.get("from_field")
if isinstance(field_name, str):
return _resolve_field(filter_input, field_name)
return value
def _match_condition(filter_input: FilterInput, condition: FieldCondition) -> bool:
current = _resolve_field(filter_input, condition.field)
expected = _expected_value(filter_input, condition.value)
if condition.op == "exists":
return (current is not None) if expected is not False else (current is None)
if condition.op == "eq":
return current == expected
if condition.op == "ne":
return current != expected
if condition.op == "in":
return current in expected if isinstance(expected, list) else False
if condition.op == "not_in":
return current not in expected if isinstance(expected, list) else False
if condition.op == "contains":
if isinstance(current, list):
return expected in current
if isinstance(current, str) and isinstance(expected, str):
return expected in current
return False
if condition.op == "overlap":
if isinstance(current, list) and isinstance(expected, list):
return bool(set(current) & set(expected))
return False
if condition.op == "gte":
return current is not None and expected is not None and current >= expected
if condition.op == "lte":
return current is not None and expected is not None and current <= expected
return False
def _rule_matches(filter_input: FilterInput, rule: FilterRule) -> bool:
if not rule.enabled:
return False
if rule.conditions_all and not all(_match_condition(filter_input, condition) for condition in rule.conditions_all):
return False
if rule.conditions_any and not any(_match_condition(filter_input, condition) for condition in rule.conditions_any):
return False
return bool(rule.conditions_all or rule.conditions_any)
def evaluate_filter_rules(filter_input: FilterInput, rules: list[FilterRule]) -> FilterDecisionResult:
matched: list[MatchedRule] = []
for rule in sorted(rules, key=lambda item: item.action.priority, reverse=True):
if not _rule_matches(filter_input, rule):
continue
matched_rule = MatchedRule(
rule_id=rule.rule_id,
decision=rule.action.decision,
reason=rule.action.reason,
labels=rule.action.labels,
priority=rule.action.priority,
)
matched.append(matched_rule)
if rule.stop_on_match:
break
if any(rule.decision == "drop" for rule in matched):
final_decision = "drop"
elif any(rule.decision == "keep" for rule in matched):
final_decision = "keep"
elif any(rule.decision == "review" for rule in matched):
final_decision = "review"
else:
final_decision = "review"
labels = sorted({label for rule in matched for label in rule.labels})
reasons = [rule.reason for rule in matched]
priorities = [rule.priority for rule in matched]
return FilterDecisionResult(
decision=final_decision,
matched_rules=[rule.rule_id for rule in matched],
reasons=reasons if reasons else ["No rule matched; defaulted to review."],
labels=labels,
priority=max(priorities, default=0),
matches=matched,
)
+1
View File
@@ -0,0 +1 @@
"""Integration helpers for upstream content sources."""
+199
View File
@@ -0,0 +1,199 @@
from __future__ import annotations
import hashlib
from datetime import UTC, datetime
from typing import Any
from urllib.parse import urlparse
import httpx
from summary_mcp.models.item import Item
def _trim_api_base_url(api_base_url: str) -> str:
return api_base_url.rstrip("/")
def _build_item_id(source_id: str, external_id: str | None, url: str, published_at: datetime | None) -> str:
seed = "|".join(
[
source_id,
external_id or "",
url,
published_at.isoformat() if published_at else "",
]
)
return f"sha256:{hashlib.sha256(seed.encode('utf-8')).hexdigest()}"
def _build_source_id(entry: dict[str, Any], url: str) -> str:
origin = entry.get("origin") or {}
stream_id = origin.get("streamId")
if isinstance(stream_id, str) and stream_id.strip():
digest = hashlib.sha256(stream_id.encode("utf-8")).hexdigest()[:16]
return f"freshrss:{digest}"
host = urlparse(url).netloc or "unknown-source"
return f"freshrss:{host}"
def _pick_entry_url(entry: dict[str, Any]) -> str:
candidates = [
entry.get("canonical"),
entry.get("alternate"),
]
for candidate_list in candidates:
if not isinstance(candidate_list, list):
continue
for candidate in candidate_list:
href = (candidate or {}).get("href")
if isinstance(href, str) and href.strip():
return href.strip()
entry_id = entry.get("id")
if isinstance(entry_id, str) and entry_id.startswith("tag:"):
return entry_id
raise ValueError("FreshRSS entry does not contain a usable URL.")
def _pick_content_block(entry: dict[str, Any], key: str) -> str | None:
block = entry.get(key)
if isinstance(block, dict):
content = block.get("content")
if isinstance(content, str) and content.strip():
return content
return None
def _pick_categories(entry: dict[str, Any]) -> list[str]:
categories = entry.get("categories")
if not isinstance(categories, list):
return []
values: list[str] = []
for category in categories:
if not isinstance(category, str):
continue
if category.startswith("user/-/label/"):
values.append(category.removeprefix("user/-/label/"))
elif category.startswith("user/-/state/com.google/"):
continue
else:
values.append(category)
return values
def _parse_datetime(timestamp: Any) -> datetime | None:
if timestamp is None:
return None
if isinstance(timestamp, (int, float)):
if timestamp > 10_000_000_000:
return datetime.fromtimestamp(timestamp / 1000, tz=UTC)
return datetime.fromtimestamp(timestamp, tz=UTC)
if isinstance(timestamp, str) and timestamp.isdigit():
value = int(timestamp)
if value > 10_000_000_000:
return datetime.fromtimestamp(value / 1000, tz=UTC)
return datetime.fromtimestamp(value, tz=UTC)
return None
def map_entry_to_item(entry: dict[str, Any]) -> Item:
url = _pick_entry_url(entry)
summary_html = _pick_content_block(entry, "summary")
content_html = _pick_content_block(entry, "content")
published_at = _parse_datetime(entry.get("published"))
source_id = _build_source_id(entry, url)
external_id = entry.get("id") if isinstance(entry.get("id"), str) else None
fetch_state = "fetched" if content_html and len(content_html.strip()) >= 500 else "pending"
metadata = {
"upstream": "freshrss",
"origin": entry.get("origin") or {},
"categories": _pick_categories(entry),
"crawled_at": _parse_datetime(entry.get("crawlTimeMsec")),
"published_epoch": entry.get("published"),
}
return Item(
item_id=_build_item_id(source_id, external_id, url, published_at),
source_id=source_id,
external_id=external_id,
title=entry.get("title"),
url=url,
author=entry.get("author"),
published_at=published_at,
discovered_at=datetime.now(tz=UTC),
content_kind="article",
language=None,
raw_summary=summary_html,
raw_content=content_html,
metadata=metadata,
fetch_state=fetch_state,
)
class FreshRSSClient:
def __init__(
self,
api_base_url: str,
username: str,
api_password: str,
timeout_seconds: float = 20.0,
) -> None:
self.api_base_url = _trim_api_base_url(api_base_url)
self.username = username
self.api_password = api_password
self.timeout_seconds = timeout_seconds
def _build_url(self, path: str) -> str:
return f"{self.api_base_url}/{path.lstrip('/')}"
def client_login(self) -> str:
data = {
"Email": self.username,
"Passwd": self.api_password,
}
with httpx.Client(timeout=self.timeout_seconds) as client:
response = client.post(self._build_url("accounts/ClientLogin"), data=data)
response.raise_for_status()
auth_token: str | None = None
for line in response.text.splitlines():
if line.startswith("Auth="):
auth_token = line.split("=", 1)[1].strip()
break
if not auth_token:
raise RuntimeError("FreshRSS ClientLogin succeeded but did not return an Auth token.")
return auth_token
def fetch_stream_contents(
self,
auth_token: str,
stream_id: str = "user/-/state/com.google/reading-list",
limit: int = 20,
continuation: str | None = None,
) -> dict[str, Any]:
params: dict[str, Any] = {
"output": "json",
"n": limit,
}
if continuation:
params["c"] = continuation
with httpx.Client(
timeout=self.timeout_seconds,
headers={"Authorization": f"GoogleLogin auth={auth_token}"},
) as client:
response = client.get(
self._build_url(f"reader/api/0/stream/contents/{stream_id}"),
params=params,
)
response.raise_for_status()
return response.json()
+67
View File
@@ -0,0 +1,67 @@
from __future__ import annotations
from datetime import datetime
from typing import Any, Literal
from pydantic import BaseModel, Field
from .document import ExtractedArticle
from .item import Item
from .llm_result import LlmSummaryResult
FilterDecision = Literal["keep", "drop", "review"]
ConditionOp = Literal["eq", "ne", "in", "not_in", "contains", "overlap", "gte", "lte", "exists"]
class FilterContext(BaseModel):
source_tags: list[str] = Field(default_factory=list)
interest_topics: list[str] = Field(default_factory=list)
interest_keywords: list[str] = Field(default_factory=list)
now: datetime | None = None
class FieldCondition(BaseModel):
field: str
op: ConditionOp
value: Any = None
class FilterAction(BaseModel):
decision: FilterDecision
reason: str
labels: list[str] = Field(default_factory=list)
priority: int = 50
class FilterRule(BaseModel):
rule_id: str
enabled: bool = True
stop_on_match: bool = False
conditions_all: list[FieldCondition] = Field(default_factory=list)
conditions_any: list[FieldCondition] = Field(default_factory=list)
action: FilterAction
class FilterInput(BaseModel):
item: Item | None = None
article: ExtractedArticle | None = None
summary: LlmSummaryResult
context: FilterContext = Field(default_factory=FilterContext)
class MatchedRule(BaseModel):
rule_id: str
decision: FilterDecision
reason: str
labels: list[str] = Field(default_factory=list)
priority: int = 0
class FilterDecisionResult(BaseModel):
decision: FilterDecision
matched_rules: list[str] = Field(default_factory=list)
reasons: list[str] = Field(default_factory=list)
labels: list[str] = Field(default_factory=list)
priority: int = 0
matches: list[MatchedRule] = Field(default_factory=list)
+30 -1
View File
@@ -1,9 +1,13 @@
from __future__ import annotations from __future__ import annotations
from mcp.server.fastmcp import FastMCP from mcp.server.fastmcp import FastMCP
from summary_mcp.core.pipeline import extract_content from summary_mcp.core.pipeline import extract_content
from summary_mcp.filters.engine import evaluate_filter_rules, load_filter_rules
from summary_mcp.models.document import ExtractedArticle
from summary_mcp.models.filtering import FilterContext, FilterInput
from summary_mcp.models.item import Item from summary_mcp.models.item import Item
from summary_mcp.models.llm_result import LlmSummaryResult
from summary_mcp.models.summary_io import ExtractionInput from summary_mcp.models.summary_io import ExtractionInput
@@ -32,6 +36,31 @@ def extract_item_content(item: dict) -> dict:
return result.model_dump(mode="json") return result.model_dump(mode="json")
@mcp.tool()
def filter_summary_result(
summary_result: dict,
extracted_article: dict | None = None,
item: dict | None = None,
context: dict | None = None,
) -> dict:
"""Apply deterministic filter rules to a structured summary result."""
parsed_summary = LlmSummaryResult.model_validate(summary_result)
parsed_article = ExtractedArticle.model_validate(extracted_article) if extracted_article else None
parsed_item = Item.model_validate(item) if item else None
parsed_context = FilterContext.model_validate(context or {})
rules = load_filter_rules()
decision = evaluate_filter_rules(
FilterInput(
item=parsed_item,
article=parsed_article,
summary=parsed_summary,
context=parsed_context,
),
rules,
)
return decision.model_dump(mode="json")
def main() -> None: def main() -> None:
mcp.run() mcp.run()