Add FreshRSS ingestion and rule filtering
This commit is contained in:
@@ -1,2 +1,4 @@
|
||||
.claude/
|
||||
.codex/
|
||||
__pycache__/
|
||||
*.pyc
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# Content Extract MCP
|
||||
# Content Extract MCP
|
||||
|
||||
Python MCP scaffold for article content extraction.
|
||||
Python MCP scaffold for article content extraction, structured summary validation, and deterministic filtering.
|
||||
|
||||
## Run
|
||||
|
||||
@@ -9,10 +9,11 @@ pip install -e .
|
||||
summary-mcp
|
||||
```
|
||||
|
||||
The server exposes two tools:
|
||||
The server exposes three tools:
|
||||
|
||||
- `extract_url_content`
|
||||
- `extract_item_content`
|
||||
- `filter_summary_result`
|
||||
|
||||
Validate an LLM summary result:
|
||||
|
||||
@@ -28,3 +29,51 @@ python scripts/run_summary_loop.py ^
|
||||
--prompt outputs/llm-summary-prompt.txt ^
|
||||
--output outputs/result.json
|
||||
```
|
||||
|
||||
Pull FreshRSS entries and map them into normalized `item` objects:
|
||||
|
||||
```bash
|
||||
set FRESHRSS_API_BASE_URL=http://127.0.0.1:8081/api/greader.php
|
||||
set FRESHRSS_USERNAME=bot
|
||||
set FRESHRSS_API_PASSWORD=your-api-password
|
||||
python scripts/pull_freshrss_items.py --limit 5
|
||||
```
|
||||
|
||||
The script writes:
|
||||
|
||||
- `outputs/freshrss.raw.json`
|
||||
- `outputs/freshrss.items.json`
|
||||
|
||||
Pull FreshRSS entries and run content extraction for each mapped item:
|
||||
|
||||
```bash
|
||||
set FRESHRSS_API_BASE_URL=http://127.0.0.1:8081/api/greader.php
|
||||
set FRESHRSS_USERNAME=osiman
|
||||
set FRESHRSS_API_PASSWORD=your-api-password
|
||||
python scripts/run_freshrss_extract.py --limit 1
|
||||
```
|
||||
|
||||
The script writes:
|
||||
|
||||
- `outputs/freshrss.raw.json`
|
||||
- `outputs/freshrss.items.json`
|
||||
- `outputs/freshrss.extracted.json`
|
||||
|
||||
Run deterministic filter rules against a structured summary result:
|
||||
|
||||
```bash
|
||||
python scripts/run_filter_rules.py ^
|
||||
--summary outputs/result.json ^
|
||||
--extracted outputs/read-flow-2026.extracted.json ^
|
||||
--output outputs/filter-decision.json
|
||||
```
|
||||
|
||||
You can optionally pass a context file to inject interest topics or source tags:
|
||||
|
||||
```bash
|
||||
python scripts/run_filter_rules.py ^
|
||||
--summary outputs/result.json ^
|
||||
--extracted outputs/read-flow-2026.extracted.json ^
|
||||
--context outputs/filter-context.json ^
|
||||
--output outputs/filter-decision.with-context.json
|
||||
```
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
## 当前状态
|
||||
|
||||
项目当前处于 `Content Extract MCP` 的 MVP 完成阶段。
|
||||
项目当前处于 `Content Extract MCP` 的 MVP 完成并进入规则过滤阶段。
|
||||
|
||||
已完成:
|
||||
|
||||
@@ -19,6 +19,10 @@
|
||||
- [x] 设计并迭代 LLM 摘要 prompt
|
||||
- [x] 验证 LLM 输出的 `result.json` 基本符合预期
|
||||
- [x] 已封装 LLM 摘要校验 workflow skill
|
||||
- [x] 接通 FreshRSS 上游并完成 `item` 映射
|
||||
- [x] 跑通 `FreshRSS -> item -> content extraction` 单条链路
|
||||
- [x] 定义规则过滤层 schema
|
||||
- [x] 实现第一版规则引擎与过滤 MCP tool
|
||||
|
||||
---
|
||||
|
||||
@@ -34,12 +38,14 @@
|
||||
|
||||
## P1 - 下一阶段推进
|
||||
|
||||
- [ ] 把上游 RSS 聚合结果映射成标准化 `item`
|
||||
- [ ] 补充 `item` 的实际样例文件
|
||||
- [ ] 定义规则过滤层的输入输出 schema
|
||||
- [x] 把上游 RSS 聚合结果映射成标准化 `item`
|
||||
- [x] 补充 `item` 的实际样例文件
|
||||
- [x] 跑通 `FreshRSS -> item -> content extraction` 单条链路
|
||||
- [x] 定义规则过滤层的输入输出 schema
|
||||
- [ ] 设计知识库入库格式
|
||||
- [ ] 设计 webhook / 推送格式
|
||||
- [ ] 给提取结果增加更多正文清洗策略,比如尾部噪音清理
|
||||
- [ ] 细化过滤规则并引入更多个性化上下文
|
||||
|
||||
---
|
||||
|
||||
@@ -56,18 +62,16 @@
|
||||
|
||||
## 当前建议的下一步
|
||||
|
||||
|
||||
优先做这两件事:
|
||||
|
||||
1. 把上游 RSS 聚合结果映射成标准化 `item`
|
||||
2. 补充一个真实 `item` 样例文件并开始设计规则过滤 schema
|
||||
1. 设计知识库入库格式
|
||||
2. 设计 webhook / 推送格式
|
||||
|
||||
---
|
||||
|
||||
## 收束上下文后建议先看
|
||||
|
||||
- docs/context-reset-brief.md
|
||||
- docs/filter-rule-engine-design.md
|
||||
- docs/content-extract-mcp-mvp-archive.md
|
||||
- TODO.md
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,102 @@
|
||||
[
|
||||
{
|
||||
"rule_id": "drop-low-content",
|
||||
"enabled": true,
|
||||
"stop_on_match": true,
|
||||
"conditions_all": [
|
||||
{ "field": "article.quality_flags.is_low_content", "op": "eq", "value": true }
|
||||
],
|
||||
"action": {
|
||||
"decision": "drop",
|
||||
"reason": "Extracted article quality is too low.",
|
||||
"labels": ["quality", "low-content"],
|
||||
"priority": 100
|
||||
}
|
||||
},
|
||||
{
|
||||
"rule_id": "review-truncated-content",
|
||||
"enabled": true,
|
||||
"stop_on_match": false,
|
||||
"conditions_all": [
|
||||
{ "field": "article.quality_flags.is_truncated", "op": "eq", "value": true }
|
||||
],
|
||||
"action": {
|
||||
"decision": "review",
|
||||
"reason": "Extracted article may be truncated.",
|
||||
"labels": ["quality", "truncated"],
|
||||
"priority": 95
|
||||
}
|
||||
},
|
||||
{
|
||||
"rule_id": "review-paywalled-content",
|
||||
"enabled": true,
|
||||
"stop_on_match": false,
|
||||
"conditions_all": [
|
||||
{ "field": "article.quality_flags.is_paywalled", "op": "eq", "value": true }
|
||||
],
|
||||
"action": {
|
||||
"decision": "review",
|
||||
"reason": "Article may be behind a paywall.",
|
||||
"labels": ["quality", "paywall"],
|
||||
"priority": 95
|
||||
}
|
||||
},
|
||||
{
|
||||
"rule_id": "drop-ephemeral-news",
|
||||
"enabled": true,
|
||||
"stop_on_match": true,
|
||||
"conditions_all": [
|
||||
{ "field": "summary.category", "op": "eq", "value": "资讯" },
|
||||
{ "field": "summary.worth_keeping", "op": "eq", "value": false }
|
||||
],
|
||||
"action": {
|
||||
"decision": "drop",
|
||||
"reason": "News-like content marked as not worth keeping.",
|
||||
"labels": ["summary", "ephemeral-news"],
|
||||
"priority": 90
|
||||
}
|
||||
},
|
||||
{
|
||||
"rule_id": "keep-worth-keeping-method",
|
||||
"enabled": true,
|
||||
"stop_on_match": false,
|
||||
"conditions_all": [
|
||||
{ "field": "summary.worth_keeping", "op": "eq", "value": true },
|
||||
{ "field": "summary.category", "op": "in", "value": ["方法论", "工具实践"] }
|
||||
],
|
||||
"action": {
|
||||
"decision": "keep",
|
||||
"reason": "Structured summary marked the content as worth keeping in a durable category.",
|
||||
"labels": ["summary", "durable"],
|
||||
"priority": 80
|
||||
}
|
||||
},
|
||||
{
|
||||
"rule_id": "keep-interest-topic",
|
||||
"enabled": true,
|
||||
"stop_on_match": false,
|
||||
"conditions_all": [
|
||||
{ "field": "summary.topics", "op": "overlap", "value": { "from_field": "context.interest_topics" } }
|
||||
],
|
||||
"action": {
|
||||
"decision": "keep",
|
||||
"reason": "Topics overlap with current interest profile.",
|
||||
"labels": ["interest", "topic-match"],
|
||||
"priority": 75
|
||||
}
|
||||
},
|
||||
{
|
||||
"rule_id": "review-worth-keeping-other",
|
||||
"enabled": true,
|
||||
"stop_on_match": false,
|
||||
"conditions_all": [
|
||||
{ "field": "summary.worth_keeping", "op": "eq", "value": true }
|
||||
],
|
||||
"action": {
|
||||
"decision": "review",
|
||||
"reason": "Worth-keeping signal is positive but no stronger keep rule matched.",
|
||||
"labels": ["summary", "needs-review"],
|
||||
"priority": 60
|
||||
}
|
||||
}
|
||||
]
|
||||
@@ -0,0 +1,59 @@
|
||||
# 文档索引
|
||||
|
||||
## 当前推荐阅读顺序
|
||||
|
||||
1. `context-reset-brief.md`
|
||||
- 当前真实进度与下一步入口
|
||||
2. `summary-mcp-service-design.md`
|
||||
- 当前 MCP 服务的职责、接口和边界
|
||||
3. `filter-rule-engine-design.md`
|
||||
- 过滤层的输入输出、规则结构与当前实现
|
||||
4. `source-schema-design.md`
|
||||
- `source -> item -> document` 的对象设计
|
||||
5. `reading-pipeline-design-notes.md`
|
||||
- 更上层的阅读流方案与阶段划分
|
||||
6. `summary-loop-explained.md`
|
||||
- 当前 LLM 摘要校验闭环的解释
|
||||
|
||||
## 当前文档分层
|
||||
|
||||
### 1. 当前状态与导航
|
||||
|
||||
- `README.md`
|
||||
- 仓库入口与脚本运行方式
|
||||
- `TODO.md`
|
||||
- 当前优先级、已完成项、下一阶段任务
|
||||
- `docs/context-reset-brief.md`
|
||||
- 当前阶段状态的最短摘要
|
||||
- `docs/README.md`
|
||||
- 文档索引与阅读顺序
|
||||
|
||||
### 2. 当前实现设计
|
||||
|
||||
- `docs/summary-mcp-service-design.md`
|
||||
- 当前内容提取 MCP 的真实设计
|
||||
- `docs/summary-core-interface-design.md`
|
||||
- 摘要/提取内核的接口抽象
|
||||
- `docs/source-schema-design.md`
|
||||
- `source`、`item`、`document` 三层 schema
|
||||
- `docs/summary-loop-explained.md`
|
||||
- 提取 JSON -> LLM 摘要 JSON -> 校验 的闭环说明
|
||||
- `docs/filter-rule-engine-design.md`
|
||||
- 第一版规则过滤引擎设计与落地位置
|
||||
|
||||
### 3. 上下游方案设计
|
||||
|
||||
- `docs/reading-pipeline-design-notes.md`
|
||||
- 整体阅读流、规则、sink、push 的方案笔记
|
||||
|
||||
### 4. 历史归档
|
||||
|
||||
- `docs/content-extract-mcp-mvp-archive.md`
|
||||
- MVP 阶段归档,部分状态已被后续进展覆盖
|
||||
|
||||
## 当前文档维护原则
|
||||
|
||||
- `context-reset-brief.md` 记录当前最新状态
|
||||
- `TODO.md` 记录任务优先级与下一步
|
||||
- `content-extract-mcp-mvp-archive.md` 只当历史快照,不再作为最新事实来源
|
||||
- 新增阶段性进展,优先更新 `README.md`、`TODO.md`、`context-reset-brief.md`
|
||||
@@ -7,9 +7,10 @@
|
||||
- 只负责内容提取
|
||||
- 不负责摘要、分类、价值判断
|
||||
- 已完成 Python MCP 骨架
|
||||
- 已实现两个 tool:
|
||||
- 已实现三个 tool:
|
||||
- `extract_url_content`
|
||||
- `extract_item_content`
|
||||
- `filter_summary_result`
|
||||
- 已完成真实 URL 提取验证
|
||||
- 已完成 LLM 摘要 prompt
|
||||
- 已完成 LLM 摘要结果 schema 校验器
|
||||
@@ -17,6 +18,12 @@
|
||||
- 已完成 LLM 摘要校验 skill 封装
|
||||
- 已清理旧的启发式 `summarizer.py`
|
||||
- 已将旧的 MCP 设计文档更新为当前“Content Extract MCP”语义
|
||||
- 已完成 FreshRSS `greader` API 接入
|
||||
- 已完成 FreshRSS entry -> `item` 映射
|
||||
- 已产出真实 `item` 样例文件
|
||||
- 已跑通 `FreshRSS -> item -> content extraction` 单条链路
|
||||
- 已定义过滤层输入输出 schema
|
||||
- 已完成第一版规则引擎、本地脚本和 MCP tool
|
||||
|
||||
## 当前关键文件
|
||||
|
||||
@@ -28,12 +35,34 @@
|
||||
- `src/summary_mcp/models/llm_result.py`
|
||||
- LLM 校验器:
|
||||
- `src/summary_mcp/validators/llm_result.py`
|
||||
- 过滤模型:
|
||||
- `src/summary_mcp/models/filtering.py`
|
||||
- 过滤引擎:
|
||||
- `src/summary_mcp/filters/engine.py`
|
||||
- 默认过滤规则:
|
||||
- `configs/filter_rules.json`
|
||||
- 校验 CLI:
|
||||
- `src/summary_mcp/validate_llm_result.py`
|
||||
- 最小闭环脚本:
|
||||
- `scripts/run_summary_loop.py`
|
||||
- FreshRSS 拉取脚本:
|
||||
- `scripts/pull_freshrss_items.py`
|
||||
- FreshRSS 提取脚本:
|
||||
- `scripts/run_freshrss_extract.py`
|
||||
- 过滤脚本:
|
||||
- `scripts/run_filter_rules.py`
|
||||
- 当前提示词:
|
||||
- `outputs/llm-summary-prompt.txt`
|
||||
- FreshRSS 原始响应样例:
|
||||
- `outputs/freshrss.raw.json`
|
||||
- FreshRSS item 样例:
|
||||
- `outputs/freshrss.items.json`
|
||||
- FreshRSS 提取结果:
|
||||
- `outputs/freshrss.extracted.json`
|
||||
- 过滤结果样例:
|
||||
- `outputs/filter-decision.json`
|
||||
- 文档索引:
|
||||
- `docs/README.md`
|
||||
- MVP 归档:
|
||||
- `docs/content-extract-mcp-mvp-archive.md`
|
||||
- 当前 TODO:
|
||||
@@ -45,21 +74,26 @@
|
||||
- LLM 可根据提取结果生成摘要 JSON
|
||||
- validator 可校验摘要 JSON
|
||||
- 最小闭环脚本可直接调用 LLM 接口并产出通过校验的结果
|
||||
- FreshRSS API 可拉取真实 entry
|
||||
- 真实 entry 可映射为标准化 `item`
|
||||
- 标准化 `item` 可继续进入内容提取流程
|
||||
- 规则引擎可对结构化摘要结果输出 `keep / drop / review` 决策
|
||||
- MCP tool 可承接“由上层 LLM/Agent 调用过滤”的模式
|
||||
|
||||
## 当前未开始的下一阶段
|
||||
|
||||
- 让 FreshRSS 作为主聚合池
|
||||
- 设计 FreshRSS entry -> `item` 的映射
|
||||
- 准备真实 `item` 样例
|
||||
- 开始定义规则过滤层 schema
|
||||
- 设计知识库入库格式
|
||||
- 设计 webhook / 推送格式
|
||||
- 把 FreshRSS 拉取与提取流程进一步批量化/调度化
|
||||
- 迭代更细的过滤规则与个性化上下文
|
||||
|
||||
## 收束后建议从这里继续
|
||||
|
||||
优先从这两个问题继续:
|
||||
|
||||
1. FreshRSS 的 entry 字段如何映射成 `item`
|
||||
2. 先做一个真实 `item` 样例文件,再讨论规则过滤
|
||||
1. 先设计知识库 sink 的输入输出格式
|
||||
2. 再决定先接 webhook / 推送还是继续细化过滤规则
|
||||
|
||||
## 一句话结论
|
||||
|
||||
当前 MVP 已完成,下一阶段不再是继续打磨 MCP,而是开始接入 FreshRSS 上游并建立 `item` 标准化入口。
|
||||
当前 MVP 已完成,FreshRSS 上游和第一版规则过滤层都已接通,下一阶段应转向下游 sink 与推送设计。
|
||||
|
||||
@@ -0,0 +1,225 @@
|
||||
# 规则过滤引擎设计
|
||||
|
||||
## 1. 目标
|
||||
|
||||
当前过滤层的定位是:
|
||||
|
||||
- 不让 LLM 直接做最终过滤决策
|
||||
- 让 LLM 只产出结构化信号
|
||||
- 由规则引擎输出最终 `keep / drop / review` 决策
|
||||
|
||||
当前链路是:
|
||||
|
||||
`item -> content extraction -> llm summary -> filter rule engine`
|
||||
|
||||
## 2. 输入输出
|
||||
|
||||
### 2.1 FilterInput
|
||||
|
||||
过滤层统一读取四类输入:
|
||||
|
||||
- `item`
|
||||
- `article`
|
||||
- `summary`
|
||||
- `context`
|
||||
|
||||
其中:
|
||||
|
||||
- `item` 表示上游标准化候选条目
|
||||
- `article` 表示正文提取结果
|
||||
- `summary` 表示结构化 LLM 摘要结果
|
||||
- `context` 表示额外的用户偏好或运行时上下文
|
||||
|
||||
### 2.2 FilterDecisionResult
|
||||
|
||||
过滤结果统一输出:
|
||||
|
||||
- `decision`
|
||||
- `keep`
|
||||
- `drop`
|
||||
- `review`
|
||||
- `matched_rules`
|
||||
- `reasons`
|
||||
- `labels`
|
||||
- `priority`
|
||||
- `matches`
|
||||
|
||||
这样后续的知识库 sink、推送层或人工审核都可以稳定消费。
|
||||
|
||||
## 3. 规则结构
|
||||
|
||||
当前规则文件位置:
|
||||
|
||||
- `configs/filter_rules.json`
|
||||
|
||||
单条规则结构为:
|
||||
|
||||
```json
|
||||
{
|
||||
"rule_id": "keep-worth-keeping-method",
|
||||
"enabled": true,
|
||||
"stop_on_match": false,
|
||||
"conditions_all": [
|
||||
{ "field": "summary.worth_keeping", "op": "eq", "value": true },
|
||||
{ "field": "summary.category", "op": "in", "value": ["方法论", "工具实践"] }
|
||||
],
|
||||
"action": {
|
||||
"decision": "keep",
|
||||
"reason": "Structured summary marked the content as worth keeping in a durable category.",
|
||||
"labels": ["summary", "durable"],
|
||||
"priority": 80
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
## 4. 当前支持的操作符
|
||||
|
||||
- `eq`
|
||||
- `ne`
|
||||
- `in`
|
||||
- `not_in`
|
||||
- `contains`
|
||||
- `overlap`
|
||||
- `gte`
|
||||
- `lte`
|
||||
- `exists`
|
||||
|
||||
当前条件组合方式:
|
||||
|
||||
- `conditions_all`
|
||||
- `conditions_any`
|
||||
|
||||
## 5. 动态上下文字段
|
||||
|
||||
规则支持从其他字段动态取值,例如:
|
||||
|
||||
```json
|
||||
{ "field": "summary.topics", "op": "overlap", "value": { "from_field": "context.interest_topics" } }
|
||||
```
|
||||
|
||||
这允许上层 Agent 在调用 `filter_summary_result` 时,把当前关注主题动态注入,而不必把偏好硬编码在规则文件里。
|
||||
|
||||
## 6. 当前默认规则思路
|
||||
|
||||
当前默认规则分为三类:
|
||||
|
||||
- 质量拦截
|
||||
- 低质量正文直接 `drop`
|
||||
- 疑似截断或付费墙进入 `review`
|
||||
- 摘要价值判断
|
||||
- `资讯 + worth_keeping=false` 直接 `drop`
|
||||
- `方法论/工具实践 + worth_keeping=true` 直接 `keep`
|
||||
- 个性化补充信号
|
||||
- `summary.topics` 与 `context.interest_topics` 重叠时提升为 `keep`
|
||||
- 其他 `worth_keeping=true` 的结果默认进入 `review`
|
||||
|
||||
## 7. 规则执行流程
|
||||
|
||||
当前过滤流程按以下顺序执行:
|
||||
|
||||
1. 上层先产出结构化 `summary`
|
||||
2. 可选附带 `item`、`article` 与 `context`
|
||||
3. 规则引擎按 `priority` 从高到低遍历规则
|
||||
4. 每条规则根据 `conditions_all` / `conditions_any` 判断是否命中
|
||||
5. 命中的规则被收集为 `matches`
|
||||
6. 最终根据命中结果收敛为 `keep / drop / review`
|
||||
|
||||
当前决策收敛规则是:
|
||||
|
||||
- 只要命中任意 `drop`,最终结果就是 `drop`
|
||||
- 否则只要命中任意 `keep`,最终结果就是 `keep`
|
||||
- 否则如果命中任意 `review`,最终结果就是 `review`
|
||||
- 如果没有任何规则命中,默认回落到 `review`
|
||||
|
||||
这样设计的原因是:
|
||||
|
||||
- `drop` 应该拥有最高约束力
|
||||
- `keep` 只在没有硬性淘汰时生效
|
||||
- 默认不自动放行未知内容
|
||||
|
||||
## 8. 为什么让 LLM 调用规则 tool,而不是直接裁决
|
||||
|
||||
当前架构故意拆成两层:
|
||||
|
||||
- LLM 负责生成结构化信号
|
||||
- 规则引擎负责输出最终过滤决策
|
||||
|
||||
原因是:
|
||||
|
||||
- LLM 适合做语义理解、归类、摘要和价值信号提取
|
||||
- 规则引擎适合做稳定、可复现、可审计的最终判断
|
||||
|
||||
因此推荐的调用方式是:
|
||||
|
||||
1. LLM 先调用提取 tool
|
||||
2. LLM 或本地脚本产出 `summary_result`
|
||||
3. LLM 再调用 `filter_summary_result`
|
||||
4. 后续根据过滤结果决定是否入库、推送或人工审核
|
||||
|
||||
这意味着:
|
||||
|
||||
- LLM 是编排者
|
||||
- rule engine 是裁决器
|
||||
|
||||
而不是让 LLM 在过滤阶段再次自由发挥。
|
||||
|
||||
## 9. 调用示例
|
||||
|
||||
### 9.1 本地脚本
|
||||
|
||||
```bash
|
||||
python scripts/run_filter_rules.py ^
|
||||
--summary outputs/result.json ^
|
||||
--extracted outputs/read-flow-2026.extracted.json ^
|
||||
--output outputs/filter-decision.json
|
||||
```
|
||||
|
||||
### 9.2 带上下文的调用
|
||||
|
||||
```json
|
||||
{
|
||||
"interest_topics": ["个人信息管理", "阅读工作流"]
|
||||
}
|
||||
```
|
||||
|
||||
将这份上下文作为 `--context` 传入后,规则就可以根据当前关注主题做加权。
|
||||
|
||||
### 9.3 MCP tool
|
||||
|
||||
`filter_summary_result` 接收:
|
||||
|
||||
- `summary_result`
|
||||
- 可选 `extracted_article`
|
||||
- 可选 `item`
|
||||
- 可选 `context`
|
||||
|
||||
返回:
|
||||
|
||||
- `decision`
|
||||
- `matched_rules`
|
||||
- `reasons`
|
||||
- `labels`
|
||||
- `priority`
|
||||
- `matches`
|
||||
|
||||
## 10. 当前落地位置
|
||||
|
||||
- 过滤模型:
|
||||
- `src/summary_mcp/models/filtering.py`
|
||||
- 规则引擎:
|
||||
- `src/summary_mcp/filters/engine.py`
|
||||
- 默认规则:
|
||||
- `configs/filter_rules.json`
|
||||
- 本地运行脚本:
|
||||
- `scripts/run_filter_rules.py`
|
||||
- MCP tool:
|
||||
- `filter_summary_result`
|
||||
|
||||
## 11. 当前阶段的设计结论
|
||||
|
||||
第一版过滤层采用“规则主导,LLM 提供信号”的架构:
|
||||
|
||||
- LLM 不直接做最终裁决
|
||||
- 上层 Agent 可以决定是否调用过滤 tool
|
||||
- 真正的过滤决策由规则引擎输出
|
||||
- 灰区内容后续再考虑是否引入第二层 LLM 辅助判断
|
||||
@@ -0,0 +1,39 @@
|
||||
{
|
||||
"decision": "keep",
|
||||
"matched_rules": [
|
||||
"keep-worth-keeping-method",
|
||||
"review-worth-keeping-other"
|
||||
],
|
||||
"reasons": [
|
||||
"Structured summary marked the content as worth keeping in a durable category.",
|
||||
"Worth-keeping signal is positive but no stronger keep rule matched."
|
||||
],
|
||||
"labels": [
|
||||
"durable",
|
||||
"needs-review",
|
||||
"summary"
|
||||
],
|
||||
"priority": 80,
|
||||
"matches": [
|
||||
{
|
||||
"rule_id": "keep-worth-keeping-method",
|
||||
"decision": "keep",
|
||||
"reason": "Structured summary marked the content as worth keeping in a durable category.",
|
||||
"labels": [
|
||||
"summary",
|
||||
"durable"
|
||||
],
|
||||
"priority": 80
|
||||
},
|
||||
{
|
||||
"rule_id": "review-worth-keeping-other",
|
||||
"decision": "review",
|
||||
"reason": "Worth-keeping signal is positive but no stronger keep rule matched.",
|
||||
"labels": [
|
||||
"summary",
|
||||
"needs-review"
|
||||
],
|
||||
"priority": 60
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,53 @@
|
||||
{
|
||||
"decision": "keep",
|
||||
"matched_rules": [
|
||||
"keep-worth-keeping-method",
|
||||
"keep-interest-topic",
|
||||
"review-worth-keeping-other"
|
||||
],
|
||||
"reasons": [
|
||||
"Structured summary marked the content as worth keeping in a durable category.",
|
||||
"Topics overlap with current interest profile.",
|
||||
"Worth-keeping signal is positive but no stronger keep rule matched."
|
||||
],
|
||||
"labels": [
|
||||
"durable",
|
||||
"interest",
|
||||
"needs-review",
|
||||
"summary",
|
||||
"topic-match"
|
||||
],
|
||||
"priority": 80,
|
||||
"matches": [
|
||||
{
|
||||
"rule_id": "keep-worth-keeping-method",
|
||||
"decision": "keep",
|
||||
"reason": "Structured summary marked the content as worth keeping in a durable category.",
|
||||
"labels": [
|
||||
"summary",
|
||||
"durable"
|
||||
],
|
||||
"priority": 80
|
||||
},
|
||||
{
|
||||
"rule_id": "keep-interest-topic",
|
||||
"decision": "keep",
|
||||
"reason": "Topics overlap with current interest profile.",
|
||||
"labels": [
|
||||
"interest",
|
||||
"topic-match"
|
||||
],
|
||||
"priority": 75
|
||||
},
|
||||
{
|
||||
"rule_id": "review-worth-keeping-other",
|
||||
"decision": "review",
|
||||
"reason": "Worth-keeping signal is positive but no stronger keep rule matched.",
|
||||
"labels": [
|
||||
"summary",
|
||||
"needs-review"
|
||||
],
|
||||
"priority": 60
|
||||
}
|
||||
]
|
||||
}
|
||||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -0,0 +1,92 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||
SRC_ROOT = REPO_ROOT / "src"
|
||||
|
||||
if str(SRC_ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(SRC_ROOT))
|
||||
|
||||
from summary_mcp.integrations.freshrss import FreshRSSClient, map_entry_to_item
|
||||
|
||||
|
||||
def _save_json(path: Path, payload: dict[str, Any] | list[Any]) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||
|
||||
|
||||
def _load_required(name: str, value: str | None) -> str:
|
||||
resolved = value or os.environ.get(name)
|
||||
if not resolved:
|
||||
cli_name = name.lower().replace("_", "-")
|
||||
raise RuntimeError(f"Missing required value: pass --{cli_name} or set {name}.")
|
||||
return resolved
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description="Pull FreshRSS entries and map them into normalized item objects.")
|
||||
parser.add_argument("--api-base-url", type=str, default=None, help="FreshRSS greader API base URL")
|
||||
parser.add_argument("--username", type=str, default=None, help="FreshRSS API username")
|
||||
parser.add_argument("--api-password", type=str, default=None, help="FreshRSS API password")
|
||||
parser.add_argument(
|
||||
"--stream-id",
|
||||
type=str,
|
||||
default="user/-/state/com.google/reading-list",
|
||||
help="Google Reader API stream id",
|
||||
)
|
||||
parser.add_argument("--limit", type=int, default=10, help="Maximum number of entries to request")
|
||||
parser.add_argument("--continuation", type=str, default=None, help="Continuation token for paging")
|
||||
parser.add_argument(
|
||||
"--raw-output",
|
||||
type=Path,
|
||||
default=REPO_ROOT / "outputs" / "freshrss.raw.json",
|
||||
help="Where to save the raw FreshRSS response",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--items-output",
|
||||
type=Path,
|
||||
default=REPO_ROOT / "outputs" / "freshrss.items.json",
|
||||
help="Where to save the mapped item list",
|
||||
)
|
||||
parser.add_argument("--timeout", type=float, default=20.0, help="Request timeout in seconds")
|
||||
args = parser.parse_args()
|
||||
|
||||
api_base_url = _load_required("FRESHRSS_API_BASE_URL", args.api_base_url)
|
||||
username = _load_required("FRESHRSS_USERNAME", args.username)
|
||||
api_password = _load_required("FRESHRSS_API_PASSWORD", args.api_password)
|
||||
|
||||
client = FreshRSSClient(
|
||||
api_base_url=api_base_url,
|
||||
username=username,
|
||||
api_password=api_password,
|
||||
timeout_seconds=args.timeout,
|
||||
)
|
||||
auth_token = client.client_login()
|
||||
payload = client.fetch_stream_contents(
|
||||
auth_token=auth_token,
|
||||
stream_id=args.stream_id,
|
||||
limit=args.limit,
|
||||
continuation=args.continuation,
|
||||
)
|
||||
|
||||
entries = payload.get("items")
|
||||
if not isinstance(entries, list):
|
||||
raise RuntimeError("FreshRSS stream response does not contain an items array.")
|
||||
|
||||
mapped_items = [map_entry_to_item(entry).model_dump(mode="json") for entry in entries]
|
||||
|
||||
_save_json(args.raw_output, payload)
|
||||
_save_json(args.items_output, mapped_items)
|
||||
|
||||
print(f"Saved {len(mapped_items)} mapped items to {args.items_output}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,77 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||
SRC_ROOT = REPO_ROOT / "src"
|
||||
|
||||
if str(SRC_ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(SRC_ROOT))
|
||||
|
||||
from summary_mcp.filters.engine import load_filter_rules, evaluate_filter_rules
|
||||
from summary_mcp.models.document import ExtractedArticle
|
||||
from summary_mcp.models.filtering import FilterContext, FilterInput
|
||||
from summary_mcp.models.item import Item
|
||||
from summary_mcp.models.llm_result import LlmSummaryResult
|
||||
|
||||
|
||||
def _load_json(path: Path) -> dict[str, Any]:
|
||||
return json.loads(path.read_text(encoding="utf-8-sig"))
|
||||
|
||||
|
||||
def _save_json(path: Path, payload: dict[str, Any]) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description="Run deterministic filter rules against a structured summary result.")
|
||||
parser.add_argument("--summary", type=Path, required=True, help="Structured LLM summary JSON file")
|
||||
parser.add_argument("--extracted", type=Path, default=None, help="Extracted article JSON file")
|
||||
parser.add_argument("--item", type=Path, default=None, help="Normalized item JSON file")
|
||||
parser.add_argument("--context", type=Path, default=None, help="Optional filter context JSON file")
|
||||
parser.add_argument(
|
||||
"--rules",
|
||||
type=Path,
|
||||
default=REPO_ROOT / "configs" / "filter_rules.json",
|
||||
help="Filter rule config JSON file",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--output",
|
||||
type=Path,
|
||||
default=REPO_ROOT / "outputs" / "filter-decision.json",
|
||||
help="Where to save the filter decision",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
summary = LlmSummaryResult.model_validate(_load_json(args.summary))
|
||||
|
||||
article = None
|
||||
if args.extracted is not None:
|
||||
extracted_payload = _load_json(args.extracted)
|
||||
article_payload = extracted_payload.get("article", extracted_payload)
|
||||
article = ExtractedArticle.model_validate(article_payload)
|
||||
|
||||
item = None
|
||||
if args.item is not None:
|
||||
item = Item.model_validate(_load_json(args.item))
|
||||
|
||||
context = FilterContext.model_validate(_load_json(args.context)) if args.context else FilterContext()
|
||||
rules = load_filter_rules(args.rules)
|
||||
decision = evaluate_filter_rules(
|
||||
FilterInput(item=item, article=article, summary=summary, context=context),
|
||||
rules,
|
||||
)
|
||||
|
||||
_save_json(args.output, decision.model_dump(mode="json"))
|
||||
print(f"Saved filter decision to {args.output}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
@@ -0,0 +1,125 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||
SRC_ROOT = REPO_ROOT / "src"
|
||||
|
||||
if str(SRC_ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(SRC_ROOT))
|
||||
|
||||
from summary_mcp.core.pipeline import extract_content
|
||||
from summary_mcp.integrations.freshrss import FreshRSSClient, map_entry_to_item
|
||||
from summary_mcp.models.summary_io import ExtractionInput
|
||||
|
||||
|
||||
def _save_json(path: Path, payload: dict[str, Any] | list[Any]) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||
|
||||
|
||||
def _load_required(name: str, value: str | None) -> str:
|
||||
resolved = value or os.environ.get(name)
|
||||
if not resolved:
|
||||
cli_name = name.lower().replace("_", "-")
|
||||
raise RuntimeError(f"Missing required value: pass --{cli_name} or set {name}.")
|
||||
return resolved
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description="Pull FreshRSS entries and run content extraction for each mapped item.")
|
||||
parser.add_argument("--api-base-url", type=str, default=None, help="FreshRSS greader API base URL")
|
||||
parser.add_argument("--username", type=str, default=None, help="FreshRSS API username")
|
||||
parser.add_argument("--api-password", type=str, default=None, help="FreshRSS API password")
|
||||
parser.add_argument(
|
||||
"--stream-id",
|
||||
type=str,
|
||||
default="user/-/state/com.google/reading-list",
|
||||
help="Google Reader API stream id",
|
||||
)
|
||||
parser.add_argument("--limit", type=int, default=5, help="Maximum number of entries to request")
|
||||
parser.add_argument("--continuation", type=str, default=None, help="Continuation token for paging")
|
||||
parser.add_argument(
|
||||
"--raw-output",
|
||||
type=Path,
|
||||
default=REPO_ROOT / "outputs" / "freshrss.raw.json",
|
||||
help="Where to save the raw FreshRSS response",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--items-output",
|
||||
type=Path,
|
||||
default=REPO_ROOT / "outputs" / "freshrss.items.json",
|
||||
help="Where to save the mapped item list",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--extracted-output",
|
||||
type=Path,
|
||||
default=REPO_ROOT / "outputs" / "freshrss.extracted.json",
|
||||
help="Where to save the extraction results",
|
||||
)
|
||||
parser.add_argument("--timeout", type=float, default=20.0, help="Request timeout in seconds")
|
||||
args = parser.parse_args()
|
||||
|
||||
api_base_url = _load_required("FRESHRSS_API_BASE_URL", args.api_base_url)
|
||||
username = _load_required("FRESHRSS_USERNAME", args.username)
|
||||
api_password = _load_required("FRESHRSS_API_PASSWORD", args.api_password)
|
||||
|
||||
client = FreshRSSClient(
|
||||
api_base_url=api_base_url,
|
||||
username=username,
|
||||
api_password=api_password,
|
||||
timeout_seconds=args.timeout,
|
||||
)
|
||||
auth_token = client.client_login()
|
||||
payload = client.fetch_stream_contents(
|
||||
auth_token=auth_token,
|
||||
stream_id=args.stream_id,
|
||||
limit=args.limit,
|
||||
continuation=args.continuation,
|
||||
)
|
||||
|
||||
entries = payload.get("items")
|
||||
if not isinstance(entries, list):
|
||||
raise RuntimeError("FreshRSS stream response does not contain an items array.")
|
||||
|
||||
mapped_items = [map_entry_to_item(entry) for entry in entries]
|
||||
extraction_results: list[dict[str, Any]] = []
|
||||
success_count = 0
|
||||
|
||||
for item in mapped_items:
|
||||
extraction = extract_content(ExtractionInput(item=item))
|
||||
extraction_results.append(
|
||||
{
|
||||
"item": item.model_dump(mode="json"),
|
||||
"extraction": extraction.model_dump(mode="json"),
|
||||
}
|
||||
)
|
||||
if extraction.success:
|
||||
success_count += 1
|
||||
|
||||
_save_json(args.raw_output, payload)
|
||||
_save_json(args.items_output, [item.model_dump(mode="json") for item in mapped_items])
|
||||
_save_json(
|
||||
args.extracted_output,
|
||||
{
|
||||
"stream_id": args.stream_id,
|
||||
"requested_limit": args.limit,
|
||||
"entry_count": len(entries),
|
||||
"extracted_success_count": success_count,
|
||||
"results": extraction_results,
|
||||
},
|
||||
)
|
||||
|
||||
print(
|
||||
f"Saved {len(mapped_items)} mapped items and {success_count} successful extractions to {args.extracted_output}"
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,3 @@
|
||||
from .engine import DEFAULT_RULES_PATH, evaluate_filter_rules, load_filter_rules
|
||||
|
||||
__all__ = ["DEFAULT_RULES_PATH", "evaluate_filter_rules", "load_filter_rules"]
|
||||
@@ -0,0 +1,136 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from summary_mcp.models.filtering import (
|
||||
FieldCondition,
|
||||
FilterDecisionResult,
|
||||
FilterInput,
|
||||
FilterRule,
|
||||
MatchedRule,
|
||||
)
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[3]
|
||||
DEFAULT_RULES_PATH = REPO_ROOT / "configs" / "filter_rules.json"
|
||||
|
||||
|
||||
def load_filter_rules(path: Path | None = None) -> list[FilterRule]:
|
||||
rules_path = path or DEFAULT_RULES_PATH
|
||||
payload = json.loads(rules_path.read_text(encoding="utf-8-sig"))
|
||||
if not isinstance(payload, list):
|
||||
raise RuntimeError("Filter rules file must contain a JSON array.")
|
||||
return [FilterRule.model_validate(item) for item in payload]
|
||||
|
||||
|
||||
def _normalize_value(value: Any) -> Any:
|
||||
if hasattr(value, "model_dump"):
|
||||
return value.model_dump(mode="json")
|
||||
return value
|
||||
|
||||
|
||||
def _resolve_field(filter_input: FilterInput, field_path: str) -> Any:
|
||||
current: Any = filter_input
|
||||
for part in field_path.split("."):
|
||||
current = _normalize_value(current)
|
||||
if isinstance(current, dict):
|
||||
if part not in current:
|
||||
return None
|
||||
current = current[part]
|
||||
continue
|
||||
return None
|
||||
return _normalize_value(current)
|
||||
|
||||
|
||||
def _expected_value(filter_input: FilterInput, value: Any) -> Any:
|
||||
if isinstance(value, dict) and "from_field" in value:
|
||||
field_name = value.get("from_field")
|
||||
if isinstance(field_name, str):
|
||||
return _resolve_field(filter_input, field_name)
|
||||
return value
|
||||
|
||||
|
||||
def _match_condition(filter_input: FilterInput, condition: FieldCondition) -> bool:
|
||||
current = _resolve_field(filter_input, condition.field)
|
||||
expected = _expected_value(filter_input, condition.value)
|
||||
|
||||
if condition.op == "exists":
|
||||
return (current is not None) if expected is not False else (current is None)
|
||||
if condition.op == "eq":
|
||||
return current == expected
|
||||
if condition.op == "ne":
|
||||
return current != expected
|
||||
if condition.op == "in":
|
||||
return current in expected if isinstance(expected, list) else False
|
||||
if condition.op == "not_in":
|
||||
return current not in expected if isinstance(expected, list) else False
|
||||
if condition.op == "contains":
|
||||
if isinstance(current, list):
|
||||
return expected in current
|
||||
if isinstance(current, str) and isinstance(expected, str):
|
||||
return expected in current
|
||||
return False
|
||||
if condition.op == "overlap":
|
||||
if isinstance(current, list) and isinstance(expected, list):
|
||||
return bool(set(current) & set(expected))
|
||||
return False
|
||||
if condition.op == "gte":
|
||||
return current is not None and expected is not None and current >= expected
|
||||
if condition.op == "lte":
|
||||
return current is not None and expected is not None and current <= expected
|
||||
return False
|
||||
|
||||
|
||||
def _rule_matches(filter_input: FilterInput, rule: FilterRule) -> bool:
|
||||
if not rule.enabled:
|
||||
return False
|
||||
if rule.conditions_all and not all(_match_condition(filter_input, condition) for condition in rule.conditions_all):
|
||||
return False
|
||||
if rule.conditions_any and not any(_match_condition(filter_input, condition) for condition in rule.conditions_any):
|
||||
return False
|
||||
return bool(rule.conditions_all or rule.conditions_any)
|
||||
|
||||
|
||||
def evaluate_filter_rules(filter_input: FilterInput, rules: list[FilterRule]) -> FilterDecisionResult:
|
||||
matched: list[MatchedRule] = []
|
||||
|
||||
for rule in sorted(rules, key=lambda item: item.action.priority, reverse=True):
|
||||
if not _rule_matches(filter_input, rule):
|
||||
continue
|
||||
|
||||
matched_rule = MatchedRule(
|
||||
rule_id=rule.rule_id,
|
||||
decision=rule.action.decision,
|
||||
reason=rule.action.reason,
|
||||
labels=rule.action.labels,
|
||||
priority=rule.action.priority,
|
||||
)
|
||||
matched.append(matched_rule)
|
||||
if rule.stop_on_match:
|
||||
break
|
||||
|
||||
if any(rule.decision == "drop" for rule in matched):
|
||||
final_decision = "drop"
|
||||
elif any(rule.decision == "keep" for rule in matched):
|
||||
final_decision = "keep"
|
||||
elif any(rule.decision == "review" for rule in matched):
|
||||
final_decision = "review"
|
||||
else:
|
||||
final_decision = "review"
|
||||
|
||||
labels = sorted({label for rule in matched for label in rule.labels})
|
||||
reasons = [rule.reason for rule in matched]
|
||||
priorities = [rule.priority for rule in matched]
|
||||
|
||||
return FilterDecisionResult(
|
||||
decision=final_decision,
|
||||
matched_rules=[rule.rule_id for rule in matched],
|
||||
reasons=reasons if reasons else ["No rule matched; defaulted to review."],
|
||||
labels=labels,
|
||||
priority=max(priorities, default=0),
|
||||
matches=matched,
|
||||
)
|
||||
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
"""Integration helpers for upstream content sources."""
|
||||
@@ -0,0 +1,199 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
from datetime import UTC, datetime
|
||||
from typing import Any
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import httpx
|
||||
|
||||
from summary_mcp.models.item import Item
|
||||
|
||||
|
||||
def _trim_api_base_url(api_base_url: str) -> str:
|
||||
return api_base_url.rstrip("/")
|
||||
|
||||
|
||||
def _build_item_id(source_id: str, external_id: str | None, url: str, published_at: datetime | None) -> str:
|
||||
seed = "|".join(
|
||||
[
|
||||
source_id,
|
||||
external_id or "",
|
||||
url,
|
||||
published_at.isoformat() if published_at else "",
|
||||
]
|
||||
)
|
||||
return f"sha256:{hashlib.sha256(seed.encode('utf-8')).hexdigest()}"
|
||||
|
||||
|
||||
def _build_source_id(entry: dict[str, Any], url: str) -> str:
|
||||
origin = entry.get("origin") or {}
|
||||
stream_id = origin.get("streamId")
|
||||
if isinstance(stream_id, str) and stream_id.strip():
|
||||
digest = hashlib.sha256(stream_id.encode("utf-8")).hexdigest()[:16]
|
||||
return f"freshrss:{digest}"
|
||||
|
||||
host = urlparse(url).netloc or "unknown-source"
|
||||
return f"freshrss:{host}"
|
||||
|
||||
|
||||
def _pick_entry_url(entry: dict[str, Any]) -> str:
|
||||
candidates = [
|
||||
entry.get("canonical"),
|
||||
entry.get("alternate"),
|
||||
]
|
||||
|
||||
for candidate_list in candidates:
|
||||
if not isinstance(candidate_list, list):
|
||||
continue
|
||||
for candidate in candidate_list:
|
||||
href = (candidate or {}).get("href")
|
||||
if isinstance(href, str) and href.strip():
|
||||
return href.strip()
|
||||
|
||||
entry_id = entry.get("id")
|
||||
if isinstance(entry_id, str) and entry_id.startswith("tag:"):
|
||||
return entry_id
|
||||
|
||||
raise ValueError("FreshRSS entry does not contain a usable URL.")
|
||||
|
||||
|
||||
def _pick_content_block(entry: dict[str, Any], key: str) -> str | None:
|
||||
block = entry.get(key)
|
||||
if isinstance(block, dict):
|
||||
content = block.get("content")
|
||||
if isinstance(content, str) and content.strip():
|
||||
return content
|
||||
return None
|
||||
|
||||
|
||||
def _pick_categories(entry: dict[str, Any]) -> list[str]:
|
||||
categories = entry.get("categories")
|
||||
if not isinstance(categories, list):
|
||||
return []
|
||||
|
||||
values: list[str] = []
|
||||
for category in categories:
|
||||
if not isinstance(category, str):
|
||||
continue
|
||||
if category.startswith("user/-/label/"):
|
||||
values.append(category.removeprefix("user/-/label/"))
|
||||
elif category.startswith("user/-/state/com.google/"):
|
||||
continue
|
||||
else:
|
||||
values.append(category)
|
||||
return values
|
||||
|
||||
|
||||
def _parse_datetime(timestamp: Any) -> datetime | None:
|
||||
if timestamp is None:
|
||||
return None
|
||||
if isinstance(timestamp, (int, float)):
|
||||
if timestamp > 10_000_000_000:
|
||||
return datetime.fromtimestamp(timestamp / 1000, tz=UTC)
|
||||
return datetime.fromtimestamp(timestamp, tz=UTC)
|
||||
if isinstance(timestamp, str) and timestamp.isdigit():
|
||||
value = int(timestamp)
|
||||
if value > 10_000_000_000:
|
||||
return datetime.fromtimestamp(value / 1000, tz=UTC)
|
||||
return datetime.fromtimestamp(value, tz=UTC)
|
||||
return None
|
||||
|
||||
|
||||
def map_entry_to_item(entry: dict[str, Any]) -> Item:
|
||||
url = _pick_entry_url(entry)
|
||||
summary_html = _pick_content_block(entry, "summary")
|
||||
content_html = _pick_content_block(entry, "content")
|
||||
published_at = _parse_datetime(entry.get("published"))
|
||||
source_id = _build_source_id(entry, url)
|
||||
external_id = entry.get("id") if isinstance(entry.get("id"), str) else None
|
||||
fetch_state = "fetched" if content_html and len(content_html.strip()) >= 500 else "pending"
|
||||
|
||||
metadata = {
|
||||
"upstream": "freshrss",
|
||||
"origin": entry.get("origin") or {},
|
||||
"categories": _pick_categories(entry),
|
||||
"crawled_at": _parse_datetime(entry.get("crawlTimeMsec")),
|
||||
"published_epoch": entry.get("published"),
|
||||
}
|
||||
|
||||
return Item(
|
||||
item_id=_build_item_id(source_id, external_id, url, published_at),
|
||||
source_id=source_id,
|
||||
external_id=external_id,
|
||||
title=entry.get("title"),
|
||||
url=url,
|
||||
author=entry.get("author"),
|
||||
published_at=published_at,
|
||||
discovered_at=datetime.now(tz=UTC),
|
||||
content_kind="article",
|
||||
language=None,
|
||||
raw_summary=summary_html,
|
||||
raw_content=content_html,
|
||||
metadata=metadata,
|
||||
fetch_state=fetch_state,
|
||||
)
|
||||
|
||||
|
||||
class FreshRSSClient:
|
||||
def __init__(
|
||||
self,
|
||||
api_base_url: str,
|
||||
username: str,
|
||||
api_password: str,
|
||||
timeout_seconds: float = 20.0,
|
||||
) -> None:
|
||||
self.api_base_url = _trim_api_base_url(api_base_url)
|
||||
self.username = username
|
||||
self.api_password = api_password
|
||||
self.timeout_seconds = timeout_seconds
|
||||
|
||||
def _build_url(self, path: str) -> str:
|
||||
return f"{self.api_base_url}/{path.lstrip('/')}"
|
||||
|
||||
def client_login(self) -> str:
|
||||
data = {
|
||||
"Email": self.username,
|
||||
"Passwd": self.api_password,
|
||||
}
|
||||
with httpx.Client(timeout=self.timeout_seconds) as client:
|
||||
response = client.post(self._build_url("accounts/ClientLogin"), data=data)
|
||||
response.raise_for_status()
|
||||
|
||||
auth_token: str | None = None
|
||||
for line in response.text.splitlines():
|
||||
if line.startswith("Auth="):
|
||||
auth_token = line.split("=", 1)[1].strip()
|
||||
break
|
||||
|
||||
if not auth_token:
|
||||
raise RuntimeError("FreshRSS ClientLogin succeeded but did not return an Auth token.")
|
||||
|
||||
return auth_token
|
||||
|
||||
def fetch_stream_contents(
|
||||
self,
|
||||
auth_token: str,
|
||||
stream_id: str = "user/-/state/com.google/reading-list",
|
||||
limit: int = 20,
|
||||
continuation: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
params: dict[str, Any] = {
|
||||
"output": "json",
|
||||
"n": limit,
|
||||
}
|
||||
if continuation:
|
||||
params["c"] = continuation
|
||||
|
||||
with httpx.Client(
|
||||
timeout=self.timeout_seconds,
|
||||
headers={"Authorization": f"GoogleLogin auth={auth_token}"},
|
||||
) as client:
|
||||
response = client.get(
|
||||
self._build_url(f"reader/api/0/stream/contents/{stream_id}"),
|
||||
params=params,
|
||||
)
|
||||
response.raise_for_status()
|
||||
return response.json()
|
||||
|
||||
|
||||
@@ -0,0 +1,67 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import datetime
|
||||
from typing import Any, Literal
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
from .document import ExtractedArticle
|
||||
from .item import Item
|
||||
from .llm_result import LlmSummaryResult
|
||||
|
||||
|
||||
FilterDecision = Literal["keep", "drop", "review"]
|
||||
ConditionOp = Literal["eq", "ne", "in", "not_in", "contains", "overlap", "gte", "lte", "exists"]
|
||||
|
||||
|
||||
class FilterContext(BaseModel):
|
||||
source_tags: list[str] = Field(default_factory=list)
|
||||
interest_topics: list[str] = Field(default_factory=list)
|
||||
interest_keywords: list[str] = Field(default_factory=list)
|
||||
now: datetime | None = None
|
||||
|
||||
|
||||
class FieldCondition(BaseModel):
|
||||
field: str
|
||||
op: ConditionOp
|
||||
value: Any = None
|
||||
|
||||
|
||||
class FilterAction(BaseModel):
|
||||
decision: FilterDecision
|
||||
reason: str
|
||||
labels: list[str] = Field(default_factory=list)
|
||||
priority: int = 50
|
||||
|
||||
|
||||
class FilterRule(BaseModel):
|
||||
rule_id: str
|
||||
enabled: bool = True
|
||||
stop_on_match: bool = False
|
||||
conditions_all: list[FieldCondition] = Field(default_factory=list)
|
||||
conditions_any: list[FieldCondition] = Field(default_factory=list)
|
||||
action: FilterAction
|
||||
|
||||
|
||||
class FilterInput(BaseModel):
|
||||
item: Item | None = None
|
||||
article: ExtractedArticle | None = None
|
||||
summary: LlmSummaryResult
|
||||
context: FilterContext = Field(default_factory=FilterContext)
|
||||
|
||||
|
||||
class MatchedRule(BaseModel):
|
||||
rule_id: str
|
||||
decision: FilterDecision
|
||||
reason: str
|
||||
labels: list[str] = Field(default_factory=list)
|
||||
priority: int = 0
|
||||
|
||||
|
||||
class FilterDecisionResult(BaseModel):
|
||||
decision: FilterDecision
|
||||
matched_rules: list[str] = Field(default_factory=list)
|
||||
reasons: list[str] = Field(default_factory=list)
|
||||
labels: list[str] = Field(default_factory=list)
|
||||
priority: int = 0
|
||||
matches: list[MatchedRule] = Field(default_factory=list)
|
||||
@@ -1,9 +1,13 @@
|
||||
from __future__ import annotations
|
||||
from __future__ import annotations
|
||||
|
||||
from mcp.server.fastmcp import FastMCP
|
||||
|
||||
from summary_mcp.core.pipeline import extract_content
|
||||
from summary_mcp.filters.engine import evaluate_filter_rules, load_filter_rules
|
||||
from summary_mcp.models.document import ExtractedArticle
|
||||
from summary_mcp.models.filtering import FilterContext, FilterInput
|
||||
from summary_mcp.models.item import Item
|
||||
from summary_mcp.models.llm_result import LlmSummaryResult
|
||||
from summary_mcp.models.summary_io import ExtractionInput
|
||||
|
||||
|
||||
@@ -32,6 +36,31 @@ def extract_item_content(item: dict) -> dict:
|
||||
return result.model_dump(mode="json")
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def filter_summary_result(
|
||||
summary_result: dict,
|
||||
extracted_article: dict | None = None,
|
||||
item: dict | None = None,
|
||||
context: dict | None = None,
|
||||
) -> dict:
|
||||
"""Apply deterministic filter rules to a structured summary result."""
|
||||
parsed_summary = LlmSummaryResult.model_validate(summary_result)
|
||||
parsed_article = ExtractedArticle.model_validate(extracted_article) if extracted_article else None
|
||||
parsed_item = Item.model_validate(item) if item else None
|
||||
parsed_context = FilterContext.model_validate(context or {})
|
||||
rules = load_filter_rules()
|
||||
decision = evaluate_filter_rules(
|
||||
FilterInput(
|
||||
item=parsed_item,
|
||||
article=parsed_article,
|
||||
summary=parsed_summary,
|
||||
context=parsed_context,
|
||||
),
|
||||
rules,
|
||||
)
|
||||
return decision.model_dump(mode="json")
|
||||
|
||||
|
||||
def main() -> None:
|
||||
mcp.run()
|
||||
|
||||
|
||||
Reference in New Issue
Block a user