From c528e0abc7efd1d4ce7bcaf592671e372c195f5a Mon Sep 17 00:00:00 2001 From: root Date: Mon, 13 Apr 2026 17:38:40 +0800 Subject: [PATCH] =?UTF-8?q?fix(article-summary):=20=E4=BF=AE=E5=A4=8D=20ar?= =?UTF-8?q?ticle=5Fsummary=20pipeline=20=E7=9A=84=203=20=E4=B8=AA=20bug?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 1. Bug 3 (文件名含斜杠导致路径错误): - safe_title 生成时用 .replace('/', '-') 处理 '/' 字符 - 同时清理连续 dash (--+) 和首尾 dash - 原本只处理空格,导致含 '/' 的标题写出路径错误 2. Bug 2 (delivery payload 格式不匹配): - 新增 '_iter_selected_items' 对 'candidates' 数组格式的支持 - 自动 normalize selected_ids 的 'cand:' 前缀 - candidate 条目同时支持 item_id 字段(新增)和 candidate_id(兼容) - 传入 delivery payload + candidate_id 时可正常匹配 3. Pipeline 补充 item_id 字段: - OpenClawCandidateInput 加 item_id 字段 - build_openclaw_candidate_input 填充 item_id - 使得后续 article-summary 可通过 item_id 关联 extracted 文件 --- src/summary_mcp/models/article_candidate.py | 2 ++ src/summary_mcp/workflows/article_summary.py | 35 ++++++++++++++++++-- 2 files changed, 34 insertions(+), 3 deletions(-) diff --git a/src/summary_mcp/models/article_candidate.py b/src/summary_mcp/models/article_candidate.py index 844d08c..91f0845 100644 --- a/src/summary_mcp/models/article_candidate.py +++ b/src/summary_mcp/models/article_candidate.py @@ -66,6 +66,7 @@ class ArticleCandidateRecord(BaseModel): class OpenClawCandidateInput(BaseModel): candidate_id: str + item_id: str | None = None title: str url: HttpUrl canonical_url: HttpUrl | None = None @@ -169,6 +170,7 @@ def build_openclaw_candidate_input(record: ArticleCandidateRecord) -> OpenClawCa return OpenClawCandidateInput( candidate_id=record.candidate_id, + item_id=item.item_id if item is not None else None, title=title, url=raw_url, canonical_url=normalize_candidate_url(raw_url), diff --git a/src/summary_mcp/workflows/article_summary.py b/src/summary_mcp/workflows/article_summary.py index 6bed434..b456639 100644 --- a/src/summary_mcp/workflows/article_summary.py +++ b/src/summary_mcp/workflows/article_summary.py @@ -110,6 +110,32 @@ def _iter_selected_items( } return + # Fallback: pipeline delivery payload with top-level "candidates" array. + # Each entry has item_id (the raw FreshRSS item_id) and candidate_id (with cand: prefix). + # Normalize selected_ids by stripping "cand:" prefix so they match item_id. + candidates = extracted_payload.get("candidates") + if isinstance(candidates, list): + norm_selected = { + sid.removeprefix("cand:") if sid.startswith("cand:") else sid + for sid in selected_ids + } + for entry in candidates: + if not isinstance(entry, Mapping): + continue + # Support both item_id field (new) and candidate_id (legacy fallback) + raw_item_id = entry.get("item_id") or "" + if not raw_item_id and entry.get("candidate_id"): + raw_item_id = entry["candidate_id"].removeprefix("cand:") + item_id = str(raw_item_id) if raw_item_id else None + if not item_id or item_id not in norm_selected: + continue + # Candidates store article fields directly, not nested under "article" + yield item_id, { + "item": {}, + "extraction": {"article": entry, "warnings": []}, + } + return + # Fallback: legacy payload with top-level "items" array. items = extracted_payload.get("items") if isinstance(items, list): @@ -120,11 +146,10 @@ def _iter_selected_items( item_id = str(raw_item_id) if raw_item_id is not None else None if not item_id or item_id not in selected_set: continue - yield item_id, item return - # Format 3: single-item extracted file produced by run_freshrss_pipeline debug mode. + # Format 3: single-item extracted file produced by run_freshrss_pipeline. # Shape: {"success": bool, "article": {"item_id": "...", ...}, "warnings": [...]} article = extracted_payload.get("article") if isinstance(article, Mapping): @@ -293,8 +318,12 @@ def summarize_selected_articles( lines.append("、".join(topics)) lines.append("") + # Normalize: collapse spaces/slashes/underscores to single dash, strip punctuation, collapse multi-dashes + normalized = str(title).lower() + for sep in (" ", "/", "_", "——", "―", "‐"): + normalized = normalized.replace(sep, "-") safe_title = "-".join( - str(title).lower().strip().replace(" ", "-").split() + part for part in normalized.split("-") if part )[:80] filename = f"{safe_title or item_id}.md" output_path = output_dir / filename