fix(article-summary): 修复 article_summary pipeline 的 3 个 bug

1. Bug 3 (文件名含斜杠导致路径错误):
   - safe_title 生成时用 .replace('/', '-') 处理 '/' 字符
   - 同时清理连续 dash (--+) 和首尾 dash
   - 原本只处理空格,导致含 '/' 的标题写出路径错误

2. Bug 2 (delivery payload 格式不匹配):
   - 新增 '_iter_selected_items' 对 'candidates' 数组格式的支持
   - 自动 normalize selected_ids 的 'cand:' 前缀
   - candidate 条目同时支持 item_id 字段(新增)和 candidate_id(兼容)
   - 传入 delivery payload + candidate_id 时可正常匹配

3. Pipeline 补充 item_id 字段:
   - OpenClawCandidateInput 加 item_id 字段
   - build_openclaw_candidate_input 填充 item_id
   - 使得后续 article-summary 可通过 item_id 关联 extracted 文件
This commit is contained in:
root
2026-04-13 17:38:40 +08:00
parent 52ce6bfdf5
commit c528e0abc7
2 changed files with 34 additions and 3 deletions
@@ -66,6 +66,7 @@ class ArticleCandidateRecord(BaseModel):
class OpenClawCandidateInput(BaseModel):
candidate_id: str
item_id: str | None = None
title: str
url: HttpUrl
canonical_url: HttpUrl | None = None
@@ -169,6 +170,7 @@ def build_openclaw_candidate_input(record: ArticleCandidateRecord) -> OpenClawCa
return OpenClawCandidateInput(
candidate_id=record.candidate_id,
item_id=item.item_id if item is not None else None,
title=title,
url=raw_url,
canonical_url=normalize_candidate_url(raw_url),
+32 -3
View File
@@ -110,6 +110,32 @@ def _iter_selected_items(
}
return
# Fallback: pipeline delivery payload with top-level "candidates" array.
# Each entry has item_id (the raw FreshRSS item_id) and candidate_id (with cand: prefix).
# Normalize selected_ids by stripping "cand:" prefix so they match item_id.
candidates = extracted_payload.get("candidates")
if isinstance(candidates, list):
norm_selected = {
sid.removeprefix("cand:") if sid.startswith("cand:") else sid
for sid in selected_ids
}
for entry in candidates:
if not isinstance(entry, Mapping):
continue
# Support both item_id field (new) and candidate_id (legacy fallback)
raw_item_id = entry.get("item_id") or ""
if not raw_item_id and entry.get("candidate_id"):
raw_item_id = entry["candidate_id"].removeprefix("cand:")
item_id = str(raw_item_id) if raw_item_id else None
if not item_id or item_id not in norm_selected:
continue
# Candidates store article fields directly, not nested under "article"
yield item_id, {
"item": {},
"extraction": {"article": entry, "warnings": []},
}
return
# Fallback: legacy payload with top-level "items" array.
items = extracted_payload.get("items")
if isinstance(items, list):
@@ -120,11 +146,10 @@ def _iter_selected_items(
item_id = str(raw_item_id) if raw_item_id is not None else None
if not item_id or item_id not in selected_set:
continue
yield item_id, item
return
# Format 3: single-item extracted file produced by run_freshrss_pipeline debug mode.
# Format 3: single-item extracted file produced by run_freshrss_pipeline.
# Shape: {"success": bool, "article": {"item_id": "...", ...}, "warnings": [...]}
article = extracted_payload.get("article")
if isinstance(article, Mapping):
@@ -293,8 +318,12 @@ def summarize_selected_articles(
lines.append("、".join(topics))
lines.append("")
# Normalize: collapse spaces/slashes/underscores to single dash, strip punctuation, collapse multi-dashes
normalized = str(title).lower()
for sep in (" ", "/", "_", "——", "―", "‐"):
normalized = normalized.replace(sep, "-")
safe_title = "-".join(
str(title).lower().strip().replace(" ", "-").split()
part for part in normalized.split("-") if part
)[:80]
filename = f"{safe_title or item_id}.md"
output_path = output_dir / filename