fix(article-summary): 修复 article_summary pipeline 的 3 个 bug
1. Bug 3 (文件名含斜杠导致路径错误):
- safe_title 生成时用 .replace('/', '-') 处理 '/' 字符
- 同时清理连续 dash (--+) 和首尾 dash
- 原本只处理空格,导致含 '/' 的标题写出路径错误
2. Bug 2 (delivery payload 格式不匹配):
- 新增 '_iter_selected_items' 对 'candidates' 数组格式的支持
- 自动 normalize selected_ids 的 'cand:' 前缀
- candidate 条目同时支持 item_id 字段(新增)和 candidate_id(兼容)
- 传入 delivery payload + candidate_id 时可正常匹配
3. Pipeline 补充 item_id 字段:
- OpenClawCandidateInput 加 item_id 字段
- build_openclaw_candidate_input 填充 item_id
- 使得后续 article-summary 可通过 item_id 关联 extracted 文件
This commit is contained in:
@@ -66,6 +66,7 @@ class ArticleCandidateRecord(BaseModel):
|
|||||||
|
|
||||||
class OpenClawCandidateInput(BaseModel):
|
class OpenClawCandidateInput(BaseModel):
|
||||||
candidate_id: str
|
candidate_id: str
|
||||||
|
item_id: str | None = None
|
||||||
title: str
|
title: str
|
||||||
url: HttpUrl
|
url: HttpUrl
|
||||||
canonical_url: HttpUrl | None = None
|
canonical_url: HttpUrl | None = None
|
||||||
@@ -169,6 +170,7 @@ def build_openclaw_candidate_input(record: ArticleCandidateRecord) -> OpenClawCa
|
|||||||
|
|
||||||
return OpenClawCandidateInput(
|
return OpenClawCandidateInput(
|
||||||
candidate_id=record.candidate_id,
|
candidate_id=record.candidate_id,
|
||||||
|
item_id=item.item_id if item is not None else None,
|
||||||
title=title,
|
title=title,
|
||||||
url=raw_url,
|
url=raw_url,
|
||||||
canonical_url=normalize_candidate_url(raw_url),
|
canonical_url=normalize_candidate_url(raw_url),
|
||||||
|
|||||||
@@ -110,6 +110,32 @@ def _iter_selected_items(
|
|||||||
}
|
}
|
||||||
return
|
return
|
||||||
|
|
||||||
|
# Fallback: pipeline delivery payload with top-level "candidates" array.
|
||||||
|
# Each entry has item_id (the raw FreshRSS item_id) and candidate_id (with cand: prefix).
|
||||||
|
# Normalize selected_ids by stripping "cand:" prefix so they match item_id.
|
||||||
|
candidates = extracted_payload.get("candidates")
|
||||||
|
if isinstance(candidates, list):
|
||||||
|
norm_selected = {
|
||||||
|
sid.removeprefix("cand:") if sid.startswith("cand:") else sid
|
||||||
|
for sid in selected_ids
|
||||||
|
}
|
||||||
|
for entry in candidates:
|
||||||
|
if not isinstance(entry, Mapping):
|
||||||
|
continue
|
||||||
|
# Support both item_id field (new) and candidate_id (legacy fallback)
|
||||||
|
raw_item_id = entry.get("item_id") or ""
|
||||||
|
if not raw_item_id and entry.get("candidate_id"):
|
||||||
|
raw_item_id = entry["candidate_id"].removeprefix("cand:")
|
||||||
|
item_id = str(raw_item_id) if raw_item_id else None
|
||||||
|
if not item_id or item_id not in norm_selected:
|
||||||
|
continue
|
||||||
|
# Candidates store article fields directly, not nested under "article"
|
||||||
|
yield item_id, {
|
||||||
|
"item": {},
|
||||||
|
"extraction": {"article": entry, "warnings": []},
|
||||||
|
}
|
||||||
|
return
|
||||||
|
|
||||||
# Fallback: legacy payload with top-level "items" array.
|
# Fallback: legacy payload with top-level "items" array.
|
||||||
items = extracted_payload.get("items")
|
items = extracted_payload.get("items")
|
||||||
if isinstance(items, list):
|
if isinstance(items, list):
|
||||||
@@ -120,11 +146,10 @@ def _iter_selected_items(
|
|||||||
item_id = str(raw_item_id) if raw_item_id is not None else None
|
item_id = str(raw_item_id) if raw_item_id is not None else None
|
||||||
if not item_id or item_id not in selected_set:
|
if not item_id or item_id not in selected_set:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
yield item_id, item
|
yield item_id, item
|
||||||
return
|
return
|
||||||
|
|
||||||
# Format 3: single-item extracted file produced by run_freshrss_pipeline debug mode.
|
# Format 3: single-item extracted file produced by run_freshrss_pipeline.
|
||||||
# Shape: {"success": bool, "article": {"item_id": "...", ...}, "warnings": [...]}
|
# Shape: {"success": bool, "article": {"item_id": "...", ...}, "warnings": [...]}
|
||||||
article = extracted_payload.get("article")
|
article = extracted_payload.get("article")
|
||||||
if isinstance(article, Mapping):
|
if isinstance(article, Mapping):
|
||||||
@@ -293,8 +318,12 @@ def summarize_selected_articles(
|
|||||||
lines.append("、".join(topics))
|
lines.append("、".join(topics))
|
||||||
lines.append("")
|
lines.append("")
|
||||||
|
|
||||||
|
# Normalize: collapse spaces/slashes/underscores to single dash, strip punctuation, collapse multi-dashes
|
||||||
|
normalized = str(title).lower()
|
||||||
|
for sep in (" ", "/", "_", "——", "―", "‐"):
|
||||||
|
normalized = normalized.replace(sep, "-")
|
||||||
safe_title = "-".join(
|
safe_title = "-".join(
|
||||||
str(title).lower().strip().replace(" ", "-").split()
|
part for part in normalized.split("-") if part
|
||||||
)[:80]
|
)[:80]
|
||||||
filename = f"{safe_title or item_id}.md"
|
filename = f"{safe_title or item_id}.md"
|
||||||
output_path = output_dir / filename
|
output_path = output_dir / filename
|
||||||
|
|||||||
Reference in New Issue
Block a user