feat: add article summary post-processing flow

This commit is contained in:
root
2026-03-28 00:08:38 +08:00
parent 3940db1443
commit 0a00a97724
7 changed files with 390 additions and 2 deletions
+3 -1
View File
@@ -2,7 +2,9 @@ from __future__ import annotations
import json
from collections import Counter
from datetime import UTC, date, datetime
from datetime import date, datetime, timezone
UTC = timezone.utc
from pathlib import Path
from typing import Iterable
+57
View File
@@ -15,6 +15,7 @@ from summary_mcp.models.item import Item
from summary_mcp.models.llm_result import LlmSummaryResult
from summary_mcp.models.summary_io import ExtractionInput
from summary_mcp.workflows import run_freshrss_pipeline
from summary_mcp.workflows.article_summary import ArticleSummaryConfig, summarize_selected_articles
mcp = FastMCP(name="content-extract-mcp")
@@ -125,6 +126,62 @@ def run_freshrss_openclaw_pipeline(
return result
@mcp.tool()
def generate_article_summaries(
*,
extracted_path: str,
selected_ids: list[str],
output_dir: str | None = None,
max_retries: int = 2,
timeout_seconds: float = 60.0,
llm_api_key: str | None = None,
llm_model: str | None = None,
llm_api_url: str | None = None,
) -> list[str]:
"""Generate Markdown summaries for selected articles from an extracted payload.
Parameters
----------
extracted_path:
Path to the extracted JSON file produced by the FreshRSS pipeline
(for example `outputs/freshrss/extracted/freshrss.extracted.json`).
selected_ids:
One or more `item_id` values from the extracted payload to summarize.
output_dir:
Optional output directory for the generated Markdown files. If omitted,
summaries are written next to the extracted file under a
`single_summaries/` subdirectory.
llm_api_key / llm_model / llm_api_url:
Optional overrides for the article-summary LLM settings. If omitted,
the workflow falls back to the ARTICLE_SUMMARY_* or main LLM_* env
variables as documented in the README.
"""
extracted_path_obj = Path(extracted_path)
if not extracted_path_obj.exists():
raise FileNotFoundError(f"extracted_path does not exist: {extracted_path}")
if output_dir is None:
default_dir = extracted_path_obj.parent / "single_summaries"
output_dir_obj = default_dir
else:
output_dir_obj = Path(output_dir)
config = ArticleSummaryConfig(max_retries=max_retries, timeout_seconds=timeout_seconds)
written_paths = summarize_selected_articles(
extracted_path=extracted_path_obj,
selected_ids=selected_ids,
output_dir=output_dir_obj,
config=config,
api_key=llm_api_key,
model=llm_model,
api_url=llm_api_url,
)
return [str(p) for p in written_paths]
def main() -> None:
mcp.run()
@@ -0,0 +1,216 @@
from __future__ import annotations
from dataclasses import dataclass
from pathlib import Path
from typing import Iterable, Mapping, Sequence
from summary_mcp.core.summary_loop import run_loop_payload
REPO_ROOT = Path(__file__).resolve().parents[3]
OUTPUT_ROOT = REPO_ROOT / "outputs"
FRESHRSS_OUTPUT_ROOT = OUTPUT_ROOT / "freshrss"
DEFAULT_PROMPT_PATH = OUTPUT_ROOT / "prompts" / "llm-summary-prompt.txt"
@dataclass
class ArticleSummaryConfig:
prompt_path: Path = DEFAULT_PROMPT_PATH
max_retries: int = 2
timeout_seconds: float = 60.0
llm_api_key: str | None = None
llm_model: str | None = None
llm_api_url: str | None = None
def _resolve_article_llm_settings(
*,
api_key: str | None = None,
model: str | None = None,
api_url: str | None = None,
) -> tuple[str | None, str | None, str | None]:
"""Resolve article-summary LLM settings with fallback to main pipeline env.
Resolution order for each field:
- explicit function argument
- ARTICLE_SUMMARY_* environment variable
- main LLM_* / OPENAI_* environment variables (handled by summary_loop.resolve_llm_settings)
This helper intentionally does not validate presence; the summary loop
will perform final validation and raise a clear error if nothing is set.
"""
import os
resolved_api_key = api_key or os.environ.get("ARTICLE_SUMMARY_LLM_API_KEY")
resolved_model = model or os.environ.get("ARTICLE_SUMMARY_LLM_MODEL")
resolved_api_url = api_url or os.environ.get("ARTICLE_SUMMARY_LLM_API_URL")
return resolved_api_key, resolved_model, resolved_api_url
def _iter_selected_items(
extracted_payload: Mapping[str, object],
selected_ids: Sequence[str],
) -> Iterable[tuple[str, Mapping[str, object]]]:
"""Yield (item_id, extracted_entry) pairs for selected items.
The reader pipeline currently produces an extracted payload shaped like:
{
"results": [
{
"item": {"item_id": "...", ...},
"extraction": {"article": {..}, "warnings": [...]},
},
...
]
}
Historically some flows used a top-level ``items`` array where each
element directly contained the ``article`` field. To keep compatibility
sane, we first try the real project format (``results``), then fall
back to the legacy ``items`` layout if present.
"""
selected_set = set(selected_ids)
# Preferred: current reader format with top-level "results" array.
results = extracted_payload.get("results")
if isinstance(results, list):
for entry in results:
if not isinstance(entry, Mapping):
continue
item = entry.get("item")
extraction = entry.get("extraction")
if not isinstance(item, Mapping) or not isinstance(extraction, Mapping):
continue
raw_item_id = item.get("item_id")
item_id = str(raw_item_id) if raw_item_id is not None else None
if not item_id or item_id not in selected_set:
continue
yield item_id, {
"item": item,
"extraction": extraction,
}
return
# Fallback: legacy payload with top-level "items" array.
items = extracted_payload.get("items")
if isinstance(items, list):
for item in items:
if not isinstance(item, Mapping):
continue
raw_item_id = item.get("item_id")
item_id = str(raw_item_id) if raw_item_id is not None else None
if not item_id or item_id not in selected_set:
continue
yield item_id, item
return
raise RuntimeError(
"Extracted payload does not contain a supported article items structure (expected 'results' or 'items').",
)
def summarize_selected_articles(
*,
extracted_path: Path,
selected_ids: Sequence[str],
output_dir: Path,
config: ArticleSummaryConfig | None = None,
api_key: str | None = None,
model: str | None = None,
api_url: str | None = None,
) -> list[Path]:
"""Generate one Markdown summary per selected article.
This operates purely on already extracted article payloads and does not
re-fetch original URLs. It expects an `extracted_path` produced by the
FreshRSS extract or pipeline steps.
The current reader format uses a top-level ``results`` array where each
entry has ``item`` and ``extraction`` (with ``extraction.article`` being
the extracted article). For compatibility, a legacy top-level ``items``
array is also supported if present.
"""
import json
cfg = config or ArticleSummaryConfig()
resolved_api_key, resolved_model, resolved_api_url = _resolve_article_llm_settings(
api_key=api_key or cfg.llm_api_key,
model=model or cfg.llm_model,
api_url=api_url or cfg.llm_api_url,
)
payload = json.loads(extracted_path.read_text(encoding="utf-8-sig"))
output_dir.mkdir(parents=True, exist_ok=True)
written_paths: list[Path] = []
from summary_mcp.core.summary_loop import build_summary_input
for item_id, entry in _iter_selected_items(payload, selected_ids):
# Prefer the real project format where each entry has ``item`` and
# ``extraction.article``; fall back to legacy layout where the
# article fields live directly on the element.
if "item" in entry or "extraction" in entry:
item = entry.get("item") or {}
extraction = entry.get("extraction") or {}
article = (extraction.get("article") or {}) if isinstance(extraction, dict) else {}
warnings = extraction.get("warnings", []) if isinstance(extraction, dict) else []
else:
item = entry
article = item.get("article") or {}
warnings = item.get("warnings", [])
extracted_payload = {
"article": article,
"warnings": warnings,
}
summary_exit_code, summary_payload, _ = run_loop_payload(
extracted_payload=extracted_payload,
prompt_path=cfg.prompt_path,
max_retries=cfg.max_retries,
timeout_seconds=cfg.timeout_seconds,
api_key=resolved_api_key,
model=resolved_model,
api_url=resolved_api_url,
output_path=None,
)
if summary_exit_code != 0 or summary_payload is None:
continue
# Render a simple Markdown file summarizing the article.
title = article.get("title") or summary_payload.get("title") or item_id
url = article.get("url")
summary_text = summary_payload.get("summary") or ""
highlights = summary_payload.get("highlights") or []
lines: list[str] = []
lines.append(f"# {title}")
if url:
lines.append("")
lines.append(f"Source: {url}")
lines.append("")
if summary_text:
lines.append(summary_text)
lines.append("")
if highlights:
lines.append("## Highlights")
lines.extend(f"- {h}" for h in highlights)
lines.append("")
safe_title = "-".join(
str(title).lower().strip().replace(" ", "-").split()
)[:80]
filename = f"{safe_title or item_id}.md"
output_path = output_dir / filename
output_path.write_text("\n".join(lines), encoding="utf-8")
written_paths.append(output_path)
return written_paths
@@ -2,7 +2,9 @@ from __future__ import annotations
import json
import os
from datetime import UTC, date, datetime
from datetime import date, datetime, timezone
UTC = timezone.utc
from pathlib import Path
from typing import Any