feat: add article summary post-processing flow
This commit is contained in:
@@ -2,7 +2,9 @@ from __future__ import annotations
|
||||
|
||||
import json
|
||||
from collections import Counter
|
||||
from datetime import UTC, date, datetime
|
||||
from datetime import date, datetime, timezone
|
||||
|
||||
UTC = timezone.utc
|
||||
from pathlib import Path
|
||||
from typing import Iterable
|
||||
|
||||
|
||||
@@ -15,6 +15,7 @@ from summary_mcp.models.item import Item
|
||||
from summary_mcp.models.llm_result import LlmSummaryResult
|
||||
from summary_mcp.models.summary_io import ExtractionInput
|
||||
from summary_mcp.workflows import run_freshrss_pipeline
|
||||
from summary_mcp.workflows.article_summary import ArticleSummaryConfig, summarize_selected_articles
|
||||
|
||||
|
||||
mcp = FastMCP(name="content-extract-mcp")
|
||||
@@ -125,6 +126,62 @@ def run_freshrss_openclaw_pipeline(
|
||||
return result
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def generate_article_summaries(
|
||||
*,
|
||||
extracted_path: str,
|
||||
selected_ids: list[str],
|
||||
output_dir: str | None = None,
|
||||
max_retries: int = 2,
|
||||
timeout_seconds: float = 60.0,
|
||||
llm_api_key: str | None = None,
|
||||
llm_model: str | None = None,
|
||||
llm_api_url: str | None = None,
|
||||
) -> list[str]:
|
||||
"""Generate Markdown summaries for selected articles from an extracted payload.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
extracted_path:
|
||||
Path to the extracted JSON file produced by the FreshRSS pipeline
|
||||
(for example `outputs/freshrss/extracted/freshrss.extracted.json`).
|
||||
selected_ids:
|
||||
One or more `item_id` values from the extracted payload to summarize.
|
||||
output_dir:
|
||||
Optional output directory for the generated Markdown files. If omitted,
|
||||
summaries are written next to the extracted file under a
|
||||
`single_summaries/` subdirectory.
|
||||
llm_api_key / llm_model / llm_api_url:
|
||||
Optional overrides for the article-summary LLM settings. If omitted,
|
||||
the workflow falls back to the ARTICLE_SUMMARY_* or main LLM_* env
|
||||
variables as documented in the README.
|
||||
"""
|
||||
|
||||
extracted_path_obj = Path(extracted_path)
|
||||
if not extracted_path_obj.exists():
|
||||
raise FileNotFoundError(f"extracted_path does not exist: {extracted_path}")
|
||||
|
||||
if output_dir is None:
|
||||
default_dir = extracted_path_obj.parent / "single_summaries"
|
||||
output_dir_obj = default_dir
|
||||
else:
|
||||
output_dir_obj = Path(output_dir)
|
||||
|
||||
config = ArticleSummaryConfig(max_retries=max_retries, timeout_seconds=timeout_seconds)
|
||||
|
||||
written_paths = summarize_selected_articles(
|
||||
extracted_path=extracted_path_obj,
|
||||
selected_ids=selected_ids,
|
||||
output_dir=output_dir_obj,
|
||||
config=config,
|
||||
api_key=llm_api_key,
|
||||
model=llm_model,
|
||||
api_url=llm_api_url,
|
||||
)
|
||||
|
||||
return [str(p) for p in written_paths]
|
||||
|
||||
|
||||
def main() -> None:
|
||||
mcp.run()
|
||||
|
||||
|
||||
@@ -0,0 +1,216 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Iterable, Mapping, Sequence
|
||||
|
||||
from summary_mcp.core.summary_loop import run_loop_payload
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[3]
|
||||
OUTPUT_ROOT = REPO_ROOT / "outputs"
|
||||
FRESHRSS_OUTPUT_ROOT = OUTPUT_ROOT / "freshrss"
|
||||
DEFAULT_PROMPT_PATH = OUTPUT_ROOT / "prompts" / "llm-summary-prompt.txt"
|
||||
|
||||
|
||||
@dataclass
|
||||
class ArticleSummaryConfig:
|
||||
prompt_path: Path = DEFAULT_PROMPT_PATH
|
||||
max_retries: int = 2
|
||||
timeout_seconds: float = 60.0
|
||||
llm_api_key: str | None = None
|
||||
llm_model: str | None = None
|
||||
llm_api_url: str | None = None
|
||||
|
||||
|
||||
def _resolve_article_llm_settings(
|
||||
*,
|
||||
api_key: str | None = None,
|
||||
model: str | None = None,
|
||||
api_url: str | None = None,
|
||||
) -> tuple[str | None, str | None, str | None]:
|
||||
"""Resolve article-summary LLM settings with fallback to main pipeline env.
|
||||
|
||||
Resolution order for each field:
|
||||
- explicit function argument
|
||||
- ARTICLE_SUMMARY_* environment variable
|
||||
- main LLM_* / OPENAI_* environment variables (handled by summary_loop.resolve_llm_settings)
|
||||
|
||||
This helper intentionally does not validate presence; the summary loop
|
||||
will perform final validation and raise a clear error if nothing is set.
|
||||
"""
|
||||
|
||||
import os
|
||||
|
||||
resolved_api_key = api_key or os.environ.get("ARTICLE_SUMMARY_LLM_API_KEY")
|
||||
resolved_model = model or os.environ.get("ARTICLE_SUMMARY_LLM_MODEL")
|
||||
resolved_api_url = api_url or os.environ.get("ARTICLE_SUMMARY_LLM_API_URL")
|
||||
return resolved_api_key, resolved_model, resolved_api_url
|
||||
|
||||
|
||||
def _iter_selected_items(
|
||||
extracted_payload: Mapping[str, object],
|
||||
selected_ids: Sequence[str],
|
||||
) -> Iterable[tuple[str, Mapping[str, object]]]:
|
||||
"""Yield (item_id, extracted_entry) pairs for selected items.
|
||||
|
||||
The reader pipeline currently produces an extracted payload shaped like:
|
||||
|
||||
{
|
||||
"results": [
|
||||
{
|
||||
"item": {"item_id": "...", ...},
|
||||
"extraction": {"article": {..}, "warnings": [...]},
|
||||
},
|
||||
...
|
||||
]
|
||||
}
|
||||
|
||||
Historically some flows used a top-level ``items`` array where each
|
||||
element directly contained the ``article`` field. To keep compatibility
|
||||
sane, we first try the real project format (``results``), then fall
|
||||
back to the legacy ``items`` layout if present.
|
||||
"""
|
||||
|
||||
selected_set = set(selected_ids)
|
||||
|
||||
# Preferred: current reader format with top-level "results" array.
|
||||
results = extracted_payload.get("results")
|
||||
if isinstance(results, list):
|
||||
for entry in results:
|
||||
if not isinstance(entry, Mapping):
|
||||
continue
|
||||
item = entry.get("item")
|
||||
extraction = entry.get("extraction")
|
||||
if not isinstance(item, Mapping) or not isinstance(extraction, Mapping):
|
||||
continue
|
||||
|
||||
raw_item_id = item.get("item_id")
|
||||
item_id = str(raw_item_id) if raw_item_id is not None else None
|
||||
if not item_id or item_id not in selected_set:
|
||||
continue
|
||||
|
||||
yield item_id, {
|
||||
"item": item,
|
||||
"extraction": extraction,
|
||||
}
|
||||
return
|
||||
|
||||
# Fallback: legacy payload with top-level "items" array.
|
||||
items = extracted_payload.get("items")
|
||||
if isinstance(items, list):
|
||||
for item in items:
|
||||
if not isinstance(item, Mapping):
|
||||
continue
|
||||
raw_item_id = item.get("item_id")
|
||||
item_id = str(raw_item_id) if raw_item_id is not None else None
|
||||
if not item_id or item_id not in selected_set:
|
||||
continue
|
||||
|
||||
yield item_id, item
|
||||
return
|
||||
|
||||
raise RuntimeError(
|
||||
"Extracted payload does not contain a supported article items structure (expected 'results' or 'items').",
|
||||
)
|
||||
|
||||
|
||||
def summarize_selected_articles(
|
||||
*,
|
||||
extracted_path: Path,
|
||||
selected_ids: Sequence[str],
|
||||
output_dir: Path,
|
||||
config: ArticleSummaryConfig | None = None,
|
||||
api_key: str | None = None,
|
||||
model: str | None = None,
|
||||
api_url: str | None = None,
|
||||
) -> list[Path]:
|
||||
"""Generate one Markdown summary per selected article.
|
||||
|
||||
This operates purely on already extracted article payloads and does not
|
||||
re-fetch original URLs. It expects an `extracted_path` produced by the
|
||||
FreshRSS extract or pipeline steps.
|
||||
|
||||
The current reader format uses a top-level ``results`` array where each
|
||||
entry has ``item`` and ``extraction`` (with ``extraction.article`` being
|
||||
the extracted article). For compatibility, a legacy top-level ``items``
|
||||
array is also supported if present.
|
||||
"""
|
||||
|
||||
import json
|
||||
|
||||
cfg = config or ArticleSummaryConfig()
|
||||
resolved_api_key, resolved_model, resolved_api_url = _resolve_article_llm_settings(
|
||||
api_key=api_key or cfg.llm_api_key,
|
||||
model=model or cfg.llm_model,
|
||||
api_url=api_url or cfg.llm_api_url,
|
||||
)
|
||||
|
||||
payload = json.loads(extracted_path.read_text(encoding="utf-8-sig"))
|
||||
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
written_paths: list[Path] = []
|
||||
|
||||
from summary_mcp.core.summary_loop import build_summary_input
|
||||
|
||||
for item_id, entry in _iter_selected_items(payload, selected_ids):
|
||||
# Prefer the real project format where each entry has ``item`` and
|
||||
# ``extraction.article``; fall back to legacy layout where the
|
||||
# article fields live directly on the element.
|
||||
if "item" in entry or "extraction" in entry:
|
||||
item = entry.get("item") or {}
|
||||
extraction = entry.get("extraction") or {}
|
||||
article = (extraction.get("article") or {}) if isinstance(extraction, dict) else {}
|
||||
warnings = extraction.get("warnings", []) if isinstance(extraction, dict) else []
|
||||
else:
|
||||
item = entry
|
||||
article = item.get("article") or {}
|
||||
warnings = item.get("warnings", [])
|
||||
|
||||
extracted_payload = {
|
||||
"article": article,
|
||||
"warnings": warnings,
|
||||
}
|
||||
|
||||
summary_exit_code, summary_payload, _ = run_loop_payload(
|
||||
extracted_payload=extracted_payload,
|
||||
prompt_path=cfg.prompt_path,
|
||||
max_retries=cfg.max_retries,
|
||||
timeout_seconds=cfg.timeout_seconds,
|
||||
api_key=resolved_api_key,
|
||||
model=resolved_model,
|
||||
api_url=resolved_api_url,
|
||||
output_path=None,
|
||||
)
|
||||
if summary_exit_code != 0 or summary_payload is None:
|
||||
continue
|
||||
|
||||
# Render a simple Markdown file summarizing the article.
|
||||
title = article.get("title") or summary_payload.get("title") or item_id
|
||||
url = article.get("url")
|
||||
summary_text = summary_payload.get("summary") or ""
|
||||
highlights = summary_payload.get("highlights") or []
|
||||
|
||||
lines: list[str] = []
|
||||
lines.append(f"# {title}")
|
||||
if url:
|
||||
lines.append("")
|
||||
lines.append(f"Source: {url}")
|
||||
lines.append("")
|
||||
if summary_text:
|
||||
lines.append(summary_text)
|
||||
lines.append("")
|
||||
if highlights:
|
||||
lines.append("## Highlights")
|
||||
lines.extend(f"- {h}" for h in highlights)
|
||||
lines.append("")
|
||||
|
||||
safe_title = "-".join(
|
||||
str(title).lower().strip().replace(" ", "-").split()
|
||||
)[:80]
|
||||
filename = f"{safe_title or item_id}.md"
|
||||
output_path = output_dir / filename
|
||||
output_path.write_text("\n".join(lines), encoding="utf-8")
|
||||
written_paths.append(output_path)
|
||||
|
||||
return written_paths
|
||||
@@ -2,7 +2,9 @@ from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
from datetime import UTC, date, datetime
|
||||
from datetime import date, datetime, timezone
|
||||
|
||||
UTC = timezone.utc
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
Reference in New Issue
Block a user