feat: add article summary post-processing flow

This commit is contained in:
root
2026-03-28 00:08:38 +08:00
parent 3940db1443
commit 0a00a97724
7 changed files with 390 additions and 2 deletions
+57
View File
@@ -15,6 +15,7 @@ from summary_mcp.models.item import Item
from summary_mcp.models.llm_result import LlmSummaryResult
from summary_mcp.models.summary_io import ExtractionInput
from summary_mcp.workflows import run_freshrss_pipeline
from summary_mcp.workflows.article_summary import ArticleSummaryConfig, summarize_selected_articles
mcp = FastMCP(name="content-extract-mcp")
@@ -125,6 +126,62 @@ def run_freshrss_openclaw_pipeline(
return result
@mcp.tool()
def generate_article_summaries(
*,
extracted_path: str,
selected_ids: list[str],
output_dir: str | None = None,
max_retries: int = 2,
timeout_seconds: float = 60.0,
llm_api_key: str | None = None,
llm_model: str | None = None,
llm_api_url: str | None = None,
) -> list[str]:
"""Generate Markdown summaries for selected articles from an extracted payload.
Parameters
----------
extracted_path:
Path to the extracted JSON file produced by the FreshRSS pipeline
(for example `outputs/freshrss/extracted/freshrss.extracted.json`).
selected_ids:
One or more `item_id` values from the extracted payload to summarize.
output_dir:
Optional output directory for the generated Markdown files. If omitted,
summaries are written next to the extracted file under a
`single_summaries/` subdirectory.
llm_api_key / llm_model / llm_api_url:
Optional overrides for the article-summary LLM settings. If omitted,
the workflow falls back to the ARTICLE_SUMMARY_* or main LLM_* env
variables as documented in the README.
"""
extracted_path_obj = Path(extracted_path)
if not extracted_path_obj.exists():
raise FileNotFoundError(f"extracted_path does not exist: {extracted_path}")
if output_dir is None:
default_dir = extracted_path_obj.parent / "single_summaries"
output_dir_obj = default_dir
else:
output_dir_obj = Path(output_dir)
config = ArticleSummaryConfig(max_retries=max_retries, timeout_seconds=timeout_seconds)
written_paths = summarize_selected_articles(
extracted_path=extracted_path_obj,
selected_ids=selected_ids,
output_dir=output_dir_obj,
config=config,
api_key=llm_api_key,
model=llm_model,
api_url=llm_api_url,
)
return [str(p) for p in written_paths]
def main() -> None:
mcp.run()