Files
reader/src/summary_mcp/server.py
T

191 lines
6.3 KiB
Python

from __future__ import annotations
import json
import tempfile
from datetime import date
from pathlib import Path
from mcp.server.fastmcp import FastMCP
from summary_mcp.core.pipeline import extract_content
from summary_mcp.filters.engine import evaluate_filter_rules, load_filter_rules
from summary_mcp.models.document import ExtractedArticle
from summary_mcp.models.filtering import FilterContext, FilterInput
from summary_mcp.models.item import Item
from summary_mcp.models.llm_result import LlmSummaryResult
from summary_mcp.models.summary_io import ExtractionInput
from summary_mcp.workflows import run_freshrss_pipeline
from summary_mcp.workflows.article_summary import ArticleSummaryConfig, summarize_selected_articles
mcp = FastMCP(name="content-extract-mcp")
@mcp.tool()
def extract_url_content(url: str, language_hint: str | None = None) -> dict:
"""Extract structured article content from a single URL."""
result = extract_content(
ExtractionInput(
url=url,
language_hint=language_hint,
)
)
return result.model_dump(mode="json")
@mcp.tool()
def extract_item_content(item: dict) -> dict:
"""Extract structured article content from a normalized item object."""
parsed_item = Item.model_validate(item)
result = extract_content(ExtractionInput(item=parsed_item))
return result.model_dump(mode="json")
@mcp.tool()
def filter_summary_result(
summary_result: dict,
extracted_article: dict | None = None,
item: dict | None = None,
context: dict | None = None,
) -> dict:
"""Apply deterministic filter rules to a structured summary result."""
parsed_summary = LlmSummaryResult.model_validate(summary_result)
parsed_article = ExtractedArticle.model_validate(extracted_article) if extracted_article else None
parsed_item = Item.model_validate(item) if item else None
parsed_context = FilterContext.model_validate(context or {})
rules = load_filter_rules()
decision = evaluate_filter_rules(
FilterInput(
item=parsed_item,
article=parsed_article,
summary=parsed_summary,
context=parsed_context,
),
rules,
)
return decision.model_dump(mode="json")
@mcp.tool()
def run_freshrss_openclaw_pipeline(
limit: int = 5,
mark_read: bool = False,
include_read: bool = False,
debug_artifacts: bool = False,
continuation: str | None = None,
timeout_seconds: float = 60.0,
max_retries: int = 2,
stream_id: str = "user/-/state/com.google/reading-list",
api_base_url: str | None = None,
username: str | None = None,
api_password: str | None = None,
llm_api_key: str | None = None,
llm_model: str | None = None,
llm_api_url: str | None = None,
context: dict | None = None,
run_id: str | None = None,
date_value: str | None = None,
output_dir: str | None = None,
include_item_reports: bool = False,
) -> dict:
"""Run the full FreshRSS -> extract -> LLM -> filter -> OpenClaw payload pipeline."""
temp_context_path: Path | None = None
try:
if context is not None:
with tempfile.NamedTemporaryFile("w", encoding="utf-8", suffix=".json", delete=False) as handle:
json.dump(context, handle, ensure_ascii=False, indent=2)
temp_context_path = Path(handle.name)
result = run_freshrss_pipeline(
api_base_url=api_base_url,
username=username,
api_password=api_password,
stream_id=stream_id,
limit=limit,
continuation=continuation,
include_read=include_read,
mark_read=mark_read,
debug_artifacts=debug_artifacts,
context_path=temp_context_path,
max_retries=max_retries,
timeout_seconds=timeout_seconds,
llm_api_key=llm_api_key,
llm_model=llm_model,
llm_api_url=llm_api_url,
run_id=run_id,
delivery_date=date.fromisoformat(date_value) if date_value else None,
output_dir=Path(output_dir) if output_dir else None,
)
finally:
if temp_context_path and temp_context_path.exists():
temp_context_path.unlink()
if not include_item_reports:
result = {key: value for key, value in result.items() if key != "items"}
return result
@mcp.tool()
def generate_article_summaries(
*,
extracted_path: str,
selected_ids: list[str],
output_dir: str | None = None,
max_retries: int = 2,
timeout_seconds: float = 60.0,
llm_api_key: str | None = None,
llm_model: str | None = None,
llm_api_url: str | None = None,
) -> list[str]:
"""Generate Markdown summaries for selected articles from an extracted payload.
Parameters
----------
extracted_path:
Path to the extracted JSON file produced by the FreshRSS pipeline
(for example `outputs/freshrss/extracted/freshrss.extracted.json`).
selected_ids:
One or more `item_id` values from the extracted payload to summarize.
output_dir:
Optional output directory for the generated Markdown files. If omitted,
summaries are written next to the extracted file under a
`single_summaries/` subdirectory.
llm_api_key / llm_model / llm_api_url:
Optional overrides for the article-summary LLM settings. If omitted,
the workflow falls back to the ARTICLE_SUMMARY_* or main LLM_* env
variables as documented in the README.
"""
extracted_path_obj = Path(extracted_path)
if not extracted_path_obj.exists():
raise FileNotFoundError(f"extracted_path does not exist: {extracted_path}")
if output_dir is None:
default_dir = extracted_path_obj.parent / "single_summaries"
output_dir_obj = default_dir
else:
output_dir_obj = Path(output_dir)
config = ArticleSummaryConfig(max_retries=max_retries, timeout_seconds=timeout_seconds)
written_paths = summarize_selected_articles(
extracted_path=extracted_path_obj,
selected_ids=selected_ids,
output_dir=output_dir_obj,
config=config,
api_key=llm_api_key,
model=llm_model,
api_url=llm_api_url,
)
return [str(p) for p in written_paths]
def main() -> None:
mcp.run()
if __name__ == "__main__":
main()