Files
reader/src/summary_mcp/server.py
T

271 lines
9.1 KiB
Python

from __future__ import annotations
# MCP 服务入口:将内容提取、过滤、FreshRSS 全链路管道暴露为 MCP 工具。
# 生产主入口是 run_freshrss_openclaw_pipeline,其余工具供单步调试使用。
from datetime import date
from pathlib import Path
from mcp.server.fastmcp import FastMCP
from summary_mcp.core.pipeline import extract_content
from summary_mcp.filters.engine import evaluate_filter_rules, load_filter_rules
from summary_mcp.models.document import ExtractedArticle
from summary_mcp.models.filtering import FilterContext, FilterInput
from summary_mcp.models.item import Item
from summary_mcp.models.llm_result import LlmSummaryResult
from summary_mcp.models.summary_io import ExtractionInput
from summary_mcp.runtime import get_delivery_payload as load_delivery_payload
from summary_mcp.runtime import get_run_status as load_run_status
from summary_mcp.runtime import get_run_report as load_run_report
from summary_mcp.runtime import list_run_artifacts as load_run_artifacts
from summary_mcp.runtime import list_runs as load_runs
from summary_mcp.runtime.article_summary_jobs import (
get_article_summary_job_result as load_article_summary_job_result,
get_article_summary_job_status as load_article_summary_job_status,
start_article_summary_job as launch_article_summary_job,
)
from summary_mcp.runtime.resume_service import resume_run as resume_existing_run
from summary_mcp.workflows import run_freshrss_pipeline
from summary_mcp.workflows.article_summary import ArticleSummaryConfig, summarize_selected_articles
mcp = FastMCP(name="content-extract-mcp")
@mcp.tool()
def extract_url_content(url: str, language_hint: str | None = None) -> dict:
# 从单个 URL 抓取并提取结构化文章内容,供单步调试使用
"""Extract structured article content from a single URL."""
result = extract_content(
ExtractionInput(
url=url,
language_hint=language_hint,
)
)
return result.model_dump(mode="json")
@mcp.tool()
def extract_item_content(item: dict) -> dict:
# 从已标准化的 item 对象提取内容(跳过网络抓取,使用 RSS 内联内容)
"""Extract structured article content from a normalized item object."""
parsed_item = Item.model_validate(item)
result = extract_content(ExtractionInput(item=parsed_item))
return result.model_dump(mode="json")
@mcp.tool()
def filter_summary_result(
summary_result: dict,
extracted_article: dict | None = None,
item: dict | None = None,
context: dict | None = None,
) -> dict:
# 对结构化摘要结果运行确定性规则引擎,返回 keep/review/drop 决策
"""Apply deterministic filter rules to a structured summary result."""
parsed_summary = LlmSummaryResult.model_validate(summary_result)
parsed_article = ExtractedArticle.model_validate(extracted_article) if extracted_article else None
parsed_item = Item.model_validate(item) if item else None
parsed_context = FilterContext.model_validate(context or {})
rules = load_filter_rules()
decision = evaluate_filter_rules(
FilterInput(
item=parsed_item,
article=parsed_article,
summary=parsed_summary,
context=parsed_context,
),
rules,
)
return decision.model_dump(mode="json")
@mcp.tool()
def run_freshrss_openclaw_pipeline(
limit: int = 5,
mark_read: bool = False,
include_read: bool = False,
debug_artifacts: bool = False,
continuation: str | None = None,
timeout_seconds: float = 60.0,
max_retries: int = 2,
stream_id: str = "user/-/state/com.google/reading-list",
api_base_url: str | None = None,
username: str | None = None,
api_password: str | None = None,
llm_api_key: str | None = None,
llm_model: str | None = None,
llm_api_url: str | None = None,
context: dict | None = None,
run_id: str | None = None,
date_value: str | None = None,
output_dir: str | None = None,
include_item_reports: bool = False,
) -> dict:
"""Run the full FreshRSS -> extract -> LLM -> filter -> OpenClaw payload pipeline."""
result = run_freshrss_pipeline(
api_base_url=api_base_url,
username=username,
api_password=api_password,
stream_id=stream_id,
limit=limit,
continuation=continuation,
include_read=include_read,
mark_read=mark_read,
debug_artifacts=debug_artifacts,
context=context,
max_retries=max_retries,
timeout_seconds=timeout_seconds,
llm_api_key=llm_api_key,
llm_model=llm_model,
llm_api_url=llm_api_url,
run_id=run_id,
delivery_date=date.fromisoformat(date_value) if date_value else None,
output_dir=Path(output_dir) if output_dir else None,
)
if not include_item_reports:
result = {key: value for key, value in result.items() if key != "items"}
return result
@mcp.tool()
def get_run_status(run_id: str) -> dict:
"""Get the current status of a workflow run by run_id."""
return load_run_status(run_id=run_id)
@mcp.tool()
def list_runs(
workflow: str | None = None,
status: str | None = None,
latest_n: int = 20,
) -> dict:
"""List recent workflow runs with optional workflow/status filters."""
return load_runs(workflow=workflow, status=status, latest_n=latest_n)
@mcp.tool()
def list_run_artifacts(run_id: str) -> dict:
"""List registered and discovered artifacts for a workflow run."""
return load_run_artifacts(run_id=run_id)
@mcp.tool()
def get_delivery_payload(run_id: str) -> dict:
"""Get the structured OpenClaw delivery payload for a workflow run."""
return load_delivery_payload(run_id=run_id)
@mcp.tool()
def get_run_report(run_id: str) -> dict:
"""Get the structured run report for a workflow run."""
return load_run_report(run_id=run_id)
@mcp.tool()
def resume_run(run_id: str) -> dict:
"""Resume a failed or interrupted FreshRSS workflow run from its latest supported recovery point."""
return resume_existing_run(run_id=run_id)
@mcp.tool()
def start_article_summary_job(
*,
extracted_path: str,
selected_ids: list[str],
output_dir: str | None = None,
max_retries: int = 2,
timeout_seconds: float = 120.0,
llm_api_key: str | None = None,
llm_model: str | None = None,
llm_api_url: str | None = None,
) -> dict:
"""Start an asynchronous article-summary job and return a job_id immediately."""
return launch_article_summary_job(
extracted_path=Path(extracted_path),
selected_ids=selected_ids,
output_dir=Path(output_dir) if output_dir else None,
max_retries=max_retries,
timeout_seconds=timeout_seconds,
llm_api_key=llm_api_key,
llm_model=llm_model,
llm_api_url=llm_api_url,
)
@mcp.tool()
def get_article_summary_job_status(job_id: str) -> dict:
"""Get the current status of an asynchronous article-summary job."""
return load_article_summary_job_status(job_id=job_id)
@mcp.tool()
def get_article_summary_job_result(job_id: str) -> dict:
"""Get the final result of an asynchronous article-summary job."""
return load_article_summary_job_result(job_id=job_id)
@mcp.tool()
def generate_article_summaries(
*,
extracted_path: str,
selected_ids: list[str],
output_dir: str | None = None,
max_retries: int = 2,
timeout_seconds: float = 120.0,
llm_api_key: str | None = None,
llm_model: str | None = None,
llm_api_url: str | None = None,
) -> list[str]:
"""Generate Markdown summaries for selected articles from an extracted payload.
Parameters
----------
extracted_path:
Path to the extracted JSON file produced by the FreshRSS pipeline
(for example `outputs/freshrss/extracted/freshrss.extracted.json`).
selected_ids:
One or more `item_id` values from the extracted payload to summarize.
output_dir:
Optional output directory for the generated Markdown files. If omitted,
summaries are written next to the extracted file under a
`single_summaries/` subdirectory.
llm_api_key / llm_model / llm_api_url:
Optional overrides for the article-summary LLM settings. If omitted,
the workflow falls back to the ARTICLE_SUMMARY_* or main LLM_* env
variables as documented in the README.
"""
extracted_path_obj = Path(extracted_path)
if not extracted_path_obj.exists():
raise FileNotFoundError(f"extracted_path does not exist: {extracted_path}")
if output_dir is None:
default_dir = extracted_path_obj.parent / "single_summaries"
output_dir_obj = default_dir
else:
output_dir_obj = Path(output_dir)
config = ArticleSummaryConfig(max_retries=max_retries, timeout_seconds=timeout_seconds)
written_paths = summarize_selected_articles(
extracted_path=extracted_path_obj,
selected_ids=selected_ids,
output_dir=output_dir_obj,
config=config,
api_key=llm_api_key,
model=llm_model,
api_url=llm_api_url,
)
return [str(p) for p in written_paths]
def main() -> None:
mcp.run()
if __name__ == "__main__":
main()