from __future__ import annotations # MCP 服务入口:将内容提取、过滤、FreshRSS 全链路管道暴露为 MCP 工具。 # 生产主入口是 run_freshrss_openclaw_pipeline,其余工具供单步调试使用。 from datetime import date from pathlib import Path from mcp.server.fastmcp import FastMCP from summary_mcp.core.pipeline import extract_content from summary_mcp.filters.engine import evaluate_filter_rules, load_filter_rules from summary_mcp.models.document import ExtractedArticle from summary_mcp.models.filtering import FilterContext, FilterInput from summary_mcp.models.item import Item from summary_mcp.models.llm_result import LlmSummaryResult from summary_mcp.models.summary_io import ExtractionInput from summary_mcp.runtime import get_delivery_payload as load_delivery_payload from summary_mcp.runtime import get_run_status as load_run_status from summary_mcp.runtime import get_run_report as load_run_report from summary_mcp.runtime import list_run_artifacts as load_run_artifacts from summary_mcp.runtime import list_runs as load_runs from summary_mcp.runtime.article_summary_jobs import ( get_article_summary_job_result as load_article_summary_job_result, get_article_summary_job_status as load_article_summary_job_status, start_article_summary_job as launch_article_summary_job, ) from summary_mcp.runtime.resume_service import resume_run as resume_existing_run from summary_mcp.workflows import run_freshrss_pipeline from summary_mcp.workflows.article_summary import ArticleSummaryConfig, summarize_selected_articles mcp = FastMCP(name="content-extract-mcp") @mcp.tool() def extract_url_content(url: str, language_hint: str | None = None) -> dict: # 从单个 URL 抓取并提取结构化文章内容,供单步调试使用 """Extract structured article content from a single URL.""" result = extract_content( ExtractionInput( url=url, language_hint=language_hint, ) ) return result.model_dump(mode="json") @mcp.tool() def extract_item_content(item: dict) -> dict: # 从已标准化的 item 对象提取内容(跳过网络抓取,使用 RSS 内联内容) """Extract structured article content from a normalized item object.""" parsed_item = Item.model_validate(item) result = extract_content(ExtractionInput(item=parsed_item)) return result.model_dump(mode="json") @mcp.tool() def filter_summary_result( summary_result: dict, extracted_article: dict | None = None, item: dict | None = None, context: dict | None = None, ) -> dict: # 对结构化摘要结果运行确定性规则引擎,返回 keep/review/drop 决策 """Apply deterministic filter rules to a structured summary result.""" parsed_summary = LlmSummaryResult.model_validate(summary_result) parsed_article = ExtractedArticle.model_validate(extracted_article) if extracted_article else None parsed_item = Item.model_validate(item) if item else None parsed_context = FilterContext.model_validate(context or {}) rules = load_filter_rules() decision = evaluate_filter_rules( FilterInput( item=parsed_item, article=parsed_article, summary=parsed_summary, context=parsed_context, ), rules, ) return decision.model_dump(mode="json") @mcp.tool() def run_freshrss_openclaw_pipeline( limit: int = 5, mark_read: bool = False, include_read: bool = False, debug_artifacts: bool = False, continuation: str | None = None, timeout_seconds: float = 60.0, max_retries: int = 2, stream_id: str = "user/-/state/com.google/reading-list", api_base_url: str | None = None, username: str | None = None, api_password: str | None = None, llm_api_key: str | None = None, llm_model: str | None = None, llm_api_url: str | None = None, context: dict | None = None, run_id: str | None = None, date_value: str | None = None, output_dir: str | None = None, include_item_reports: bool = False, ) -> dict: """Run the full FreshRSS -> extract -> LLM -> filter -> OpenClaw payload pipeline.""" result = run_freshrss_pipeline( api_base_url=api_base_url, username=username, api_password=api_password, stream_id=stream_id, limit=limit, continuation=continuation, include_read=include_read, mark_read=mark_read, debug_artifacts=debug_artifacts, context=context, max_retries=max_retries, timeout_seconds=timeout_seconds, llm_api_key=llm_api_key, llm_model=llm_model, llm_api_url=llm_api_url, run_id=run_id, delivery_date=date.fromisoformat(date_value) if date_value else None, output_dir=Path(output_dir) if output_dir else None, ) if not include_item_reports: result = {key: value for key, value in result.items() if key != "items"} return result @mcp.tool() def get_run_status(run_id: str) -> dict: """Get the current status of a workflow run by run_id.""" return load_run_status(run_id=run_id) @mcp.tool() def list_runs( workflow: str | None = None, status: str | None = None, latest_n: int = 20, ) -> dict: """List recent workflow runs with optional workflow/status filters.""" return load_runs(workflow=workflow, status=status, latest_n=latest_n) @mcp.tool() def list_run_artifacts(run_id: str) -> dict: """List registered and discovered artifacts for a workflow run.""" return load_run_artifacts(run_id=run_id) @mcp.tool() def get_delivery_payload(run_id: str) -> dict: """Get the structured OpenClaw delivery payload for a workflow run.""" return load_delivery_payload(run_id=run_id) @mcp.tool() def get_run_report(run_id: str) -> dict: """Get the structured run report for a workflow run.""" return load_run_report(run_id=run_id) @mcp.tool() def resume_run(run_id: str) -> dict: """Resume a failed or interrupted FreshRSS workflow run from its latest supported recovery point.""" return resume_existing_run(run_id=run_id) @mcp.tool() def start_article_summary_job( *, extracted_path: str, selected_ids: list[str], output_dir: str | None = None, max_retries: int = 2, timeout_seconds: float = 120.0, llm_api_key: str | None = None, llm_model: str | None = None, llm_api_url: str | None = None, ) -> dict: """Start an asynchronous article-summary job and return a job_id immediately.""" return launch_article_summary_job( extracted_path=Path(extracted_path), selected_ids=selected_ids, output_dir=Path(output_dir) if output_dir else None, max_retries=max_retries, timeout_seconds=timeout_seconds, llm_api_key=llm_api_key, llm_model=llm_model, llm_api_url=llm_api_url, ) @mcp.tool() def get_article_summary_job_status(job_id: str) -> dict: """Get the current status of an asynchronous article-summary job.""" return load_article_summary_job_status(job_id=job_id) @mcp.tool() def get_article_summary_job_result(job_id: str) -> dict: """Get the final result of an asynchronous article-summary job.""" return load_article_summary_job_result(job_id=job_id) @mcp.tool() def generate_article_summaries( *, extracted_path: str, selected_ids: list[str], output_dir: str | None = None, max_retries: int = 2, timeout_seconds: float = 120.0, llm_api_key: str | None = None, llm_model: str | None = None, llm_api_url: str | None = None, ) -> list[str]: """Generate Markdown summaries for selected articles from an extracted payload. Parameters ---------- extracted_path: Path to the extracted JSON file produced by the FreshRSS pipeline (for example `outputs/freshrss/extracted/freshrss.extracted.json`). selected_ids: One or more `item_id` values from the extracted payload to summarize. output_dir: Optional output directory for the generated Markdown files. If omitted, summaries are written next to the extracted file under a `single_summaries/` subdirectory. llm_api_key / llm_model / llm_api_url: Optional overrides for the article-summary LLM settings. If omitted, the workflow falls back to the ARTICLE_SUMMARY_* or main LLM_* env variables as documented in the README. """ extracted_path_obj = Path(extracted_path) if not extracted_path_obj.exists(): raise FileNotFoundError(f"extracted_path does not exist: {extracted_path}") if output_dir is None: default_dir = extracted_path_obj.parent / "single_summaries" output_dir_obj = default_dir else: output_dir_obj = Path(output_dir) config = ArticleSummaryConfig(max_retries=max_retries, timeout_seconds=timeout_seconds) written_paths = summarize_selected_articles( extracted_path=extracted_path_obj, selected_ids=selected_ids, output_dir=output_dir_obj, config=config, api_key=llm_api_key, model=llm_model, api_url=llm_api_url, ) return [str(p) for p in written_paths] def main() -> None: mcp.run() if __name__ == "__main__": main()