271 lines
9.1 KiB
Python
271 lines
9.1 KiB
Python
from __future__ import annotations
|
|
|
|
# MCP 服务入口:将内容提取、过滤、FreshRSS 全链路管道暴露为 MCP 工具。
|
|
# 生产主入口是 run_freshrss_openclaw_pipeline,其余工具供单步调试使用。
|
|
|
|
from datetime import date
|
|
from pathlib import Path
|
|
|
|
from mcp.server.fastmcp import FastMCP
|
|
|
|
from summary_mcp.core.pipeline import extract_content
|
|
from summary_mcp.filters.engine import evaluate_filter_rules, load_filter_rules
|
|
from summary_mcp.models.document import ExtractedArticle
|
|
from summary_mcp.models.filtering import FilterContext, FilterInput
|
|
from summary_mcp.models.item import Item
|
|
from summary_mcp.models.llm_result import LlmSummaryResult
|
|
from summary_mcp.models.summary_io import ExtractionInput
|
|
from summary_mcp.runtime import get_delivery_payload as load_delivery_payload
|
|
from summary_mcp.runtime import get_run_status as load_run_status
|
|
from summary_mcp.runtime import get_run_report as load_run_report
|
|
from summary_mcp.runtime import list_run_artifacts as load_run_artifacts
|
|
from summary_mcp.runtime import list_runs as load_runs
|
|
from summary_mcp.runtime.article_summary_jobs import (
|
|
get_article_summary_job_result as load_article_summary_job_result,
|
|
get_article_summary_job_status as load_article_summary_job_status,
|
|
start_article_summary_job as launch_article_summary_job,
|
|
)
|
|
from summary_mcp.runtime.resume_service import resume_run as resume_existing_run
|
|
from summary_mcp.workflows import run_freshrss_pipeline
|
|
from summary_mcp.workflows.article_summary import ArticleSummaryConfig, summarize_selected_articles
|
|
|
|
|
|
mcp = FastMCP(name="content-extract-mcp")
|
|
|
|
|
|
@mcp.tool()
|
|
def extract_url_content(url: str, language_hint: str | None = None) -> dict:
|
|
# 从单个 URL 抓取并提取结构化文章内容,供单步调试使用
|
|
"""Extract structured article content from a single URL."""
|
|
result = extract_content(
|
|
ExtractionInput(
|
|
url=url,
|
|
language_hint=language_hint,
|
|
)
|
|
)
|
|
return result.model_dump(mode="json")
|
|
|
|
|
|
@mcp.tool()
|
|
def extract_item_content(item: dict) -> dict:
|
|
# 从已标准化的 item 对象提取内容(跳过网络抓取,使用 RSS 内联内容)
|
|
"""Extract structured article content from a normalized item object."""
|
|
parsed_item = Item.model_validate(item)
|
|
result = extract_content(ExtractionInput(item=parsed_item))
|
|
return result.model_dump(mode="json")
|
|
|
|
|
|
@mcp.tool()
|
|
def filter_summary_result(
|
|
summary_result: dict,
|
|
extracted_article: dict | None = None,
|
|
item: dict | None = None,
|
|
context: dict | None = None,
|
|
) -> dict:
|
|
# 对结构化摘要结果运行确定性规则引擎,返回 keep/review/drop 决策
|
|
"""Apply deterministic filter rules to a structured summary result."""
|
|
parsed_summary = LlmSummaryResult.model_validate(summary_result)
|
|
parsed_article = ExtractedArticle.model_validate(extracted_article) if extracted_article else None
|
|
parsed_item = Item.model_validate(item) if item else None
|
|
parsed_context = FilterContext.model_validate(context or {})
|
|
rules = load_filter_rules()
|
|
decision = evaluate_filter_rules(
|
|
FilterInput(
|
|
item=parsed_item,
|
|
article=parsed_article,
|
|
summary=parsed_summary,
|
|
context=parsed_context,
|
|
),
|
|
rules,
|
|
)
|
|
return decision.model_dump(mode="json")
|
|
|
|
|
|
@mcp.tool()
|
|
def run_freshrss_openclaw_pipeline(
|
|
limit: int = 5,
|
|
mark_read: bool = False,
|
|
include_read: bool = False,
|
|
debug_artifacts: bool = False,
|
|
continuation: str | None = None,
|
|
timeout_seconds: float = 60.0,
|
|
max_retries: int = 2,
|
|
stream_id: str = "user/-/state/com.google/reading-list",
|
|
api_base_url: str | None = None,
|
|
username: str | None = None,
|
|
api_password: str | None = None,
|
|
llm_api_key: str | None = None,
|
|
llm_model: str | None = None,
|
|
llm_api_url: str | None = None,
|
|
context: dict | None = None,
|
|
run_id: str | None = None,
|
|
date_value: str | None = None,
|
|
output_dir: str | None = None,
|
|
include_item_reports: bool = False,
|
|
) -> dict:
|
|
"""Run the full FreshRSS -> extract -> LLM -> filter -> OpenClaw payload pipeline."""
|
|
result = run_freshrss_pipeline(
|
|
api_base_url=api_base_url,
|
|
username=username,
|
|
api_password=api_password,
|
|
stream_id=stream_id,
|
|
limit=limit,
|
|
continuation=continuation,
|
|
include_read=include_read,
|
|
mark_read=mark_read,
|
|
debug_artifacts=debug_artifacts,
|
|
context=context,
|
|
max_retries=max_retries,
|
|
timeout_seconds=timeout_seconds,
|
|
llm_api_key=llm_api_key,
|
|
llm_model=llm_model,
|
|
llm_api_url=llm_api_url,
|
|
run_id=run_id,
|
|
delivery_date=date.fromisoformat(date_value) if date_value else None,
|
|
output_dir=Path(output_dir) if output_dir else None,
|
|
)
|
|
|
|
if not include_item_reports:
|
|
result = {key: value for key, value in result.items() if key != "items"}
|
|
return result
|
|
|
|
|
|
@mcp.tool()
|
|
def get_run_status(run_id: str) -> dict:
|
|
"""Get the current status of a workflow run by run_id."""
|
|
return load_run_status(run_id=run_id)
|
|
|
|
|
|
@mcp.tool()
|
|
def list_runs(
|
|
workflow: str | None = None,
|
|
status: str | None = None,
|
|
latest_n: int = 20,
|
|
) -> dict:
|
|
"""List recent workflow runs with optional workflow/status filters."""
|
|
return load_runs(workflow=workflow, status=status, latest_n=latest_n)
|
|
|
|
|
|
@mcp.tool()
|
|
def list_run_artifacts(run_id: str) -> dict:
|
|
"""List registered and discovered artifacts for a workflow run."""
|
|
return load_run_artifacts(run_id=run_id)
|
|
|
|
|
|
@mcp.tool()
|
|
def get_delivery_payload(run_id: str) -> dict:
|
|
"""Get the structured OpenClaw delivery payload for a workflow run."""
|
|
return load_delivery_payload(run_id=run_id)
|
|
|
|
|
|
@mcp.tool()
|
|
def get_run_report(run_id: str) -> dict:
|
|
"""Get the structured run report for a workflow run."""
|
|
return load_run_report(run_id=run_id)
|
|
|
|
|
|
@mcp.tool()
|
|
def resume_run(run_id: str) -> dict:
|
|
"""Resume a failed or interrupted FreshRSS workflow run from its latest supported recovery point."""
|
|
return resume_existing_run(run_id=run_id)
|
|
|
|
@mcp.tool()
|
|
def start_article_summary_job(
|
|
*,
|
|
extracted_path: str,
|
|
selected_ids: list[str],
|
|
output_dir: str | None = None,
|
|
max_retries: int = 2,
|
|
timeout_seconds: float = 120.0,
|
|
llm_api_key: str | None = None,
|
|
llm_model: str | None = None,
|
|
llm_api_url: str | None = None,
|
|
) -> dict:
|
|
"""Start an asynchronous article-summary job and return a job_id immediately."""
|
|
return launch_article_summary_job(
|
|
extracted_path=Path(extracted_path),
|
|
selected_ids=selected_ids,
|
|
output_dir=Path(output_dir) if output_dir else None,
|
|
max_retries=max_retries,
|
|
timeout_seconds=timeout_seconds,
|
|
llm_api_key=llm_api_key,
|
|
llm_model=llm_model,
|
|
llm_api_url=llm_api_url,
|
|
)
|
|
|
|
|
|
@mcp.tool()
|
|
def get_article_summary_job_status(job_id: str) -> dict:
|
|
"""Get the current status of an asynchronous article-summary job."""
|
|
return load_article_summary_job_status(job_id=job_id)
|
|
|
|
|
|
@mcp.tool()
|
|
def get_article_summary_job_result(job_id: str) -> dict:
|
|
"""Get the final result of an asynchronous article-summary job."""
|
|
return load_article_summary_job_result(job_id=job_id)
|
|
|
|
|
|
@mcp.tool()
|
|
def generate_article_summaries(
|
|
*,
|
|
extracted_path: str,
|
|
selected_ids: list[str],
|
|
output_dir: str | None = None,
|
|
max_retries: int = 2,
|
|
timeout_seconds: float = 120.0,
|
|
llm_api_key: str | None = None,
|
|
llm_model: str | None = None,
|
|
llm_api_url: str | None = None,
|
|
) -> list[str]:
|
|
"""Generate Markdown summaries for selected articles from an extracted payload.
|
|
|
|
Parameters
|
|
----------
|
|
extracted_path:
|
|
Path to the extracted JSON file produced by the FreshRSS pipeline
|
|
(for example `outputs/freshrss/extracted/freshrss.extracted.json`).
|
|
selected_ids:
|
|
One or more `item_id` values from the extracted payload to summarize.
|
|
output_dir:
|
|
Optional output directory for the generated Markdown files. If omitted,
|
|
summaries are written next to the extracted file under a
|
|
`single_summaries/` subdirectory.
|
|
llm_api_key / llm_model / llm_api_url:
|
|
Optional overrides for the article-summary LLM settings. If omitted,
|
|
the workflow falls back to the ARTICLE_SUMMARY_* or main LLM_* env
|
|
variables as documented in the README.
|
|
"""
|
|
|
|
extracted_path_obj = Path(extracted_path)
|
|
if not extracted_path_obj.exists():
|
|
raise FileNotFoundError(f"extracted_path does not exist: {extracted_path}")
|
|
|
|
if output_dir is None:
|
|
default_dir = extracted_path_obj.parent / "single_summaries"
|
|
output_dir_obj = default_dir
|
|
else:
|
|
output_dir_obj = Path(output_dir)
|
|
|
|
config = ArticleSummaryConfig(max_retries=max_retries, timeout_seconds=timeout_seconds)
|
|
|
|
written_paths = summarize_selected_articles(
|
|
extracted_path=extracted_path_obj,
|
|
selected_ids=selected_ids,
|
|
output_dir=output_dir_obj,
|
|
config=config,
|
|
api_key=llm_api_key,
|
|
model=llm_model,
|
|
api_url=llm_api_url,
|
|
)
|
|
|
|
return [str(p) for p in written_paths]
|
|
|
|
|
|
def main() -> None:
|
|
mcp.run()
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|