From 52ce6bfdf5d06e3db9453a1690e573a851aff3a2 Mon Sep 17 00:00:00 2001 From: root Date: Sat, 11 Apr 2026 23:04:13 +0800 Subject: [PATCH] fix: validate article summary extracted path input --- src/summary_mcp/runtime/article_summary_jobs.py | 17 +++++++++++++---- src/summary_mcp/server.py | 4 ++-- 2 files changed, 15 insertions(+), 6 deletions(-) diff --git a/src/summary_mcp/runtime/article_summary_jobs.py b/src/summary_mcp/runtime/article_summary_jobs.py index 630c3a0..4db2cdb 100644 --- a/src/summary_mcp/runtime/article_summary_jobs.py +++ b/src/summary_mcp/runtime/article_summary_jobs.py @@ -76,6 +76,17 @@ def _load_result(job_id: str) -> dict[str, Any] | None: return json.loads(path.read_text(encoding="utf-8-sig")) +def _validate_article_summary_extracted_path(extracted_path: Path) -> None: + if not extracted_path.exists(): + raise FileNotFoundError(f"extracted_path does not exist: {extracted_path}") + if extracted_path.is_dir(): + raise ValueError( + "extracted_path must be a JSON file, not a directory. " + "Pass a single extracted file like outputs/freshrss/rerun//extracted/item-01.extracted.json, " + "or a batch extracted JSON file." + ) + + def start_article_summary_job( *, extracted_path: Path, @@ -87,8 +98,7 @@ def start_article_summary_job( llm_model: str | None = None, llm_api_url: str | None = None, ) -> dict[str, Any]: - if not extracted_path.exists(): - raise FileNotFoundError(f"extracted_path does not exist: {extracted_path}") + _validate_article_summary_extracted_path(extracted_path) if not selected_ids: raise ValueError("selected_ids must not be empty") @@ -166,8 +176,7 @@ def run_article_summary_job(*, job_id: str) -> dict[str, Any]: extracted_path = Path(input_payload["extracted_path"]) output_dir = Path(input_payload["output_dir"]) selected_ids = list(input_payload["selected_ids"]) - if not extracted_path.exists(): - raise FileNotFoundError(f"extracted_path does not exist: {extracted_path}") + _validate_article_summary_extracted_path(extracted_path) if not selected_ids: raise ValueError("selected_ids must not be empty") store.finish_stage( diff --git a/src/summary_mcp/server.py b/src/summary_mcp/server.py index 2b9d0ed..8239ac8 100644 --- a/src/summary_mcp/server.py +++ b/src/summary_mcp/server.py @@ -27,6 +27,7 @@ from summary_mcp.runtime.article_summary_jobs import ( get_article_summary_job_result as load_article_summary_job_result, get_article_summary_job_status as load_article_summary_job_status, start_article_summary_job as launch_article_summary_job, + _validate_article_summary_extracted_path, ) from summary_mcp.runtime.resume_service import resume_run as resume_existing_run from summary_mcp.workflows import run_freshrss_pipeline @@ -300,8 +301,7 @@ def generate_article_summaries( """ extracted_path_obj = Path(extracted_path) - if not extracted_path_obj.exists(): - raise FileNotFoundError(f"extracted_path does not exist: {extracted_path}") + _validate_article_summary_extracted_path(extracted_path_obj) if output_dir is None: default_dir = extracted_path_obj.parent / "single_summaries"