diff --git a/.gitignore b/.gitignore index 9f2b3b0..eaf30dc 100644 --- a/.gitignore +++ b/.gitignore @@ -4,6 +4,7 @@ __pycache__/ *.pyc *.egg-info/ .env.local +.env findings.md progress.md task_plan.md diff --git a/README.md b/README.md index bd54e8b..db9ed20 100644 --- a/README.md +++ b/README.md @@ -16,6 +16,11 @@ The server exposes four tools: - `filter_summary_result` - `run_freshrss_openclaw_pipeline` +Article-summary post-processing (separate LLM optional): + +- `article-summary` MCP tool (operates on existing extracted payloads) +- `scripts/run_article_summaries.py` CLI helper + Validate an LLM summary result: ```bash @@ -214,3 +219,40 @@ python scripts/apply_term_suggestions.py ^ ``` Remove `--dry-run` to write the accepted changes. The script can also apply accepted `alias`, `stopword`, and `interest keyword` suggestions through `--accept-alias`, `--accept-stopword`, and `--accept-interest`. Accepted watch terms are written into `configs/term_watchlist.json`, and every applied action is appended into `configs/term_change_log.json`. + + +## Article-summary LLM configuration + +Set a dedicated model for post-processing summaries without affecting the main pipeline: + +- `ARTICLE_SUMMARY_LLM_API_URL` +- `ARTICLE_SUMMARY_LLM_MODEL` +- `ARTICLE_SUMMARY_LLM_API_KEY` + +If these are not set, the summarizer falls back to the main `LLM_*` / `OPENAI_*` settings used elsewhere. + +Example (PowerShell style): + +```bash +set ARTICLE_SUMMARY_LLM_API_URL=https://api.deepseek.com +set ARTICLE_SUMMARY_LLM_API_KEY=your-article-summary-key +set ARTICLE_SUMMARY_LLM_MODEL=deepseek-chat +``` + +Then run, for example, either via the CLI script: + +```bash +python scripts/run_article_summaries.py ^ + --extracted outputs/freshrss/extracted/freshrss.extracted.json ^ + --ids 12345 67890 ^ + --output-dir outputs/freshrss/single_summaries +``` + +…or through the MCP server tool `generate_article_summaries` exposed by `summary_mcp.server`: + +- `extracted_path` (string): path to the extracted JSON, for example `outputs/freshrss/extracted/freshrss.extracted.json`. +- `selected_ids` (array of strings): one or more `item_id` values from the extracted payload to summarize. +- `output_dir` (optional string): directory to write Markdown summaries. If omitted, summaries are written under `single_summaries/` next to the extracted file. +- `llm_api_key` / `llm_model` / `llm_api_url` (optional strings): overrides for article-summary LLM settings. If omitted, the tool falls back to `ARTICLE_SUMMARY_*` or main `LLM_*` env vars as described above. + +The tool returns a JSON array of file paths for the generated Markdown summaries. diff --git a/scripts/run_article_summaries.py b/scripts/run_article_summaries.py new file mode 100755 index 0000000..ccd7423 --- /dev/null +++ b/scripts/run_article_summaries.py @@ -0,0 +1,68 @@ +from __future__ import annotations + +import argparse +import sys +from pathlib import Path + + +REPO_ROOT = Path(__file__).resolve().parents[1] +SRC_ROOT = REPO_ROOT / "src" + +if str(SRC_ROOT) not in sys.path: + sys.path.insert(0, str(SRC_ROOT)) + +from summary_mcp.workflows.article_summary import summarize_selected_articles + + +def main() -> None: + parser = argparse.ArgumentParser( + description="Generate Markdown summaries for selected articles from an extracted payload.", + ) + parser.add_argument( + "--extracted", + type=Path, + required=True, + help="Path to extracted JSON (e.g. outputs/freshrss/extracted/freshrss.extracted.json)", + ) + parser.add_argument( + "--ids", + nargs="+", + required=True, + help="One or more item_ids to summarize", + ) + parser.add_argument( + "--output-dir", + type=Path, + default=REPO_ROOT / "outputs" / "freshrss" / "single_summaries", + help="Directory to write per-article Markdown summaries", + ) + parser.add_argument("--max-retries", type=int, default=2, help="Maximum LLM retry attempts per article") + parser.add_argument("--timeout", type=float, default=60.0, help="LLM request timeout in seconds") + parser.add_argument("--api-key", type=str, default=None, help="Override article-summary LLM API key") + parser.add_argument("--model", type=str, default=None, help="Override article-summary LLM model") + parser.add_argument("--api-url", type=str, default=None, help="Override article-summary LLM API URL/base URL") + args = parser.parse_args() + + from summary_mcp.workflows.article_summary import ArticleSummaryConfig + + config = ArticleSummaryConfig( + max_retries=args.max_retries, + timeout_seconds=args.timeout, + ) + + paths = summarize_selected_articles( + extracted_path=args.extracted, + selected_ids=args.ids, + output_dir=args.output_dir, + config=config, + api_key=args.api_key, + model=args.model, + api_url=args.api_url, + ) + + for path in paths: + print(path) + + +if __name__ == "__main__": + main() diff --git a/src/summary_mcp/core/keyword_index.py b/src/summary_mcp/core/keyword_index.py index cfce533..9cf8372 100644 --- a/src/summary_mcp/core/keyword_index.py +++ b/src/summary_mcp/core/keyword_index.py @@ -2,7 +2,9 @@ from __future__ import annotations import json from collections import Counter -from datetime import UTC, date, datetime +from datetime import date, datetime, timezone + +UTC = timezone.utc from pathlib import Path from typing import Iterable diff --git a/src/summary_mcp/server.py b/src/summary_mcp/server.py index 8ea6398..245f642 100644 --- a/src/summary_mcp/server.py +++ b/src/summary_mcp/server.py @@ -15,6 +15,7 @@ from summary_mcp.models.item import Item from summary_mcp.models.llm_result import LlmSummaryResult from summary_mcp.models.summary_io import ExtractionInput from summary_mcp.workflows import run_freshrss_pipeline +from summary_mcp.workflows.article_summary import ArticleSummaryConfig, summarize_selected_articles mcp = FastMCP(name="content-extract-mcp") @@ -125,6 +126,62 @@ def run_freshrss_openclaw_pipeline( return result +@mcp.tool() +def generate_article_summaries( + *, + extracted_path: str, + selected_ids: list[str], + output_dir: str | None = None, + max_retries: int = 2, + timeout_seconds: float = 60.0, + llm_api_key: str | None = None, + llm_model: str | None = None, + llm_api_url: str | None = None, +) -> list[str]: + """Generate Markdown summaries for selected articles from an extracted payload. + + Parameters + ---------- + extracted_path: + Path to the extracted JSON file produced by the FreshRSS pipeline + (for example `outputs/freshrss/extracted/freshrss.extracted.json`). + selected_ids: + One or more `item_id` values from the extracted payload to summarize. + output_dir: + Optional output directory for the generated Markdown files. If omitted, + summaries are written next to the extracted file under a + `single_summaries/` subdirectory. + llm_api_key / llm_model / llm_api_url: + Optional overrides for the article-summary LLM settings. If omitted, + the workflow falls back to the ARTICLE_SUMMARY_* or main LLM_* env + variables as documented in the README. + """ + + extracted_path_obj = Path(extracted_path) + if not extracted_path_obj.exists(): + raise FileNotFoundError(f"extracted_path does not exist: {extracted_path}") + + if output_dir is None: + default_dir = extracted_path_obj.parent / "single_summaries" + output_dir_obj = default_dir + else: + output_dir_obj = Path(output_dir) + + config = ArticleSummaryConfig(max_retries=max_retries, timeout_seconds=timeout_seconds) + + written_paths = summarize_selected_articles( + extracted_path=extracted_path_obj, + selected_ids=selected_ids, + output_dir=output_dir_obj, + config=config, + api_key=llm_api_key, + model=llm_model, + api_url=llm_api_url, + ) + + return [str(p) for p in written_paths] + + def main() -> None: mcp.run() diff --git a/src/summary_mcp/workflows/article_summary.py b/src/summary_mcp/workflows/article_summary.py new file mode 100644 index 0000000..ed12506 --- /dev/null +++ b/src/summary_mcp/workflows/article_summary.py @@ -0,0 +1,216 @@ +from __future__ import annotations + +from dataclasses import dataclass +from pathlib import Path +from typing import Iterable, Mapping, Sequence + +from summary_mcp.core.summary_loop import run_loop_payload + + +REPO_ROOT = Path(__file__).resolve().parents[3] +OUTPUT_ROOT = REPO_ROOT / "outputs" +FRESHRSS_OUTPUT_ROOT = OUTPUT_ROOT / "freshrss" +DEFAULT_PROMPT_PATH = OUTPUT_ROOT / "prompts" / "llm-summary-prompt.txt" + + +@dataclass +class ArticleSummaryConfig: + prompt_path: Path = DEFAULT_PROMPT_PATH + max_retries: int = 2 + timeout_seconds: float = 60.0 + llm_api_key: str | None = None + llm_model: str | None = None + llm_api_url: str | None = None + + +def _resolve_article_llm_settings( + *, + api_key: str | None = None, + model: str | None = None, + api_url: str | None = None, +) -> tuple[str | None, str | None, str | None]: + """Resolve article-summary LLM settings with fallback to main pipeline env. + + Resolution order for each field: + - explicit function argument + - ARTICLE_SUMMARY_* environment variable + - main LLM_* / OPENAI_* environment variables (handled by summary_loop.resolve_llm_settings) + + This helper intentionally does not validate presence; the summary loop + will perform final validation and raise a clear error if nothing is set. + """ + + import os + + resolved_api_key = api_key or os.environ.get("ARTICLE_SUMMARY_LLM_API_KEY") + resolved_model = model or os.environ.get("ARTICLE_SUMMARY_LLM_MODEL") + resolved_api_url = api_url or os.environ.get("ARTICLE_SUMMARY_LLM_API_URL") + return resolved_api_key, resolved_model, resolved_api_url + + +def _iter_selected_items( + extracted_payload: Mapping[str, object], + selected_ids: Sequence[str], +) -> Iterable[tuple[str, Mapping[str, object]]]: + """Yield (item_id, extracted_entry) pairs for selected items. + + The reader pipeline currently produces an extracted payload shaped like: + + { + "results": [ + { + "item": {"item_id": "...", ...}, + "extraction": {"article": {..}, "warnings": [...]}, + }, + ... + ] + } + + Historically some flows used a top-level ``items`` array where each + element directly contained the ``article`` field. To keep compatibility + sane, we first try the real project format (``results``), then fall + back to the legacy ``items`` layout if present. + """ + + selected_set = set(selected_ids) + + # Preferred: current reader format with top-level "results" array. + results = extracted_payload.get("results") + if isinstance(results, list): + for entry in results: + if not isinstance(entry, Mapping): + continue + item = entry.get("item") + extraction = entry.get("extraction") + if not isinstance(item, Mapping) or not isinstance(extraction, Mapping): + continue + + raw_item_id = item.get("item_id") + item_id = str(raw_item_id) if raw_item_id is not None else None + if not item_id or item_id not in selected_set: + continue + + yield item_id, { + "item": item, + "extraction": extraction, + } + return + + # Fallback: legacy payload with top-level "items" array. + items = extracted_payload.get("items") + if isinstance(items, list): + for item in items: + if not isinstance(item, Mapping): + continue + raw_item_id = item.get("item_id") + item_id = str(raw_item_id) if raw_item_id is not None else None + if not item_id or item_id not in selected_set: + continue + + yield item_id, item + return + + raise RuntimeError( + "Extracted payload does not contain a supported article items structure (expected 'results' or 'items').", + ) + + +def summarize_selected_articles( + *, + extracted_path: Path, + selected_ids: Sequence[str], + output_dir: Path, + config: ArticleSummaryConfig | None = None, + api_key: str | None = None, + model: str | None = None, + api_url: str | None = None, +) -> list[Path]: + """Generate one Markdown summary per selected article. + + This operates purely on already extracted article payloads and does not + re-fetch original URLs. It expects an `extracted_path` produced by the + FreshRSS extract or pipeline steps. + + The current reader format uses a top-level ``results`` array where each + entry has ``item`` and ``extraction`` (with ``extraction.article`` being + the extracted article). For compatibility, a legacy top-level ``items`` + array is also supported if present. + """ + + import json + + cfg = config or ArticleSummaryConfig() + resolved_api_key, resolved_model, resolved_api_url = _resolve_article_llm_settings( + api_key=api_key or cfg.llm_api_key, + model=model or cfg.llm_model, + api_url=api_url or cfg.llm_api_url, + ) + + payload = json.loads(extracted_path.read_text(encoding="utf-8-sig")) + + output_dir.mkdir(parents=True, exist_ok=True) + written_paths: list[Path] = [] + + from summary_mcp.core.summary_loop import build_summary_input + + for item_id, entry in _iter_selected_items(payload, selected_ids): + # Prefer the real project format where each entry has ``item`` and + # ``extraction.article``; fall back to legacy layout where the + # article fields live directly on the element. + if "item" in entry or "extraction" in entry: + item = entry.get("item") or {} + extraction = entry.get("extraction") or {} + article = (extraction.get("article") or {}) if isinstance(extraction, dict) else {} + warnings = extraction.get("warnings", []) if isinstance(extraction, dict) else [] + else: + item = entry + article = item.get("article") or {} + warnings = item.get("warnings", []) + + extracted_payload = { + "article": article, + "warnings": warnings, + } + + summary_exit_code, summary_payload, _ = run_loop_payload( + extracted_payload=extracted_payload, + prompt_path=cfg.prompt_path, + max_retries=cfg.max_retries, + timeout_seconds=cfg.timeout_seconds, + api_key=resolved_api_key, + model=resolved_model, + api_url=resolved_api_url, + output_path=None, + ) + if summary_exit_code != 0 or summary_payload is None: + continue + + # Render a simple Markdown file summarizing the article. + title = article.get("title") or summary_payload.get("title") or item_id + url = article.get("url") + summary_text = summary_payload.get("summary") or "" + highlights = summary_payload.get("highlights") or [] + + lines: list[str] = [] + lines.append(f"# {title}") + if url: + lines.append("") + lines.append(f"Source: {url}") + lines.append("") + if summary_text: + lines.append(summary_text) + lines.append("") + if highlights: + lines.append("## Highlights") + lines.extend(f"- {h}" for h in highlights) + lines.append("") + + safe_title = "-".join( + str(title).lower().strip().replace(" ", "-").split() + )[:80] + filename = f"{safe_title or item_id}.md" + output_path = output_dir / filename + output_path.write_text("\n".join(lines), encoding="utf-8") + written_paths.append(output_path) + + return written_paths diff --git a/src/summary_mcp/workflows/freshrss_pipeline.py b/src/summary_mcp/workflows/freshrss_pipeline.py index b271eaa..eb7ee5f 100644 --- a/src/summary_mcp/workflows/freshrss_pipeline.py +++ b/src/summary_mcp/workflows/freshrss_pipeline.py @@ -2,7 +2,9 @@ from __future__ import annotations import json import os -from datetime import UTC, date, datetime +from datetime import date, datetime, timezone + +UTC = timezone.utc from pathlib import Path from typing import Any