from __future__ import annotations import json import os from datetime import date, datetime, timezone from pathlib import Path from typing import Any from dotenv import dotenv_values from summary_mcp.core.keyword_index import persist_keyword_indexes from summary_mcp.core.pipeline import extract_content from summary_mcp.core.summary_loop import resolve_llm_settings, run_loop_payload from summary_mcp.filters.engine import evaluate_filter_rules, load_filter_rules from summary_mcp.integrations.freshrss import READ_TAG, FreshRSSClient, map_entry_to_item from summary_mcp.models.article_candidate import ( CandidateMetadata, CandidateSourceRefs, OpenClawCandidateInput, build_article_candidate_record, build_openclaw_candidate_input, ) from summary_mcp.models.filtering import FilterContext, FilterInput from summary_mcp.models.llm_result import LlmSummaryResult from summary_mcp.models.openclaw_delivery import ( OpenClawDeliveryPayload, build_openclaw_delivery_payload, build_openclaw_digest_brief, ) from summary_mcp.models.summary_io import ExtractionInput from summary_mcp.runtime import RunStore UTC = timezone.utc REPO_ROOT = Path(__file__).resolve().parents[3] OUTPUT_ROOT = REPO_ROOT / "outputs" FRESHRSS_OUTPUT_ROOT = OUTPUT_ROOT / "freshrss" DATA_ROOT = REPO_ROOT / "data" / "term_index" DEFAULT_PROMPT_PATH = OUTPUT_ROOT / "prompts" / "llm-summary-prompt.txt" DEFAULT_RULES_PATH = REPO_ROOT / "configs" / "filter_rules.json" DEFAULT_TERM_ALIASES_PATH = REPO_ROOT / "configs" / "term_aliases.json" DEFAULT_TERM_STOPWORDS_PATH = REPO_ROOT / "configs" / "term_stopwords.json" DEFAULT_TERM_DAILY_DIR = DATA_ROOT / "daily" DEFAULT_TERM_STATS_PATH = DATA_ROOT / "term_stats.json" DEFAULT_DOTENV_PATH = REPO_ROOT / ".env" WORKFLOW_NAME = "freshrss_daily_digest" RUN_TYPE = "daily_digest" FETCH_STAGE = "fetch_feed" EXTRACT_STAGE = "extract_articles" SUMMARY_STAGE = "generate_summaries" FILTER_STAGE = "apply_filters" DELIVERY_STAGE = "build_delivery_payload" REPORT_STAGE = "write_run_report" SUMMARY_BATCH_ARTIFACT = "summary_batch" CANDIDATE_BATCH_ARTIFACT = "candidate_batch" SUMMARY_BATCH_FILENAME = "summary-batch.json" CANDIDATE_BATCH_FILENAME = "candidate-batch.json" def _save_json(path: Path, payload: dict[str, Any] | list[Any]) -> None: path.parent.mkdir(parents=True, exist_ok=True) path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8") def _load_json(path: Path) -> dict[str, Any]: return json.loads(path.read_text(encoding="utf-8-sig")) def _load_repo_dotenv() -> dict[str, str]: if not DEFAULT_DOTENV_PATH.exists(): return {} return { key: value for key, value in dotenv_values(DEFAULT_DOTENV_PATH).items() if isinstance(key, str) and isinstance(value, str) and value } def _load_required_env(name: str, value: str | None, dotenv_map: dict[str, str] | None = None) -> str: if value: return value env_value = os.environ.get(name) if env_value: return env_value dotenv_value = (dotenv_map or {}).get(name) if dotenv_value: return dotenv_value raise RuntimeError( f"Missing required value '{name}': not passed as argument, not set as environment variable, and not found in {DEFAULT_DOTENV_PATH}." ) def default_output_dir() -> Path: return FRESHRSS_OUTPUT_ROOT / "rerun" / datetime.now(tz=UTC).strftime("%Y%m%d-%H%M%S") def _maybe_path(enabled: bool, path: Path) -> Path | None: return path if enabled else None def _build_item_context(*, index: int, item: Any, resolved_output_dir: Path, debug_artifacts: bool) -> dict[str, Any]: item_key = f"item-{index:02d}" item_path = _maybe_path(debug_artifacts, resolved_output_dir / "items" / f"{item_key}.item.json") extracted_path = resolved_output_dir / "extracted" / f"{item_key}.extracted.json" summary_output = _maybe_path(debug_artifacts, resolved_output_dir / "summary" / item_key / "result.loop.json") filter_path = _maybe_path(debug_artifacts, resolved_output_dir / "filter" / f"{item_key}.filter.json") record_path = _maybe_path(debug_artifacts, resolved_output_dir / "candidates" / f"{item_key}.article-candidate-record.json") openclaw_path = _maybe_path(debug_artifacts, resolved_output_dir / "candidates" / f"{item_key}.openclaw-candidate-input.json") item_report: dict[str, Any] = { "item_key": item_key, "item_id": item.item_id, "external_id": item.external_id, "url": str(item.url), "title": item.title, "status": "pulled", } if debug_artifacts: item_report["paths"] = { "item": str(item_path) if item_path else None, "extracted": str(extracted_path), "summary": str(summary_output) if summary_output else None, "filter": str(filter_path) if filter_path else None, "article_candidate": str(record_path) if record_path else None, "openclaw_candidate": str(openclaw_path) if openclaw_path else None, } return { "item": item, "item_key": item_key, "item_path": item_path, "extracted_path": extracted_path, "summary_output": summary_output, "filter_path": filter_path, "record_path": record_path, "openclaw_path": openclaw_path, "item_report": item_report, "extraction": None, "extracted_payload": None, "summary_payload": None, } def _summary_batch_output(run_dir: Path) -> Path: return run_dir / "summary" / SUMMARY_BATCH_FILENAME def _candidate_batch_output(run_dir: Path) -> Path: return run_dir / "candidates" / CANDIDATE_BATCH_FILENAME def _build_summary_batch_payload(*, run_id: str, item_contexts: list[dict[str, Any]]) -> dict[str, Any]: items: list[dict[str, Any]] = [] for item_context in item_contexts: summary_payload = item_context.get("summary_payload") if summary_payload is None: continue item = item_context["item"] items.append( { "item_key": item_context["item_key"], "item_id": item.item_id, "summary": summary_payload, } ) return { "run_id": run_id, "summary_count": len(items), "items": items, } def _build_candidate_batch_payload(*, run_id: str, item_contexts: list[dict[str, Any]]) -> dict[str, Any]: items: list[dict[str, Any]] = [] for item_context in item_contexts: candidate = item_context.get("candidate") if candidate is None: continue item = item_context["item"] items.append( { "item_key": item_context["item_key"], "item_id": item.item_id, "candidate_id": candidate.candidate_id, "candidate": candidate.model_dump(mode="json"), } ) return { "run_id": run_id, "candidate_count": len(items), "items": items, } def _persist_summary_batch_artifact(*, run_store: RunStore, run_dir: Path, item_contexts: list[dict[str, Any]]) -> Path: output_path = _summary_batch_output(run_dir) _save_json( output_path, _build_summary_batch_payload(run_id=run_store.state.run_id, item_contexts=item_contexts), ) run_store.register_artifact(name=SUMMARY_BATCH_ARTIFACT, path=output_path, kind="json", stage=SUMMARY_STAGE) return output_path def _persist_candidate_batch_artifact(*, run_store: RunStore, run_dir: Path, item_contexts: list[dict[str, Any]]) -> Path: output_path = _candidate_batch_output(run_dir) _save_json( output_path, _build_candidate_batch_payload(run_id=run_store.state.run_id, item_contexts=item_contexts), ) run_store.register_artifact(name=CANDIDATE_BATCH_ARTIFACT, path=output_path, kind="json", stage=FILTER_STAGE) return output_path def _build_run_report( *, resolved_run_id: str, started_at: datetime, limit: int, items: list[Any], delivered_candidates: list[OpenClawCandidateInput], marked_count: int, mark_read: bool, debug_artifacts: bool, raw_output: Path, delivery_output: Path, digest_brief_output: Path, keyword_index_result: dict[str, Any], item_reports: list[dict[str, Any]], ) -> dict[str, Any]: status_counts: dict[str, int] = {} for item_report in item_reports: status = str(item_report["status"]) status_counts[status] = status_counts.get(status, 0) + 1 return { "run_id": resolved_run_id, "started_at": started_at.isoformat(), "completed_at": datetime.now(tz=UTC).isoformat(), "requested_limit": limit, "pulled_count": len(items), "delivered_count": len(delivered_candidates), "marked_read_count": marked_count, "mark_read_requested": mark_read, "debug_artifacts": debug_artifacts, "raw_output": str(raw_output), "delivery_output": str(delivery_output), "digest_brief_output": str(digest_brief_output), "keyword_index": keyword_index_result, "status_counts": status_counts, "items": item_reports, } def _final_run_status(item_reports: list[dict[str, Any]]) -> str: if any(item_report.get("status") in {"extract_failed", "summary_failed"} for item_report in item_reports): return "partial" return "success" def run_freshrss_pipeline( *, api_base_url: str | None = None, username: str | None = None, api_password: str | None = None, stream_id: str = "user/-/state/com.google/reading-list", limit: int = 5, continuation: str | None = None, include_read: bool = False, mark_read: bool = False, debug_artifacts: bool = False, prompt: Path | None = None, rules: Path | None = None, context: dict[str, Any] | None = None, context_path: Path | None = None, max_retries: int = 2, timeout_seconds: float = 60.0, llm_api_key: str | None = None, llm_model: str | None = None, llm_api_url: str | None = None, run_id: str | None = None, delivery_date: date | None = None, output_dir: Path | None = None, ) -> dict[str, Any]: started_at = datetime.now(tz=UTC) resolved_output_dir = output_dir or default_output_dir() run_stamp = started_at.strftime("%Y%m%d-%H%M%S") resolved_run_id = run_id or f"freshrss-pipeline-{run_stamp}" resolved_delivery_date = delivery_date or datetime.now(tz=UTC).date() resolved_prompt_path = prompt or DEFAULT_PROMPT_PATH resolved_rules_path = rules or DEFAULT_RULES_PATH raw_output = resolved_output_dir / "raw" / "freshrss.raw.json" delivery_output = resolved_output_dir / "candidates" / "openclaw-delivery-payload.json" digest_brief_output = delivery_output.with_name("digest-brief.json") report_output = resolved_output_dir / "run-report.json" run_state_output = resolved_output_dir / "run-state.json" items_list_output = _maybe_path(debug_artifacts, resolved_output_dir / "items" / "freshrss.items.json") run_store = RunStore.create( path=run_state_output, run_id=resolved_run_id, workflow=WORKFLOW_NAME, run_type=RUN_TYPE, started_at=started_at, input_payload={ "limit": limit, "mark_read": mark_read, "include_read": include_read, "debug_artifacts": debug_artifacts, "continuation": continuation, "stream_id": stream_id, "timeout_seconds": timeout_seconds, "max_retries": max_retries, "delivery_date": resolved_delivery_date.isoformat(), "prompt": str(resolved_prompt_path), "rules": str(resolved_rules_path), "context": context, "context_path": str(context_path) if context_path else None, }, repo_root=REPO_ROOT, ) run_store.save() client: FreshRSSClient | None = None auth_token: str | None = None items: list[Any] = [] item_contexts: list[dict[str, Any]] = [] item_reports: list[dict[str, Any]] = [] delivered_candidates: list[OpenClawCandidateInput] = [] delivered_item_ids: list[str] = [] keyword_index_result: dict[str, Any] = {} marked_count = 0 try: run_store.start_stage(FETCH_STAGE, outputs={"output_dir": str(resolved_output_dir)}) dotenv_map = _load_repo_dotenv() resolved_api_base_url = _load_required_env("FRESHRSS_API_BASE_URL", api_base_url, dotenv_map) resolved_username = _load_required_env("FRESHRSS_USERNAME", username, dotenv_map) resolved_api_password = _load_required_env("FRESHRSS_API_PASSWORD", api_password, dotenv_map) resolved_llm_api_key, resolved_llm_model, resolved_llm_api_url = resolve_llm_settings( api_key=llm_api_key, model=llm_model, api_url=llm_api_url, ) client = FreshRSSClient( api_base_url=resolved_api_base_url, username=resolved_username, api_password=resolved_api_password, timeout_seconds=timeout_seconds, ) auth_token = client.client_login() payload = client.fetch_stream_contents( auth_token=auth_token, stream_id=stream_id, limit=limit, continuation=continuation, exclude_targets=[] if include_read else [READ_TAG], ) entries = payload.get("items") if not isinstance(entries, list): raise RuntimeError("FreshRSS stream response does not contain an items array.") _save_json(raw_output, payload) run_store.register_artifact(name="raw_output", path=raw_output, kind="json", stage=FETCH_STAGE) items = [map_entry_to_item(entry) for entry in entries] if items_list_output is not None: _save_json(items_list_output, [item.model_dump(mode="json") for item in items]) run_store.register_artifact(name="items_output", path=items_list_output, kind="json", stage=FETCH_STAGE) loaded_rules = load_filter_rules(resolved_rules_path) if context is not None: filter_context = FilterContext.model_validate(context) elif context_path is not None: filter_context = FilterContext.model_validate(_load_json(context_path)) else: filter_context = FilterContext() run_store.finish_stage( FETCH_STAGE, outputs={ "pulled_count": len(items), "raw_output": str(raw_output), "items_output": str(items_list_output) if items_list_output else None, }, ) run_store.start_stage( EXTRACT_STAGE, outputs={ "expected_items": len(items), "completed_items": 0, "success_count": 0, "failed_count": 0, }, ) extracted_success_count = 0 extracted_failed_count = 0 for index, item in enumerate(items, start=1): item_context = _build_item_context( index=index, item=item, resolved_output_dir=resolved_output_dir, debug_artifacts=debug_artifacts, ) item_contexts.append(item_context) item_report = item_context["item_report"] item_reports.append(item_report) item_path = item_context["item_path"] if item_path is not None: _save_json(item_path, item.model_dump(mode="json")) extraction = extract_content(ExtractionInput(item=item)) extracted_payload = extraction.model_dump(mode="json") item_context["extraction"] = extraction item_context["extracted_payload"] = extracted_payload _save_json(item_context["extracted_path"], extracted_payload) run_store.register_artifact(name="extracted_dir", path=resolved_output_dir / "extracted", kind="directory", stage=EXTRACT_STAGE) if not extraction.success or extraction.article is None: item_report["status"] = "extract_failed" item_report["error"] = extraction.error.model_dump(mode="json") if extraction.error else None extracted_failed_count += 1 else: item_report["status"] = "extracted" extracted_success_count += 1 run_store.update_stage( EXTRACT_STAGE, outputs={ "expected_items": len(items), "completed_items": extracted_success_count + extracted_failed_count, "success_count": extracted_success_count, "failed_count": extracted_failed_count, }, ) run_store.finish_stage( EXTRACT_STAGE, outputs={ "expected_items": len(items), "completed_items": extracted_success_count + extracted_failed_count, "success_count": extracted_success_count, "failed_count": extracted_failed_count, "extracted_dir": str(resolved_output_dir / "extracted"), }, ) run_store.start_stage( SUMMARY_STAGE, outputs={ "expected_items": extracted_success_count, "completed_items": 0, "success_count": 0, "failed_count": 0, }, ) summary_success_count = 0 summary_failed_count = 0 summary_candidates = [ctx for ctx in item_contexts if ctx["extraction"] is not None and ctx["extraction"].success] for item_context in summary_candidates: item_report = item_context["item_report"] summary_exit_code, summary_payload, summary_report = run_loop_payload( extracted_payload=item_context["extracted_payload"], prompt_path=resolved_prompt_path, output_path=item_context["summary_output"], max_retries=max_retries, timeout_seconds=timeout_seconds, api_key=resolved_llm_api_key, model=resolved_llm_model, api_url=resolved_llm_api_url, ) if summary_exit_code != 0 or summary_payload is None: item_report["status"] = "summary_failed" if summary_report is not None: item_report["summary_errors"] = summary_report.errors summary_failed_count += 1 else: item_context["summary_payload"] = summary_payload item_report["status"] = "summarized" summary_success_count += 1 run_store.update_stage( SUMMARY_STAGE, outputs={ "expected_items": extracted_success_count, "completed_items": summary_success_count + summary_failed_count, "success_count": summary_success_count, "failed_count": summary_failed_count, }, ) summary_batch_output = _persist_summary_batch_artifact( run_store=run_store, run_dir=resolved_output_dir, item_contexts=item_contexts, ) if debug_artifacts and (resolved_output_dir / "summary").exists(): run_store.register_artifact(name="summary_dir", path=resolved_output_dir / "summary", kind="directory", stage=SUMMARY_STAGE) run_store.finish_stage( SUMMARY_STAGE, outputs={ "expected_items": extracted_success_count, "completed_items": summary_success_count + summary_failed_count, "success_count": summary_success_count, "failed_count": summary_failed_count, "summary_batch_output": str(summary_batch_output), }, ) run_store.start_stage( FILTER_STAGE, outputs={ "expected_items": summary_success_count, "completed_items": 0, "candidate_count": 0, "keep_count": 0, "review_count": 0, "drop_count": 0, }, ) filter_completed_count = 0 keep_count = 0 review_count = 0 drop_count = 0 for item_context in [ctx for ctx in item_contexts if ctx["summary_payload"] is not None]: item = item_context["item"] item_report = item_context["item_report"] extraction = item_context["extraction"] summary = LlmSummaryResult.model_validate(item_context["summary_payload"]) decision = evaluate_filter_rules( FilterInput(item=item, article=extraction.article, summary=summary, context=filter_context), loaded_rules, ) filter_path = item_context["filter_path"] if filter_path is not None: _save_json(filter_path, decision.model_dump(mode="json")) record = build_article_candidate_record( summary=summary, article=extraction.article, filter_result=decision, item=item, source_refs=CandidateSourceRefs( item_path=str(item_context["item_path"]) if item_context["item_path"] else None, extracted_path=str(item_context["extracted_path"]), summary_path=str(item_context["summary_output"]) if item_context["summary_output"] else None, filter_path=str(filter_path) if filter_path else None, ), metadata=CandidateMetadata( generated_at=datetime.now(tz=UTC), producer="run_freshrss_pipeline", run_id=resolved_run_id, ), ) openclaw_input = build_openclaw_candidate_input(record) item_context["candidate"] = openclaw_input if item_context["record_path"] is not None: _save_json(item_context["record_path"], record.model_dump(mode="json")) if item_context["openclaw_path"] is not None: _save_json(item_context["openclaw_path"], openclaw_input.model_dump(mode="json")) item_report["status"] = "delivered" item_report["selection_decision"] = decision.decision item_report["candidate_id"] = openclaw_input.candidate_id delivered_candidates.append(openclaw_input) if item.external_id: delivered_item_ids.append(item.external_id) if decision.decision == "keep": keep_count += 1 elif decision.decision == "review": review_count += 1 elif decision.decision == "drop": drop_count += 1 filter_completed_count += 1 run_store.update_stage( FILTER_STAGE, outputs={ "expected_items": summary_success_count, "completed_items": filter_completed_count, "candidate_count": len(delivered_candidates), "keep_count": keep_count, "review_count": review_count, "drop_count": drop_count, }, ) candidate_batch_output = _persist_candidate_batch_artifact( run_store=run_store, run_dir=resolved_output_dir, item_contexts=item_contexts, ) if debug_artifacts and (resolved_output_dir / "candidates").exists(): run_store.register_artifact(name="candidate_dir", path=resolved_output_dir / "candidates", kind="directory", stage=FILTER_STAGE) run_store.finish_stage( FILTER_STAGE, outputs={ "expected_items": summary_success_count, "completed_items": filter_completed_count, "candidate_count": len(delivered_candidates), "keep_count": keep_count, "review_count": review_count, "drop_count": drop_count, "candidate_batch_output": str(candidate_batch_output), }, ) run_store.start_stage(DELIVERY_STAGE, outputs={"candidate_count": len(delivered_candidates)}) delivered_candidates.sort(key=lambda candidate: candidate.digest_rank, reverse=True) delivery_payload = build_openclaw_delivery_payload( delivered_candidates, run_id=resolved_run_id, for_date=resolved_delivery_date, ) _save_json(delivery_output, delivery_payload.model_dump(mode="json")) run_store.register_artifact(name="delivery_payload", path=delivery_output, kind="json", stage=DELIVERY_STAGE) digest_brief = build_openclaw_digest_brief(delivery_payload) _save_json(digest_brief_output, digest_brief.model_dump(mode="json")) run_store.register_artifact(name="digest_brief", path=digest_brief_output, kind="json", stage=DELIVERY_STAGE) keyword_index_result = persist_keyword_indexes( delivery_payload.candidates, for_date=delivery_payload.date, digest_id=delivery_payload.run_id, source="openclaw_delivery_payload", daily_dir=DEFAULT_TERM_DAILY_DIR, stats_path=DEFAULT_TERM_STATS_PATH, aliases_path=DEFAULT_TERM_ALIASES_PATH, stopwords_path=DEFAULT_TERM_STOPWORDS_PATH, ) run_store.register_artifact( name="keyword_daily_index", path=Path(str(keyword_index_result["daily_output"])), kind="json", stage=DELIVERY_STAGE, ) run_store.register_artifact( name="keyword_stats_index", path=Path(str(keyword_index_result["stats_output"])), kind="json", stage=DELIVERY_STAGE, ) run_store.finish_stage( DELIVERY_STAGE, outputs={ "candidate_count": len(delivered_candidates), "delivery_output": str(delivery_output), "digest_brief_output": str(digest_brief_output), "keyword_daily_output": str(keyword_index_result["daily_output"]), "keyword_stats_output": str(keyword_index_result["stats_output"]), }, ) run_store.start_stage(REPORT_STAGE, outputs={"mark_read_requested": mark_read}) if mark_read and delivered_item_ids: client.mark_items_as_read(auth_token=auth_token, item_ids=delivered_item_ids) marked_count = len({item_id for item_id in delivered_item_ids if item_id}) report = _build_run_report( resolved_run_id=resolved_run_id, started_at=started_at, limit=limit, items=items, delivered_candidates=delivered_candidates, marked_count=marked_count, mark_read=mark_read, debug_artifacts=debug_artifacts, raw_output=raw_output, delivery_output=delivery_output, digest_brief_output=digest_brief_output, keyword_index_result=keyword_index_result, item_reports=item_reports, ) _save_json(report_output, report) run_store.register_artifact(name="run_report", path=report_output, kind="json", stage=REPORT_STAGE) run_store.finish_stage( REPORT_STAGE, outputs={ "marked_read_count": marked_count, "report_output": str(report_output), }, ) run_store.finish_run(status=_final_run_status(item_reports)) return { "run_id": resolved_run_id, "output_dir": str(resolved_output_dir), "raw_output": str(raw_output), "delivery_output": str(delivery_output), "digest_brief_output": str(digest_brief_output), "report_output": str(report_output), "keyword_index": keyword_index_result, "pulled_count": len(items), "delivered_count": len(delivered_candidates), "marked_read_count": marked_count, "status_counts": report["status_counts"], "debug_artifacts": debug_artifacts, "delivery_payload": delivery_payload.model_dump(mode="json"), "items": item_reports, } except Exception as error: failed_stage = run_store.state.current_stage or FETCH_STAGE run_store.fail_stage(failed_stage, error=error) raise def read_delivery_payload(path: Path) -> OpenClawDeliveryPayload: return OpenClawDeliveryPayload.model_validate(_load_json(path))