Add FreshRSS ingestion and rule filtering
This commit is contained in:
@@ -0,0 +1,125 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||
SRC_ROOT = REPO_ROOT / "src"
|
||||
|
||||
if str(SRC_ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(SRC_ROOT))
|
||||
|
||||
from summary_mcp.core.pipeline import extract_content
|
||||
from summary_mcp.integrations.freshrss import FreshRSSClient, map_entry_to_item
|
||||
from summary_mcp.models.summary_io import ExtractionInput
|
||||
|
||||
|
||||
def _save_json(path: Path, payload: dict[str, Any] | list[Any]) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||
|
||||
|
||||
def _load_required(name: str, value: str | None) -> str:
|
||||
resolved = value or os.environ.get(name)
|
||||
if not resolved:
|
||||
cli_name = name.lower().replace("_", "-")
|
||||
raise RuntimeError(f"Missing required value: pass --{cli_name} or set {name}.")
|
||||
return resolved
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description="Pull FreshRSS entries and run content extraction for each mapped item.")
|
||||
parser.add_argument("--api-base-url", type=str, default=None, help="FreshRSS greader API base URL")
|
||||
parser.add_argument("--username", type=str, default=None, help="FreshRSS API username")
|
||||
parser.add_argument("--api-password", type=str, default=None, help="FreshRSS API password")
|
||||
parser.add_argument(
|
||||
"--stream-id",
|
||||
type=str,
|
||||
default="user/-/state/com.google/reading-list",
|
||||
help="Google Reader API stream id",
|
||||
)
|
||||
parser.add_argument("--limit", type=int, default=5, help="Maximum number of entries to request")
|
||||
parser.add_argument("--continuation", type=str, default=None, help="Continuation token for paging")
|
||||
parser.add_argument(
|
||||
"--raw-output",
|
||||
type=Path,
|
||||
default=REPO_ROOT / "outputs" / "freshrss.raw.json",
|
||||
help="Where to save the raw FreshRSS response",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--items-output",
|
||||
type=Path,
|
||||
default=REPO_ROOT / "outputs" / "freshrss.items.json",
|
||||
help="Where to save the mapped item list",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--extracted-output",
|
||||
type=Path,
|
||||
default=REPO_ROOT / "outputs" / "freshrss.extracted.json",
|
||||
help="Where to save the extraction results",
|
||||
)
|
||||
parser.add_argument("--timeout", type=float, default=20.0, help="Request timeout in seconds")
|
||||
args = parser.parse_args()
|
||||
|
||||
api_base_url = _load_required("FRESHRSS_API_BASE_URL", args.api_base_url)
|
||||
username = _load_required("FRESHRSS_USERNAME", args.username)
|
||||
api_password = _load_required("FRESHRSS_API_PASSWORD", args.api_password)
|
||||
|
||||
client = FreshRSSClient(
|
||||
api_base_url=api_base_url,
|
||||
username=username,
|
||||
api_password=api_password,
|
||||
timeout_seconds=args.timeout,
|
||||
)
|
||||
auth_token = client.client_login()
|
||||
payload = client.fetch_stream_contents(
|
||||
auth_token=auth_token,
|
||||
stream_id=args.stream_id,
|
||||
limit=args.limit,
|
||||
continuation=args.continuation,
|
||||
)
|
||||
|
||||
entries = payload.get("items")
|
||||
if not isinstance(entries, list):
|
||||
raise RuntimeError("FreshRSS stream response does not contain an items array.")
|
||||
|
||||
mapped_items = [map_entry_to_item(entry) for entry in entries]
|
||||
extraction_results: list[dict[str, Any]] = []
|
||||
success_count = 0
|
||||
|
||||
for item in mapped_items:
|
||||
extraction = extract_content(ExtractionInput(item=item))
|
||||
extraction_results.append(
|
||||
{
|
||||
"item": item.model_dump(mode="json"),
|
||||
"extraction": extraction.model_dump(mode="json"),
|
||||
}
|
||||
)
|
||||
if extraction.success:
|
||||
success_count += 1
|
||||
|
||||
_save_json(args.raw_output, payload)
|
||||
_save_json(args.items_output, [item.model_dump(mode="json") for item in mapped_items])
|
||||
_save_json(
|
||||
args.extracted_output,
|
||||
{
|
||||
"stream_id": args.stream_id,
|
||||
"requested_limit": args.limit,
|
||||
"entry_count": len(entries),
|
||||
"extracted_success_count": success_count,
|
||||
"results": extraction_results,
|
||||
},
|
||||
)
|
||||
|
||||
print(
|
||||
f"Saved {len(mapped_items)} mapped items and {success_count} successful extractions to {args.extracted_output}"
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user