feat: add freshrss openclaw pipeline and clean repo

This commit is contained in:
zhuyongxin
2026-03-26 16:48:19 +08:00
parent 27fe1e8882
commit 100044e1f7
143 changed files with 1776 additions and 7293 deletions
+22 -2
View File
@@ -16,7 +16,7 @@ FRESHRSS_OUTPUT_ROOT = OUTPUT_ROOT / "freshrss"
if str(SRC_ROOT) not in sys.path:
sys.path.insert(0, str(SRC_ROOT))
from summary_mcp.integrations.freshrss import FreshRSSClient, map_entry_to_item
from summary_mcp.integrations.freshrss import READ_TAG, FreshRSSClient, map_entry_to_item
def _save_json(path: Path, payload: dict[str, Any] | list[Any]) -> None:
@@ -45,6 +45,16 @@ def main() -> None:
)
parser.add_argument("--limit", type=int, default=10, help="Maximum number of entries to request")
parser.add_argument("--continuation", type=str, default=None, help="Continuation token for paging")
parser.add_argument(
"--include-read",
action="store_true",
help="Do not exclude items already tagged as read.",
)
parser.add_argument(
"--mark-read",
action="store_true",
help="Mark fetched entries as read after this script finishes successfully.",
)
parser.add_argument(
"--raw-output",
type=Path,
@@ -76,6 +86,7 @@ def main() -> None:
stream_id=args.stream_id,
limit=args.limit,
continuation=args.continuation,
exclude_targets=[] if args.include_read else [READ_TAG],
)
entries = payload.get("items")
@@ -87,8 +98,17 @@ def main() -> None:
_save_json(args.raw_output, payload)
_save_json(args.items_output, mapped_items)
marked_count = 0
if args.mark_read:
item_ids = [entry.get("external_id") for entry in mapped_items if isinstance(entry.get("external_id"), str)]
if item_ids:
client.mark_items_as_read(auth_token=auth_token, item_ids=item_ids)
marked_count = len(item_ids)
print(f"Saved {len(mapped_items)} mapped items to {args.items_output}")
if args.mark_read:
print(f"Marked {marked_count} FreshRSS entries as read")
if __name__ == "__main__":
main()
main()
+23 -2
View File
@@ -17,7 +17,7 @@ if str(SRC_ROOT) not in sys.path:
sys.path.insert(0, str(SRC_ROOT))
from summary_mcp.core.pipeline import extract_content
from summary_mcp.integrations.freshrss import FreshRSSClient, map_entry_to_item
from summary_mcp.integrations.freshrss import READ_TAG, FreshRSSClient, map_entry_to_item
from summary_mcp.models.summary_io import ExtractionInput
@@ -47,6 +47,16 @@ def main() -> None:
)
parser.add_argument("--limit", type=int, default=5, help="Maximum number of entries to request")
parser.add_argument("--continuation", type=str, default=None, help="Continuation token for paging")
parser.add_argument(
"--include-read",
action="store_true",
help="Do not exclude items already tagged as read.",
)
parser.add_argument(
"--mark-read",
action="store_true",
help="Mark entries as read after successful extraction for each item.",
)
parser.add_argument(
"--raw-output",
type=Path,
@@ -84,6 +94,7 @@ def main() -> None:
stream_id=args.stream_id,
limit=args.limit,
continuation=args.continuation,
exclude_targets=[] if args.include_read else [READ_TAG],
)
entries = payload.get("items")
@@ -93,6 +104,7 @@ def main() -> None:
mapped_items = [map_entry_to_item(entry) for entry in entries]
extraction_results: list[dict[str, Any]] = []
success_count = 0
mark_read_ids: list[str] = []
for item in mapped_items:
extraction = extract_content(ExtractionInput(item=item))
@@ -104,6 +116,8 @@ def main() -> None:
)
if extraction.success:
success_count += 1
if item.external_id:
mark_read_ids.append(item.external_id)
_save_json(args.raw_output, payload)
_save_json(args.items_output, [item.model_dump(mode="json") for item in mapped_items])
@@ -118,10 +132,17 @@ def main() -> None:
},
)
marked_count = 0
if args.mark_read and mark_read_ids:
client.mark_items_as_read(auth_token=auth_token, item_ids=mark_read_ids)
marked_count = len(mark_read_ids)
print(
f"Saved {len(mapped_items)} mapped items and {success_count} successful extractions to {args.extracted_output}"
)
if args.mark_read:
print(f"Marked {marked_count} FreshRSS entries as read")
if __name__ == "__main__":
main()
main()
+96
View File
@@ -0,0 +1,96 @@
from __future__ import annotations
import argparse
import sys
from datetime import date
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parents[1]
SRC_ROOT = REPO_ROOT / "src"
if str(SRC_ROOT) not in sys.path:
sys.path.insert(0, str(SRC_ROOT))
from summary_mcp.workflows import run_freshrss_pipeline
def main() -> None:
parser = argparse.ArgumentParser(
description="Run the full FreshRSS -> extract -> LLM -> filter -> OpenClaw payload pipeline."
)
parser.add_argument("--api-base-url", type=str, default=None, help="FreshRSS greader API base URL")
parser.add_argument("--username", type=str, default=None, help="FreshRSS API username")
parser.add_argument("--api-password", type=str, default=None, help="FreshRSS API password")
parser.add_argument(
"--stream-id",
type=str,
default="user/-/state/com.google/reading-list",
help="Google Reader API stream id",
)
parser.add_argument("--limit", type=int, default=5, help="Maximum number of entries to request")
parser.add_argument("--continuation", type=str, default=None, help="Continuation token for paging")
parser.add_argument(
"--include-read",
action="store_true",
help="Do not exclude items already tagged as read.",
)
parser.add_argument(
"--mark-read",
action="store_true",
help="Mark only successfully delivered items as read after the final payload is written.",
)
parser.add_argument(
"--debug-artifacts",
action="store_true",
help="Persist per-item intermediate files for debugging.",
)
parser.add_argument("--prompt", type=Path, default=None, help="LLM prompt template file")
parser.add_argument("--rules", type=Path, default=None, help="Filter rule config JSON file")
parser.add_argument("--context", type=Path, default=None, help="Optional filter context JSON file")
parser.add_argument("--max-retries", type=int, default=2, help="Number of repair retries after the initial attempt")
parser.add_argument("--timeout", type=float, default=60.0, help="Request timeout in seconds")
parser.add_argument("--llm-api-key", type=str, default=None, help="LLM API key")
parser.add_argument("--llm-model", type=str, default=None, help="LLM model name")
parser.add_argument("--llm-api-url", type=str, default=None, help="LLM chat completions API URL or base URL")
parser.add_argument("--run-id", type=str, default=None, help="Optional pipeline run id")
parser.add_argument("--date", type=str, default=None, help="Optional delivery date in YYYY-MM-DD format")
parser.add_argument(
"--output-dir",
type=Path,
default=None,
help="Run output directory. Defaults to outputs/freshrss/rerun/<timestamp>",
)
args = parser.parse_args()
result = run_freshrss_pipeline(
api_base_url=args.api_base_url,
username=args.username,
api_password=args.api_password,
stream_id=args.stream_id,
limit=args.limit,
continuation=args.continuation,
include_read=args.include_read,
mark_read=args.mark_read,
debug_artifacts=args.debug_artifacts,
prompt=args.prompt,
rules=args.rules,
context_path=args.context,
max_retries=args.max_retries,
timeout_seconds=args.timeout,
llm_api_key=args.llm_api_key,
llm_model=args.llm_model,
llm_api_url=args.llm_api_url,
run_id=args.run_id,
delivery_date=date.fromisoformat(args.date) if args.date else None,
output_dir=args.output_dir,
)
print(f"Saved FreshRSS pipeline run to {result['output_dir']}")
print(f"Pulled {result['pulled_count']} items, delivered {result['delivered_count']} candidates")
if args.mark_read:
print(f"Marked {result['marked_read_count']} FreshRSS entries as read")
if __name__ == "__main__":
main()
+2 -193
View File
@@ -1,14 +1,8 @@
from __future__ import annotations
import argparse
import json
import os
import re
import sys
from pathlib import Path
from typing import Any
import httpx
REPO_ROOT = Path(__file__).resolve().parents[1]
@@ -17,192 +11,7 @@ SRC_ROOT = REPO_ROOT / "src"
if str(SRC_ROOT) not in sys.path:
sys.path.insert(0, str(SRC_ROOT))
from summary_mcp.validators.llm_result import validate_llm_result
JSON_BLOCK_RE = re.compile(r"```(?:json)?\s*(\{.*\})\s*```", re.DOTALL)
def load_text(path: Path) -> str:
return path.read_text(encoding="utf-8")
def load_json(path: Path) -> dict[str, Any]:
return json.loads(load_text(path))
def save_json(path: Path, payload: dict[str, Any]) -> None:
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
def build_summary_input(extracted: dict[str, Any]) -> dict[str, Any]:
article = extracted.get('article') or {}
return {
'article': {
'title': article.get('title'),
'url': article.get('url'),
'plain_text': article.get('plain_text'),
'quality_flags': article.get('quality_flags'),
},
'warnings': extracted.get('warnings', []),
}
def build_initial_prompt(prompt_template: str, extracted: dict[str, Any]) -> str:
summary_input = build_summary_input(extracted)
return (
f"{prompt_template}\n\n"
"Below is the structured extracted article input. Generate the final summary JSON from it.\n\n"
f"{json.dumps(summary_input, ensure_ascii=False, indent=2)}"
)
def build_repair_prompt(
errors: list[str],
extracted: dict[str, Any],
result_json: dict[str, Any],
) -> str:
summary_input = build_summary_input(extracted)
return (
"Please repair the following invalid summary JSON.\n\n"
"Requirements:\n"
"- Output valid JSON only\n"
"- Keep fields that are already correct\n"
"- Fix only the validator-reported errors\n"
"- Do not add explanations\n\n"
f"validator errors:\n{json.dumps(errors, ensure_ascii=False, indent=2)}\n\n"
f"Extracted article input:\n{json.dumps(summary_input, ensure_ascii=False, indent=2)}\n\n"
f"Current summary JSON:\n{json.dumps(result_json, ensure_ascii=False, indent=2)}\n"
)
def extract_json_text(raw_text: str) -> str:
fenced = JSON_BLOCK_RE.search(raw_text)
if fenced:
return fenced.group(1)
stripped = raw_text.strip()
start = stripped.find("{")
end = stripped.rfind("}")
if start == -1 or end == -1 or end <= start:
raise ValueError("Model output does not contain a JSON object.")
return stripped[start : end + 1]
def call_llm(
prompt: str,
timeout_seconds: float,
api_key: str | None,
model: str | None,
api_url: str | None,
) -> str:
api_key = api_key or os.environ.get("LLM_API_KEY") or os.environ.get("OPENAI_API_KEY")
model = model or os.environ.get("LLM_MODEL") or os.environ.get("OPENAI_MODEL")
api_url = api_url or os.environ.get("LLM_API_URL", "https://api.openai.com/v1/chat/completions")
if not api_key:
raise RuntimeError("Missing LLM_API_KEY or OPENAI_API_KEY, or pass --api-key.")
if not model:
raise RuntimeError("Missing LLM_MODEL or OPENAI_MODEL, or pass --model.")
headers = {
"Authorization": f"Bearer {api_key}",
"Content-Type": "application/json",
}
payload = {
"model": model,
"messages": [
{
"role": "system",
"content": "You are a precise JSON generator. Always output a single valid JSON object.",
},
{"role": "user", "content": prompt},
],
"temperature": 0.2,
}
with httpx.Client(timeout=timeout_seconds) as client:
response = client.post(api_url, headers=headers, json=payload)
response.raise_for_status()
data = response.json()
try:
return data["choices"][0]["message"]["content"]
except (KeyError, IndexError, TypeError) as exc:
raise RuntimeError(f"Unexpected LLM response shape: {json.dumps(data, ensure_ascii=False)[:1000]}") from exc
def run_loop(
extracted_path: Path,
prompt_path: Path,
output_path: Path,
max_retries: int,
timeout_seconds: float,
api_key: str | None,
model: str | None,
api_url: str | None,
) -> int:
extracted = load_json(extracted_path)
prompt_template = load_text(prompt_path)
output_path.parent.mkdir(parents=True, exist_ok=True)
last_errors: list[str] = []
last_result: dict[str, Any] | None = None
for attempt in range(1, max_retries + 2):
if attempt == 1:
prompt = build_initial_prompt(prompt_template, extracted)
else:
assert last_result is not None
prompt = build_repair_prompt(last_errors, extracted, last_result)
raw_output = call_llm(prompt, timeout_seconds, api_key, model, api_url)
raw_path = output_path.with_name(f"{output_path.stem}.attempt-{attempt}.raw.txt")
raw_path.write_text(raw_output, encoding="utf-8")
try:
result_payload = json.loads(extract_json_text(raw_output))
except (json.JSONDecodeError, ValueError) as exc:
last_errors = [f"Model output is not valid JSON: {exc}"]
last_result = {"raw_output": raw_output}
validation_path = output_path.with_name(f"{output_path.stem}.attempt-{attempt}.validation.json")
save_json(
validation_path,
{
"valid": False,
"errors": last_errors,
"warnings": [],
"normalized_result": None,
},
)
if attempt > max_retries:
output_path.write_text(raw_output, encoding="utf-8")
return 1
continue
attempt_path = output_path.with_name(f"{output_path.stem}.attempt-{attempt}.json")
save_json(attempt_path, result_payload)
save_json(output_path, result_payload)
report = validate_llm_result(output_path, extracted_path)
validation_path = output_path.with_name(f"{output_path.stem}.attempt-{attempt}.validation.json")
save_json(
validation_path,
{
"valid": report.valid,
"errors": report.errors,
"warnings": report.warnings,
"normalized_result": report.normalized_result,
},
)
if report.valid:
return 0
last_errors = report.errors
last_result = result_payload
return 1
from summary_mcp.core.summary_loop import run_loop
def main() -> None:
@@ -214,7 +23,7 @@ def main() -> None:
parser.add_argument("--timeout", type=float, default=60.0, help="LLM request timeout in seconds")
parser.add_argument("--api-key", type=str, default=None, help="LLM API key")
parser.add_argument("--model", type=str, default=None, help="LLM model name")
parser.add_argument("--api-url", type=str, default=None, help="LLM chat completions API URL")
parser.add_argument("--api-url", type=str, default=None, help="LLM chat completions API URL or base URL")
args = parser.parse_args()
raise SystemExit(