feat: add freshrss openclaw pipeline and clean repo
This commit is contained in:
@@ -16,7 +16,7 @@ FRESHRSS_OUTPUT_ROOT = OUTPUT_ROOT / "freshrss"
|
||||
if str(SRC_ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(SRC_ROOT))
|
||||
|
||||
from summary_mcp.integrations.freshrss import FreshRSSClient, map_entry_to_item
|
||||
from summary_mcp.integrations.freshrss import READ_TAG, FreshRSSClient, map_entry_to_item
|
||||
|
||||
|
||||
def _save_json(path: Path, payload: dict[str, Any] | list[Any]) -> None:
|
||||
@@ -45,6 +45,16 @@ def main() -> None:
|
||||
)
|
||||
parser.add_argument("--limit", type=int, default=10, help="Maximum number of entries to request")
|
||||
parser.add_argument("--continuation", type=str, default=None, help="Continuation token for paging")
|
||||
parser.add_argument(
|
||||
"--include-read",
|
||||
action="store_true",
|
||||
help="Do not exclude items already tagged as read.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--mark-read",
|
||||
action="store_true",
|
||||
help="Mark fetched entries as read after this script finishes successfully.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--raw-output",
|
||||
type=Path,
|
||||
@@ -76,6 +86,7 @@ def main() -> None:
|
||||
stream_id=args.stream_id,
|
||||
limit=args.limit,
|
||||
continuation=args.continuation,
|
||||
exclude_targets=[] if args.include_read else [READ_TAG],
|
||||
)
|
||||
|
||||
entries = payload.get("items")
|
||||
@@ -87,8 +98,17 @@ def main() -> None:
|
||||
_save_json(args.raw_output, payload)
|
||||
_save_json(args.items_output, mapped_items)
|
||||
|
||||
marked_count = 0
|
||||
if args.mark_read:
|
||||
item_ids = [entry.get("external_id") for entry in mapped_items if isinstance(entry.get("external_id"), str)]
|
||||
if item_ids:
|
||||
client.mark_items_as_read(auth_token=auth_token, item_ids=item_ids)
|
||||
marked_count = len(item_ids)
|
||||
|
||||
print(f"Saved {len(mapped_items)} mapped items to {args.items_output}")
|
||||
if args.mark_read:
|
||||
print(f"Marked {marked_count} FreshRSS entries as read")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
main()
|
||||
|
||||
@@ -17,7 +17,7 @@ if str(SRC_ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(SRC_ROOT))
|
||||
|
||||
from summary_mcp.core.pipeline import extract_content
|
||||
from summary_mcp.integrations.freshrss import FreshRSSClient, map_entry_to_item
|
||||
from summary_mcp.integrations.freshrss import READ_TAG, FreshRSSClient, map_entry_to_item
|
||||
from summary_mcp.models.summary_io import ExtractionInput
|
||||
|
||||
|
||||
@@ -47,6 +47,16 @@ def main() -> None:
|
||||
)
|
||||
parser.add_argument("--limit", type=int, default=5, help="Maximum number of entries to request")
|
||||
parser.add_argument("--continuation", type=str, default=None, help="Continuation token for paging")
|
||||
parser.add_argument(
|
||||
"--include-read",
|
||||
action="store_true",
|
||||
help="Do not exclude items already tagged as read.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--mark-read",
|
||||
action="store_true",
|
||||
help="Mark entries as read after successful extraction for each item.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--raw-output",
|
||||
type=Path,
|
||||
@@ -84,6 +94,7 @@ def main() -> None:
|
||||
stream_id=args.stream_id,
|
||||
limit=args.limit,
|
||||
continuation=args.continuation,
|
||||
exclude_targets=[] if args.include_read else [READ_TAG],
|
||||
)
|
||||
|
||||
entries = payload.get("items")
|
||||
@@ -93,6 +104,7 @@ def main() -> None:
|
||||
mapped_items = [map_entry_to_item(entry) for entry in entries]
|
||||
extraction_results: list[dict[str, Any]] = []
|
||||
success_count = 0
|
||||
mark_read_ids: list[str] = []
|
||||
|
||||
for item in mapped_items:
|
||||
extraction = extract_content(ExtractionInput(item=item))
|
||||
@@ -104,6 +116,8 @@ def main() -> None:
|
||||
)
|
||||
if extraction.success:
|
||||
success_count += 1
|
||||
if item.external_id:
|
||||
mark_read_ids.append(item.external_id)
|
||||
|
||||
_save_json(args.raw_output, payload)
|
||||
_save_json(args.items_output, [item.model_dump(mode="json") for item in mapped_items])
|
||||
@@ -118,10 +132,17 @@ def main() -> None:
|
||||
},
|
||||
)
|
||||
|
||||
marked_count = 0
|
||||
if args.mark_read and mark_read_ids:
|
||||
client.mark_items_as_read(auth_token=auth_token, item_ids=mark_read_ids)
|
||||
marked_count = len(mark_read_ids)
|
||||
|
||||
print(
|
||||
f"Saved {len(mapped_items)} mapped items and {success_count} successful extractions to {args.extracted_output}"
|
||||
)
|
||||
if args.mark_read:
|
||||
print(f"Marked {marked_count} FreshRSS entries as read")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
main()
|
||||
|
||||
@@ -0,0 +1,96 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import sys
|
||||
from datetime import date
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||
SRC_ROOT = REPO_ROOT / "src"
|
||||
|
||||
if str(SRC_ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(SRC_ROOT))
|
||||
|
||||
from summary_mcp.workflows import run_freshrss_pipeline
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Run the full FreshRSS -> extract -> LLM -> filter -> OpenClaw payload pipeline."
|
||||
)
|
||||
parser.add_argument("--api-base-url", type=str, default=None, help="FreshRSS greader API base URL")
|
||||
parser.add_argument("--username", type=str, default=None, help="FreshRSS API username")
|
||||
parser.add_argument("--api-password", type=str, default=None, help="FreshRSS API password")
|
||||
parser.add_argument(
|
||||
"--stream-id",
|
||||
type=str,
|
||||
default="user/-/state/com.google/reading-list",
|
||||
help="Google Reader API stream id",
|
||||
)
|
||||
parser.add_argument("--limit", type=int, default=5, help="Maximum number of entries to request")
|
||||
parser.add_argument("--continuation", type=str, default=None, help="Continuation token for paging")
|
||||
parser.add_argument(
|
||||
"--include-read",
|
||||
action="store_true",
|
||||
help="Do not exclude items already tagged as read.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--mark-read",
|
||||
action="store_true",
|
||||
help="Mark only successfully delivered items as read after the final payload is written.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--debug-artifacts",
|
||||
action="store_true",
|
||||
help="Persist per-item intermediate files for debugging.",
|
||||
)
|
||||
parser.add_argument("--prompt", type=Path, default=None, help="LLM prompt template file")
|
||||
parser.add_argument("--rules", type=Path, default=None, help="Filter rule config JSON file")
|
||||
parser.add_argument("--context", type=Path, default=None, help="Optional filter context JSON file")
|
||||
parser.add_argument("--max-retries", type=int, default=2, help="Number of repair retries after the initial attempt")
|
||||
parser.add_argument("--timeout", type=float, default=60.0, help="Request timeout in seconds")
|
||||
parser.add_argument("--llm-api-key", type=str, default=None, help="LLM API key")
|
||||
parser.add_argument("--llm-model", type=str, default=None, help="LLM model name")
|
||||
parser.add_argument("--llm-api-url", type=str, default=None, help="LLM chat completions API URL or base URL")
|
||||
parser.add_argument("--run-id", type=str, default=None, help="Optional pipeline run id")
|
||||
parser.add_argument("--date", type=str, default=None, help="Optional delivery date in YYYY-MM-DD format")
|
||||
parser.add_argument(
|
||||
"--output-dir",
|
||||
type=Path,
|
||||
default=None,
|
||||
help="Run output directory. Defaults to outputs/freshrss/rerun/<timestamp>",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
result = run_freshrss_pipeline(
|
||||
api_base_url=args.api_base_url,
|
||||
username=args.username,
|
||||
api_password=args.api_password,
|
||||
stream_id=args.stream_id,
|
||||
limit=args.limit,
|
||||
continuation=args.continuation,
|
||||
include_read=args.include_read,
|
||||
mark_read=args.mark_read,
|
||||
debug_artifacts=args.debug_artifacts,
|
||||
prompt=args.prompt,
|
||||
rules=args.rules,
|
||||
context_path=args.context,
|
||||
max_retries=args.max_retries,
|
||||
timeout_seconds=args.timeout,
|
||||
llm_api_key=args.llm_api_key,
|
||||
llm_model=args.llm_model,
|
||||
llm_api_url=args.llm_api_url,
|
||||
run_id=args.run_id,
|
||||
delivery_date=date.fromisoformat(args.date) if args.date else None,
|
||||
output_dir=args.output_dir,
|
||||
)
|
||||
|
||||
print(f"Saved FreshRSS pipeline run to {result['output_dir']}")
|
||||
print(f"Pulled {result['pulled_count']} items, delivered {result['delivered_count']} candidates")
|
||||
if args.mark_read:
|
||||
print(f"Marked {result['marked_read_count']} FreshRSS entries as read")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
+2
-193
@@ -1,14 +1,8 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||
@@ -17,192 +11,7 @@ SRC_ROOT = REPO_ROOT / "src"
|
||||
if str(SRC_ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(SRC_ROOT))
|
||||
|
||||
from summary_mcp.validators.llm_result import validate_llm_result
|
||||
|
||||
|
||||
JSON_BLOCK_RE = re.compile(r"```(?:json)?\s*(\{.*\})\s*```", re.DOTALL)
|
||||
|
||||
|
||||
def load_text(path: Path) -> str:
|
||||
return path.read_text(encoding="utf-8")
|
||||
|
||||
|
||||
def load_json(path: Path) -> dict[str, Any]:
|
||||
return json.loads(load_text(path))
|
||||
|
||||
|
||||
def save_json(path: Path, payload: dict[str, Any]) -> None:
|
||||
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||
|
||||
|
||||
def build_summary_input(extracted: dict[str, Any]) -> dict[str, Any]:
|
||||
article = extracted.get('article') or {}
|
||||
return {
|
||||
'article': {
|
||||
'title': article.get('title'),
|
||||
'url': article.get('url'),
|
||||
'plain_text': article.get('plain_text'),
|
||||
'quality_flags': article.get('quality_flags'),
|
||||
},
|
||||
'warnings': extracted.get('warnings', []),
|
||||
}
|
||||
|
||||
|
||||
def build_initial_prompt(prompt_template: str, extracted: dict[str, Any]) -> str:
|
||||
summary_input = build_summary_input(extracted)
|
||||
return (
|
||||
f"{prompt_template}\n\n"
|
||||
"Below is the structured extracted article input. Generate the final summary JSON from it.\n\n"
|
||||
f"{json.dumps(summary_input, ensure_ascii=False, indent=2)}"
|
||||
)
|
||||
|
||||
|
||||
def build_repair_prompt(
|
||||
errors: list[str],
|
||||
extracted: dict[str, Any],
|
||||
result_json: dict[str, Any],
|
||||
) -> str:
|
||||
summary_input = build_summary_input(extracted)
|
||||
return (
|
||||
"Please repair the following invalid summary JSON.\n\n"
|
||||
"Requirements:\n"
|
||||
"- Output valid JSON only\n"
|
||||
"- Keep fields that are already correct\n"
|
||||
"- Fix only the validator-reported errors\n"
|
||||
"- Do not add explanations\n\n"
|
||||
f"validator errors:\n{json.dumps(errors, ensure_ascii=False, indent=2)}\n\n"
|
||||
f"Extracted article input:\n{json.dumps(summary_input, ensure_ascii=False, indent=2)}\n\n"
|
||||
f"Current summary JSON:\n{json.dumps(result_json, ensure_ascii=False, indent=2)}\n"
|
||||
)
|
||||
|
||||
|
||||
def extract_json_text(raw_text: str) -> str:
|
||||
fenced = JSON_BLOCK_RE.search(raw_text)
|
||||
if fenced:
|
||||
return fenced.group(1)
|
||||
|
||||
stripped = raw_text.strip()
|
||||
start = stripped.find("{")
|
||||
end = stripped.rfind("}")
|
||||
if start == -1 or end == -1 or end <= start:
|
||||
raise ValueError("Model output does not contain a JSON object.")
|
||||
return stripped[start : end + 1]
|
||||
|
||||
|
||||
def call_llm(
|
||||
prompt: str,
|
||||
timeout_seconds: float,
|
||||
api_key: str | None,
|
||||
model: str | None,
|
||||
api_url: str | None,
|
||||
) -> str:
|
||||
api_key = api_key or os.environ.get("LLM_API_KEY") or os.environ.get("OPENAI_API_KEY")
|
||||
model = model or os.environ.get("LLM_MODEL") or os.environ.get("OPENAI_MODEL")
|
||||
api_url = api_url or os.environ.get("LLM_API_URL", "https://api.openai.com/v1/chat/completions")
|
||||
|
||||
if not api_key:
|
||||
raise RuntimeError("Missing LLM_API_KEY or OPENAI_API_KEY, or pass --api-key.")
|
||||
if not model:
|
||||
raise RuntimeError("Missing LLM_MODEL or OPENAI_MODEL, or pass --model.")
|
||||
|
||||
headers = {
|
||||
"Authorization": f"Bearer {api_key}",
|
||||
"Content-Type": "application/json",
|
||||
}
|
||||
payload = {
|
||||
"model": model,
|
||||
"messages": [
|
||||
{
|
||||
"role": "system",
|
||||
"content": "You are a precise JSON generator. Always output a single valid JSON object.",
|
||||
},
|
||||
{"role": "user", "content": prompt},
|
||||
],
|
||||
"temperature": 0.2,
|
||||
}
|
||||
|
||||
with httpx.Client(timeout=timeout_seconds) as client:
|
||||
response = client.post(api_url, headers=headers, json=payload)
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
|
||||
try:
|
||||
return data["choices"][0]["message"]["content"]
|
||||
except (KeyError, IndexError, TypeError) as exc:
|
||||
raise RuntimeError(f"Unexpected LLM response shape: {json.dumps(data, ensure_ascii=False)[:1000]}") from exc
|
||||
|
||||
|
||||
def run_loop(
|
||||
extracted_path: Path,
|
||||
prompt_path: Path,
|
||||
output_path: Path,
|
||||
max_retries: int,
|
||||
timeout_seconds: float,
|
||||
api_key: str | None,
|
||||
model: str | None,
|
||||
api_url: str | None,
|
||||
) -> int:
|
||||
extracted = load_json(extracted_path)
|
||||
prompt_template = load_text(prompt_path)
|
||||
output_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
last_errors: list[str] = []
|
||||
last_result: dict[str, Any] | None = None
|
||||
|
||||
for attempt in range(1, max_retries + 2):
|
||||
if attempt == 1:
|
||||
prompt = build_initial_prompt(prompt_template, extracted)
|
||||
else:
|
||||
assert last_result is not None
|
||||
prompt = build_repair_prompt(last_errors, extracted, last_result)
|
||||
|
||||
raw_output = call_llm(prompt, timeout_seconds, api_key, model, api_url)
|
||||
raw_path = output_path.with_name(f"{output_path.stem}.attempt-{attempt}.raw.txt")
|
||||
raw_path.write_text(raw_output, encoding="utf-8")
|
||||
|
||||
try:
|
||||
result_payload = json.loads(extract_json_text(raw_output))
|
||||
except (json.JSONDecodeError, ValueError) as exc:
|
||||
last_errors = [f"Model output is not valid JSON: {exc}"]
|
||||
last_result = {"raw_output": raw_output}
|
||||
validation_path = output_path.with_name(f"{output_path.stem}.attempt-{attempt}.validation.json")
|
||||
save_json(
|
||||
validation_path,
|
||||
{
|
||||
"valid": False,
|
||||
"errors": last_errors,
|
||||
"warnings": [],
|
||||
"normalized_result": None,
|
||||
},
|
||||
)
|
||||
if attempt > max_retries:
|
||||
output_path.write_text(raw_output, encoding="utf-8")
|
||||
return 1
|
||||
continue
|
||||
|
||||
attempt_path = output_path.with_name(f"{output_path.stem}.attempt-{attempt}.json")
|
||||
save_json(attempt_path, result_payload)
|
||||
save_json(output_path, result_payload)
|
||||
|
||||
report = validate_llm_result(output_path, extracted_path)
|
||||
validation_path = output_path.with_name(f"{output_path.stem}.attempt-{attempt}.validation.json")
|
||||
save_json(
|
||||
validation_path,
|
||||
{
|
||||
"valid": report.valid,
|
||||
"errors": report.errors,
|
||||
"warnings": report.warnings,
|
||||
"normalized_result": report.normalized_result,
|
||||
},
|
||||
)
|
||||
|
||||
if report.valid:
|
||||
return 0
|
||||
|
||||
last_errors = report.errors
|
||||
last_result = result_payload
|
||||
|
||||
return 1
|
||||
from summary_mcp.core.summary_loop import run_loop
|
||||
|
||||
|
||||
def main() -> None:
|
||||
@@ -214,7 +23,7 @@ def main() -> None:
|
||||
parser.add_argument("--timeout", type=float, default=60.0, help="LLM request timeout in seconds")
|
||||
parser.add_argument("--api-key", type=str, default=None, help="LLM API key")
|
||||
parser.add_argument("--model", type=str, default=None, help="LLM model name")
|
||||
parser.add_argument("--api-url", type=str, default=None, help="LLM chat completions API URL")
|
||||
parser.add_argument("--api-url", type=str, default=None, help="LLM chat completions API URL or base URL")
|
||||
args = parser.parse_args()
|
||||
|
||||
raise SystemExit(
|
||||
|
||||
Reference in New Issue
Block a user