feat: add freshrss openclaw pipeline and clean repo

This commit is contained in:
zhuyongxin
2026-03-26 16:48:19 +08:00
parent 27fe1e8882
commit 100044e1f7
143 changed files with 1776 additions and 7293 deletions
+12 -2
View File
@@ -6,14 +6,24 @@ from summary_mcp.core.errors import SummaryError
from summary_mcp.models.summary_io import ExtractionInput
MIN_INLINE_CONTENT_LENGTH = 500
def _has_usable_content(value: str | None) -> bool:
return bool(value and len(value.strip()) >= MIN_INLINE_CONTENT_LENGTH)
def choose_inline_content(extraction_input: ExtractionInput) -> tuple[str | None, str]:
if extraction_input.raw_html:
return extraction_input.raw_html, "raw_html"
if extraction_input.item and extraction_input.item.raw_content and len(extraction_input.item.raw_content.strip()) >= 500:
if extraction_input.item and _has_usable_content(extraction_input.item.raw_content):
return extraction_input.item.raw_content, "item.raw_content"
if extraction_input.rss_content and len(extraction_input.rss_content.strip()) >= 500:
if extraction_input.item and _has_usable_content(extraction_input.item.raw_summary):
return extraction_input.item.raw_summary, "item.raw_summary"
if _has_usable_content(extraction_input.rss_content):
return extraction_input.rss_content, "rss_content"
return None, "none"
+23 -1
View File
@@ -9,6 +9,19 @@ from summary_mcp.core.quality_checker import assess_quality
from summary_mcp.models.summary_io import DebugInfo, ExtractionInput, ExtractionOutput
RSS_ONLY_UPSTREAMS = {"freshrss"}
def _should_skip_fetch(extraction_input: ExtractionInput) -> bool:
item = extraction_input.item
if item is None:
return False
metadata = item.metadata if isinstance(item.metadata, dict) else {}
upstream = metadata.get("upstream")
return isinstance(upstream, str) and upstream in RSS_ONLY_UPSTREAMS
def extract_content(extraction_input: ExtractionInput) -> ExtractionOutput:
try:
normalized = normalize_input(extraction_input)
@@ -16,11 +29,20 @@ def extract_content(extraction_input: ExtractionInput) -> ExtractionOutput:
inline_content, content_source = choose_inline_content(normalized)
if inline_content is None:
if _should_skip_fetch(normalized):
raise SummaryError(
code="RSS_CONTENT_MISSING",
message="Skipping item because RSS content is unavailable.",
retryable=False,
stage="extract",
details={"url": str(normalized.item.url), "upstream": normalized.item.metadata.get("upstream")},
)
inline_content = fetch_html(str(normalized.item.url))
content_source = "fetched_html"
if normalized.item.title is None and (
content_source in {"raw_html", "fetched_html"} or inline_content.lstrip().startswith("<")
content_source in {"raw_html", "fetched_html", "item.raw_content", "item.raw_summary", "rss_content"}
or inline_content.lstrip().startswith("<")
):
normalized.item.title = extract_title(inline_content)
+273
View File
@@ -0,0 +1,273 @@
from __future__ import annotations
import json
import os
import re
from pathlib import Path
from typing import Any
import httpx
from summary_mcp.validators.llm_result import ValidationReport
from summary_mcp.validators.llm_result import validate_llm_result as validate_llm_result_from_path
from summary_mcp.validators.llm_result import validate_llm_result_payload
JSON_BLOCK_RE = re.compile(r"```(?:json)?\s*(\{.*\})\s*```", re.DOTALL)
DEFAULT_CHAT_COMPLETIONS_URL = "https://api.openai.com/v1/chat/completions"
def load_text(path: Path) -> str:
return path.read_text(encoding="utf-8")
def load_json(path: Path) -> dict[str, Any]:
return json.loads(load_text(path))
def save_json(path: Path, payload: dict[str, Any]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
def build_summary_input(extracted: dict[str, Any]) -> dict[str, Any]:
article = extracted.get("article") or {}
return {
"article": {
"title": article.get("title"),
"url": article.get("url"),
"plain_text": article.get("plain_text"),
"quality_flags": article.get("quality_flags"),
},
"warnings": extracted.get("warnings", []),
}
def build_initial_prompt(prompt_template: str, extracted: dict[str, Any]) -> str:
summary_input = build_summary_input(extracted)
return (
f"{prompt_template}\n\n"
"Below is the structured extracted article input. Generate the final summary JSON from it.\n\n"
f"{json.dumps(summary_input, ensure_ascii=False, indent=2)}"
)
def build_repair_prompt(
errors: list[str],
extracted: dict[str, Any],
result_json: dict[str, Any],
) -> str:
summary_input = build_summary_input(extracted)
return (
"Please repair the following invalid summary JSON.\n\n"
"Requirements:\n"
"- Output valid JSON only\n"
"- Keep fields that are already correct\n"
"- Fix only the validator-reported errors\n"
"- Do not add explanations\n\n"
f"validator errors:\n{json.dumps(errors, ensure_ascii=False, indent=2)}\n\n"
f"Extracted article input:\n{json.dumps(summary_input, ensure_ascii=False, indent=2)}\n\n"
f"Current summary JSON:\n{json.dumps(result_json, ensure_ascii=False, indent=2)}\n"
)
def extract_json_text(raw_text: str) -> str:
fenced = JSON_BLOCK_RE.search(raw_text)
if fenced:
return fenced.group(1)
stripped = raw_text.strip()
start = stripped.find("{")
end = stripped.rfind("}")
if start == -1 or end == -1 or end <= start:
raise ValueError("Model output does not contain a JSON object.")
return stripped[start : end + 1]
def normalize_chat_completions_url(api_url: str | None) -> str | None:
if api_url is None:
return None
normalized = api_url.strip().rstrip("/")
if not normalized:
return None
if normalized.endswith("/chat/completions"):
return normalized
return f"{normalized}/chat/completions"
def resolve_llm_settings(
*,
api_key: str | None = None,
model: str | None = None,
api_url: str | None = None,
) -> tuple[str, str, str]:
resolved_api_key = api_key or os.environ.get("LLM_API_KEY") or os.environ.get("OPENAI_API_KEY")
if not resolved_api_key:
raise RuntimeError("Missing LLM_API_KEY or OPENAI_API_KEY, or pass an API key.")
resolved_model = model or os.environ.get("LLM_MODEL") or os.environ.get("OPENAI_MODEL")
if not resolved_model:
raise RuntimeError("Missing LLM_MODEL or OPENAI_MODEL, or pass a model.")
resolved_api_url = normalize_chat_completions_url(
api_url or os.environ.get("LLM_API_URL") or os.environ.get("OPENAI_API_URL") or DEFAULT_CHAT_COMPLETIONS_URL
)
if not resolved_api_url:
raise RuntimeError("Missing LLM_API_URL, OPENAI_API_URL, or pass an API URL.")
return resolved_api_key, resolved_model, resolved_api_url
def call_llm(
prompt: str,
timeout_seconds: float,
api_key: str | None,
model: str | None,
api_url: str | None,
) -> str:
resolved_api_key, resolved_model, resolved_api_url = resolve_llm_settings(
api_key=api_key,
model=model,
api_url=api_url,
)
headers = {
"Authorization": f"Bearer {resolved_api_key}",
"Content-Type": "application/json",
}
payload = {
"model": resolved_model,
"messages": [
{
"role": "system",
"content": "You are a precise JSON generator. Always output a single valid JSON object.",
},
{"role": "user", "content": prompt},
],
"temperature": 0.2,
}
with httpx.Client(timeout=timeout_seconds) as client:
response = client.post(resolved_api_url, headers=headers, json=payload)
response.raise_for_status()
data = response.json()
try:
return data["choices"][0]["message"]["content"]
except (KeyError, IndexError, TypeError) as exc:
raise RuntimeError(f"Unexpected LLM response shape: {json.dumps(data, ensure_ascii=False)[:1000]}") from exc
def _save_attempt_artifact(base_output_path: Path | None, suffix: str, payload: str | dict[str, Any]) -> None:
if base_output_path is None:
return
path = base_output_path.with_name(f"{base_output_path.stem}.{suffix}")
if isinstance(payload, str):
path.write_text(payload, encoding="utf-8")
else:
save_json(path, payload)
def run_loop_payload(
*,
extracted_payload: dict[str, Any],
prompt_path: Path,
max_retries: int,
timeout_seconds: float,
api_key: str | None,
model: str | None,
api_url: str | None,
output_path: Path | None = None,
) -> tuple[int, dict[str, Any] | None, ValidationReport | None]:
prompt_template = load_text(prompt_path)
if output_path is not None:
output_path.parent.mkdir(parents=True, exist_ok=True)
last_errors: list[str] = []
last_result: dict[str, Any] | None = None
for attempt in range(1, max_retries + 2):
if attempt == 1:
prompt = build_initial_prompt(prompt_template, extracted_payload)
else:
assert last_result is not None
prompt = build_repair_prompt(last_errors, extracted_payload, last_result)
raw_output = call_llm(prompt, timeout_seconds, api_key, model, api_url)
_save_attempt_artifact(output_path, f"attempt-{attempt}.raw.txt", raw_output)
try:
result_payload = json.loads(extract_json_text(raw_output))
except (json.JSONDecodeError, ValueError) as exc:
last_errors = [f"Model output is not valid JSON: {exc}"]
last_result = {"raw_output": raw_output}
report_payload = {
"valid": False,
"errors": last_errors,
"warnings": [],
"normalized_result": None,
}
_save_attempt_artifact(output_path, f"attempt-{attempt}.validation.json", report_payload)
if attempt > max_retries:
if output_path is not None:
output_path.write_text(raw_output, encoding="utf-8")
return 1, None, ValidationReport(valid=False, errors=last_errors)
continue
_save_attempt_artifact(output_path, f"attempt-{attempt}.json", result_payload)
if output_path is not None:
save_json(output_path, result_payload)
report = validate_llm_result_payload(result_payload, extracted_payload)
_save_attempt_artifact(
output_path,
f"attempt-{attempt}.validation.json",
{
"valid": report.valid,
"errors": report.errors,
"warnings": report.warnings,
"normalized_result": report.normalized_result,
},
)
if report.valid:
return 0, result_payload, report
last_errors = report.errors
last_result = result_payload
return 1, None, ValidationReport(valid=False, errors=last_errors)
def run_loop(
extracted_path: Path,
prompt_path: Path,
output_path: Path,
max_retries: int,
timeout_seconds: float,
api_key: str | None,
model: str | None,
api_url: str | None,
) -> int:
extracted = load_json(extracted_path)
exit_code, _, report = run_loop_payload(
extracted_payload=extracted,
prompt_path=prompt_path,
output_path=output_path,
max_retries=max_retries,
timeout_seconds=timeout_seconds,
api_key=api_key,
model=model,
api_url=api_url,
)
if exit_code == 0:
return 0
if report is not None and output_path.exists() and extracted_path.exists():
fallback_report = validate_llm_result_from_path(output_path, extracted_path)
if fallback_report.valid:
return 0
return exit_code