feat: add freshrss openclaw pipeline and clean repo
This commit is contained in:
@@ -6,14 +6,24 @@ from summary_mcp.core.errors import SummaryError
|
||||
from summary_mcp.models.summary_io import ExtractionInput
|
||||
|
||||
|
||||
MIN_INLINE_CONTENT_LENGTH = 500
|
||||
|
||||
|
||||
def _has_usable_content(value: str | None) -> bool:
|
||||
return bool(value and len(value.strip()) >= MIN_INLINE_CONTENT_LENGTH)
|
||||
|
||||
|
||||
def choose_inline_content(extraction_input: ExtractionInput) -> tuple[str | None, str]:
|
||||
if extraction_input.raw_html:
|
||||
return extraction_input.raw_html, "raw_html"
|
||||
|
||||
if extraction_input.item and extraction_input.item.raw_content and len(extraction_input.item.raw_content.strip()) >= 500:
|
||||
if extraction_input.item and _has_usable_content(extraction_input.item.raw_content):
|
||||
return extraction_input.item.raw_content, "item.raw_content"
|
||||
|
||||
if extraction_input.rss_content and len(extraction_input.rss_content.strip()) >= 500:
|
||||
if extraction_input.item and _has_usable_content(extraction_input.item.raw_summary):
|
||||
return extraction_input.item.raw_summary, "item.raw_summary"
|
||||
|
||||
if _has_usable_content(extraction_input.rss_content):
|
||||
return extraction_input.rss_content, "rss_content"
|
||||
|
||||
return None, "none"
|
||||
|
||||
@@ -9,6 +9,19 @@ from summary_mcp.core.quality_checker import assess_quality
|
||||
from summary_mcp.models.summary_io import DebugInfo, ExtractionInput, ExtractionOutput
|
||||
|
||||
|
||||
RSS_ONLY_UPSTREAMS = {"freshrss"}
|
||||
|
||||
|
||||
def _should_skip_fetch(extraction_input: ExtractionInput) -> bool:
|
||||
item = extraction_input.item
|
||||
if item is None:
|
||||
return False
|
||||
|
||||
metadata = item.metadata if isinstance(item.metadata, dict) else {}
|
||||
upstream = metadata.get("upstream")
|
||||
return isinstance(upstream, str) and upstream in RSS_ONLY_UPSTREAMS
|
||||
|
||||
|
||||
def extract_content(extraction_input: ExtractionInput) -> ExtractionOutput:
|
||||
try:
|
||||
normalized = normalize_input(extraction_input)
|
||||
@@ -16,11 +29,20 @@ def extract_content(extraction_input: ExtractionInput) -> ExtractionOutput:
|
||||
|
||||
inline_content, content_source = choose_inline_content(normalized)
|
||||
if inline_content is None:
|
||||
if _should_skip_fetch(normalized):
|
||||
raise SummaryError(
|
||||
code="RSS_CONTENT_MISSING",
|
||||
message="Skipping item because RSS content is unavailable.",
|
||||
retryable=False,
|
||||
stage="extract",
|
||||
details={"url": str(normalized.item.url), "upstream": normalized.item.metadata.get("upstream")},
|
||||
)
|
||||
inline_content = fetch_html(str(normalized.item.url))
|
||||
content_source = "fetched_html"
|
||||
|
||||
if normalized.item.title is None and (
|
||||
content_source in {"raw_html", "fetched_html"} or inline_content.lstrip().startswith("<")
|
||||
content_source in {"raw_html", "fetched_html", "item.raw_content", "item.raw_summary", "rss_content"}
|
||||
or inline_content.lstrip().startswith("<")
|
||||
):
|
||||
normalized.item.title = extract_title(inline_content)
|
||||
|
||||
|
||||
@@ -0,0 +1,273 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
|
||||
from summary_mcp.validators.llm_result import ValidationReport
|
||||
from summary_mcp.validators.llm_result import validate_llm_result as validate_llm_result_from_path
|
||||
from summary_mcp.validators.llm_result import validate_llm_result_payload
|
||||
|
||||
|
||||
JSON_BLOCK_RE = re.compile(r"```(?:json)?\s*(\{.*\})\s*```", re.DOTALL)
|
||||
DEFAULT_CHAT_COMPLETIONS_URL = "https://api.openai.com/v1/chat/completions"
|
||||
|
||||
|
||||
def load_text(path: Path) -> str:
|
||||
return path.read_text(encoding="utf-8")
|
||||
|
||||
|
||||
def load_json(path: Path) -> dict[str, Any]:
|
||||
return json.loads(load_text(path))
|
||||
|
||||
|
||||
def save_json(path: Path, payload: dict[str, Any]) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||
|
||||
|
||||
def build_summary_input(extracted: dict[str, Any]) -> dict[str, Any]:
|
||||
article = extracted.get("article") or {}
|
||||
return {
|
||||
"article": {
|
||||
"title": article.get("title"),
|
||||
"url": article.get("url"),
|
||||
"plain_text": article.get("plain_text"),
|
||||
"quality_flags": article.get("quality_flags"),
|
||||
},
|
||||
"warnings": extracted.get("warnings", []),
|
||||
}
|
||||
|
||||
|
||||
def build_initial_prompt(prompt_template: str, extracted: dict[str, Any]) -> str:
|
||||
summary_input = build_summary_input(extracted)
|
||||
return (
|
||||
f"{prompt_template}\n\n"
|
||||
"Below is the structured extracted article input. Generate the final summary JSON from it.\n\n"
|
||||
f"{json.dumps(summary_input, ensure_ascii=False, indent=2)}"
|
||||
)
|
||||
|
||||
|
||||
def build_repair_prompt(
|
||||
errors: list[str],
|
||||
extracted: dict[str, Any],
|
||||
result_json: dict[str, Any],
|
||||
) -> str:
|
||||
summary_input = build_summary_input(extracted)
|
||||
return (
|
||||
"Please repair the following invalid summary JSON.\n\n"
|
||||
"Requirements:\n"
|
||||
"- Output valid JSON only\n"
|
||||
"- Keep fields that are already correct\n"
|
||||
"- Fix only the validator-reported errors\n"
|
||||
"- Do not add explanations\n\n"
|
||||
f"validator errors:\n{json.dumps(errors, ensure_ascii=False, indent=2)}\n\n"
|
||||
f"Extracted article input:\n{json.dumps(summary_input, ensure_ascii=False, indent=2)}\n\n"
|
||||
f"Current summary JSON:\n{json.dumps(result_json, ensure_ascii=False, indent=2)}\n"
|
||||
)
|
||||
|
||||
|
||||
def extract_json_text(raw_text: str) -> str:
|
||||
fenced = JSON_BLOCK_RE.search(raw_text)
|
||||
if fenced:
|
||||
return fenced.group(1)
|
||||
|
||||
stripped = raw_text.strip()
|
||||
start = stripped.find("{")
|
||||
end = stripped.rfind("}")
|
||||
if start == -1 or end == -1 or end <= start:
|
||||
raise ValueError("Model output does not contain a JSON object.")
|
||||
return stripped[start : end + 1]
|
||||
|
||||
|
||||
def normalize_chat_completions_url(api_url: str | None) -> str | None:
|
||||
if api_url is None:
|
||||
return None
|
||||
|
||||
normalized = api_url.strip().rstrip("/")
|
||||
if not normalized:
|
||||
return None
|
||||
if normalized.endswith("/chat/completions"):
|
||||
return normalized
|
||||
return f"{normalized}/chat/completions"
|
||||
|
||||
|
||||
def resolve_llm_settings(
|
||||
*,
|
||||
api_key: str | None = None,
|
||||
model: str | None = None,
|
||||
api_url: str | None = None,
|
||||
) -> tuple[str, str, str]:
|
||||
resolved_api_key = api_key or os.environ.get("LLM_API_KEY") or os.environ.get("OPENAI_API_KEY")
|
||||
if not resolved_api_key:
|
||||
raise RuntimeError("Missing LLM_API_KEY or OPENAI_API_KEY, or pass an API key.")
|
||||
|
||||
resolved_model = model or os.environ.get("LLM_MODEL") or os.environ.get("OPENAI_MODEL")
|
||||
if not resolved_model:
|
||||
raise RuntimeError("Missing LLM_MODEL or OPENAI_MODEL, or pass a model.")
|
||||
|
||||
resolved_api_url = normalize_chat_completions_url(
|
||||
api_url or os.environ.get("LLM_API_URL") or os.environ.get("OPENAI_API_URL") or DEFAULT_CHAT_COMPLETIONS_URL
|
||||
)
|
||||
if not resolved_api_url:
|
||||
raise RuntimeError("Missing LLM_API_URL, OPENAI_API_URL, or pass an API URL.")
|
||||
|
||||
return resolved_api_key, resolved_model, resolved_api_url
|
||||
|
||||
|
||||
def call_llm(
|
||||
prompt: str,
|
||||
timeout_seconds: float,
|
||||
api_key: str | None,
|
||||
model: str | None,
|
||||
api_url: str | None,
|
||||
) -> str:
|
||||
resolved_api_key, resolved_model, resolved_api_url = resolve_llm_settings(
|
||||
api_key=api_key,
|
||||
model=model,
|
||||
api_url=api_url,
|
||||
)
|
||||
|
||||
headers = {
|
||||
"Authorization": f"Bearer {resolved_api_key}",
|
||||
"Content-Type": "application/json",
|
||||
}
|
||||
payload = {
|
||||
"model": resolved_model,
|
||||
"messages": [
|
||||
{
|
||||
"role": "system",
|
||||
"content": "You are a precise JSON generator. Always output a single valid JSON object.",
|
||||
},
|
||||
{"role": "user", "content": prompt},
|
||||
],
|
||||
"temperature": 0.2,
|
||||
}
|
||||
|
||||
with httpx.Client(timeout=timeout_seconds) as client:
|
||||
response = client.post(resolved_api_url, headers=headers, json=payload)
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
|
||||
try:
|
||||
return data["choices"][0]["message"]["content"]
|
||||
except (KeyError, IndexError, TypeError) as exc:
|
||||
raise RuntimeError(f"Unexpected LLM response shape: {json.dumps(data, ensure_ascii=False)[:1000]}") from exc
|
||||
|
||||
|
||||
def _save_attempt_artifact(base_output_path: Path | None, suffix: str, payload: str | dict[str, Any]) -> None:
|
||||
if base_output_path is None:
|
||||
return
|
||||
|
||||
path = base_output_path.with_name(f"{base_output_path.stem}.{suffix}")
|
||||
if isinstance(payload, str):
|
||||
path.write_text(payload, encoding="utf-8")
|
||||
else:
|
||||
save_json(path, payload)
|
||||
|
||||
|
||||
def run_loop_payload(
|
||||
*,
|
||||
extracted_payload: dict[str, Any],
|
||||
prompt_path: Path,
|
||||
max_retries: int,
|
||||
timeout_seconds: float,
|
||||
api_key: str | None,
|
||||
model: str | None,
|
||||
api_url: str | None,
|
||||
output_path: Path | None = None,
|
||||
) -> tuple[int, dict[str, Any] | None, ValidationReport | None]:
|
||||
prompt_template = load_text(prompt_path)
|
||||
if output_path is not None:
|
||||
output_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
last_errors: list[str] = []
|
||||
last_result: dict[str, Any] | None = None
|
||||
|
||||
for attempt in range(1, max_retries + 2):
|
||||
if attempt == 1:
|
||||
prompt = build_initial_prompt(prompt_template, extracted_payload)
|
||||
else:
|
||||
assert last_result is not None
|
||||
prompt = build_repair_prompt(last_errors, extracted_payload, last_result)
|
||||
|
||||
raw_output = call_llm(prompt, timeout_seconds, api_key, model, api_url)
|
||||
_save_attempt_artifact(output_path, f"attempt-{attempt}.raw.txt", raw_output)
|
||||
|
||||
try:
|
||||
result_payload = json.loads(extract_json_text(raw_output))
|
||||
except (json.JSONDecodeError, ValueError) as exc:
|
||||
last_errors = [f"Model output is not valid JSON: {exc}"]
|
||||
last_result = {"raw_output": raw_output}
|
||||
report_payload = {
|
||||
"valid": False,
|
||||
"errors": last_errors,
|
||||
"warnings": [],
|
||||
"normalized_result": None,
|
||||
}
|
||||
_save_attempt_artifact(output_path, f"attempt-{attempt}.validation.json", report_payload)
|
||||
if attempt > max_retries:
|
||||
if output_path is not None:
|
||||
output_path.write_text(raw_output, encoding="utf-8")
|
||||
return 1, None, ValidationReport(valid=False, errors=last_errors)
|
||||
continue
|
||||
|
||||
_save_attempt_artifact(output_path, f"attempt-{attempt}.json", result_payload)
|
||||
if output_path is not None:
|
||||
save_json(output_path, result_payload)
|
||||
|
||||
report = validate_llm_result_payload(result_payload, extracted_payload)
|
||||
_save_attempt_artifact(
|
||||
output_path,
|
||||
f"attempt-{attempt}.validation.json",
|
||||
{
|
||||
"valid": report.valid,
|
||||
"errors": report.errors,
|
||||
"warnings": report.warnings,
|
||||
"normalized_result": report.normalized_result,
|
||||
},
|
||||
)
|
||||
|
||||
if report.valid:
|
||||
return 0, result_payload, report
|
||||
|
||||
last_errors = report.errors
|
||||
last_result = result_payload
|
||||
|
||||
return 1, None, ValidationReport(valid=False, errors=last_errors)
|
||||
|
||||
|
||||
def run_loop(
|
||||
extracted_path: Path,
|
||||
prompt_path: Path,
|
||||
output_path: Path,
|
||||
max_retries: int,
|
||||
timeout_seconds: float,
|
||||
api_key: str | None,
|
||||
model: str | None,
|
||||
api_url: str | None,
|
||||
) -> int:
|
||||
extracted = load_json(extracted_path)
|
||||
exit_code, _, report = run_loop_payload(
|
||||
extracted_payload=extracted,
|
||||
prompt_path=prompt_path,
|
||||
output_path=output_path,
|
||||
max_retries=max_retries,
|
||||
timeout_seconds=timeout_seconds,
|
||||
api_key=api_key,
|
||||
model=model,
|
||||
api_url=api_url,
|
||||
)
|
||||
|
||||
if exit_code == 0:
|
||||
return 0
|
||||
|
||||
if report is not None and output_path.exists() and extracted_path.exists():
|
||||
fallback_report = validate_llm_result_from_path(output_path, extracted_path)
|
||||
if fallback_report.valid:
|
||||
return 0
|
||||
return exit_code
|
||||
Reference in New Issue
Block a user