Files
reader/src/summary_mcp/core/content_loader.py
T

49 lines
1.6 KiB
Python

from __future__ import annotations
import httpx
from summary_mcp.core.errors import SummaryError
from summary_mcp.models.summary_io import ExtractionInput
MIN_INLINE_CONTENT_LENGTH = 500
def _has_usable_content(value: str | None) -> bool:
return bool(value and len(value.strip()) >= MIN_INLINE_CONTENT_LENGTH)
def choose_inline_content(extraction_input: ExtractionInput) -> tuple[str | None, str]:
if extraction_input.raw_html:
return extraction_input.raw_html, "raw_html"
if extraction_input.item and _has_usable_content(extraction_input.item.raw_content):
return extraction_input.item.raw_content, "item.raw_content"
if extraction_input.item and _has_usable_content(extraction_input.item.raw_summary):
return extraction_input.item.raw_summary, "item.raw_summary"
if _has_usable_content(extraction_input.rss_content):
return extraction_input.rss_content, "rss_content"
return None, "none"
def fetch_html(url: str) -> str:
headers = {
"User-Agent": "summary-mcp/0.1 (+https://modelcontextprotocol.io/)",
}
try:
with httpx.Client(follow_redirects=True, timeout=15.0, headers=headers) as client:
response = client.get(url)
response.raise_for_status()
return response.text
except httpx.HTTPError as exc:
raise SummaryError(
code="CONTENT_FETCH_FAILED",
message="Failed to fetch article content",
retryable=True,
stage="fetch",
details={"url": url, "reason": str(exc)},
) from exc