first commit
This commit is contained in:
@@ -0,0 +1 @@
|
||||
"""Core pipeline for the summary MCP service."""
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,38 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import httpx
|
||||
|
||||
from summary_mcp.core.errors import SummaryError
|
||||
from summary_mcp.models.summary_io import ExtractionInput
|
||||
|
||||
|
||||
def choose_inline_content(extraction_input: ExtractionInput) -> tuple[str | None, str]:
|
||||
if extraction_input.raw_html:
|
||||
return extraction_input.raw_html, "raw_html"
|
||||
|
||||
if extraction_input.item and extraction_input.item.raw_content and len(extraction_input.item.raw_content.strip()) >= 500:
|
||||
return extraction_input.item.raw_content, "item.raw_content"
|
||||
|
||||
if extraction_input.rss_content and len(extraction_input.rss_content.strip()) >= 500:
|
||||
return extraction_input.rss_content, "rss_content"
|
||||
|
||||
return None, "none"
|
||||
|
||||
|
||||
def fetch_html(url: str) -> str:
|
||||
headers = {
|
||||
"User-Agent": "summary-mcp/0.1 (+https://modelcontextprotocol.io/)",
|
||||
}
|
||||
try:
|
||||
with httpx.Client(follow_redirects=True, timeout=15.0, headers=headers) as client:
|
||||
response = client.get(url)
|
||||
response.raise_for_status()
|
||||
return response.text
|
||||
except httpx.HTTPError as exc:
|
||||
raise SummaryError(
|
||||
code="CONTENT_FETCH_FAILED",
|
||||
message="Failed to fetch article content",
|
||||
retryable=True,
|
||||
stage="fetch",
|
||||
details={"url": url, "reason": str(exc)},
|
||||
) from exc
|
||||
@@ -0,0 +1,24 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any
|
||||
|
||||
from summary_mcp.models.summary_io import ErrorInfo
|
||||
|
||||
|
||||
@dataclass
|
||||
class SummaryError(Exception):
|
||||
code: str
|
||||
message: str
|
||||
retryable: bool
|
||||
stage: str
|
||||
details: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
def to_error_info(self) -> ErrorInfo:
|
||||
return ErrorInfo(
|
||||
code=self.code,
|
||||
message=self.message,
|
||||
retryable=self.retryable,
|
||||
stage=self.stage,
|
||||
details=self.details,
|
||||
)
|
||||
@@ -0,0 +1,63 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from bs4 import BeautifulSoup
|
||||
import trafilatura
|
||||
|
||||
from summary_mcp.core.errors import SummaryError
|
||||
|
||||
|
||||
def extract_title(content: str) -> str | None:
|
||||
soup = BeautifulSoup(content, "html.parser")
|
||||
og_title = soup.find("meta", attrs={"property": "og:title"})
|
||||
if og_title and og_title.get("content"):
|
||||
return og_title["content"].strip()
|
||||
|
||||
if soup.title and soup.title.string:
|
||||
raw_title = soup.title.string.strip()
|
||||
return raw_title.split("|", 1)[0].strip()
|
||||
|
||||
heading = soup.find(["h1", "h2"])
|
||||
if heading:
|
||||
heading_text = heading.get_text(" ", strip=True)
|
||||
if heading_text:
|
||||
return heading_text
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def _dedupe_leading_lines(text: str) -> str:
|
||||
lines = [line.strip() for line in text.splitlines() if line.strip()]
|
||||
if len(lines) >= 2 and lines[0] == lines[1]:
|
||||
lines.pop(0)
|
||||
return "\n".join(lines).strip()
|
||||
|
||||
|
||||
def extract_plain_text(content: str, content_source: str) -> tuple[str, str]:
|
||||
if content_source == "raw_html" or content.lstrip().startswith("<"):
|
||||
extracted = trafilatura.extract(content, include_links=False, include_formatting=False)
|
||||
if extracted and len(extracted.strip()) >= 200:
|
||||
return _dedupe_leading_lines(extracted), "trafilatura"
|
||||
|
||||
soup = BeautifulSoup(content, "html.parser")
|
||||
fallback = " ".join(soup.stripped_strings)
|
||||
if len(fallback.strip()) >= 200:
|
||||
return fallback.strip(), "beautifulsoup"
|
||||
|
||||
raise SummaryError(
|
||||
code="CONTENT_EXTRACTION_FAILED",
|
||||
message="Failed to extract article body from HTML",
|
||||
retryable=False,
|
||||
stage="extract",
|
||||
)
|
||||
|
||||
if len(content.strip()) < 200:
|
||||
raise SummaryError(
|
||||
code="CONTENT_TOO_SHORT",
|
||||
message="Content is too short to summarize reliably",
|
||||
retryable=False,
|
||||
stage="extract",
|
||||
details={"length": len(content.strip())},
|
||||
)
|
||||
|
||||
return content.strip(), "inline"
|
||||
|
||||
@@ -0,0 +1,40 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
|
||||
from summary_mcp.models.document import ExtractedArticle, QualityFlags
|
||||
from summary_mcp.models.item import Item
|
||||
|
||||
|
||||
def _extract_id(item: Item, plain_text: str) -> str:
|
||||
seed = f"{item.url}|{item.title or ''}|{len(plain_text)}"
|
||||
digest = hashlib.sha256(seed.encode("utf-8")).hexdigest()
|
||||
return f"sha256:{digest}"
|
||||
|
||||
|
||||
def build_article(
|
||||
item: Item,
|
||||
plain_text: str,
|
||||
quality_flags: QualityFlags,
|
||||
content_source: str,
|
||||
extractor_name: str,
|
||||
) -> ExtractedArticle:
|
||||
return ExtractedArticle(
|
||||
extract_id=_extract_id(item, plain_text),
|
||||
item_id=item.item_id,
|
||||
source_id=item.source_id,
|
||||
url=item.url,
|
||||
title=item.title,
|
||||
author=item.author,
|
||||
published_at=item.published_at,
|
||||
language=item.language,
|
||||
content_kind=item.content_kind,
|
||||
plain_text=plain_text,
|
||||
quality_flags=quality_flags,
|
||||
metadata={
|
||||
"content_source": content_source,
|
||||
"extractor": extractor_name,
|
||||
"char_count": len(plain_text),
|
||||
},
|
||||
pipeline_state="extracted",
|
||||
)
|
||||
@@ -0,0 +1,25 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from summary_mcp.core.errors import SummaryError
|
||||
from summary_mcp.models.item import Item
|
||||
from summary_mcp.models.summary_io import ExtractionInput
|
||||
|
||||
|
||||
def normalize_input(extraction_input: ExtractionInput) -> ExtractionInput:
|
||||
if extraction_input.item is not None:
|
||||
return extraction_input
|
||||
|
||||
if extraction_input.url:
|
||||
return ExtractionInput(
|
||||
item=Item(url=extraction_input.url, title=None),
|
||||
raw_html=extraction_input.raw_html,
|
||||
rss_content=extraction_input.rss_content,
|
||||
language_hint=extraction_input.language_hint,
|
||||
)
|
||||
|
||||
raise SummaryError(
|
||||
code="INVALID_INPUT",
|
||||
message="Either item or url must be provided",
|
||||
retryable=False,
|
||||
stage="normalize",
|
||||
)
|
||||
@@ -0,0 +1,46 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from summary_mcp.core.content_loader import choose_inline_content, fetch_html
|
||||
from summary_mcp.core.errors import SummaryError
|
||||
from summary_mcp.core.extractor import extract_plain_text, extract_title
|
||||
from summary_mcp.core.mapper import build_article
|
||||
from summary_mcp.core.normalizer import normalize_input
|
||||
from summary_mcp.core.quality_checker import assess_quality
|
||||
from summary_mcp.models.summary_io import DebugInfo, ExtractionInput, ExtractionOutput
|
||||
|
||||
|
||||
def extract_content(extraction_input: ExtractionInput) -> ExtractionOutput:
|
||||
try:
|
||||
normalized = normalize_input(extraction_input)
|
||||
assert normalized.item is not None
|
||||
|
||||
inline_content, content_source = choose_inline_content(normalized)
|
||||
if inline_content is None:
|
||||
inline_content = fetch_html(str(normalized.item.url))
|
||||
content_source = "fetched_html"
|
||||
|
||||
if normalized.item.title is None and (
|
||||
content_source in {"raw_html", "fetched_html"} or inline_content.lstrip().startswith("<")
|
||||
):
|
||||
normalized.item.title = extract_title(inline_content)
|
||||
|
||||
plain_text, extractor_name = extract_plain_text(inline_content, content_source)
|
||||
quality_flags, warnings = assess_quality(plain_text)
|
||||
article = build_article(
|
||||
normalized.item,
|
||||
plain_text,
|
||||
quality_flags,
|
||||
content_source,
|
||||
extractor_name,
|
||||
)
|
||||
return ExtractionOutput(
|
||||
success=True,
|
||||
article=article,
|
||||
debug=DebugInfo(
|
||||
content_source=content_source,
|
||||
extractor=extractor_name,
|
||||
),
|
||||
warnings=warnings,
|
||||
)
|
||||
except SummaryError as exc:
|
||||
return ExtractionOutput(success=False, error=exc.to_error_info(), warnings=[])
|
||||
@@ -0,0 +1,25 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from summary_mcp.models.document import QualityFlags
|
||||
|
||||
|
||||
PAYWALL_HINTS = ("subscribe to read", "会员", "付费", "订阅后查看", "sign in to continue")
|
||||
|
||||
|
||||
def assess_quality(text: str) -> tuple[QualityFlags, list[str]]:
|
||||
lowered = text.lower()
|
||||
flags = QualityFlags(
|
||||
is_paywalled=any(hint in lowered for hint in PAYWALL_HINTS),
|
||||
is_truncated=text.endswith("...") or text.endswith("……"),
|
||||
is_low_content=len(text.strip()) < 500,
|
||||
)
|
||||
|
||||
warnings: list[str] = []
|
||||
if flags.is_paywalled:
|
||||
warnings.append("Potential paywall detected in content.")
|
||||
if flags.is_truncated:
|
||||
warnings.append("Content may be truncated.")
|
||||
if flags.is_low_content:
|
||||
warnings.append("Content has low information density.")
|
||||
|
||||
return flags, warnings
|
||||
Reference in New Issue
Block a user