first commit
This commit is contained in:
@@ -0,0 +1,2 @@
|
||||
"""Summary MCP service package."""
|
||||
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -0,0 +1 @@
|
||||
"""Core pipeline for the summary MCP service."""
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,38 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import httpx
|
||||
|
||||
from summary_mcp.core.errors import SummaryError
|
||||
from summary_mcp.models.summary_io import ExtractionInput
|
||||
|
||||
|
||||
def choose_inline_content(extraction_input: ExtractionInput) -> tuple[str | None, str]:
|
||||
if extraction_input.raw_html:
|
||||
return extraction_input.raw_html, "raw_html"
|
||||
|
||||
if extraction_input.item and extraction_input.item.raw_content and len(extraction_input.item.raw_content.strip()) >= 500:
|
||||
return extraction_input.item.raw_content, "item.raw_content"
|
||||
|
||||
if extraction_input.rss_content and len(extraction_input.rss_content.strip()) >= 500:
|
||||
return extraction_input.rss_content, "rss_content"
|
||||
|
||||
return None, "none"
|
||||
|
||||
|
||||
def fetch_html(url: str) -> str:
|
||||
headers = {
|
||||
"User-Agent": "summary-mcp/0.1 (+https://modelcontextprotocol.io/)",
|
||||
}
|
||||
try:
|
||||
with httpx.Client(follow_redirects=True, timeout=15.0, headers=headers) as client:
|
||||
response = client.get(url)
|
||||
response.raise_for_status()
|
||||
return response.text
|
||||
except httpx.HTTPError as exc:
|
||||
raise SummaryError(
|
||||
code="CONTENT_FETCH_FAILED",
|
||||
message="Failed to fetch article content",
|
||||
retryable=True,
|
||||
stage="fetch",
|
||||
details={"url": url, "reason": str(exc)},
|
||||
) from exc
|
||||
@@ -0,0 +1,24 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any
|
||||
|
||||
from summary_mcp.models.summary_io import ErrorInfo
|
||||
|
||||
|
||||
@dataclass
|
||||
class SummaryError(Exception):
|
||||
code: str
|
||||
message: str
|
||||
retryable: bool
|
||||
stage: str
|
||||
details: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
def to_error_info(self) -> ErrorInfo:
|
||||
return ErrorInfo(
|
||||
code=self.code,
|
||||
message=self.message,
|
||||
retryable=self.retryable,
|
||||
stage=self.stage,
|
||||
details=self.details,
|
||||
)
|
||||
@@ -0,0 +1,63 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from bs4 import BeautifulSoup
|
||||
import trafilatura
|
||||
|
||||
from summary_mcp.core.errors import SummaryError
|
||||
|
||||
|
||||
def extract_title(content: str) -> str | None:
|
||||
soup = BeautifulSoup(content, "html.parser")
|
||||
og_title = soup.find("meta", attrs={"property": "og:title"})
|
||||
if og_title and og_title.get("content"):
|
||||
return og_title["content"].strip()
|
||||
|
||||
if soup.title and soup.title.string:
|
||||
raw_title = soup.title.string.strip()
|
||||
return raw_title.split("|", 1)[0].strip()
|
||||
|
||||
heading = soup.find(["h1", "h2"])
|
||||
if heading:
|
||||
heading_text = heading.get_text(" ", strip=True)
|
||||
if heading_text:
|
||||
return heading_text
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def _dedupe_leading_lines(text: str) -> str:
|
||||
lines = [line.strip() for line in text.splitlines() if line.strip()]
|
||||
if len(lines) >= 2 and lines[0] == lines[1]:
|
||||
lines.pop(0)
|
||||
return "\n".join(lines).strip()
|
||||
|
||||
|
||||
def extract_plain_text(content: str, content_source: str) -> tuple[str, str]:
|
||||
if content_source == "raw_html" or content.lstrip().startswith("<"):
|
||||
extracted = trafilatura.extract(content, include_links=False, include_formatting=False)
|
||||
if extracted and len(extracted.strip()) >= 200:
|
||||
return _dedupe_leading_lines(extracted), "trafilatura"
|
||||
|
||||
soup = BeautifulSoup(content, "html.parser")
|
||||
fallback = " ".join(soup.stripped_strings)
|
||||
if len(fallback.strip()) >= 200:
|
||||
return fallback.strip(), "beautifulsoup"
|
||||
|
||||
raise SummaryError(
|
||||
code="CONTENT_EXTRACTION_FAILED",
|
||||
message="Failed to extract article body from HTML",
|
||||
retryable=False,
|
||||
stage="extract",
|
||||
)
|
||||
|
||||
if len(content.strip()) < 200:
|
||||
raise SummaryError(
|
||||
code="CONTENT_TOO_SHORT",
|
||||
message="Content is too short to summarize reliably",
|
||||
retryable=False,
|
||||
stage="extract",
|
||||
details={"length": len(content.strip())},
|
||||
)
|
||||
|
||||
return content.strip(), "inline"
|
||||
|
||||
@@ -0,0 +1,40 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
|
||||
from summary_mcp.models.document import ExtractedArticle, QualityFlags
|
||||
from summary_mcp.models.item import Item
|
||||
|
||||
|
||||
def _extract_id(item: Item, plain_text: str) -> str:
|
||||
seed = f"{item.url}|{item.title or ''}|{len(plain_text)}"
|
||||
digest = hashlib.sha256(seed.encode("utf-8")).hexdigest()
|
||||
return f"sha256:{digest}"
|
||||
|
||||
|
||||
def build_article(
|
||||
item: Item,
|
||||
plain_text: str,
|
||||
quality_flags: QualityFlags,
|
||||
content_source: str,
|
||||
extractor_name: str,
|
||||
) -> ExtractedArticle:
|
||||
return ExtractedArticle(
|
||||
extract_id=_extract_id(item, plain_text),
|
||||
item_id=item.item_id,
|
||||
source_id=item.source_id,
|
||||
url=item.url,
|
||||
title=item.title,
|
||||
author=item.author,
|
||||
published_at=item.published_at,
|
||||
language=item.language,
|
||||
content_kind=item.content_kind,
|
||||
plain_text=plain_text,
|
||||
quality_flags=quality_flags,
|
||||
metadata={
|
||||
"content_source": content_source,
|
||||
"extractor": extractor_name,
|
||||
"char_count": len(plain_text),
|
||||
},
|
||||
pipeline_state="extracted",
|
||||
)
|
||||
@@ -0,0 +1,25 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from summary_mcp.core.errors import SummaryError
|
||||
from summary_mcp.models.item import Item
|
||||
from summary_mcp.models.summary_io import ExtractionInput
|
||||
|
||||
|
||||
def normalize_input(extraction_input: ExtractionInput) -> ExtractionInput:
|
||||
if extraction_input.item is not None:
|
||||
return extraction_input
|
||||
|
||||
if extraction_input.url:
|
||||
return ExtractionInput(
|
||||
item=Item(url=extraction_input.url, title=None),
|
||||
raw_html=extraction_input.raw_html,
|
||||
rss_content=extraction_input.rss_content,
|
||||
language_hint=extraction_input.language_hint,
|
||||
)
|
||||
|
||||
raise SummaryError(
|
||||
code="INVALID_INPUT",
|
||||
message="Either item or url must be provided",
|
||||
retryable=False,
|
||||
stage="normalize",
|
||||
)
|
||||
@@ -0,0 +1,46 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from summary_mcp.core.content_loader import choose_inline_content, fetch_html
|
||||
from summary_mcp.core.errors import SummaryError
|
||||
from summary_mcp.core.extractor import extract_plain_text, extract_title
|
||||
from summary_mcp.core.mapper import build_article
|
||||
from summary_mcp.core.normalizer import normalize_input
|
||||
from summary_mcp.core.quality_checker import assess_quality
|
||||
from summary_mcp.models.summary_io import DebugInfo, ExtractionInput, ExtractionOutput
|
||||
|
||||
|
||||
def extract_content(extraction_input: ExtractionInput) -> ExtractionOutput:
|
||||
try:
|
||||
normalized = normalize_input(extraction_input)
|
||||
assert normalized.item is not None
|
||||
|
||||
inline_content, content_source = choose_inline_content(normalized)
|
||||
if inline_content is None:
|
||||
inline_content = fetch_html(str(normalized.item.url))
|
||||
content_source = "fetched_html"
|
||||
|
||||
if normalized.item.title is None and (
|
||||
content_source in {"raw_html", "fetched_html"} or inline_content.lstrip().startswith("<")
|
||||
):
|
||||
normalized.item.title = extract_title(inline_content)
|
||||
|
||||
plain_text, extractor_name = extract_plain_text(inline_content, content_source)
|
||||
quality_flags, warnings = assess_quality(plain_text)
|
||||
article = build_article(
|
||||
normalized.item,
|
||||
plain_text,
|
||||
quality_flags,
|
||||
content_source,
|
||||
extractor_name,
|
||||
)
|
||||
return ExtractionOutput(
|
||||
success=True,
|
||||
article=article,
|
||||
debug=DebugInfo(
|
||||
content_source=content_source,
|
||||
extractor=extractor_name,
|
||||
),
|
||||
warnings=warnings,
|
||||
)
|
||||
except SummaryError as exc:
|
||||
return ExtractionOutput(success=False, error=exc.to_error_info(), warnings=[])
|
||||
@@ -0,0 +1,25 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from summary_mcp.models.document import QualityFlags
|
||||
|
||||
|
||||
PAYWALL_HINTS = ("subscribe to read", "会员", "付费", "订阅后查看", "sign in to continue")
|
||||
|
||||
|
||||
def assess_quality(text: str) -> tuple[QualityFlags, list[str]]:
|
||||
lowered = text.lower()
|
||||
flags = QualityFlags(
|
||||
is_paywalled=any(hint in lowered for hint in PAYWALL_HINTS),
|
||||
is_truncated=text.endswith("...") or text.endswith("……"),
|
||||
is_low_content=len(text.strip()) < 500,
|
||||
)
|
||||
|
||||
warnings: list[str] = []
|
||||
if flags.is_paywalled:
|
||||
warnings.append("Potential paywall detected in content.")
|
||||
if flags.is_truncated:
|
||||
warnings.append("Content may be truncated.")
|
||||
if flags.is_low_content:
|
||||
warnings.append("Content has low information density.")
|
||||
|
||||
return flags, warnings
|
||||
@@ -0,0 +1 @@
|
||||
"""Shared models for the summary MCP service."""
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,33 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import datetime
|
||||
from typing import Any, Literal
|
||||
|
||||
from pydantic import BaseModel, Field, HttpUrl
|
||||
|
||||
from .item import ContentKind
|
||||
|
||||
|
||||
PipelineState = Literal["ingested", "extracted", "filtered", "stored", "pushed", "dropped"]
|
||||
|
||||
|
||||
class QualityFlags(BaseModel):
|
||||
is_paywalled: bool = False
|
||||
is_truncated: bool = False
|
||||
is_low_content: bool = False
|
||||
|
||||
|
||||
class ExtractedArticle(BaseModel):
|
||||
extract_id: str
|
||||
item_id: str | None = None
|
||||
source_id: str | None = None
|
||||
url: HttpUrl
|
||||
title: str | None = None
|
||||
author: str | None = None
|
||||
published_at: datetime | None = None
|
||||
language: str | None = None
|
||||
content_kind: ContentKind = "article"
|
||||
plain_text: str
|
||||
quality_flags: QualityFlags = Field(default_factory=QualityFlags)
|
||||
metadata: dict[str, Any] = Field(default_factory=dict)
|
||||
pipeline_state: PipelineState = "extracted"
|
||||
@@ -0,0 +1,27 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import datetime
|
||||
from typing import Any, Literal
|
||||
|
||||
from pydantic import BaseModel, Field, HttpUrl
|
||||
|
||||
|
||||
ContentKind = Literal["article", "thread", "release", "changelog", "video", "mixed"]
|
||||
FetchState = Literal["pending", "fetched", "failed", "skipped"]
|
||||
|
||||
|
||||
class Item(BaseModel):
|
||||
item_id: str | None = None
|
||||
source_id: str | None = None
|
||||
external_id: str | None = None
|
||||
title: str | None = None
|
||||
url: HttpUrl
|
||||
author: str | None = None
|
||||
published_at: datetime | None = None
|
||||
discovered_at: datetime | None = None
|
||||
content_kind: ContentKind = "article"
|
||||
language: str | None = None
|
||||
raw_summary: str | None = None
|
||||
raw_content: str | None = None
|
||||
metadata: dict[str, Any] = Field(default_factory=dict)
|
||||
fetch_state: FetchState = "pending"
|
||||
@@ -0,0 +1,30 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Literal
|
||||
|
||||
from pydantic import BaseModel, Field, HttpUrl, field_validator
|
||||
|
||||
|
||||
Category = Literal["资讯", "方法论", "工具实践", "观点评论"]
|
||||
|
||||
|
||||
class LlmSummaryResult(BaseModel):
|
||||
title: str = Field(min_length=1)
|
||||
url: HttpUrl
|
||||
summary: str = Field(min_length=20, max_length=140)
|
||||
highlights: list[str] = Field(min_length=3, max_length=5)
|
||||
keywords: list[str] = Field(min_length=5, max_length=8)
|
||||
topics: list[str] = Field(min_length=3, max_length=5)
|
||||
category: Category
|
||||
worth_keeping: bool
|
||||
reason: str = Field(min_length=1)
|
||||
|
||||
@field_validator("title", "summary", "reason")
|
||||
@classmethod
|
||||
def normalize_text_fields(cls, value: str) -> str:
|
||||
return value.strip()
|
||||
|
||||
@field_validator("highlights", "keywords", "topics")
|
||||
@classmethod
|
||||
def normalize_list_fields(cls, values: list[str]) -> list[str]:
|
||||
return [value.strip() for value in values if value.strip()]
|
||||
@@ -0,0 +1,37 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
from .document import ExtractedArticle
|
||||
from .item import Item
|
||||
|
||||
|
||||
class ExtractionInput(BaseModel):
|
||||
item: Item | None = None
|
||||
raw_html: str | None = None
|
||||
rss_content: str | None = None
|
||||
language_hint: str | None = None
|
||||
url: str | None = None
|
||||
|
||||
|
||||
class ErrorInfo(BaseModel):
|
||||
code: str
|
||||
message: str
|
||||
retryable: bool
|
||||
stage: str
|
||||
details: dict[str, Any] = Field(default_factory=dict)
|
||||
|
||||
|
||||
class DebugInfo(BaseModel):
|
||||
content_source: str | None = None
|
||||
extractor: str | None = None
|
||||
|
||||
|
||||
class ExtractionOutput(BaseModel):
|
||||
success: bool
|
||||
article: ExtractedArticle | None = None
|
||||
error: ErrorInfo | None = None
|
||||
debug: DebugInfo | None = None
|
||||
warnings: list[str] = Field(default_factory=list)
|
||||
@@ -0,0 +1,40 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from mcp.server.fastmcp import FastMCP
|
||||
|
||||
from summary_mcp.core.pipeline import extract_content
|
||||
from summary_mcp.models.item import Item
|
||||
from summary_mcp.models.summary_io import ExtractionInput
|
||||
|
||||
|
||||
mcp = FastMCP(name="content-extract-mcp")
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def extract_url_content(url: str, language_hint: str | None = None) -> dict:
|
||||
"""Extract structured article content from a single URL."""
|
||||
result = extract_content(
|
||||
ExtractionInput(
|
||||
url=url,
|
||||
language_hint=language_hint,
|
||||
)
|
||||
)
|
||||
return result.model_dump(mode="json")
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def extract_item_content(item: dict) -> dict:
|
||||
"""Extract structured article content from a normalized item object."""
|
||||
parsed_item = Item.model_validate(item)
|
||||
result = extract_content(
|
||||
ExtractionInput(item=parsed_item)
|
||||
)
|
||||
return result.model_dump(mode="json")
|
||||
|
||||
|
||||
def main() -> None:
|
||||
mcp.run()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,33 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from summary_mcp.validators.llm_result import validate_llm_result
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description="Validate an LLM summary JSON result.")
|
||||
parser.add_argument("result", type=Path, help="Path to the LLM result JSON file")
|
||||
parser.add_argument(
|
||||
"--extracted",
|
||||
type=Path,
|
||||
default=None,
|
||||
help="Optional extracted article JSON used for title/url consistency checks",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
report = validate_llm_result(args.result, args.extracted)
|
||||
payload = {
|
||||
"valid": report.valid,
|
||||
"errors": report.errors,
|
||||
"warnings": report.warnings,
|
||||
"normalized_result": report.normalized_result,
|
||||
}
|
||||
print(json.dumps(payload, ensure_ascii=False, indent=2))
|
||||
raise SystemExit(0 if report.valid else 1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1 @@
|
||||
"""Validation helpers for LLM outputs."""
|
||||
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,84 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from pydantic import ValidationError
|
||||
|
||||
from summary_mcp.models.llm_result import LlmSummaryResult
|
||||
|
||||
|
||||
@dataclass
|
||||
class ValidationReport:
|
||||
valid: bool
|
||||
errors: list[str] = field(default_factory=list)
|
||||
warnings: list[str] = field(default_factory=list)
|
||||
normalized_result: dict[str, Any] | None = None
|
||||
|
||||
|
||||
def _load_json(path: Path) -> dict[str, Any]:
|
||||
with path.open("r", encoding="utf-8") as handle:
|
||||
return json.load(handle)
|
||||
|
||||
|
||||
def _validate_business_rules(result: LlmSummaryResult, extracted: dict[str, Any] | None) -> tuple[list[str], list[str]]:
|
||||
errors: list[str] = []
|
||||
warnings: list[str] = []
|
||||
|
||||
keyword_overlap = set(result.keywords) & set(result.topics)
|
||||
if keyword_overlap:
|
||||
errors.append(f"`keywords` and `topics` must not overlap: {sorted(keyword_overlap)}")
|
||||
|
||||
if len(set(result.highlights)) != len(result.highlights):
|
||||
errors.append("`highlights` contains duplicate entries")
|
||||
if len(set(result.keywords)) != len(result.keywords):
|
||||
errors.append("`keywords` contains duplicate entries")
|
||||
if len(set(result.topics)) != len(result.topics):
|
||||
errors.append("`topics` contains duplicate entries")
|
||||
|
||||
if extracted:
|
||||
article = extracted.get("article") or {}
|
||||
extracted_title = article.get("title")
|
||||
extracted_url = article.get("url")
|
||||
if extracted_title and result.title != extracted_title:
|
||||
errors.append("`title` does not match extracted article title")
|
||||
if extracted_url and str(result.url) != extracted_url:
|
||||
errors.append("`url` does not match extracted article url")
|
||||
|
||||
if result.category == "\u8d44\u8baf" and result.worth_keeping:
|
||||
warnings.append("`??` category marked as worth keeping; check if this is intentional.")
|
||||
|
||||
return errors, warnings
|
||||
|
||||
|
||||
def validate_llm_result(
|
||||
result_path: Path,
|
||||
extracted_path: Path | None = None,
|
||||
) -> ValidationReport:
|
||||
try:
|
||||
raw_result = _load_json(result_path)
|
||||
except json.JSONDecodeError as exc:
|
||||
return ValidationReport(valid=False, errors=[f"Invalid JSON: {exc}"])
|
||||
|
||||
extracted: dict[str, Any] | None = None
|
||||
if extracted_path is not None:
|
||||
try:
|
||||
extracted = _load_json(extracted_path)
|
||||
except json.JSONDecodeError as exc:
|
||||
return ValidationReport(valid=False, errors=[f"Invalid extracted JSON: {exc}"])
|
||||
|
||||
try:
|
||||
parsed = LlmSummaryResult.model_validate(raw_result)
|
||||
except ValidationError as exc:
|
||||
errors = [f"{'.'.join(str(part) for part in error['loc'])}: {error['msg']}" for error in exc.errors()]
|
||||
return ValidationReport(valid=False, errors=errors)
|
||||
|
||||
errors, warnings = _validate_business_rules(parsed, extracted)
|
||||
return ValidationReport(
|
||||
valid=not errors,
|
||||
errors=errors,
|
||||
warnings=warnings,
|
||||
normalized_result=parsed.model_dump(mode="json"),
|
||||
)
|
||||
Reference in New Issue
Block a user