first commit

This commit is contained in:
zhuyongxin
2026-03-24 17:01:35 +08:00
commit 1dfae8ca19
68 changed files with 3898 additions and 0 deletions
+2
View File
@@ -0,0 +1,2 @@
"""Summary MCP service package."""
Binary file not shown.
Binary file not shown.
+1
View File
@@ -0,0 +1 @@
"""Core pipeline for the summary MCP service."""
+38
View File
@@ -0,0 +1,38 @@
from __future__ import annotations
import httpx
from summary_mcp.core.errors import SummaryError
from summary_mcp.models.summary_io import ExtractionInput
def choose_inline_content(extraction_input: ExtractionInput) -> tuple[str | None, str]:
if extraction_input.raw_html:
return extraction_input.raw_html, "raw_html"
if extraction_input.item and extraction_input.item.raw_content and len(extraction_input.item.raw_content.strip()) >= 500:
return extraction_input.item.raw_content, "item.raw_content"
if extraction_input.rss_content and len(extraction_input.rss_content.strip()) >= 500:
return extraction_input.rss_content, "rss_content"
return None, "none"
def fetch_html(url: str) -> str:
headers = {
"User-Agent": "summary-mcp/0.1 (+https://modelcontextprotocol.io/)",
}
try:
with httpx.Client(follow_redirects=True, timeout=15.0, headers=headers) as client:
response = client.get(url)
response.raise_for_status()
return response.text
except httpx.HTTPError as exc:
raise SummaryError(
code="CONTENT_FETCH_FAILED",
message="Failed to fetch article content",
retryable=True,
stage="fetch",
details={"url": url, "reason": str(exc)},
) from exc
+24
View File
@@ -0,0 +1,24 @@
from __future__ import annotations
from dataclasses import dataclass, field
from typing import Any
from summary_mcp.models.summary_io import ErrorInfo
@dataclass
class SummaryError(Exception):
code: str
message: str
retryable: bool
stage: str
details: dict[str, Any] = field(default_factory=dict)
def to_error_info(self) -> ErrorInfo:
return ErrorInfo(
code=self.code,
message=self.message,
retryable=self.retryable,
stage=self.stage,
details=self.details,
)
+63
View File
@@ -0,0 +1,63 @@
from __future__ import annotations
from bs4 import BeautifulSoup
import trafilatura
from summary_mcp.core.errors import SummaryError
def extract_title(content: str) -> str | None:
soup = BeautifulSoup(content, "html.parser")
og_title = soup.find("meta", attrs={"property": "og:title"})
if og_title and og_title.get("content"):
return og_title["content"].strip()
if soup.title and soup.title.string:
raw_title = soup.title.string.strip()
return raw_title.split("|", 1)[0].strip()
heading = soup.find(["h1", "h2"])
if heading:
heading_text = heading.get_text(" ", strip=True)
if heading_text:
return heading_text
return None
def _dedupe_leading_lines(text: str) -> str:
lines = [line.strip() for line in text.splitlines() if line.strip()]
if len(lines) >= 2 and lines[0] == lines[1]:
lines.pop(0)
return "\n".join(lines).strip()
def extract_plain_text(content: str, content_source: str) -> tuple[str, str]:
if content_source == "raw_html" or content.lstrip().startswith("<"):
extracted = trafilatura.extract(content, include_links=False, include_formatting=False)
if extracted and len(extracted.strip()) >= 200:
return _dedupe_leading_lines(extracted), "trafilatura"
soup = BeautifulSoup(content, "html.parser")
fallback = " ".join(soup.stripped_strings)
if len(fallback.strip()) >= 200:
return fallback.strip(), "beautifulsoup"
raise SummaryError(
code="CONTENT_EXTRACTION_FAILED",
message="Failed to extract article body from HTML",
retryable=False,
stage="extract",
)
if len(content.strip()) < 200:
raise SummaryError(
code="CONTENT_TOO_SHORT",
message="Content is too short to summarize reliably",
retryable=False,
stage="extract",
details={"length": len(content.strip())},
)
return content.strip(), "inline"
+40
View File
@@ -0,0 +1,40 @@
from __future__ import annotations
import hashlib
from summary_mcp.models.document import ExtractedArticle, QualityFlags
from summary_mcp.models.item import Item
def _extract_id(item: Item, plain_text: str) -> str:
seed = f"{item.url}|{item.title or ''}|{len(plain_text)}"
digest = hashlib.sha256(seed.encode("utf-8")).hexdigest()
return f"sha256:{digest}"
def build_article(
item: Item,
plain_text: str,
quality_flags: QualityFlags,
content_source: str,
extractor_name: str,
) -> ExtractedArticle:
return ExtractedArticle(
extract_id=_extract_id(item, plain_text),
item_id=item.item_id,
source_id=item.source_id,
url=item.url,
title=item.title,
author=item.author,
published_at=item.published_at,
language=item.language,
content_kind=item.content_kind,
plain_text=plain_text,
quality_flags=quality_flags,
metadata={
"content_source": content_source,
"extractor": extractor_name,
"char_count": len(plain_text),
},
pipeline_state="extracted",
)
+25
View File
@@ -0,0 +1,25 @@
from __future__ import annotations
from summary_mcp.core.errors import SummaryError
from summary_mcp.models.item import Item
from summary_mcp.models.summary_io import ExtractionInput
def normalize_input(extraction_input: ExtractionInput) -> ExtractionInput:
if extraction_input.item is not None:
return extraction_input
if extraction_input.url:
return ExtractionInput(
item=Item(url=extraction_input.url, title=None),
raw_html=extraction_input.raw_html,
rss_content=extraction_input.rss_content,
language_hint=extraction_input.language_hint,
)
raise SummaryError(
code="INVALID_INPUT",
message="Either item or url must be provided",
retryable=False,
stage="normalize",
)
+46
View File
@@ -0,0 +1,46 @@
from __future__ import annotations
from summary_mcp.core.content_loader import choose_inline_content, fetch_html
from summary_mcp.core.errors import SummaryError
from summary_mcp.core.extractor import extract_plain_text, extract_title
from summary_mcp.core.mapper import build_article
from summary_mcp.core.normalizer import normalize_input
from summary_mcp.core.quality_checker import assess_quality
from summary_mcp.models.summary_io import DebugInfo, ExtractionInput, ExtractionOutput
def extract_content(extraction_input: ExtractionInput) -> ExtractionOutput:
try:
normalized = normalize_input(extraction_input)
assert normalized.item is not None
inline_content, content_source = choose_inline_content(normalized)
if inline_content is None:
inline_content = fetch_html(str(normalized.item.url))
content_source = "fetched_html"
if normalized.item.title is None and (
content_source in {"raw_html", "fetched_html"} or inline_content.lstrip().startswith("<")
):
normalized.item.title = extract_title(inline_content)
plain_text, extractor_name = extract_plain_text(inline_content, content_source)
quality_flags, warnings = assess_quality(plain_text)
article = build_article(
normalized.item,
plain_text,
quality_flags,
content_source,
extractor_name,
)
return ExtractionOutput(
success=True,
article=article,
debug=DebugInfo(
content_source=content_source,
extractor=extractor_name,
),
warnings=warnings,
)
except SummaryError as exc:
return ExtractionOutput(success=False, error=exc.to_error_info(), warnings=[])
+25
View File
@@ -0,0 +1,25 @@
from __future__ import annotations
from summary_mcp.models.document import QualityFlags
PAYWALL_HINTS = ("subscribe to read", "会员", "付费", "订阅后查看", "sign in to continue")
def assess_quality(text: str) -> tuple[QualityFlags, list[str]]:
lowered = text.lower()
flags = QualityFlags(
is_paywalled=any(hint in lowered for hint in PAYWALL_HINTS),
is_truncated=text.endswith("...") or text.endswith("……"),
is_low_content=len(text.strip()) < 500,
)
warnings: list[str] = []
if flags.is_paywalled:
warnings.append("Potential paywall detected in content.")
if flags.is_truncated:
warnings.append("Content may be truncated.")
if flags.is_low_content:
warnings.append("Content has low information density.")
return flags, warnings
+1
View File
@@ -0,0 +1 @@
"""Shared models for the summary MCP service."""
+33
View File
@@ -0,0 +1,33 @@
from __future__ import annotations
from datetime import datetime
from typing import Any, Literal
from pydantic import BaseModel, Field, HttpUrl
from .item import ContentKind
PipelineState = Literal["ingested", "extracted", "filtered", "stored", "pushed", "dropped"]
class QualityFlags(BaseModel):
is_paywalled: bool = False
is_truncated: bool = False
is_low_content: bool = False
class ExtractedArticle(BaseModel):
extract_id: str
item_id: str | None = None
source_id: str | None = None
url: HttpUrl
title: str | None = None
author: str | None = None
published_at: datetime | None = None
language: str | None = None
content_kind: ContentKind = "article"
plain_text: str
quality_flags: QualityFlags = Field(default_factory=QualityFlags)
metadata: dict[str, Any] = Field(default_factory=dict)
pipeline_state: PipelineState = "extracted"
+27
View File
@@ -0,0 +1,27 @@
from __future__ import annotations
from datetime import datetime
from typing import Any, Literal
from pydantic import BaseModel, Field, HttpUrl
ContentKind = Literal["article", "thread", "release", "changelog", "video", "mixed"]
FetchState = Literal["pending", "fetched", "failed", "skipped"]
class Item(BaseModel):
item_id: str | None = None
source_id: str | None = None
external_id: str | None = None
title: str | None = None
url: HttpUrl
author: str | None = None
published_at: datetime | None = None
discovered_at: datetime | None = None
content_kind: ContentKind = "article"
language: str | None = None
raw_summary: str | None = None
raw_content: str | None = None
metadata: dict[str, Any] = Field(default_factory=dict)
fetch_state: FetchState = "pending"
+30
View File
@@ -0,0 +1,30 @@
from __future__ import annotations
from typing import Literal
from pydantic import BaseModel, Field, HttpUrl, field_validator
Category = Literal["资讯", "方法论", "工具实践", "观点评论"]
class LlmSummaryResult(BaseModel):
title: str = Field(min_length=1)
url: HttpUrl
summary: str = Field(min_length=20, max_length=140)
highlights: list[str] = Field(min_length=3, max_length=5)
keywords: list[str] = Field(min_length=5, max_length=8)
topics: list[str] = Field(min_length=3, max_length=5)
category: Category
worth_keeping: bool
reason: str = Field(min_length=1)
@field_validator("title", "summary", "reason")
@classmethod
def normalize_text_fields(cls, value: str) -> str:
return value.strip()
@field_validator("highlights", "keywords", "topics")
@classmethod
def normalize_list_fields(cls, values: list[str]) -> list[str]:
return [value.strip() for value in values if value.strip()]
+37
View File
@@ -0,0 +1,37 @@
from __future__ import annotations
from typing import Any
from pydantic import BaseModel, Field
from .document import ExtractedArticle
from .item import Item
class ExtractionInput(BaseModel):
item: Item | None = None
raw_html: str | None = None
rss_content: str | None = None
language_hint: str | None = None
url: str | None = None
class ErrorInfo(BaseModel):
code: str
message: str
retryable: bool
stage: str
details: dict[str, Any] = Field(default_factory=dict)
class DebugInfo(BaseModel):
content_source: str | None = None
extractor: str | None = None
class ExtractionOutput(BaseModel):
success: bool
article: ExtractedArticle | None = None
error: ErrorInfo | None = None
debug: DebugInfo | None = None
warnings: list[str] = Field(default_factory=list)
+40
View File
@@ -0,0 +1,40 @@
from __future__ import annotations
from mcp.server.fastmcp import FastMCP
from summary_mcp.core.pipeline import extract_content
from summary_mcp.models.item import Item
from summary_mcp.models.summary_io import ExtractionInput
mcp = FastMCP(name="content-extract-mcp")
@mcp.tool()
def extract_url_content(url: str, language_hint: str | None = None) -> dict:
"""Extract structured article content from a single URL."""
result = extract_content(
ExtractionInput(
url=url,
language_hint=language_hint,
)
)
return result.model_dump(mode="json")
@mcp.tool()
def extract_item_content(item: dict) -> dict:
"""Extract structured article content from a normalized item object."""
parsed_item = Item.model_validate(item)
result = extract_content(
ExtractionInput(item=parsed_item)
)
return result.model_dump(mode="json")
def main() -> None:
mcp.run()
if __name__ == "__main__":
main()
+33
View File
@@ -0,0 +1,33 @@
from __future__ import annotations
import argparse
import json
from pathlib import Path
from summary_mcp.validators.llm_result import validate_llm_result
def main() -> None:
parser = argparse.ArgumentParser(description="Validate an LLM summary JSON result.")
parser.add_argument("result", type=Path, help="Path to the LLM result JSON file")
parser.add_argument(
"--extracted",
type=Path,
default=None,
help="Optional extracted article JSON used for title/url consistency checks",
)
args = parser.parse_args()
report = validate_llm_result(args.result, args.extracted)
payload = {
"valid": report.valid,
"errors": report.errors,
"warnings": report.warnings,
"normalized_result": report.normalized_result,
}
print(json.dumps(payload, ensure_ascii=False, indent=2))
raise SystemExit(0 if report.valid else 1)
if __name__ == "__main__":
main()
+1
View File
@@ -0,0 +1 @@
"""Validation helpers for LLM outputs."""
+84
View File
@@ -0,0 +1,84 @@
from __future__ import annotations
import json
from dataclasses import dataclass, field
from pathlib import Path
from typing import Any
from pydantic import ValidationError
from summary_mcp.models.llm_result import LlmSummaryResult
@dataclass
class ValidationReport:
valid: bool
errors: list[str] = field(default_factory=list)
warnings: list[str] = field(default_factory=list)
normalized_result: dict[str, Any] | None = None
def _load_json(path: Path) -> dict[str, Any]:
with path.open("r", encoding="utf-8") as handle:
return json.load(handle)
def _validate_business_rules(result: LlmSummaryResult, extracted: dict[str, Any] | None) -> tuple[list[str], list[str]]:
errors: list[str] = []
warnings: list[str] = []
keyword_overlap = set(result.keywords) & set(result.topics)
if keyword_overlap:
errors.append(f"`keywords` and `topics` must not overlap: {sorted(keyword_overlap)}")
if len(set(result.highlights)) != len(result.highlights):
errors.append("`highlights` contains duplicate entries")
if len(set(result.keywords)) != len(result.keywords):
errors.append("`keywords` contains duplicate entries")
if len(set(result.topics)) != len(result.topics):
errors.append("`topics` contains duplicate entries")
if extracted:
article = extracted.get("article") or {}
extracted_title = article.get("title")
extracted_url = article.get("url")
if extracted_title and result.title != extracted_title:
errors.append("`title` does not match extracted article title")
if extracted_url and str(result.url) != extracted_url:
errors.append("`url` does not match extracted article url")
if result.category == "\u8d44\u8baf" and result.worth_keeping:
warnings.append("`??` category marked as worth keeping; check if this is intentional.")
return errors, warnings
def validate_llm_result(
result_path: Path,
extracted_path: Path | None = None,
) -> ValidationReport:
try:
raw_result = _load_json(result_path)
except json.JSONDecodeError as exc:
return ValidationReport(valid=False, errors=[f"Invalid JSON: {exc}"])
extracted: dict[str, Any] | None = None
if extracted_path is not None:
try:
extracted = _load_json(extracted_path)
except json.JSONDecodeError as exc:
return ValidationReport(valid=False, errors=[f"Invalid extracted JSON: {exc}"])
try:
parsed = LlmSummaryResult.model_validate(raw_result)
except ValidationError as exc:
errors = [f"{'.'.join(str(part) for part in error['loc'])}: {error['msg']}" for error in exc.errors()]
return ValidationReport(valid=False, errors=errors)
errors, warnings = _validate_business_rules(parsed, extracted)
return ValidationReport(
valid=not errors,
errors=errors,
warnings=warnings,
normalized_result=parsed.model_dump(mode="json"),
)