Add inline comments to core modules and code review notes
Added explanatory comments to server.py, freshrss_pipeline.py, pipeline.py, freshrss.py, keyword_index.py, summary_loop.py, and filters/engine.py. Also added CODE_REVIEW.md with prioritized improvement suggestions.
This commit is contained in:
@@ -1,5 +1,8 @@
|
||||
from __future__ import annotations
|
||||
|
||||
# 日报级关键词索引:从当日 delivery 候选中提取关键词,
|
||||
# 应用别名归并和停用词过滤,输出 DailyKeywordIndex 并聚合全局 KeywordStatsIndex。
|
||||
|
||||
import json
|
||||
from collections import Counter
|
||||
from datetime import UTC, date, datetime
|
||||
@@ -83,6 +86,7 @@ def normalize_keyword(
|
||||
aliases: dict[str, str],
|
||||
stopwords: set[str],
|
||||
) -> str | None:
|
||||
# 对单个关键词做别名替换 + 停用词过滤,返回 None 表示丢弃
|
||||
raw_value = keyword.strip()
|
||||
if not raw_value:
|
||||
return None
|
||||
@@ -200,6 +204,7 @@ def persist_keyword_indexes(
|
||||
stopwords_path: Path | None = None,
|
||||
include_decisions: set[str] | None = None,
|
||||
) -> dict[str, object]:
|
||||
# 构建并写入当日词元索引,同时全量重建全局词频统计;每次 pipeline 运行后自动调用
|
||||
aliases = load_term_aliases(aliases_path)
|
||||
stopwords = load_term_stopwords(stopwords_path)
|
||||
daily_index = build_daily_keyword_index(
|
||||
|
||||
@@ -1,5 +1,8 @@
|
||||
from __future__ import annotations
|
||||
|
||||
# 内容提取主流程:根据 item 来源决定使用内联 RSS 内容还是回源抓取,
|
||||
# 再经过文本提取、质量评估、文章构建,输出标准化 ExtractionOutput。
|
||||
|
||||
from summary_mcp.core.content_loader import choose_inline_content, fetch_html
|
||||
from summary_mcp.core.errors import SummaryError
|
||||
from summary_mcp.core.extractor import extract_plain_text, extract_title
|
||||
@@ -9,10 +12,12 @@ from summary_mcp.core.quality_checker import assess_quality
|
||||
from summary_mcp.models.summary_io import DebugInfo, ExtractionInput, ExtractionOutput
|
||||
|
||||
|
||||
# FreshRSS 来源的 item 不回源抓网页,直接使用 RSS 内联内容
|
||||
RSS_ONLY_UPSTREAMS = {"freshrss"}
|
||||
|
||||
|
||||
def _should_skip_fetch(extraction_input: ExtractionInput) -> bool:
|
||||
# 判断当前 item 是否属于 RSS-only 上游,若是则禁止回源抓取
|
||||
item = extraction_input.item
|
||||
if item is None:
|
||||
return False
|
||||
|
||||
@@ -1,5 +1,8 @@
|
||||
from __future__ import annotations
|
||||
|
||||
# LLM 摘要循环:构造提示词 -> 调用 LLM -> 解析 JSON -> 校验 -> 失败时生成修复提示词重试。
|
||||
# 核心入口:run_loop_payload(),最多重试 max_retries 次。
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
|
||||
Reference in New Issue
Block a user