Merge remote-tracking branch 'origin/main'
This commit is contained in:
@@ -4,6 +4,7 @@ __pycache__/
|
|||||||
*.pyc
|
*.pyc
|
||||||
*.egg-info/
|
*.egg-info/
|
||||||
.env.local
|
.env.local
|
||||||
|
.env
|
||||||
findings.md
|
findings.md
|
||||||
progress.md
|
progress.md
|
||||||
task_plan.md
|
task_plan.md
|
||||||
|
|||||||
@@ -16,6 +16,11 @@ The server exposes four tools:
|
|||||||
- `filter_summary_result`
|
- `filter_summary_result`
|
||||||
- `run_freshrss_openclaw_pipeline`
|
- `run_freshrss_openclaw_pipeline`
|
||||||
|
|
||||||
|
Article-summary post-processing (separate LLM optional):
|
||||||
|
|
||||||
|
- `article-summary` MCP tool (operates on existing extracted payloads)
|
||||||
|
- `scripts/run_article_summaries.py` CLI helper
|
||||||
|
|
||||||
Validate an LLM summary result:
|
Validate an LLM summary result:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
@@ -91,13 +96,15 @@ This is the recommended production entrypoint. By default it writes only:
|
|||||||
- `outputs/freshrss/rerun/<timestamp>/raw/freshrss.raw.json`
|
- `outputs/freshrss/rerun/<timestamp>/raw/freshrss.raw.json`
|
||||||
- `outputs/freshrss/rerun/<timestamp>/candidates/openclaw-delivery-payload.json`
|
- `outputs/freshrss/rerun/<timestamp>/candidates/openclaw-delivery-payload.json`
|
||||||
- `outputs/freshrss/rerun/<timestamp>/run-report.json`
|
- `outputs/freshrss/rerun/<timestamp>/run-report.json`
|
||||||
|
- `outputs/freshrss/rerun/<timestamp>/extracted/item-XX.extracted.json` (one per item)
|
||||||
|
|
||||||
It also updates the daily keyword index runtime data:
|
It also updates the daily keyword index runtime data:
|
||||||
|
|
||||||
- `data/term_index/daily/YYYY-MM-DD.json`
|
- `data/term_index/daily/YYYY-MM-DD.json`
|
||||||
- `data/term_index/term_stats.json`
|
- `data/term_index/term_stats.json`
|
||||||
|
|
||||||
If you need per-item intermediates, add `--debug-artifacts`.
|
The main pipeline does not emit a batch-level `freshrss.extracted.json` file by default.
|
||||||
|
If you need additional per-item intermediates such as normalized items, summaries, filter decisions, candidate records, or candidate inputs, add `--debug-artifacts`.
|
||||||
|
|
||||||
When OpenClaw is connected to the MCP server, it should call `run_freshrss_openclaw_pipeline` for the same behavior directly through MCP. The tool also supports `debug_artifacts=true` when deeper inspection is needed.
|
When OpenClaw is connected to the MCP server, it should call `run_freshrss_openclaw_pipeline` for the same behavior directly through MCP. The tool also supports `debug_artifacts=true` when deeper inspection is needed.
|
||||||
|
|
||||||
@@ -214,3 +221,40 @@ python scripts/apply_term_suggestions.py ^
|
|||||||
```
|
```
|
||||||
|
|
||||||
Remove `--dry-run` to write the accepted changes. The script can also apply accepted `alias`, `stopword`, and `interest keyword` suggestions through `--accept-alias`, `--accept-stopword`, and `--accept-interest`. Accepted watch terms are written into `configs/term_watchlist.json`, and every applied action is appended into `configs/term_change_log.json`.
|
Remove `--dry-run` to write the accepted changes. The script can also apply accepted `alias`, `stopword`, and `interest keyword` suggestions through `--accept-alias`, `--accept-stopword`, and `--accept-interest`. Accepted watch terms are written into `configs/term_watchlist.json`, and every applied action is appended into `configs/term_change_log.json`.
|
||||||
|
|
||||||
|
|
||||||
|
## Article-summary LLM configuration
|
||||||
|
|
||||||
|
Set a dedicated model for post-processing summaries without affecting the main pipeline:
|
||||||
|
|
||||||
|
- `ARTICLE_SUMMARY_LLM_API_URL`
|
||||||
|
- `ARTICLE_SUMMARY_LLM_MODEL`
|
||||||
|
- `ARTICLE_SUMMARY_LLM_API_KEY`
|
||||||
|
|
||||||
|
If these are not set, the summarizer falls back to the main `LLM_*` / `OPENAI_*` settings used elsewhere.
|
||||||
|
|
||||||
|
Example (PowerShell style):
|
||||||
|
|
||||||
|
```bash
|
||||||
|
set ARTICLE_SUMMARY_LLM_API_URL=https://api.deepseek.com
|
||||||
|
set ARTICLE_SUMMARY_LLM_API_KEY=your-article-summary-key
|
||||||
|
set ARTICLE_SUMMARY_LLM_MODEL=deepseek-chat
|
||||||
|
```
|
||||||
|
|
||||||
|
Then run, for example, either via the CLI script:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python scripts/run_article_summaries.py ^
|
||||||
|
--extracted outputs/freshrss/extracted/freshrss.extracted.json ^
|
||||||
|
--ids 12345 67890 ^
|
||||||
|
--output-dir outputs/freshrss/single_summaries
|
||||||
|
```
|
||||||
|
|
||||||
|
…or through the MCP server tool `generate_article_summaries` exposed by `summary_mcp.server`:
|
||||||
|
|
||||||
|
- `extracted_path` (string): path to the extracted JSON, for example `outputs/freshrss/extracted/freshrss.extracted.json`.
|
||||||
|
- `selected_ids` (array of strings): one or more `item_id` values from the extracted payload to summarize.
|
||||||
|
- `output_dir` (optional string): directory to write Markdown summaries. If omitted, summaries are written under `single_summaries/` next to the extracted file.
|
||||||
|
- `llm_api_key` / `llm_model` / `llm_api_url` (optional strings): overrides for article-summary LLM settings. If omitted, the tool falls back to `ARTICLE_SUMMARY_*` or main `LLM_*` env vars as described above.
|
||||||
|
|
||||||
|
The tool returns a JSON array of file paths for the generated Markdown summaries.
|
||||||
|
|||||||
@@ -102,13 +102,16 @@ By default the pipeline writes only:
|
|||||||
- `outputs/freshrss/rerun/<run_id>/raw/freshrss.raw.json`
|
- `outputs/freshrss/rerun/<run_id>/raw/freshrss.raw.json`
|
||||||
- `outputs/freshrss/rerun/<run_id>/candidates/openclaw-delivery-payload.json`
|
- `outputs/freshrss/rerun/<run_id>/candidates/openclaw-delivery-payload.json`
|
||||||
- `outputs/freshrss/rerun/<run_id>/run-report.json`
|
- `outputs/freshrss/rerun/<run_id>/run-report.json`
|
||||||
|
- `outputs/freshrss/rerun/<run_id>/extracted/item-XX.extracted.json` (one per item)
|
||||||
|
|
||||||
It also updates local runtime keyword data:
|
It also updates local runtime keyword data:
|
||||||
|
|
||||||
- `data/term_index/daily/YYYY-MM-DD.json`
|
- `data/term_index/daily/YYYY-MM-DD.json`
|
||||||
- `data/term_index/term_stats.json`
|
- `data/term_index/term_stats.json`
|
||||||
|
|
||||||
If `debug_artifacts=true`, the pipeline additionally writes per-item intermediate files.
|
Per-item extracted files live under `extracted/` and are always written.
|
||||||
|
If `debug_artifacts=true`, the pipeline additionally writes normalized items, summaries, filter decisions, candidate records, and candidate inputs.
|
||||||
|
The main pipeline does not emit a batch-level `freshrss.extracted.json` file by default.
|
||||||
|
|
||||||
## Payload Specs
|
## Payload Specs
|
||||||
|
|
||||||
|
|||||||
+10
-3
@@ -4,7 +4,7 @@
|
|||||||
|
|
||||||
## 生产默认输出
|
## 生产默认输出
|
||||||
|
|
||||||
对于 FreshRSS 主流水线,默认只保留这 3 类文件:
|
对于 FreshRSS 主流水线,默认保留这 4 类文件:
|
||||||
|
|
||||||
- `outputs/freshrss/rerun/<run_id>/raw/freshrss.raw.json`
|
- `outputs/freshrss/rerun/<run_id>/raw/freshrss.raw.json`
|
||||||
- FreshRSS 原始响应快照
|
- FreshRSS 原始响应快照
|
||||||
@@ -12,6 +12,8 @@
|
|||||||
- 发给 OpenClaw 的最终批量 payload
|
- 发给 OpenClaw 的最终批量 payload
|
||||||
- `outputs/freshrss/rerun/<run_id>/run-report.json`
|
- `outputs/freshrss/rerun/<run_id>/run-report.json`
|
||||||
- 本次批处理的运行报告
|
- 本次批处理的运行报告
|
||||||
|
- `outputs/freshrss/rerun/<run_id>/extracted/item-XX.extracted.json`
|
||||||
|
- 单篇文章的正文提取结果(每篇一份)
|
||||||
|
|
||||||
这是当前推荐的生产模式。
|
这是当前推荐的生产模式。
|
||||||
|
|
||||||
@@ -28,12 +30,15 @@
|
|||||||
|
|
||||||
## 调试扩展输出
|
## 调试扩展输出
|
||||||
|
|
||||||
|
默认正式产物已经包含:
|
||||||
|
|
||||||
|
- `outputs/freshrss/rerun/<run_id>/extracted/item-XX.extracted.json`
|
||||||
|
- 单篇正文提取结果
|
||||||
|
|
||||||
当启用 `debug_artifacts` 时,才会额外写出这些中间文件:
|
当启用 `debug_artifacts` 时,才会额外写出这些中间文件:
|
||||||
|
|
||||||
- `outputs/freshrss/rerun/<run_id>/items/`
|
- `outputs/freshrss/rerun/<run_id>/items/`
|
||||||
- 标准化 `item` 中间文件
|
- 标准化 `item` 中间文件
|
||||||
- `outputs/freshrss/rerun/<run_id>/extracted/`
|
|
||||||
- 提取结果
|
|
||||||
- `outputs/freshrss/rerun/<run_id>/summary/`
|
- `outputs/freshrss/rerun/<run_id>/summary/`
|
||||||
- LLM 总结结果及调试尝试文件
|
- LLM 总结结果及调试尝试文件
|
||||||
- `outputs/freshrss/rerun/<run_id>/filter/`
|
- `outputs/freshrss/rerun/<run_id>/filter/`
|
||||||
@@ -43,6 +48,8 @@
|
|||||||
- `outputs/freshrss/rerun/<run_id>/candidates/*.openclaw-candidate-input.json`
|
- `outputs/freshrss/rerun/<run_id>/candidates/*.openclaw-candidate-input.json`
|
||||||
- 单篇 OpenClaw 输入对象
|
- 单篇 OpenClaw 输入对象
|
||||||
|
|
||||||
|
主流水线默认不产出批量聚合版 `freshrss.extracted.json`。
|
||||||
|
|
||||||
## 何时启用调试产物
|
## 何时启用调试产物
|
||||||
|
|
||||||
只在这些场景启用:
|
只在这些场景启用:
|
||||||
|
|||||||
Executable
+68
@@ -0,0 +1,68 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
|
REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||||
|
SRC_ROOT = REPO_ROOT / "src"
|
||||||
|
|
||||||
|
if str(SRC_ROOT) not in sys.path:
|
||||||
|
sys.path.insert(0, str(SRC_ROOT))
|
||||||
|
|
||||||
|
from summary_mcp.workflows.article_summary import summarize_selected_articles
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
parser = argparse.ArgumentParser(
|
||||||
|
description="Generate Markdown summaries for selected articles from an extracted payload.",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--extracted",
|
||||||
|
type=Path,
|
||||||
|
required=True,
|
||||||
|
help="Path to extracted JSON (e.g. outputs/freshrss/extracted/freshrss.extracted.json)",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--ids",
|
||||||
|
nargs="+",
|
||||||
|
required=True,
|
||||||
|
help="One or more item_ids to summarize",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--output-dir",
|
||||||
|
type=Path,
|
||||||
|
default=REPO_ROOT / "outputs" / "freshrss" / "single_summaries",
|
||||||
|
help="Directory to write per-article Markdown summaries",
|
||||||
|
)
|
||||||
|
parser.add_argument("--max-retries", type=int, default=2, help="Maximum LLM retry attempts per article")
|
||||||
|
parser.add_argument("--timeout", type=float, default=60.0, help="LLM request timeout in seconds")
|
||||||
|
parser.add_argument("--api-key", type=str, default=None, help="Override article-summary LLM API key")
|
||||||
|
parser.add_argument("--model", type=str, default=None, help="Override article-summary LLM model")
|
||||||
|
parser.add_argument("--api-url", type=str, default=None, help="Override article-summary LLM API URL/base URL")
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
from summary_mcp.workflows.article_summary import ArticleSummaryConfig
|
||||||
|
|
||||||
|
config = ArticleSummaryConfig(
|
||||||
|
max_retries=args.max_retries,
|
||||||
|
timeout_seconds=args.timeout,
|
||||||
|
)
|
||||||
|
|
||||||
|
paths = summarize_selected_articles(
|
||||||
|
extracted_path=args.extracted,
|
||||||
|
selected_ids=args.ids,
|
||||||
|
output_dir=args.output_dir,
|
||||||
|
config=config,
|
||||||
|
api_key=args.api_key,
|
||||||
|
model=args.model,
|
||||||
|
api_url=args.api_url,
|
||||||
|
)
|
||||||
|
|
||||||
|
for path in paths:
|
||||||
|
print(path)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -5,7 +5,9 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import json
|
import json
|
||||||
from collections import Counter
|
from collections import Counter
|
||||||
from datetime import UTC, date, datetime
|
from datetime import date, datetime, timezone
|
||||||
|
|
||||||
|
UTC = timezone.utc
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Iterable
|
from typing import Iterable
|
||||||
|
|
||||||
|
|||||||
@@ -18,6 +18,7 @@ from summary_mcp.models.item import Item
|
|||||||
from summary_mcp.models.llm_result import LlmSummaryResult
|
from summary_mcp.models.llm_result import LlmSummaryResult
|
||||||
from summary_mcp.models.summary_io import ExtractionInput
|
from summary_mcp.models.summary_io import ExtractionInput
|
||||||
from summary_mcp.workflows import run_freshrss_pipeline
|
from summary_mcp.workflows import run_freshrss_pipeline
|
||||||
|
from summary_mcp.workflows.article_summary import ArticleSummaryConfig, summarize_selected_articles
|
||||||
|
|
||||||
|
|
||||||
mcp = FastMCP(name="content-extract-mcp")
|
mcp = FastMCP(name="content-extract-mcp")
|
||||||
@@ -132,6 +133,62 @@ def run_freshrss_openclaw_pipeline(
|
|||||||
return result
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
@mcp.tool()
|
||||||
|
def generate_article_summaries(
|
||||||
|
*,
|
||||||
|
extracted_path: str,
|
||||||
|
selected_ids: list[str],
|
||||||
|
output_dir: str | None = None,
|
||||||
|
max_retries: int = 2,
|
||||||
|
timeout_seconds: float = 60.0,
|
||||||
|
llm_api_key: str | None = None,
|
||||||
|
llm_model: str | None = None,
|
||||||
|
llm_api_url: str | None = None,
|
||||||
|
) -> list[str]:
|
||||||
|
"""Generate Markdown summaries for selected articles from an extracted payload.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
extracted_path:
|
||||||
|
Path to the extracted JSON file produced by the FreshRSS pipeline
|
||||||
|
(for example `outputs/freshrss/extracted/freshrss.extracted.json`).
|
||||||
|
selected_ids:
|
||||||
|
One or more `item_id` values from the extracted payload to summarize.
|
||||||
|
output_dir:
|
||||||
|
Optional output directory for the generated Markdown files. If omitted,
|
||||||
|
summaries are written next to the extracted file under a
|
||||||
|
`single_summaries/` subdirectory.
|
||||||
|
llm_api_key / llm_model / llm_api_url:
|
||||||
|
Optional overrides for the article-summary LLM settings. If omitted,
|
||||||
|
the workflow falls back to the ARTICLE_SUMMARY_* or main LLM_* env
|
||||||
|
variables as documented in the README.
|
||||||
|
"""
|
||||||
|
|
||||||
|
extracted_path_obj = Path(extracted_path)
|
||||||
|
if not extracted_path_obj.exists():
|
||||||
|
raise FileNotFoundError(f"extracted_path does not exist: {extracted_path}")
|
||||||
|
|
||||||
|
if output_dir is None:
|
||||||
|
default_dir = extracted_path_obj.parent / "single_summaries"
|
||||||
|
output_dir_obj = default_dir
|
||||||
|
else:
|
||||||
|
output_dir_obj = Path(output_dir)
|
||||||
|
|
||||||
|
config = ArticleSummaryConfig(max_retries=max_retries, timeout_seconds=timeout_seconds)
|
||||||
|
|
||||||
|
written_paths = summarize_selected_articles(
|
||||||
|
extracted_path=extracted_path_obj,
|
||||||
|
selected_ids=selected_ids,
|
||||||
|
output_dir=output_dir_obj,
|
||||||
|
config=config,
|
||||||
|
api_key=llm_api_key,
|
||||||
|
model=llm_model,
|
||||||
|
api_url=llm_api_url,
|
||||||
|
)
|
||||||
|
|
||||||
|
return [str(p) for p in written_paths]
|
||||||
|
|
||||||
|
|
||||||
def main() -> None:
|
def main() -> None:
|
||||||
mcp.run()
|
mcp.run()
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,216 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Iterable, Mapping, Sequence
|
||||||
|
|
||||||
|
from summary_mcp.core.summary_loop import run_loop_payload
|
||||||
|
|
||||||
|
|
||||||
|
REPO_ROOT = Path(__file__).resolve().parents[3]
|
||||||
|
OUTPUT_ROOT = REPO_ROOT / "outputs"
|
||||||
|
FRESHRSS_OUTPUT_ROOT = OUTPUT_ROOT / "freshrss"
|
||||||
|
DEFAULT_PROMPT_PATH = OUTPUT_ROOT / "prompts" / "llm-summary-prompt.txt"
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class ArticleSummaryConfig:
|
||||||
|
prompt_path: Path = DEFAULT_PROMPT_PATH
|
||||||
|
max_retries: int = 2
|
||||||
|
timeout_seconds: float = 60.0
|
||||||
|
llm_api_key: str | None = None
|
||||||
|
llm_model: str | None = None
|
||||||
|
llm_api_url: str | None = None
|
||||||
|
|
||||||
|
|
||||||
|
def _resolve_article_llm_settings(
|
||||||
|
*,
|
||||||
|
api_key: str | None = None,
|
||||||
|
model: str | None = None,
|
||||||
|
api_url: str | None = None,
|
||||||
|
) -> tuple[str | None, str | None, str | None]:
|
||||||
|
"""Resolve article-summary LLM settings with fallback to main pipeline env.
|
||||||
|
|
||||||
|
Resolution order for each field:
|
||||||
|
- explicit function argument
|
||||||
|
- ARTICLE_SUMMARY_* environment variable
|
||||||
|
- main LLM_* / OPENAI_* environment variables (handled by summary_loop.resolve_llm_settings)
|
||||||
|
|
||||||
|
This helper intentionally does not validate presence; the summary loop
|
||||||
|
will perform final validation and raise a clear error if nothing is set.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
|
|
||||||
|
resolved_api_key = api_key or os.environ.get("ARTICLE_SUMMARY_LLM_API_KEY")
|
||||||
|
resolved_model = model or os.environ.get("ARTICLE_SUMMARY_LLM_MODEL")
|
||||||
|
resolved_api_url = api_url or os.environ.get("ARTICLE_SUMMARY_LLM_API_URL")
|
||||||
|
return resolved_api_key, resolved_model, resolved_api_url
|
||||||
|
|
||||||
|
|
||||||
|
def _iter_selected_items(
|
||||||
|
extracted_payload: Mapping[str, object],
|
||||||
|
selected_ids: Sequence[str],
|
||||||
|
) -> Iterable[tuple[str, Mapping[str, object]]]:
|
||||||
|
"""Yield (item_id, extracted_entry) pairs for selected items.
|
||||||
|
|
||||||
|
The reader pipeline currently produces an extracted payload shaped like:
|
||||||
|
|
||||||
|
{
|
||||||
|
"results": [
|
||||||
|
{
|
||||||
|
"item": {"item_id": "...", ...},
|
||||||
|
"extraction": {"article": {..}, "warnings": [...]},
|
||||||
|
},
|
||||||
|
...
|
||||||
|
]
|
||||||
|
}
|
||||||
|
|
||||||
|
Historically some flows used a top-level ``items`` array where each
|
||||||
|
element directly contained the ``article`` field. To keep compatibility
|
||||||
|
sane, we first try the real project format (``results``), then fall
|
||||||
|
back to the legacy ``items`` layout if present.
|
||||||
|
"""
|
||||||
|
|
||||||
|
selected_set = set(selected_ids)
|
||||||
|
|
||||||
|
# Preferred: current reader format with top-level "results" array.
|
||||||
|
results = extracted_payload.get("results")
|
||||||
|
if isinstance(results, list):
|
||||||
|
for entry in results:
|
||||||
|
if not isinstance(entry, Mapping):
|
||||||
|
continue
|
||||||
|
item = entry.get("item")
|
||||||
|
extraction = entry.get("extraction")
|
||||||
|
if not isinstance(item, Mapping) or not isinstance(extraction, Mapping):
|
||||||
|
continue
|
||||||
|
|
||||||
|
raw_item_id = item.get("item_id")
|
||||||
|
item_id = str(raw_item_id) if raw_item_id is not None else None
|
||||||
|
if not item_id or item_id not in selected_set:
|
||||||
|
continue
|
||||||
|
|
||||||
|
yield item_id, {
|
||||||
|
"item": item,
|
||||||
|
"extraction": extraction,
|
||||||
|
}
|
||||||
|
return
|
||||||
|
|
||||||
|
# Fallback: legacy payload with top-level "items" array.
|
||||||
|
items = extracted_payload.get("items")
|
||||||
|
if isinstance(items, list):
|
||||||
|
for item in items:
|
||||||
|
if not isinstance(item, Mapping):
|
||||||
|
continue
|
||||||
|
raw_item_id = item.get("item_id")
|
||||||
|
item_id = str(raw_item_id) if raw_item_id is not None else None
|
||||||
|
if not item_id or item_id not in selected_set:
|
||||||
|
continue
|
||||||
|
|
||||||
|
yield item_id, item
|
||||||
|
return
|
||||||
|
|
||||||
|
raise RuntimeError(
|
||||||
|
"Extracted payload does not contain a supported article items structure (expected 'results' or 'items').",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def summarize_selected_articles(
|
||||||
|
*,
|
||||||
|
extracted_path: Path,
|
||||||
|
selected_ids: Sequence[str],
|
||||||
|
output_dir: Path,
|
||||||
|
config: ArticleSummaryConfig | None = None,
|
||||||
|
api_key: str | None = None,
|
||||||
|
model: str | None = None,
|
||||||
|
api_url: str | None = None,
|
||||||
|
) -> list[Path]:
|
||||||
|
"""Generate one Markdown summary per selected article.
|
||||||
|
|
||||||
|
This operates purely on already extracted article payloads and does not
|
||||||
|
re-fetch original URLs. It expects an `extracted_path` produced by the
|
||||||
|
FreshRSS extract or pipeline steps.
|
||||||
|
|
||||||
|
The current reader format uses a top-level ``results`` array where each
|
||||||
|
entry has ``item`` and ``extraction`` (with ``extraction.article`` being
|
||||||
|
the extracted article). For compatibility, a legacy top-level ``items``
|
||||||
|
array is also supported if present.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import json
|
||||||
|
|
||||||
|
cfg = config or ArticleSummaryConfig()
|
||||||
|
resolved_api_key, resolved_model, resolved_api_url = _resolve_article_llm_settings(
|
||||||
|
api_key=api_key or cfg.llm_api_key,
|
||||||
|
model=model or cfg.llm_model,
|
||||||
|
api_url=api_url or cfg.llm_api_url,
|
||||||
|
)
|
||||||
|
|
||||||
|
payload = json.loads(extracted_path.read_text(encoding="utf-8-sig"))
|
||||||
|
|
||||||
|
output_dir.mkdir(parents=True, exist_ok=True)
|
||||||
|
written_paths: list[Path] = []
|
||||||
|
|
||||||
|
from summary_mcp.core.summary_loop import build_summary_input
|
||||||
|
|
||||||
|
for item_id, entry in _iter_selected_items(payload, selected_ids):
|
||||||
|
# Prefer the real project format where each entry has ``item`` and
|
||||||
|
# ``extraction.article``; fall back to legacy layout where the
|
||||||
|
# article fields live directly on the element.
|
||||||
|
if "item" in entry or "extraction" in entry:
|
||||||
|
item = entry.get("item") or {}
|
||||||
|
extraction = entry.get("extraction") or {}
|
||||||
|
article = (extraction.get("article") or {}) if isinstance(extraction, dict) else {}
|
||||||
|
warnings = extraction.get("warnings", []) if isinstance(extraction, dict) else []
|
||||||
|
else:
|
||||||
|
item = entry
|
||||||
|
article = item.get("article") or {}
|
||||||
|
warnings = item.get("warnings", [])
|
||||||
|
|
||||||
|
extracted_payload = {
|
||||||
|
"article": article,
|
||||||
|
"warnings": warnings,
|
||||||
|
}
|
||||||
|
|
||||||
|
summary_exit_code, summary_payload, _ = run_loop_payload(
|
||||||
|
extracted_payload=extracted_payload,
|
||||||
|
prompt_path=cfg.prompt_path,
|
||||||
|
max_retries=cfg.max_retries,
|
||||||
|
timeout_seconds=cfg.timeout_seconds,
|
||||||
|
api_key=resolved_api_key,
|
||||||
|
model=resolved_model,
|
||||||
|
api_url=resolved_api_url,
|
||||||
|
output_path=None,
|
||||||
|
)
|
||||||
|
if summary_exit_code != 0 or summary_payload is None:
|
||||||
|
continue
|
||||||
|
|
||||||
|
# Render a simple Markdown file summarizing the article.
|
||||||
|
title = article.get("title") or summary_payload.get("title") or item_id
|
||||||
|
url = article.get("url")
|
||||||
|
summary_text = summary_payload.get("summary") or ""
|
||||||
|
highlights = summary_payload.get("highlights") or []
|
||||||
|
|
||||||
|
lines: list[str] = []
|
||||||
|
lines.append(f"# {title}")
|
||||||
|
if url:
|
||||||
|
lines.append("")
|
||||||
|
lines.append(f"Source: {url}")
|
||||||
|
lines.append("")
|
||||||
|
if summary_text:
|
||||||
|
lines.append(summary_text)
|
||||||
|
lines.append("")
|
||||||
|
if highlights:
|
||||||
|
lines.append("## Highlights")
|
||||||
|
lines.extend(f"- {h}" for h in highlights)
|
||||||
|
lines.append("")
|
||||||
|
|
||||||
|
safe_title = "-".join(
|
||||||
|
str(title).lower().strip().replace(" ", "-").split()
|
||||||
|
)[:80]
|
||||||
|
filename = f"{safe_title or item_id}.md"
|
||||||
|
output_path = output_dir / filename
|
||||||
|
output_path.write_text("\n".join(lines), encoding="utf-8")
|
||||||
|
written_paths.append(output_path)
|
||||||
|
|
||||||
|
return written_paths
|
||||||
@@ -6,7 +6,9 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import json
|
import json
|
||||||
import os
|
import os
|
||||||
from datetime import UTC, date, datetime
|
from datetime import date, datetime, timezone
|
||||||
|
|
||||||
|
UTC = timezone.utc
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any
|
from typing import Any
|
||||||
|
|
||||||
@@ -145,7 +147,7 @@ def run_freshrss_pipeline(
|
|||||||
# 逐条处理:提取 -> 摘要 -> 过滤 -> 候选构建;任一步骤失败则记录状态后 continue
|
# 逐条处理:提取 -> 摘要 -> 过滤 -> 候选构建;任一步骤失败则记录状态后 continue
|
||||||
item_key = f"item-{index:02d}"
|
item_key = f"item-{index:02d}"
|
||||||
item_path = _maybe_path(debug_artifacts, resolved_output_dir / "items" / f"{item_key}.item.json")
|
item_path = _maybe_path(debug_artifacts, resolved_output_dir / "items" / f"{item_key}.item.json")
|
||||||
extracted_path = _maybe_path(debug_artifacts, resolved_output_dir / "extracted" / f"{item_key}.extracted.json")
|
extracted_path = resolved_output_dir / "extracted" / f"{item_key}.extracted.json"
|
||||||
summary_output = _maybe_path(debug_artifacts, resolved_output_dir / "summary" / item_key / "result.loop.json")
|
summary_output = _maybe_path(debug_artifacts, resolved_output_dir / "summary" / item_key / "result.loop.json")
|
||||||
filter_path = _maybe_path(debug_artifacts, resolved_output_dir / "filter" / f"{item_key}.filter.json")
|
filter_path = _maybe_path(debug_artifacts, resolved_output_dir / "filter" / f"{item_key}.filter.json")
|
||||||
record_path = _maybe_path(debug_artifacts, resolved_output_dir / "candidates" / f"{item_key}.article-candidate-record.json")
|
record_path = _maybe_path(debug_artifacts, resolved_output_dir / "candidates" / f"{item_key}.article-candidate-record.json")
|
||||||
|
|||||||
Reference in New Issue
Block a user