Files
reader/scripts/run_article_candidate.py
T

124 lines
4.4 KiB
Python

from __future__ import annotations
import argparse
import json
import sys
from datetime import UTC, datetime
from pathlib import Path
from typing import Any
REPO_ROOT = Path(__file__).resolve().parents[1]
SRC_ROOT = REPO_ROOT / "src"
OUTPUT_ROOT = REPO_ROOT / "outputs"
REFERENCE_OUTPUT_ROOT = OUTPUT_ROOT / "reference"
if str(SRC_ROOT) not in sys.path:
sys.path.insert(0, str(SRC_ROOT))
from summary_mcp.models.article_candidate import (
CandidateMetadata,
CandidateSourceRefs,
build_article_candidate_record,
build_openclaw_candidate_input,
)
from summary_mcp.models.document import ExtractedArticle
from summary_mcp.models.filtering import FilterDecisionResult
from summary_mcp.models.item import Item
from summary_mcp.models.llm_result import LlmSummaryResult
def _load_json(path: Path) -> dict[str, Any]:
return json.loads(path.read_text(encoding="utf-8-sig"))
def _save_json(path: Path, payload: dict[str, Any]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
def main() -> None:
parser = argparse.ArgumentParser(
description="Build an internal ArticleCandidateRecord and a slim OpenClawCandidateInput from pipeline outputs."
)
parser.add_argument("--summary", type=Path, required=True, help="Structured LLM summary JSON file")
parser.add_argument("--extracted", type=Path, required=True, help="Extracted article JSON file")
parser.add_argument("--filter", type=Path, required=True, help="Filter decision JSON file")
parser.add_argument("--item", type=Path, default=None, help="Optional normalized item JSON file")
parser.add_argument(
"--section-hint",
default=None,
choices=[
"top_news",
"tools_and_workflows",
"risk_and_security",
"open_source",
"insights",
"deep_dive",
],
help="Optional digest section hint for downstream aggregation",
)
parser.add_argument(
"--rank",
type=int,
default=None,
help="Optional digest rank override. Defaults to filter priority when omitted.",
)
parser.add_argument(
"--rendered-markdown",
type=Path,
default=None,
help="Optional markdown file to embed as rendered_markdown",
)
parser.add_argument(
"--output",
dest="record_output",
type=Path,
default=REFERENCE_OUTPUT_ROOT / "candidates" / "article-candidate-record.json",
help="Where to save the internal ArticleCandidateRecord payload",
)
parser.add_argument(
"--openclaw-output",
type=Path,
default=REFERENCE_OUTPUT_ROOT / "candidates" / "openclaw-candidate-input.json",
help="Where to save the slim OpenClawCandidateInput payload",
)
args = parser.parse_args()
summary = LlmSummaryResult.model_validate(_load_json(args.summary))
extracted_payload = _load_json(args.extracted)
article = ExtractedArticle.model_validate(extracted_payload.get("article", extracted_payload))
decision = FilterDecisionResult.model_validate(_load_json(args.filter))
item = Item.model_validate(_load_json(args.item)) if args.item else None
rendered_markdown = args.rendered_markdown.read_text(encoding="utf-8-sig") if args.rendered_markdown else None
record = build_article_candidate_record(
summary=summary,
article=article,
filter_result=decision,
item=item,
digest_section_hint=args.section_hint,
digest_rank=args.rank,
rendered_markdown=rendered_markdown,
source_refs=CandidateSourceRefs(
item_path=str(args.item) if args.item else None,
extracted_path=str(args.extracted),
summary_path=str(args.summary),
filter_path=str(args.filter),
),
metadata=CandidateMetadata(
generated_at=datetime.now(tz=UTC),
producer="run_article_candidate.py",
run_id=datetime.now(tz=UTC).strftime("candidate-%Y%m%d-%H%M%S"),
),
)
openclaw_input = build_openclaw_candidate_input(record)
_save_json(args.record_output, record.model_dump(mode="json"))
_save_json(args.openclaw_output, openclaw_input.model_dump(mode="json"))
print(f"Saved article candidate record to {args.record_output}")
print(f"Saved OpenClaw candidate input to {args.openclaw_output}")
if __name__ == "__main__":
main()