Andrew Ridgway bec1eaac87
Some checks failed
Test / test (push) Has been cancelled
first pass at the newspaper builder
2026-09-14 11:57:22 +10:00

81 lines
2.6 KiB
Python

"""generator config + content ingestion.
Per BR1.1/BR1.2, US5/US7: reads config.json (MastheadConfig) and the content/
folder, parsing markdown/text into Article entities with fidelity (NFR7).
"""
from __future__ import annotations
import json
import re
from pathlib import Path
from .model import Article, MastheadConfig
META_KEYS = {"title", "couple_names", "date_line", "issue_number", "volume"}
def load_config(path: Path) -> MastheadConfig:
"""Read and validate config.json (BR1.1/BR1.2)."""
data = json.loads(path.read_text(encoding="utf-8"))
return MastheadConfig(
title=str(data.get("title", "The Wedding Times")),
couple_names=str(data.get("couple_names", "The Happy Couple")),
date_line=str(data.get("date_line", "")),
issue_number=str(data.get("issue_number", "1")),
volume=str(data.get("volume", "1")),
)
def _read_frontmatter(text: str) -> dict[str, str]:
"""Very light front-matter parse: ---key: value--- lines at the top."""
meta: dict[str, str] = {}
if text.startswith("---"):
m = re.match(r"^---\s*\n(.*?)\n---", text, re.S)
if m:
for line in m.group(1).splitlines():
if ":" in line:
k, _, v = line.partition(":")
if k.strip() in META_KEYS:
meta[k.strip()] = v.strip()
return meta
return meta
def _strip_frontmatter(text: str) -> str:
if text.startswith("---"):
m = re.match(r"^---\s*\n.*?\n---\s*\n?", text, re.S)
if m:
return text[m.end():]
return text
def parse_article(path: Path, article_id: str) -> Article:
"""Parse one .md/.txt content file into an Article (BR1.1, NFR7 fidelity)."""
raw = path.read_text(encoding="utf-8")
meta = _read_frontmatter(raw)
body = _strip_frontmatter(raw)
first_line = next((ln for ln in body.splitlines() if ln.strip()), "")
type_ = meta.get("type", "article")
section = meta.get("section", "News")
return Article(
article_id=article_id,
type=type_,
headline=meta.get("headline", first_line.strip(" #")),
body=body.strip(),
byline=meta.get("byline", ""),
section=section,
)
def load_content(content_dir: Path) -> list[Article]:
"""Read every .md/.txt file in the folder, in sorted order (US5)."""
articles: list[Article] = []
files = sorted(
p for p in content_dir.rglob("*")
if p.is_file() and p.suffix.lower() in {".md", ".txt"}
)
for i, path in enumerate(files, start=1):
articles.append(parse_article(path, f"article-{i}"))
return articles