"""generator config + content ingestion. Per BR1.1/BR1.2, US5/US7: reads config.json (MastheadConfig) and the content/ folder, parsing markdown/text into Article entities with fidelity (NFR7). """ from __future__ import annotations import json import re from pathlib import Path from .model import Article, MastheadConfig META_KEYS = {"title", "couple_names", "date_line", "issue_number", "volume"} def load_config(path: Path) -> MastheadConfig: """Read and validate config.json (BR1.1/BR1.2).""" data = json.loads(path.read_text(encoding="utf-8")) return MastheadConfig( title=str(data.get("title", "The Wedding Times")), couple_names=str(data.get("couple_names", "The Happy Couple")), date_line=str(data.get("date_line", "")), issue_number=str(data.get("issue_number", "1")), volume=str(data.get("volume", "1")), ) def _read_frontmatter(text: str) -> dict[str, str]: """Very light front-matter parse: ---key: value--- lines at the top.""" meta: dict[str, str] = {} if text.startswith("---"): m = re.match(r"^---\s*\n(.*?)\n---", text, re.S) if m: for line in m.group(1).splitlines(): if ":" in line: k, _, v = line.partition(":") if k.strip() in META_KEYS: meta[k.strip()] = v.strip() return meta return meta def _strip_frontmatter(text: str) -> str: if text.startswith("---"): m = re.match(r"^---\s*\n.*?\n---\s*\n?", text, re.S) if m: return text[m.end():] return text def parse_article(path: Path, article_id: str) -> Article: """Parse one .md/.txt content file into an Article (BR1.1, NFR7 fidelity).""" raw = path.read_text(encoding="utf-8") meta = _read_frontmatter(raw) body = _strip_frontmatter(raw) first_line = next((ln for ln in body.splitlines() if ln.strip()), "") type_ = meta.get("type", "article") section = meta.get("section", "News") return Article( article_id=article_id, type=type_, headline=meta.get("headline", first_line.strip(" #")), body=body.strip(), byline=meta.get("byline", ""), section=section, ) def load_content(content_dir: Path) -> list[Article]: """Read every .md/.txt file in the folder, in sorted order (US5).""" articles: list[Article] = [] files = sorted( p for p in content_dir.rglob("*") if p.is_file() and p.suffix.lower() in {".md", ".txt"} ) for i, path in enumerate(files, start=1): articles.append(parse_article(path, f"article-{i}")) return articles