81 lines
2.6 KiB
Python
81 lines
2.6 KiB
Python
"""generator config + content ingestion.
|
|
|
|
Per BR1.1/BR1.2, US5/US7: reads config.json (MastheadConfig) and the content/
|
|
folder, parsing markdown/text into Article entities with fidelity (NFR7).
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import re
|
|
from pathlib import Path
|
|
|
|
from .model import Article, MastheadConfig
|
|
|
|
META_KEYS = {"title", "couple_names", "date_line", "issue_number", "volume"}
|
|
|
|
|
|
def load_config(path: Path) -> MastheadConfig:
|
|
"""Read and validate config.json (BR1.1/BR1.2)."""
|
|
data = json.loads(path.read_text(encoding="utf-8"))
|
|
return MastheadConfig(
|
|
title=str(data.get("title", "The Wedding Times")),
|
|
couple_names=str(data.get("couple_names", "The Happy Couple")),
|
|
date_line=str(data.get("date_line", "")),
|
|
issue_number=str(data.get("issue_number", "1")),
|
|
volume=str(data.get("volume", "1")),
|
|
)
|
|
|
|
|
|
def _read_frontmatter(text: str) -> dict[str, str]:
|
|
"""Very light front-matter parse: ---key: value--- lines at the top."""
|
|
meta: dict[str, str] = {}
|
|
if text.startswith("---"):
|
|
m = re.match(r"^---\s*\n(.*?)\n---", text, re.S)
|
|
if m:
|
|
for line in m.group(1).splitlines():
|
|
if ":" in line:
|
|
k, _, v = line.partition(":")
|
|
if k.strip() in META_KEYS:
|
|
meta[k.strip()] = v.strip()
|
|
return meta
|
|
return meta
|
|
|
|
|
|
def _strip_frontmatter(text: str) -> str:
|
|
if text.startswith("---"):
|
|
m = re.match(r"^---\s*\n.*?\n---\s*\n?", text, re.S)
|
|
if m:
|
|
return text[m.end():]
|
|
return text
|
|
|
|
|
|
def parse_article(path: Path, article_id: str) -> Article:
|
|
"""Parse one .md/.txt content file into an Article (BR1.1, NFR7 fidelity)."""
|
|
raw = path.read_text(encoding="utf-8")
|
|
meta = _read_frontmatter(raw)
|
|
body = _strip_frontmatter(raw)
|
|
first_line = next((ln for ln in body.splitlines() if ln.strip()), "")
|
|
type_ = meta.get("type", "article")
|
|
section = meta.get("section", "News")
|
|
return Article(
|
|
article_id=article_id,
|
|
type=type_,
|
|
headline=meta.get("headline", first_line.strip(" #")),
|
|
body=body.strip(),
|
|
byline=meta.get("byline", ""),
|
|
section=section,
|
|
)
|
|
|
|
|
|
def load_content(content_dir: Path) -> list[Article]:
|
|
"""Read every .md/.txt file in the folder, in sorted order (US5)."""
|
|
articles: list[Article] = []
|
|
files = sorted(
|
|
p for p in content_dir.rglob("*")
|
|
if p.is_file() and p.suffix.lower() in {".md", ".txt"}
|
|
)
|
|
for i, path in enumerate(files, start=1):
|
|
articles.append(parse_article(path, f"article-{i}"))
|
|
return articles
|