commit 883e264c42dda0ab4bb89487a34b45b5703e4c9b Author: Hermes Agent Date: Thu Sep 24 11:45:57 2026 +1000 chore: seed gitea-repo-template cookie-cutter Reusable repo template for homelab services on git.aridgwayweb.com. Bakes in (all verified live on armistace/wedding-photos): - Hard commit guard: shared pre-commit hook (git config core.hooksPath ~/dev/git-hooks) + master branch protection (push whitelist [armistace], merge whitelist [hermes, armistace]). - Gitea Actions CI (.gitea/workflows/build_push.yml): test + build-deploy, persistent remote buildkit cache, registry push, idempotent deploy that preserves hand-provisioned Secrets, cluster injection from repo secrets/vars via scripts/reconcile-cluster-inject.sh. - Persistent buildkit cache (ci/buildkit/): single-replica Longhorn backing. - scripts/reconcile-cluster-inject.sh: reconcile live Secret/ConfigMap from Gitea secrets/vars without clobbering hand-provisioned values. - RUNBOOK.md: handoff-complete ops doc. Placeholders ( ) are filled per-service on repo creation. diff --git a/.gitea/workflows/build_push.yml b/.gitea/workflows/build_push.yml new file mode 100644 index 0000000..ca3dd33 --- /dev/null +++ b/.gitea/workflows/build_push.yml @@ -0,0 +1,105 @@ +# Gitea Actions workflow for a single-service homelab app. +# +# THIS IS A TEMPLATE. On repo creation from this template, Gitea copies this file +# verbatim. Fill in the placeholders below (, , ) to match your +# service. See RUNBOOK.md §2 for the one-time per-repo setup (repo secrets/vars). +# +# Proven patterns baked in (all verified live on armistace/wedding-photos): +# - buildx driver: remote -> persistent buildkit daemon (NOT docker-container, +# which can't reach Docker Hub on this runner). endpoint MUST be top-level, +# not under driver-opts (that yields "no remote endpoint provided"). +# - Kubeconfig from CI secret, kubectl installed in-job. +# - Deploy does NOT delete the namespace (preserves hand-provisioned Secrets). +# - reconcile-cluster-inject.sh keeps the live Secret/ConfigMap in sync with +# repo secrets/vars without clobbering hand-provisioned values. +# - linux/amd64 only (arm64 via qemu OOMs the buildkit pod on the NUCs). +# - Do NOT use ghcr.io images in COPY --from (buildkit can't reach it); use the +# ECR base and install uv via pip. + +name: build-and-deploy + +on: + push: + branches: [master] + workflow_dispatch: + +jobs: + test: + runs-on: ubuntu-latest + steps: + - name: Checkout + uses: actions/checkout@v4 + - name: Install uv + uses: astral-sh/setup-uv@v5 + - name: Unit tests + working-directory: backend + run: | + uv python install 3.12 + uv sync --all-groups --frozen + uv run python -m pytest -q + + build-deploy: + needs: test + runs-on: ubuntu-latest + container: catthehacker/ubuntu:act-latest + steps: + - name: Checkout + uses: actions/checkout@v4 + + - name: Create Kubeconfig + run: | + mkdir -p $HOME/.kube + echo "${{ secrets.KUBEC_CONFIG_BUILDX_NEW_2 }}" > $HOME/.kube/config + + - name: Set up Docker Buildx (remote -> persistent buildkit daemon) + uses: docker/setup-buildx-action@v3 + with: + driver: remote + # endpoint must be TOP-LEVEL (the action passes it positionally to + # `buildx create --driver remote `). Under driver-opts it + # fails with "ERROR: no remote endpoint provided". + endpoint: tcp://buildkit.gitea-runner.svc:1234 + + - name: Login to Gitea registry + uses: docker/login-action@v3 + with: + registry: git.aridgwayweb.com + username: + password: ${{ secrets.REG_PASSWORD }} + + - name: Build & push image + uses: docker/build-push-action@v5 + with: + context: ./backend + push: true + platforms: linux/amd64 + tags: | + git.aridgwayweb.com//:latest + git.aridgwayweb.com//:${{ github.sha }} + + - name: Deploy + run: | + echo "Installing Kubectl" + apt-get update + apt-get install -y apt-transport-https ca-certificates curl gnupg + curl -fsSL https://pkgs.k8s.io/core:/stable:/v1.33/deb/Release.key | gpg --dearmor -o /etc/apt/keyrings/kubernetes-apt-keyring.gpg + chmod 644 /etc/apt/keyrings/kubernetes-apt-keyring.gpg + echo 'deb [signed-by=/etc/apt/keyrings/kubernetes-apt-keyring.gpg] https://pkgs.k8s.io/core:/stable:/v1.33/deb/ /' | tee /etc/apt/sources.list.d/kubernetes.list + chmod 644 /etc/apt/sources.list.d/kubernetes.list + apt-get update + apt-get install -y kubectl + # Do NOT delete the namespace — that destroys the live Secret (DB creds, + # JWT, etc.) which is provisioned by hand (see RUNBOOK.md). + kubectl apply -f kube/_namespace.yaml + kubectl create secret docker-registry regcred --docker-server=git.aridgwayweb.com --docker-username= --docker-password='${{ secrets.REG_PASSWORD }}' --docker-email=@aridgwayweb.com --namespace= --dry-run=client -o yaml | kubectl apply -f - + # Apply everything EXCEPT the placeholder Secret (which would clobber the live one). + kubectl apply -f kube/_configmap.yaml + kubectl apply -f kube/-backend_deploy.yaml + kubectl apply -f kube/-backend_service.yaml + # Reconcile the live Secret + ConfigMap from Gitea repo secrets/vars so + # every env key the deployments reference exists (prevents + # CreateContainerConfigError). See scripts/reconcile-cluster-inject.sh. + export ="${{ secrets. }}" + export ="${{ secrets. }}" + bash scripts/reconcile-cluster-inject.sh -secret -config + kubectl rollout status deployment/-backend -n --timeout=180s diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..37ab408 --- /dev/null +++ b/.gitignore @@ -0,0 +1,8 @@ +# Local runtime artifacts — never commit +__pycache__/ +*.pyc +.venv/ +node_modules/ +dist/ +*.env +.env diff --git a/README.md b/README.md new file mode 100644 index 0000000..37d60aa --- /dev/null +++ b/README.md @@ -0,0 +1,29 @@ +# Gitea Repo Template + +Reusable cookie-cutter for new homelab repos on `git.aridgwayweb.com`. Creating a +repo from this template pre-installs the guards + CI patterns that took several +iterations to converge on (see `armistace/wedding-photos`), so every future repo +gets them for free. + +## What you get for free + +- **Hard commit guard** — local pre-commit hook (blocks direct commits to + `master`/`main`, detach-HEAD) configured via `git config core.hooksPath + ~/dev/git-hooks` + Gitea branch protection on `master` (push whitelist + `[armistace]`, merge whitelist `[hermes, armistace]`). +- **Gitea Actions CI** (`.gitea/workflows/build_push.yml`) — test + build-deploy: + persistent remote buildkit cache, Gitea registry push, idempotent deploy that + preserves hand-provisioned Secrets, and cluster injection from repo + secrets/vars via `scripts/reconcile-cluster-inject.sh`. +- **Persistent buildkit cache** (`ci/buildkit/`) — single-replica Longhorn + backing, disk-pressure watchdog referenced from the control host. +- **RUNBOOK.md** — handoff-complete ops doc. + +## Create a repo from this template + +Gitea UI: `+` → *New Repository* → **"Copy to new repository"** → source +`hermes/gitea-repo-template`. +Automated: `~/.hermes/scripts/gitea-bootstrap-new-repo.sh` (see that script). + +Then fill the `` and follow RUNBOOK §1-2. **Ops/watchdog scripts do +NOT belong in repos — keep them in `~/.hermes/scripts/`.** diff --git a/RUNBOOK.md b/RUNBOOK.md new file mode 100644 index 0000000..0a6a4e1 --- /dev/null +++ b/RUNBOOK.md @@ -0,0 +1,140 @@ +# Gitea Repo Template — Operations Runbook + +This is the **cookie-cutter** for new homelab repos. On Gitea you create a new +repo **from this template** and it copies `.gitea/`, `kube/`, `ci/buildkit/`, +`scripts/`, and this RUNBOOK. Fill in the `` per service, then +follow the bootstrap flow below. Everything documented here is proven live on +`armistace/wedding-photos`. + +--- + +## 1. Create a repo from this template + +- Gitea UI: `+` → *New Repository* → **"Copy to new repository"** source = `hermes/gitea-repo-template`. +- Bot/automation: `POST /api/v1/repos/{owner}/{repo}/generate` (See `~/.hermes/scripts/gitea-bootstrap-new-repo.sh`). + +After creation, commit+push a branch filling in: +- `.gitea/workflows/build_push.yml` → ``, ``, ``, `` +- `scripts/reconcile-cluster-inject.sh` → same placeholders +- `kube/*` , `ci/buildkit/*` → your image/service names +- This RUNBOOK's placeholders → actual namespace/app + +--- + +## 2. One-time per-repo setup (do this once, after the first commit) + +### 2.1 Repo secrets + vars (Gitea Settings → Actions/secrets) +Set on the repo (or reuse the global `armistace`-scoped ones): +- `KUBEC_CONFIG_BUILDX_NEW_2` — kubeconfig for the cluster (buildx secret) +- `REG_PASSWORD` — Gitea registry password for `` +- every `` referenced by `reconcile-cluster-inject.sh` SECRET_KEYS +- vars for CFG: e.g. `DB_HOST`, `S3_ENDPOINT_URL`, etc. + +### 2.2 Branch protection (hard guard, server-side) +Created automatically by the bootstrap script (agent runs it with the bot token +on `hermes/`-owned repos). Rule on `master`: +- `enable_push + enable_push_whitelist=true`, push whitelist `[armistace]` + → only Andrew can DIRECT-push master; agent direct push is blocked even with `--no-verify` +- `enable_merge_whitelist=true`, merge whitelist `[hermes, armistace]` + → agent can still MERGE PRs (merge ≠ direct push) + +### 2.3 Local pre-commit guard +Point the repo at the shared git hook so it blocks direct-to-master commits: +```bash +git config core.hooksPath ~/dev/git-hooks +``` +(Shared hook lives at `~/dev/git-hooks/pre-commit` on the control host; the +master-guard binary is bundled in the `gitea-pr-workflow` skill.) + +--- + +## 3. Architecture & storage + +### 3.1 Cluster injection via Gitea repo secrets + vars +`scripts/reconcile-cluster-inject.sh` reconciles the live Secret/ConfigMap each +deploy with three rules (priority order): +1. Gitea supplies a **non-empty** value → that value wins (seed / rotate). +2. Key **missing** from the live resource → write placeholder (env ref always resolves). +3. Otherwise → **preserve** the live value (hand-provisioned creds never clobbered). + +The patch is built by Python `json.dumps` (no shell interpolation). Add every +key the deployments reference to `SECRET_KEYS` / `CM_KEYS`; list new repo +secrets/vars here so a handoff never loses them. + +### 3.2 Persistent buildkit cache (optional, durable CI cache) +`ci/buildkit/` installs a single-replica Longhorn-backed buildkit daemon so CI +image builds reuse cache. Manifests are shipped but NOT auto-applied — apply +once: +```bash +kubectl apply -f ci/buildkit/01-storageclass.yaml +kubectl apply -f ci/buildkit/02-statefulset.yaml +kubectl apply -f ci/buildkit/03-service.yaml +``` +The remote-driver workflow points at `buildkit.gitea-runner.svc:1234`. +Disk watchdog + revert scripts live on the control host +(`~/.hermes/scripts/buildkit-cache-monitor.sh`, `buildkit-re-enable.sh`) — NOT +in this repo (ops tooling doesn't go in app repos). + +--- + +## 4. Deploying / re-deploying from scratch + +### 4.1 Provision the namespace + live Secret ONCE by hand +CI will `kubectl apply` manifests and reconcile the Secret, but the FIRST seed +of real credentials is manual (CI only writes placeholders when a key is absent): +```bash +kubectl create namespace +kubectl create secret docker-registry regcred \ + --docker-server=git.aridgwayweb.com --docker-username= \ + --docker-password='' --docker-email=@aridgwayweb.com \ + -n --dry-run=client -o yaml | kubectl apply -f - +kubectl -n create secret generic -secret \ + --from-literal=='' \ + --from-literal=='' +``` + +### 4.2 Set the ConfigMap +```bash +kubectl -n apply -f kube/_configmap.yaml +``` + +### 4.3 After CI deploy — verify +```bash +kubectl -n get pods +kubectl -n rollout status deployment/-backend --timeout=180s +``` + +--- + +## 5. Common operations + +- **Rotate a secret**: update the Gitea repo secret → push an empty commit or + re-run `workflow_dispatch` → reconcile applies the new value → `kubectl -n rollout restart deployment/-backend`. +- **If pods CrashLoopBackOff after a secret change**: the env ref is likely + unresolvable — check `kubectl get event` / secret keys; add the missing key to + `reconcile-cluster-inject.sh` SECRET_KEYS and the workflow export block. +- **Never delete the namespace** on deploy (CI does not; don't start) — it + destroys the live Secret. + +--- + +## 6. Troubleshooting + +- **Build-deploy fails ~36s, "no remote endpoint provided"**: the buildx + `endpoint` was put under `driver-opts`. Move it to a top-level `with:` input. +- **Multi-arch build OOMs/crashes the buildkit pod**: build `linux/amd64` only. +- **`COPY --from=ghcr.io/...` TLS timeout**: buildkit can't reach ghcr.io. Use + an ECR base + `pip install uv`. +- **PR files API returns empty `patch` fields**: build review payload from local + `git diff` — never trust the API `patch`. + +--- + +## 7. Handoff checklist + +- [ ] All `` filled (workflow, reconcile script, kube/, this RUNBOOK) +- [ ] Repo secrets + vars set; keys match `reconcile-cluster-inject.sh` + workflow `export` block +- [ ] Branch protection on `master` enabled (push whitelist `[armistace]`, merge whitelist `[hermes, armistace]`) +- [ ] `git config core.hooksPath ~/dev/git-hooks` set in the new repo +- [ ] First deploy: namespace seeded with a real Secret (RUNBOOK §4.1) before relying on CI reconcile +- [ ] Any new env key added to §3.1 table + §2.1 list so a handoff doesn't lose it diff --git a/ci/buildkit/01-storageclass.yaml b/ci/buildkit/01-storageclass.yaml new file mode 100644 index 0000000..4f77ba9 --- /dev/null +++ b/ci/buildkit/01-storageclass.yaml @@ -0,0 +1,27 @@ +# StorageClass for the persistent buildkit cache. +# - numberOfReplicas: 1 -> the cache is written ONCE across the cluster (1x disk, +# NOT the 3x triple-replication of the default 'longhorn' SC). This is deliberate: +# a buildkit cache is throwable/recreatable, so replicating it 3x wastes disk that +# node 2 (already 76%) can't afford. 1 replica survives single-node loss via +# dataLocality: best-effort (re-replicates only when a node actually dies). +# - dataLocality: best-effort -> keep the single replica on the same node as the +# pod (fast local reads), replicate only if that node fails. +# Auto-revert safety net: the buildkit-cache-monitor cron reverts buildkit to the +# ephemeral (no-PVC) driver if any node crosses the critical disk threshold. +# See monitoring/buildkit-cache-monitor.sh. +apiVersion: storage.k8s.io/v1 +kind: StorageClass +metadata: + name: buildkit-single-1r + annotations: + description: "Single-replica Longhorn (buildkit cache) - 1x disk, durable across node loss" +provisioner: driver.longhorn.io +allowVolumeExpansion: true +reclaimPolicy: Delete +volumeBindingMode: Immediate +parameters: + numberOfReplicas: "1" + staleReplicaTimeout: "30" + fsType: "ext4" + dataLocality: "best-effort" + dataEngine: "v1" diff --git a/ci/buildkit/02-statefulset.yaml b/ci/buildkit/02-statefulset.yaml new file mode 100644 index 0000000..b993b02 --- /dev/null +++ b/ci/buildkit/02-statefulset.yaml @@ -0,0 +1,77 @@ +# Dedicated long-lived buildkit daemon for CI image builds. +# +# Persists its build cache on a single-replica Longhorn volume (StorageClass +# buildkit-single-1r) so repeated CI image builds reuse the layer cache instead of +# re-pulling/building every time. The cache stays ONE copy on disk (not the 3x +# default replication) and is bounded by the 10Gi PVC. The workflow's buildx uses +# the REMOTE driver to point at this daemon via the ClusterIP Service below. +# +# Disk-pressure safety net: the buildkit-cache-monitor cron (monitoring/) reverts +# CI to the ephemeral buildx driver if any node crosses critical disk usage. +# +# NOTE: no toleration for the gitea-builder taint on archlinux-k3s-1, so this pod +# schedules on a worker node (2 or 3) away from the high-CPU control plane. +apiVersion: apps/v1 +kind: StatefulSet +metadata: + name: buildkit + namespace: gitea-runner +spec: + serviceName: buildkit + replicas: 1 + selector: + matchLabels: + app: buildkit + template: + metadata: + labels: + app: buildkit + spec: + containers: + - name: buildkitd + image: moby/buildkit:buildx-stable-1 + args: + - --addr + # Listen for the remote driver over TCP on 1234 (no TLS - internal cluster traffic). + - tcp://0.0.0.0:1234 + # Keep the cache (do NOT use --oci-worker-no-process-sandbox or other + # flags that break rootless cache persistence). + ports: + - name: daemon + containerPort: 1234 + resources: + requests: + cpu: 500m + memory: 1Gi + limits: + cpu: "2" + memory: 4Gi + ephemeral-storage: 2Gi + volumeMounts: + - name: cache + mountPath: /var/lib/buildkit + # BuildKit needs this for the CA/root store even when not using TLS. + - name: certs + mountPath: /etc/buildkit/certs + - name: config + mountPath: /etc/buildkit + securityContext: + # BuildKit's OCI/runc worker must bind-mount build contexts & layers. + # The old driver-spawned (working) buildkit pods ran privileged:true; + # false here caused "failed to mount snapshot ... operation not + # permitted" (runc-native can't mount). Match the proven-working pods. + privileged: true + volumes: + - name: certs + emptyDir: {} + - name: config + emptyDir: {} + volumeClaimTemplates: + - metadata: + name: cache + spec: + accessModes: ["ReadWriteOnce"] + storageClassName: buildkit-single-1r + resources: + requests: + storage: 10Gi diff --git a/ci/buildkit/03-service.yaml b/ci/buildkit/03-service.yaml new file mode 100644 index 0000000..986e834 --- /dev/null +++ b/ci/buildkit/03-service.yaml @@ -0,0 +1,16 @@ +# ClusterIP service exposing the buildkit daemon to the CI runners. +# The workflow's buildx remote driver connects to: tcp://buildkit.gitea-runner.svc:1234 +apiVersion: v1 +kind: Service +metadata: + name: buildkit + namespace: gitea-runner +spec: + selector: + app: buildkit + ports: + - name: daemon + port: 1234 + targetPort: 1234 + protocol: TCP + type: ClusterIP diff --git a/ci/buildkit/04-networkpolicy.yaml b/ci/buildkit/04-networkpolicy.yaml new file mode 100644 index 0000000..1098a61 --- /dev/null +++ b/ci/buildkit/04-networkpolicy.yaml @@ -0,0 +1,26 @@ +# NetworkPolicy: only the Gitea action runners may reach the buildkit daemon. +# Closes the reviewer's High: buildkit listens on plaintext tcp:1234 with no +# auth — without this policy ANY cluster pod could submit arbitrary build +# requests (lateral-movement / resource-abuse surface). Restricting ingress to +# the gitea-runner namespace (where the runners live) removes that exposure. +# Homelab note: full TLS+auth on the buildkit socket is deferred (documented) — +# the runner and daemon are on the private cluster network; the policy closes the +# pod-to-pod surface that TLS alone wouldn't. +apiVersion: networking.k8s.io/v1 +kind: NetworkPolicy +metadata: + name: buildkit-allow-runner-only + namespace: gitea-runner +spec: + podSelector: + matchLabels: + app: buildkit + policyTypes: ["Ingress"] + ingress: + - from: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: gitea-runner + ports: + - protocol: TCP + port: 1234 diff --git a/ci/buildkit/README.md b/ci/buildkit/README.md new file mode 100644 index 0000000..d61d1ca --- /dev/null +++ b/ci/buildkit/README.md @@ -0,0 +1,44 @@ +# Persistent Buildkit Cache for CI + +Dedicated, long-lived buildkit daemon that persists its layer cache on a +single-replica Longhorn PVC so repeated CI image builds reuse cache instead of +re-building every time. This manifest set is **ported verbatim from +`armistace/wedding-photos`** — verified live. + +## Why single-replica Longhorn (not the default SC) + +The buildx **kubernetes driver's PVC option creates the volume with NO +`storageClassName`** — it falls back to the cluster default SC, which here is +`longhorn` at **3 replicas**. That would triple every cache byte across +disk-constrained nodes. So this uses a dedicated `buildkit-single-1r` SC +(`numberOfReplicas: 1`, `dataLocality: best-effort`): the cache is **1× on +disk** and only replicates if the hosting node actually dies. + +## Components + +| Path | Purpose | +|---|---| +| `ci/buildkit/01-storageclass.yaml` | `buildkit-single-1r` SC (1 replica, best-effort locality) | +| `ci/buildkit/02-statefulset.yaml` | `buildkit` StatefulSet, pod schedules off the control plane, listens `tcp://0.0.0.0:1234`, 10Gi cache PVC | +| `ci/buildkit/03-service.yaml` | ClusterIP `buildkit.gitea-runner.svc:1234` | +| `ci/buildkit/04-networkpolicy.yaml` | restrict buildkit to the gitea-runner namespace only | +| `.gitea/workflows/build_push.yml` | `setup-buildx-action` → `driver: remote`, `endpoint=tcp://buildkit.gitea-runner.svc:1234` | + +> **Ops tooling is NOT in this repo.** The disk watchdog + revert/restore +> scripts live on the control machine (`~/.hermes/scripts/buildkit-cache-monitor.sh`, +> `buildkit-re-enable.sh`) and run from the `buildkit-disk-monitor` cron. They are +> homelab ops, not app code. + +## Apply (one-time, per cluster — NOT per repo) + +```bash +kubectl apply -f ci/buildkit/01-storageclass.yaml +kubectl apply -f ci/buildkit/02-statefulset.yaml +kubectl apply -f ci/buildkit/03-service.yaml +kubectl apply -f ci/buildkit/04-networkpolicy.yaml +kubectl -n gitea-runner rollout status statefulset/buildkit --timeout=180s +``` + +Buildkit is a **cluster-wide shared resource** — it should be installed once, not +once per repo. If your repo doesn't own the cluster, coordinate with whoever +does (it lives in the `gitea-runner` namespace and only needs applying once). diff --git a/kube/APP-backend_deploy.yaml b/kube/APP-backend_deploy.yaml new file mode 100644 index 0000000..2e72e8c --- /dev/null +++ b/kube/APP-backend_deploy.yaml @@ -0,0 +1,25 @@ +# Template placeholder deployment. Convention: -backend_deploy.yaml +# CRITICAL: every pod spec must reference the regcred imagePullSecret or pods +# hit ImagePullBackOff pulling from the private registry. +apiVersion: apps/v1 +kind: Deployment +metadata: + name: -backend + namespace: +spec: + replicas: 1 + selector: + matchLabels: + app: -backend + template: + metadata: + labels: + app: -backend + spec: + imagePullSecrets: + - name: regcred + containers: + - name: backend + image: git.aridgwayweb.com//:latest + ports: + - containerPort: 8000 diff --git a/kube/APP-backend_service.yaml b/kube/APP-backend_service.yaml new file mode 100644 index 0000000..07eba4c --- /dev/null +++ b/kube/APP-backend_service.yaml @@ -0,0 +1,12 @@ +# Template placeholder service. Convention: -backend_service.yaml +apiVersion: v1 +kind: Service +metadata: + name: -backend + namespace: +spec: + selector: + app: -backend + ports: + - port: 8000 + targetPort: 8000 diff --git a/kube/NS_configmap.yaml b/kube/NS_configmap.yaml new file mode 100644 index 0000000..ae4928a --- /dev/null +++ b/kube/NS_configmap.yaml @@ -0,0 +1,9 @@ +# Template placeholder ConfigMap. Add keys matching reconcile-cluster-inject.sh +# CM_KEYS entries. Convention: _configmap.yaml +apiVersion: v1 +kind: ConfigMap +metadata: + name: -config + namespace: +data: + : "" diff --git a/kube/NS_namespace.yaml b/kube/NS_namespace.yaml new file mode 100644 index 0000000..d6e1ee6 --- /dev/null +++ b/kube/NS_namespace.yaml @@ -0,0 +1,6 @@ +# Template placeholder file — replace with your app's namespace manifest. +# Convention: _namespace.yaml (matches the workflow's apply step). +apiVersion: v1 +kind: Namespace +metadata: + name: diff --git a/scripts/reconcile-cluster-inject.sh b/scripts/reconcile-cluster-inject.sh new file mode 100644 index 0000000..13b7b3c --- /dev/null +++ b/scripts/reconcile-cluster-inject.sh @@ -0,0 +1,103 @@ +#!/usr/bin/env bash +# Reconciles the live Secret and ConfigMap from Gitea repo secrets/vars, +# WITHOUT ever clobbering hand-provisioned real credentials. +# +# Design rationale (post CreateContainerConfigError lesson): +# The deployments reference DB creds / JWT / etc. unconditionally. If any key +# is missing from the live Secret, the pod fails to start (rollout timeout in +# CI). The Secret has historically been provisioned by hand (RUNBOOK §4.2); +# git holds placeholders only. +# +# This script makes CI the reconciler. For each managed key it applies EXACTLY +# one rule (priority order): +# 1. Gitea supplies a NON-EMPTY value -> that value wins (seed / rotate). +# 2. The key MISSING from the live resource -> write the placeholder so the +# app's env refs always resolve and the rollout never breaks. +# 3. Otherwise (key present, Gitea empty or absent) -> PRESERVE the live +# value untouched. A hand-provisioned credential is never overwritten by +# an empty Gitea secret. +# +# The patch payload is built by Python's json module (no shell interpolation), +# so arbitrary values — quotes, newlines, unicode — cannot corrupt the JSON or +# inject flags into kubectl. + +set -euo pipefail + +SECRET="${1:?secret name}" +CM="${2:?configmap name}" +NS="${3:?namespace}" +KUBECTL="${KUBECTL:-kubectl}" + +# key -> Gitea-var-env:placeholder (placeholder applies to the Secret) +# EDIT ME: add every Secret key your deployments reference. +declare -A SECRET_KEYS=( + []=:placeholder + []=:"" +) + +# key -> Gitea-var-env:placeholder (placeholders apply to the ConfigMap) +# EDIT ME: add every ConfigMap key your deployments reference. +declare -A CM_KEYS=( + []=:"" +) + +live_has_key() { # kind name key -> 0 if the resource has the key + local v + v="$($KUBECTL get "$1" "$2" -n "$NS" -o "jsonpath={.data.$3}" 2>/dev/null || true)" + [[ -n "$v" ]] +} + +# Build a JSON patch body {"data": {...}} safely via Python json. +# Args: kind newline-joined "keyvalue" lines. +build_patch() { + python3 -c ' +import sys, json, base64 +kind = sys.argv[1] +data = {} +for line in sys.argv[2].split("\n"): + if not line: + continue + key, _, val = line.partition("\t") + data[key] = base64.b64encode(val.encode("utf-8")).decode("ascii") if kind == "secret" else val +print(json.dumps({"data": data}, ensure_ascii=False)) +' "$1" "$2" +} + +# emit_patch KIND assoc-name resource-kind resource-name +emit_patch() { + local kind="$1" declare_var="$2" reskind="$3" resname="$4" + local -n MAP="$declare_var" + local lines=() key envvar placeholder val + for key in "${!MAP[@]}"; do + envvar="${MAP[$key]%%:*}" + placeholder="${MAP[$key]#*:}" + val="${!envvar:-}" + if [[ -n "$val" ]]; then + lines+=("$key"$'\t'"$val") + elif ! live_has_key "$reskind" "$resname" "$key"; then + lines+=("$key"$'\t'"$placeholder") + fi + done + if [[ ${#lines[@]} -gt 0 ]]; then + build_patch "$kind" "$(printf '%s\n' "${lines[@]}")" + else + echo '{}' + fi +} + +apply_secret() { + local body + body="$(emit_patch secret SECRET_KEYS secret "$SECRET")" + [[ "$body" == '{}' ]] || $KUBECTL patch secret "$SECRET" -n "$NS" --type merge -p "$body" >/dev/null + echo "reconciled secret $SECRET ($NS): non-empty Gitea applied, missing seeded, present preserved" +} + +apply_cm() { + local body + body="$(emit_patch configmap CM_KEYS configmap "$CM")" + [[ "$body" == '{}' ]] || $KUBECTL patch configmap "$CM" -n "$NS" --type merge -p "$body" >/dev/null + echo "reconciled configmap $CM ($NS): non-empty Gitea applied, missing seeded, present preserved" +} + +apply_secret +apply_cm