diff --git a/.claude/settings.local.json b/.claude/settings.local.json index 886971a..fd86264 100644 --- a/.claude/settings.local.json +++ b/.claude/settings.local.json @@ -3,7 +3,22 @@ "allow": [ "WebSearch", "WebFetch(domain:docs.opentakserver.io)", - "WebFetch(domain:docs.goauthentik.io)" + "WebFetch(domain:docs.goauthentik.io)", + "Bash(grep -nE '^#{1,3} ' fleet-patch-audit.md)", + "Bash(ssh *)", + "Bash(tailscale debug *)", + "Bash(tailscale status *)", + "Bash(python3 -c \"import sys,json; d=json.load\\(sys.stdin\\); print\\('ControlURL:', d.get\\('Self',{}\\).get\\('ControlURL','n/a'\\) if 'Self' in d else 'no Self key'\\); [print\\(k,':',v\\) for k,v in d.items\\(\\) if 'control' in k.lower\\(\\)]\")", + "Read(//etc/default/**)", + "Bash(systemctl cat *)", + "Bash(timeout 5 bash -c 'cat < /dev/null > /dev/tcp/5.189.158.149/25')", + "Bash(dig +short vpn.echo6.co)", + "Bash(nslookup vpn.echo6.co)", + "Bash(timeout 5 bash -c 'cat < /dev/null > /dev/tcp/5.189.158.149/443')", + "Bash(dig +short MX echo6.co)", + "Bash(dig +short mail.echo6.co)", + "Bash(dig +short A mail.echo6.co)", + "Bash(dig +short TXT echo6.co)" ] } } diff --git a/engine/lint-report.md b/engine/lint-report.md index ccf550d..a65aa38 100644 --- a/engine/lint-report.md +++ b/engine/lint-report.md @@ -1,6 +1,6 @@ # Vault Lint Report -Generated: 2026-06-19T15:24:59Z | Docs scanned: 90 | Elapsed: 0.0s +Generated: 2026-06-19T18:00:06Z | Docs scanned: 90 | Elapsed: 0.0s ## Summary diff --git a/vault/.obsidian/workspace.json b/vault/.obsidian/workspace.json index 32351bd..0ee75fd 100644 --- a/vault/.obsidian/workspace.json +++ b/vault/.obsidian/workspace.json @@ -13,12 +13,12 @@ "state": { "type": "markdown", "state": { - "file": "docs/hardware/environment.md", + "file": "CLAUDE-baseline.md", "mode": "source", "source": false }, "icon": "lucide-file", - "title": "environment" + "title": "CLAUDE-baseline" } } ] @@ -197,19 +197,22 @@ "obsidian-livesync:Show Customization sync": false } }, - "active": "17bd4a6166f789d0", + "active": "8d53cdb6c257e685", "lastOpenFiles": [ - "runbooks/lxc-service-migration.md.tmp.40509.93172ea74019", - "runbooks/lxc-service-migration.md.tmp.40509.fec60cab0893", - "docs/hardware/ip-allocation.md.tmp.40509.ead91ed93b27", - "docs/hardware/ip-allocation.md.tmp.40509.ff3af85e1214", - "docs/hardware/ip-allocation.md.tmp.40509.7fb9dbdcf6e1", - "docs/hardware/environment.md.tmp.40509.799383b151c7", - "docs/hardware/environment.md.tmp.40509.013cd4d4a1fe", - "docs/hardware/environment.md.tmp.40509.2dba3a981e4a", - "docs/hardware/environment.md.tmp.40509.bbb40212be9a", - "docs/hardware/environment.md.tmp.40509.23a2fa131d9b", - "docs/hardware/environment.md.tmp.40509.d90c5612dafd", + "projects/nominatim-v5-reimport.md", + "projects/nominatim-v5-reimport.md.tmp.1493418.334c5df83cc6", + "projects/fleet-patch-audit.md.tmp.1493418.c5907f7438c0", + "projects/fleet-patch-audit.md.tmp.1493418.06cd7896c26f", + "projects/fleet-patch-audit.md.tmp.1493418.38b71845197f", + "projects/fleet-patch-audit.md.tmp.1493418.f77ca61905a7", + "projects/fleet-patch-audit.md.tmp.1493418.07bf87858a4f", + "projects/fleet-patch-audit.md.tmp.1493418.dff12cfa9a42", + "projects/fleet-patch-audit.md.tmp.1493418.73d7194d7615", + "projects/fleet-patch-audit.md.tmp.1493418.b6f7dc14e8ae", + "projects/fleet-patch-audit.md.tmp.1493418.a531165239ca", + "2026-06-19.md", + "Untitled.canvas", + "docs/hardware/environment.md", "projects/fleet-patch-audit.md", "rules/proxmox.md", "docs/services/services.md", @@ -224,7 +227,6 @@ "concepts/vector-database.md", "concepts/meshtastic.md", "concepts/ocr.md", - "Untitled.canvas", "archive/projects/mmud/mmud-prompts/mmud-prompts/01-update-planned.md", "INDEX.md", "glossary.md", @@ -234,8 +236,6 @@ "concepts/raspberry-pi.md", "concepts/lora.md", "concepts/knowledge-extraction.md", - "concepts/firewall.md", - "entities/mesh-bridge.md", "assets/echo6yellow_logo_422x422_square.png", "assets/echo6yellow_logo_422x81.png", "assets/echo6_logo.png", diff --git a/vault/.trash/2026-06-19.md b/vault/.trash/2026-06-19.md new file mode 100644 index 0000000..e69de29 diff --git a/vault/.trash/Untitled.canvas b/vault/.trash/Untitled.canvas new file mode 100644 index 0000000..e69de29 diff --git a/vault/CLAUDE-baseline.md b/vault/CLAUDE-baseline.md deleted file mode 120000 index 85d20f1..0000000 --- a/vault/CLAUDE-baseline.md +++ /dev/null @@ -1 +0,0 @@ -/home/zvx/.claude/CLAUDE.md \ No newline at end of file diff --git a/vault/CLAUDE-baseline.md b/vault/CLAUDE-baseline.md new file mode 100644 index 0000000..0b1f3bb --- /dev/null +++ b/vault/CLAUDE-baseline.md @@ -0,0 +1,74 @@ +--- +title: CLAUDE baseline (global rules) +type: reference +tags: [proxmox] +status: auto-generated +updated: 2026-06-19 +--- + +> [!info] Auto-generated mirror of `~/.claude/CLAUDE.md` on cortex. Edit the source, not this file — it refreshes automatically. + +# Echo6 Infrastructure — Claude Guidelines + +## Locale +Timezone: America/Boise (Mountain Time) + +--- + +## Working model — how we operate + +**Matt guides → Opus orchestrates → Sonnet executes.** + +- **Matt guides.** Sets the goal and makes the decisions. +- **Opus orchestrates — and *only* orchestrates.** Plans the work, breaks it into tight surgical tasks, dispatches **Sonnet** subagents to do them, reviews their output, and reports back. Opus does **not** do hands-on work itself — no editing files, running commands, or deploying directly. It plans, dispatches, verifies. +- **Sonnet executes — and *only* the prompt.** Runs one tightly-scoped task exactly as written. No scope creep, no initiative beyond the prompt. If the task is ambiguous or needs a decision → stop and report back to Opus; never guess. + +Flow: **Matt → Opus plans → dispatches Sonnet (tight prompt) → Sonnet executes → Opus reviews → reports to Matt.** + +--- + +## Critical policies — always apply + +- **Gemini:** `gemini-2.5-flash-lite` only, every call. No exceptions. +- **Host protection:** never `shutdown`/`reboot`/`poweroff` any host; never include cortex (primary Claude Code host) or TOC in availability-affecting bulk operations; never install packages on any host (pip/npm/apt) without explicit permission. +- **No changes without approval:** never deploy, change service ports, or change network/firewall config without explicit approval. If a target is unreachable or blocked → **STOP and report**; never redirect to an alternate host. +- **Resilience:** every deployment must survive a reboot. +- **Credentials:** source from `.ref/credentials`; never commit secrets to a git-tracked file — *except* the private `echo6-docs` Forge repo (the one documented exception). +- **Git:** GitHub `origin` is the source of truth and push target — *except* `echo6-docs`, which lives on Forge directly. Branch off the default branch before committing. Commit/push only when asked. +- **When unsure → ASK.** Never assume, never improvise. + +--- + +## Infra cheat-sheet + +| Host | Local IP | Tailscale | Role | +|------|----------|-----------|------| +| data | 192.168.1.240 | 100.64.0.6 | databases | +| utility | 192.168.1.241 | 100.64.0.5 | utility / monitoring | +| cloud | 192.168.1.242 | 100.64.0.4 | cloud / personal | +| media | 192.168.1.243 | 100.64.0.3 | media / *arr | +| toc | 192.168.1.244 | 100.64.0.13 | GPU host (passthrough → cortex) | +| **cortex** (VM 150) | 192.168.1.150 | 100.64.0.14 | GPU compute, **Claude Code**, AI | +| recon-vm (VM 1130) | 192.168.1.130 | 100.64.0.24 | recon pipeline | +| **edge1** (Contabo, rebuilt) | 5.189.158.149 | 100.64.0.40 | Proxmox edge node (PVE 8, LXC-only) — **Mail only** (Mailcow in CT 101 → 10.10.10.2) | +| **edge2** (Contabo) | 184.174.35.153 | 100.64.0.26 | Proxmox edge node (PVE 8, LXC-only) — **front door** for Auth, Forge, Notes, Matrix, Element, VPN, Vault; also hosts **PDM** (CT 100 → 100.64.0.28:8443) | +| pi-nas | 192.168.1.245 | 100.64.0.21 | NAS | + +- **SSH:** `ssh zvx@` (key auth) for most; `root@` for Proxmox hosts + edge1/edge2. Password-auth exceptions (aida-nebra, mt-isr, toc, matt-desktop) → see `environment.md`. +- **dns targets:** mail/autodiscover/autoconfig → **edge1** `5.189.158.149`; auth, forge, notes, vpn, vault, matrix, element → **edge2** `184.174.35.153`; home services (echo6.co, ai, jellyfin, immich, nextcloud, recon, stream) → `199.6.36.163` (via utility caddy). + +--- + +## Where the detail lives — load when needed + +- **Docs vault** → `.ref/vault/` (Obsidian docs library; category-tagged, maintained by `.ref/engine/` — see `.ref/CLAUDE.md`) +- **Procedures / runbooks** → `.ref/vault/runbooks/` +- **Per-project context** → `.ref/vault/projects/.md` +- **System conventions** (path-scoped) → `~/.claude/rules/` +- **Credentials** → `.ref/credentials` +- **Editing vault docs:** category tags + `[[links]]` to *existing* docs only; no entity/concept pages, no INDEX. Engine handles it. Details → `.ref/CLAUDE.md`. + +--- + +## Default behavior +When unsure → **ASK**. Default to internal access. Document everything. Never assume; never improvise. diff --git a/vault/projects/fleet-patch-audit.md b/vault/projects/fleet-patch-audit.md index c167fad..f94f108 100644 --- a/vault/projects/fleet-patch-audit.md +++ b/vault/projects/fleet-patch-audit.md @@ -13,7 +13,7 @@ status: active Read-only audit snapshot as of 2026-06-19. **Nothing has been applied — this is a planning document to build the patch plan from.** -**Topology note:** the old Contabo VPS has been rebuilt as **edge1 (mail-only)**; **edge2 is now the front door for everything else**. edge1 is excluded from this audit (mid-rebuild/maintenance). The `CLAUDE.md` cheat-sheet still lists the old Contabo layout and is **stale** — refreshing it is a follow-up task (see Open Decisions). +**Topology note:** the old Contabo VPS has been rebuilt as **edge1 (mail-only)**; **edge2 is now the front door for everything else**. edge1 is excluded from this audit (mid-rebuild/maintenance). **Headscale:** edge2 CT107 is the main fleet tailnet (34 nodes, `vpn.echo6.co`, self-hosted Headscale 0.28.0); utility CT106 is a separate IdahoMesh sub-tailnet (`vpn.idahomesh.com`, 3 nodes, low-risk). No services route through old-Contabo. **Mailcow CT108:** confirmed decommissioned — MX/A for mail.echo6.co point to edge1 (5.189.158.149, active), CT108 is stopped, backup at `/opt/mailcow-backup/mailcow-2026-06-19-03-59-45/`. --- @@ -164,19 +164,23 @@ Lowest-risk changes first; everything reboot-bearing deferred to scheduled windo --- -## Open Decisions (for tomorrow's plan) +## Open Decisions 1. **Phase 1 scope** — all non-protected guests at once, or staged worst-first? -2. **Phase 2** — patch the four hypervisor host OSes now (no reboot), or fold into the Phase 4 reboot window? -3. **Reboot-window scheduling** — order of the 5 PVE-9 nodes; **toc + cortex must be done together** (toc reboot drops cortex). edge2 needs none. -4. **Tier-2 app upgrades** — which to take on: authentik major (2025→2026, migration-heavy), Matrix/Synapse, Forgejo, headscale 0.29, mailcow. Each is its own task. -5. **data disk at 92%** — remediate before/independently of patching (operational risk regardless). -6. **CT118 archivist** rpcbind on `0.0.0.0:111` with no Tailscale/firewall — treat as a separate exposure fix. -7. **Refresh the stale `CLAUDE.md` cheat-sheet** to the edge1/edge2 topology — separate doc task. -8. **Daemon-restart tolerance** — confirm brief blips are acceptable for the stateful services (central PG16/NATS, opentakserver, peertube) during Phase 1/2. -9. **edge2 CT108 mailcow is stopped** — confirm it's superseded by the edge1 mail node and decommission it, vs. it being an unintended outage. -10. **edge2 host-kernel CVEs (DirtyFrag / copy.fail)** — verify whether the flagged in-the-wild LPEs actually apply given the host shows fully patched; if real, this is a host-kernel + reboot action on edge2 (which otherwise needs none). -11. **App-currency CVE IDs are post-cutoff** — verify the specific advisories (Authentik waves, Valkey, Immich, RabbitMQ) against primary sources before using them to justify urgency. +2. **Phase 2 host-OS sequencing** — ~~fold the four cluster hosts' OS-security apt into their Phase 3 reboot window~~ — **RESOLVED:** folded into Phase 3 per runbook (no mixed-state across Phase 2 app campaign). +3. **Reboot-window scheduling** — maintenance window dates/times in America/Boise; announce user-facing blips (Authentik SSO, home Caddy, matrix, immich/nextcloud, media). +4. **Tier-2 app upgrade scope sign-off** — which apps to take on in this campaign vs. defer: Forgejo 14→15 branch migration, Matrix/Synapse, Nominatim 4→5. Each is its own task. +5. **data disk at 92%** — remediate before Phase 1; confirm which artifacts are safe to prune (stale vzdump/snapshots/ISOs, Docker layers on recon-vm except pinned `nominatim:4.5`). +6. **CT118 archivist** rpcbind on `0.0.0.0:111` with no Tailscale/firewall — approved remediation option: disable rpcbind / bind localhost+TS / firewall. +7. **Daemon-restart tolerance** — explicit sign-off that needrestart blips are acceptable for: central CT104 (PG16/NATS/JetStream), opentakserver CT109, peertube CT110, matrix Synapse, recon-vm VM1130 PG16 + navi-backend; or lock those guests to `NEEDRESTART_MODE=l` (runbook already carves them out as stateful; this is the approval gate). +8. ~~**edge2 CT108 mailcow is stopped**~~ — **RESOLVED:** confirmed superseded by edge1. Backup verified at `/opt/mailcow-backup/mailcow-2026-06-19-03-59-45/`. Pending: verify backup stored durably off CT108, then `pct destroy 108` on approval. +9. **edge2 host-kernel CVEs (DirtyFrag / copy.fail)** — verify whether flagged in-the-wild LPEs apply given host shows fully patched; resolves whether edge2 needs a Phase 3 reboot (otherwise none required). +10. **App-currency CVE IDs are post-cutoff** — verify specific advisories (Authentik May-2026 waves, Valkey 9.1.0, Immich 2.6/2.7, RabbitMQ 4.x) against primary sources before using to justify urgency; assign owner + primary-source URL per claim. +11. ~~**Headscale location**~~ — **RESOLVED:** main fleet tailnet = edge2 CT107 (Headscale 0.28.0, `vpn.echo6.co`, 34 nodes); IdahoMesh sub-tailnet = utility CT106 (3 nodes, `vpn.idahomesh.com`). No Contabo routing. +12. **RabbitMQ 3.12→4.x scope sign-off** — confirm the separate RabbitMQ decoupled upgrade (Erlang→3.13.x→feature-flags→4.x) is in scope for this campaign, not deferred. +13. **Break-glass proof gate** — log in with akadmin in a private browser session; confirm each protected app (Forgejo, Nextcloud, PDM, Vaultwarden) has a working local-admin fallback before Phase 2 starts. +14. **Nominatim 4→5 scope** — separate project (full re-import, separate DB/instance, not in this campaign); confirm exclusion from patch campaign scope. +15. **edge2 host kernel decision** — pin `uname -r` / `proxmox-boot-tool kernel list`, check no-sub repo kernel vs. CVE-fixed version; decide reboot yes/no before Phase 3 planning. --- @@ -189,61 +193,141 @@ Step-by-step plan to bring every application and package current. Per-app target - **Protected hosts — cortex & toc — never in a bulk pass.** They get a dedicated, manual window. Note the coupling: **toc reboot drops cortex (VM150), which is the Claude Code host** — expect to lose the control session during that window; drive it from elsewhere or accept the outage. - **One target at a time.** Verify health before moving to the next. - **Snapshot/backup before each mutating step:** `pct snapshot` / `qm snapshot` (or `vzdump`) the guest first; DB dump before any DB-bearing upgrade; run the app's own backup where it has one (Nextcloud AIO, mailcow). -- **No host reboots outside the explicit Phase 3 windows.** Security upgrades that set `reboot-required` are applied but activation deferred to Phase 3. -- **Rollback = restore the snapshot / redeploy the previous image tag.** Record before/after version in the tracking table. +- **No host reboots outside the explicit Phase 3 windows.** Security upgrades that set `reboot-required` are applied but activation deferred to Phase 3. Host OS-security apt for the four cluster nodes is NOT applied in Phase 1 — it rides the Phase 3 full-upgrade window. +- **Rollback for DB-bearing apps = quiesced logical dump + old binary together.** For authentik, forgejo, nextcloud, peertube, and immich the rollback unit is a `pg_dump` / AIO borg / peertube dump taken with the app stopped and the old binary still in place. "Redeploy previous image tag" is NOT valid after forward-only migrations — Django/TypeORM migrations do not reverse, and a `pct snapshot` of a hot separate-volume DB may restore torn. Record before/after version in the tracking table. - Every result must survive a reboot (standing infra policy). ### Load-bearing dependencies (drive the ordering) - **Authentik (edge2 CT105) = SSO.** Upgrading it briefly breaks login to everything behind it. Confirm **break-glass admin** access first; do in a low-traffic window; verify dependent-service logins after. - **Caddy (utility CT101) = home-services ingress.** A restart blips all home services — fold its update into a deliberate moment, not mid-day. -- **Headscale ×2 (edge2 CT107 main + utility CT106 mesh) = tailnet coordination.** Existing tunnels usually persist a restart, but edge2 is the front door and you may be managing *over* Tailscale — back up the DB and verify nodes stay joined. +- **Headscale ×2:** edge2 CT107 = main fleet tailnet (34 nodes, `vpn.echo6.co`, ControlURL `https://vpn.echo6.co` → 184.174.35.153) — **HIGH lockout risk**. utility CT106 = IdahoMesh sub-tailnet (`vpn.idahomesh.com`, 3 nodes) — low risk. Upgrade CT106 first as rehearsal. Edge2 CT107 upgrade MUST be driven from an out-of-band path (edge2 host console / direct SSH to 184.174.35.153, NOT over `vpn.echo6.co`); extend node key-expiry first; dump DB off-tailnet before touching it. - **Nominatim 4→5 is NOT a routine update** — it's a re-import project (see Phase 2C). ### Phase 0 — Pre-flight (no app changes yet) -1. **Free space on `data`** (92% full) so snapshots/image pulls have room. -2. **Resolve edge2 CT108 mailcow** — confirm it's superseded by edge1; if so, back up then stop/destroy → removes it from scope. If it's an *unintended* outage, that's a separate incident. -3. **Verify break-glass:** local admin creds for Authentik-protected apps; PVE console access independent of the services being changed. -4. **Confirm snapshot headroom** on each PVE node's storage. -5. **Resolve the edge2 host-kernel CVE question** (DirtyFrag/copy.fail) → decides whether edge2 needs a Phase 3 reboot after all. -6. **Freeze the baseline** — this doc's versions are the pre-change record. +**STEP 1 (FIRST — gates everything mutating): Verify break-glass and out-of-band access.** Confirm a non-Tailscale, non-Caddy path to every node: Proxmox/Contabo console for edge2, LAN `192.168.1.x` SSH for home nodes, PVE noVNC/`pct enter` for each CT. Confirm PVE/Proxmox web auth is PAM/local, NOT behind Authentik. Prove it: log into Contabo/PVE console; `pct enter` 105/106/107; confirm LAN SSH; locate break-glass creds in `.ref/credentials`. Verify Authentik akadmin (private browser session) + each protected app has a local-admin fallback (Forgejo `admin user`, Nextcloud `occ`, PDM/Vaultwarden local). Every node + CT shell must be reachable WITHOUT Authentik/headscale/Caddy. Losing the management path mid-run is the single highest-consequence failure mode. -### Phase 1 — OS security packages (no reboot) +**STEP 2: Measure snapshot headroom (per node).** Run `pvesm status` + `df -h` on data/utility/cloud/media/toc/edge2. Record the storage backend per node (LVM-thin/ZFS/dir-qcow2) and its over-provision failure mode. Hard numeric rule: never start a snapshot below ~20% free; require ≥2× the largest guest's expected write-delta. Cloud node CT120/CT121 free space must be measured now — not assumed. Identify where vzdump lands; if it's `data`, freeing `data` is a fleet-wide rollback prerequisite. -Guests first (low blast radius), then hypervisor hosts. Per target: `apt-get update` → snapshot → apply **security-pocket** upgrades → `needrestart` to bounce affected daemons → verify service health and that the security-upgradable count hits 0. +**STEP 3: Free space on `data`** (92% full, ~73 GB free) — sized to the largest planned snapshot (VM1130 recon-vm / nominatim+overture+padus). Prune stale vzdump/snapshots/ISOs; `docker image prune` on recon-vm (do NOT remove the pinned `nominatim:4.5` image). Confirm nothing deleted is the only copy of a backup or live NAS data. Verify: usage <~85% on snapshot-backing storage; test snapshot succeeds then remove. **Blocks data host Phase 3 and canary use.** -- **Worst-first guests:** utility CT119 (91 sec), CT108 (79), cloud CT120/CT121 guest-OS (101/75), media CT110 (37), utility CT109 (31), CT104 (38, incl. PG 16.13→16.14). -- **Remaining guests:** all other utility/cloud/media/edge2 CTs + data VM1130 (recon-vm). -- **Hosts (one at a time):** data, utility, cloud, media. **edge2 host already patched. toc excluded** (Phase 3, with cortex). -- Kernel/libc security that flags `reboot-required` → applied, reboot deferred to Phase 3. +**STEP 4: Verify off-host restorable backups** (distinct from per-step snapshots) for every stateful guest: central PG/NATS (CT104), opentakserver PG + RabbitMQ definitions (CT109), forgejo PG + repos (CT103), matrix Synapse PG (CT106), nextcloud AIO borg (CT121), edge2 livesync CouchDB (CT104). At least one copy must live off the node being changed. + +**STEP 5: Capture pre-change baseline HEALTH snapshot.** Record per-target up/down, container states (`docker ps`/`pct`/`qm status`), key endpoint 200 checks, and `apt list --upgradable`/security counts. Note known oddities: jellyseerr on preview-OIDC dev tag, mailcow CT108 already STOPPED, pinned-stale nominatim, CT118 archivist rpcbind on `0.0.0.0:111`. + +**STEP 6: Decision gates.** Close genuinely-open decisions before Phase 1/2/3 can start: daemon-restart tolerance sign-off (Open Dec #7); verify post-cutoff CVE claims against primary advisories with owner+URL per claim; resolve edge2 host-kernel CVE question fully (pin `uname -r`, check repo kernel availability, decide reboot yes/no — not left conditional); define and announce maintenance windows in America/Boise naming user-facing blips. + +**STEP 7: edge2 CT108 mailcow — decommission (CONFIRMED safe).** CT108 is stopped, edge1 is actively serving SMTP (MX/A → 5.189.158.149), backup exists at `/opt/mailcow-backup/mailcow-2026-06-19-03-59-45/`. Action: verify that backup is stored durably OFF CT108 (copy off-host), confirm no MX points at edge2, then `pct destroy 108` ON APPROVAL. Removes CT108 from all later scope. + +### Phase 1 — Guest OS security packages (no reboot; hosts folded into Phase 3) + +Guests only in Phase 1 (low blast radius). **The four cluster hosts' OS-security apt rides their Phase 3 full-upgrade window** — do NOT patch data/utility/cloud/media host OS in Phase 1 (avoids a mixed-state libc/openssl across the entire Phase 2 app campaign; both phases get cleaned up in one reboot anyway). edge2 host is already fully patched — no action. toc excluded (Phase 3, with cortex). + +Per guest target: `apt-get update` → snapshot → apply **security-pocket** upgrades → needrestart policy (see below) → verify service health and security-upgradable count hits 0. Kernel/libc security that flags `reboot-required` → applied, reboot activation deferred to Phase 3. + +**needrestart policy:** +- **Stateful guests — `NEEDRESTART_MODE=l` (list-only), then deliberate bounce:** CT104 central (PG16/NATS/JetStream), CT109 opentakserver, CT110 peertube, VM1130 recon-vm. Stage the packages; take a logical DB dump with the app quiesced; bounce each service deliberately in controlled order. A JetStream restart mid-write loses messages on at-most-once pipelines; a RabbitMQ/redis bounce mid-job is not a clean "blip." +- **Lockout-critical daemons — `NEEDRESTART_MODE=l` + deliberate controlled bounce:** Authentik CT105, headscale instance(s), Caddy CT101/CT111, sshd on any host you're connected through. Run `needrestart -r l` to see what would restart; bounce deliberately so a libc/openssl bump can't auto-bounce auth/tailnet/SSH daemons out from under the operator. +- **All other guests:** `needrestart -r a` (automatic) is acceptable — seconds of blip, no data risk. + +**Ordering:** +1. **Canary first** (utility CT112 cobalt idle, or CT102 searxng) — prove the full procedure on a low-stakes guest before touching never-patched targets. +2. **Worst-first (after canary proven):** utility CT119 mesh-territory (91 sec, never patched), CT108 meshai (79 sec), cloud CT120/CT121 guest-OS (101/75 sec), media CT110 peertube (37 sec), utility CT109 opentakserver (31 sec), CT104 central (38 sec, incl. PG 16.13→16.14 + NATS). +3. **Remaining guests:** all other utility/cloud/media/edge2 CTs + data VM1130 (recon-vm). Gate data VM1130 step on data disk remediation verified DONE. +4. **edge2 CT106 matrix / CT107 headscale (NO Tailscale client):** drive via `pct exec` from the edge2 HOST (`root@184.174.35.153`) ONLY — never over service ingress. Treat as own singleton steps, not part of edge2 batch. `needrestart -r l`; do NOT restart headscaled/synapse unless required. +5. **Control-plane singletons LAST:** home-ingress Caddy CT101 and headscale instance(s) — each with DB backup + node-stays-joined verification, never inside a batch. +6. **cortex VM150 (PROTECTED):** security apt manually in the toc+cortex window (Phase 3 coordination); not in bulk. +7. **Cross-cutting:** Tailscale 1.94→1.98 and Docker CE→29.6 ride along Phase 1 under the same snapshots; EXCLUDE edge2 CT106/CT107 from Tailscale bump (no TS client). +8. **pi-nas:** Debian security pocket with `linux-image-*` and `openmediavault*` packages held; kernel upgrade rides Phase 3. ### Phase 2 — Application updates -**2A — Self-managed updaters** (snapshot → run native updater → verify): -- **OpenTAKServer (utility CT109)** — update per OTS upgrade procedure; **this is the agreed fix for the EOL RabbitMQ** plus MediaMTX/Mumble currency. Snapshot first; verify TAK clients reconnect. -- **Immich (cloud CT120)** — `docker compose pull && up -d` in its dir (also clears Valkey 9.0.2→9.1.0). Verify web, mobile sync, ML. -- **Nextcloud AIO (cloud CT121)** — run AIO backup → update mastercontainer → trigger child update via AIO UI (:8080). Verify `occ status`, apps. +**Hard isolation rule:** {Headscale, Authentik, Caddy, Forgejo} are NEVER in the same maintenance window — at least one of {tailnet, SSO, ingress, git} must always be a known-good recovery path. -**2B — Versioned apps** (snapshot → bump/upgrade → verify): -- **Authentik (edge2 CT105)** — **sequential, no skipping:** 2025.12.4 → **2025.12.6** (backported security, lowest risk) → 2026.2.x → **2026.5.3**, running DB migrations at each hop. SSO window + break-glass. Verify dependent logins after each hop. -- **Forgejo (edge2 CT103)** — **branch migration 14.x → 15.0.3**, not a patch: read 15.0 breaking changes, back up repos + DB, bump tag, verify. -- **Headscale ×2 (edge2 CT107, utility CT106)** — follow the 0.28→0.29 upgrade guide; back up DB; **one at a time**; verify nodes stay connected. -- **Media stack (media VM105)** — `docker compose pull && up -d` for jellyfin / sonarr / radarr / prowlarr / navidrome; **read release notes for the majors** — SABnzbd 4→5 and Lidarr 2→3; **move jellyseerr off the `preview-OIDC` dev tag to stable 3.3.0.** Snapshot VM first; verify each UI. -- **PeerTube (media CT110)** — native install: back up DB → run PeerTube's upgrade script 8.0.2→8.2.1 → verify. -- **cortex AI stack (PROTECTED — manual, own window):** ollama 0.16→0.30 (14 versions — check model compat), open-webui 0.8→0.9, qdrant, tei. Pull images, verify. -- **Lower-urgency batch:** caddy, couchdb (livesync), valhalla, photon, kiwix, navidrome, NATS, valkey-8 sidecar — snapshot + update as convenient. -- Vaultwarden, PDM, WordPress, Synapse/Element/MAS, obsidian, recon-vm Postgres → **already current, skip.** +**2A — Load-bearing SSO and control-plane apps (in this order):** -**2C — Separate project (not routine):** -- **Nominatim 4.5 → 5.3 (recon-vm)** — requires the `mediagis/nominatim:5.x` image + **full OSM re-import**, coordinated with **Photon 1.1→1.2** (1.2 reads the v5 data format). Schedule as its own effort with disk/time budget. +- **Authentik (edge2 CT105) — FIRST, sequential hops, pg_dump per hop.** + - **Pre-flight before any hop:** (1) Prove break-glass: log in as akadmin in a private browser session; confirm every protected app has a working local-admin fallback (Forgejo `admin user`, Nextcloud `occ`, PDM/Vaultwarden local, PVE PAM). (2) Audit/rename duplicate group names — the Django migration fails loudly on dupes. (3) If local `/media` storage is used, stop the service, `mv ./media ./data/media`, rewrite compose volume, before starting any new version. + - **Hop sequence (no skipping):** 2025.12.4 → **2025.12.6** (backported security) → **2026.2.x** → **2026.5.3**. One hop at a time: `pct snapshot` + `pg_dump` (labelled per hop) → pull new tag → `compose up -d` → migrations run → verified-login gate (all dependent SSO logins tested) → abort+restore on FIRST migration error. Verify all dependent SSO logins BEFORE upgrading any of the dependents. + - Rollback: `compose down` → restore prior tag + dump; or `pct rollback` for the full CT. + +- **Caddy (utility CT101 + media CT111) — early in Phase 2, before home-service app verification.** Pair CT101/CT111 in one deliberate low-traffic window (separate from Authentik, Forgejo, and headscale windows). Validate `caddy validate`; spot-check Caddy↔Authentik path; document a direct Caddy-bypassing Authentik admin URL as break-glass. Verify all home services reachable through the new ingress before proceeding with other app upgrades. + +**2B — Self-managed updaters** (snapshot → run native updater → verify): + +- **OpenTAKServer (utility CT109) — app upgrade first, RabbitMQ SEPARATE:** + - Run OTS upgrade script (upgrades OTS + webUI + schema only). Snapshot first. Verify OTS web/API up; CoT works; TAK clients reconnect. + - **RabbitMQ 3.12.1→4.2.8 is a SEPARATE sub-task (decoupled, after OTS app step).** The OTS updater does NOT migrate RabbitMQ, Erlang, or feature flags — the EOL RabbitMQ 3.12 would silently remain. Decoupled upgrade: `pct snapshot` + `rabbitmqctl export_definitions` → upgrade Erlang to ≥26 → 3.12→3.13.x → `rabbitmqctl enable_feature_flag all` (confirm all enabled, no classic-queue-mirroring config remains) → only then 4.x. A naive 3.12→4.x jump refuses to boot. Gate "OTS done" on RabbitMQ 4.x up + TAK clients reconnecting before accepting the bundled MediaMTX/Mumble bumps. + - Rollback: restore snapshot + definitions export. + +- **Immich (cloud CT120)** — prune old images + `docker image prune` before snapshot; disable Immich background regeneration until free space is confirmed; `docker compose pull && up -d`. Do NOT interrupt first-boot 2.5→2.7 DB migration. Also clears Valkey 9.0.2→9.1.0. Rollback = restore `pct snapshot` + `pg_dump` taken with the app stopped (not a hot-DB snapshot alone). Verify web, mobile sync, ML. + +- **Nextcloud AIO (cloud CT121)** — AIO borg backup → stop → update mastercontainer → trigger child update via AIO UI (:8080). A stale AIO may need two cycles. Rollback unit = AIO borg backup (mastercontainer downgrade is not supported — borg is the only path back). Verify `occ status` 32.0.11; 12 containers green; integrity clean. + +**2C — Versioned apps** (snapshot → bump/upgrade → verify): + +- **Forgejo (edge2 CT103) — branch migration 14.x→15.0.3 (separate window from Authentik).** + - Pre-steps: `git clone --mirror` echo6-docs (and other critical repos) to an OFF-edge2 location; pause the `echo6-docs-autocommit` cron so recovery docs survive an edge2/Forgejo failure; run `forgejo doctor check --all [--fix]` to repair stopwatch/tracked_time inconsistencies BEFORE backup+bump. Read 15.0 breaking changes + forward-only FK migration notes. + - `pct snapshot 103` + repos volume + `pg_dump` + app.ini → bump tag 14→15 → pull/up → migrations → verify. + - Rollback: restore 14.0.5 image + volume + dump. + +- **Headscale 0.28→0.29.1 — utility CT106 FIRST (rehearsal), edge2 CT107 LAST.** + - CT106 (IdahoMesh, 3 nodes, low risk): standard 0.28→0.29 upgrade guide; `pct snapshot` + DB backup; stop → upgrade binary → config migrate → start. Verify `nodes list` all joined; verify headplane compatibility. + - CT107 (main fleet, 34 nodes, HIGH lockout risk — edge2 front door): extend/disable node key-expiry on all 34 nodes BEFORE starting; `pct snapshot 107` + dump headscale DB off-tailnet (NOT via vpn.echo6.co); keep a second authenticated SSH session open to 184.174.35.153. Drive entirely from edge2 host console / direct SSH to `root@184.174.35.153` — NOT over the tailnet being upgraded. Verify boot chain: edge2 host → CT107 autostart → headscale up → `vpn.echo6.co` resolves → Caddy proxies. Rollback: restore snapshot via console (not via tailnet). + - NOT in the same window as Caddy, Authentik, or Forgejo. + +- **Media stack (media VM105):** + - `qm snapshot 105` before batch. SABnzbd 4.5→5.0: pause/empty queue; audit custom post-proc scripts (`empty_postproc` removed in 5.x, scripts now run on failed jobs); pin 5.0.4 tag; pull/up; verify queue intact. Lidarr 2.x→3.x: pin v3; pull/up; DB migration on start; verify library+indexers+download client. Sonarr/Radarr/Prowlarr (1-2 ver): routine bumps after SABnzbd/Lidarr. Jellyfin 10.11.6→10.11.11: pin tag; pull/up. Navidrome: routine bump. + - **Jellyseerr preview-OIDC→stable 3.3.0: DECISION GATE before retag.** preview-OIDC → stable 3.3.0 is a schema DOWNGRADE; TypeORM stable may refuse to start against a dev-migrated DB. Confirm the correct destination artifact (jellyseerr stable vs "Seerr" successor) and, if OIDC is in active use, confirm 3.3.0 parity. Test stable image against a COPY of the DB first. If parity unclear → STOP and report. + - Verify each UI; snapshot before, per-app snapshots for majors. + +- **PeerTube (media CT110)** — quiesce import/transcode jobs; `pct snapshot 110` + `pg_dump peertube_prod`; run PeerTube upgrade script 8.0.2→8.2.1; restart; verify. Rollback = `pct rollback` + restore dump. + +- **cortex AI stack (PROTECTED — manual, own window):** ollama 0.16.1→0.30.10 (snapshot models volume — the only rollback after pulling new models; upgrade; verify existing `ollama run` without re-pull; verify vault-tagger at localhost:11434 and TEI at localhost:8090 still work); open-webui 0.8.1→0.9.6; qdrant 1.16→1.18; tei 1.7→1.9. Pull images, verify. + +**2D — Lower-urgency batch:** caddy (minor), couchdb livesync (3.4→3.5.2), valhalla, photon, kiwix, NATS 2.14.0→2.14.2, valkey-8 sidecar (8.1.5→8.1.8) — snapshot + update as convenient. + +Vaultwarden, PDM, WordPress, Synapse/Element/MAS, obsidian, recon-vm Postgres → **already current, skip.** + +**2E — Separate project (not in this campaign):** +- **Nominatim 4.5→5.3 + Photon 1.1→1.2 (recon-vm)** — requires full OSM re-import into a SEPARATE DB/instance; keep `nominatim:4.5` image + data until 5.x validated; transient import disk (flat-nodes tens of GB) cannot live on the 92%-full `data` pool; Photon 1.2 AFTER nominatim 5.x validated, never concurrently. Schedule as its own effort. ### Phase 3 — Platform & reboot windows (coordinated, approval-gated) -- **PVE-9 nodes** (data, utility, cloud, media, toc): PVE 9.1.1→9.2, QEMU 10→11, LXC 6→7, kernel 6.17.2→6.17.13. Per node: snapshot/backup guests → `apt full-upgrade` the proxmox stack → reboot → verify all guests return. **One node at a time.** Validate on a low-stakes node (data or media) before utility/cloud. -- **toc + cortex = the special window, done LAST and together:** toc gets PVE 9.2 + reboot; cortex gets NVIDIA 580.159→580.167 + DKMS + container-toolkit 1.18→1.19 + apt + reboot. **The Claude Code host goes down here** — run this step from a different control point. -- **pi-nas** (standalone window): kernel 6.12→6.18 + OMV 8.1→8.4 (via OMV's update path) → reboot. -- **edge2**: reboot only if Phase 0 step 5 confirmed it needs a newer kernel; otherwise none. +**Corosync cluster rules (apply to EVERY node reboot in this section):** +- The five PVE nodes (data, utility, cloud, media, toc) are ONE `echo6-cluster`, quorum = 3 of 5. Losing quorum makes `/etc/pve` read-only fleet-wide and freezes all VM/CT start/stop/config on every surviving node. +- Before EACH node reboot: `pvecm status` must show `Quorate: Yes` and `Expected votes: 5`. Also run `ha-manager status` — if any guest is HA-managed, the reboot triggers fencing/auto-migration rather than a clean local stop/start. +- Reboot ONE node at a time. Wait for full 5/5 rejoin before the next node. +- Disable HA migration and NO live migration during the mixed-version window (QEMU 10/11 boundary — running VMs keep QEMU 10 machine type until cold-started; cross-boundary migration fails). + +**Per-node procedure:** `apt update && apt full-upgrade` (so proxmox-ve/qemu/lxc metapackages pull) → this also clears the held host OS-security packages → reboot → verify `pveversion` + all guests return with `onboot=1`. Confirm `pvecm status` shows 5/5 before proceeding. + +**Pre-window for each node:** vzdump all guests to an EXTERNAL target (not the local pool, never `data`) so a guest that won't start under LXC 7 can be restored elsewhere. Read PVE 9.2 + LXC 7 release notes for unprivileged-CT/cgroupv2/apparmor breaking changes before the first node. + +**Canary node order (lowest blast-radius with headroom first):** media (single arr VM + peertube, non-auth/non-DB-critical) → cloud (immich/nextcloud, after media proven) → data (after disk remediation verified; its near-full pool can fail the mandatory pre-reboot snapshot — do NOT use as canary until remediated) → **utility LAST of the four** (carries Caddy ingress + mesh/central stack — deliberately-scheduled full home-ingress + mesh-coordination outage, announce in advance) → **toc+cortex LAST AND ALONE** (see below). + +**Post-reboot re-verification gate (per node):** after each node boots into 9.2/LXC7, re-run the Phase-2 health checks for every guest that received a major app upgrade on that node (CT120 immich, CT121 nextcloud, CT110 peertube, VM105 arr, CT109 OTS). The substrate changed under them. + +**toc + cortex — DONE LAST AND ALONE, after all four other nodes are quorate:** +- toc is the 5th cluster vote AND the GPU host for cortex (VM150) AND the management/Claude Code host. Rebooting toc removes a vote and kills the box you'd diagnose from. It must NEVER overlap any other node reboot (toc down + one other = 3/5 → one corosync flap loses quorum). +- Name and TEST a concrete alternate control point (a non-cortex host with keys + tooling) BEFORE this window. Stabilize the tailnet well before it — never in the same period as a headscale upgrade. +- cortex steps IN ORDER: + 1. `qm snapshot 150 pre-nvidia` + vzdump (from toc, before anything changes) + 2. `apt --only-upgrade nvidia-container-toolkit`; `nvidia-ctk runtime configure`; restart docker → verify `--version`=1.19 + `docker run --gpus all nvidia-smi` + 3. `apt --only-upgrade nvidia-driver-580 nvidia-dkms-580`; DKMS rebuild → verify `dkms status` shows installed + `nvidia-smi`=580.167 BEFORE rebooting toc. Keep the old driver package installed as a reinstall fallback. + 4. cortex Phase 1 security apt (if not already done in its own window) +- Then reboot toc → verify toc rejoins (`pvecm status` 5/5) → verify VM150 autostart + cortex GPU passthrough return → final AI-stack health check (driver 580.167; all 5 containers; ollama/tei/qdrant/open-webui; vault-tagger+TEI engines; TS online; Docker 29.6). +- **The Claude Code host goes down here** — run this step from the alternate control point. + +**pi-nas — standalone window, TWO reboots:** +- Confirm physical/serial console access (no remote KVM on an RPi); export `config.xml` off-box; confirm pi-nas is not mid-sync as a Syncthing/backup target before each reboot. Back up `/boot` + `/boot/firmware`; keep the old kernel installed as fallback. +- Reboot 1: OMV 8.1→8.4 via OMV's own update path (NOT raw `apt full-upgrade` — use OMV UI Update Mgmt or `omv-upgrade`) → reboot → verify shares/SMB/NFS/omv-salt healthy. +- Reboot 2 (AFTER reboot 1 verified): kernel 6.12→6.18 — `apt install linux-image-arm64` (unhold); update bootloader; reboot ON APPROVAL → verify `uname -r`=6.18; LAN+tailnet up; shares mount; Docker up. Rollback: boot retained 6.12; restore `/boot`; on-site SD reflash. +- Independent of cluster windows. + +**edge2 — CONDITIONAL:** +- Reboot only if Phase 0 resolved the DirtyFrag/copy.fail CVE question as "yes, needs a newer kernel." If reboot required: give it a dedicated window AFTER Authentik CT105 + headscale CT107 are upgraded and verified; announce SSO+tailnet downtime; drive from a non-tailnet path; confirm all CTs auto-start. edge2 is a single SPOF for ingress — no failover. Do NOT fold into cluster windows. ### Phase 4 — Cross-cutting (fold into earlier phases) @@ -261,7 +345,7 @@ Guests first (low blast radius), then hypervisor hosts. Per target: `apt-get upd | 0 | `data` host — disk remediation (92% full, ~73 GB) | Free space sized to largest planned snapshot (VM1130) | None (cleanup; confirm deletions are cache/backup not live) | Prune stale vzdump/snapshots/ISO; `docker image prune` on recon-vm (keep `nominatim:4.5`) | Usage <~85% on snapshot-backing storage; test snapshot succeeds then removed | Restore from Forge/Syncthing/backup if a needed file removed | Free-space capture; BLOCKS data host step + canary use | | 0 | Off-host restorable backups (stateful guests) | Verify last-good vzdump/PBS off the node for central, OTS, forgejo, matrix, nextcloud, edge2 livesync CouchDB | N/A | Read-only verification + on-demand `pg_dump`/app backup copied off-host | ≥1 backup off the changing node, restore-testable | N/A | Precedes all mutating stateful steps | | 0 | Baseline HEALTH capture (all targets) | Record up/down, container states, endpoint 200s, `apt upgradable`/security counts; note oddities | None | `pct/qm status`, `docker ps`, curl probes, `apt list --upgradable` | Baseline recorded; oddities logged (jellyseerr dev tag, CT108 stopped, nominatim pin, CT118 rpcbind) | N/A | Precedes Phase 1 | -| 0 | Decision gates | Close Open Dec #8 (needrestart tolerance); verify post-cutoff CVE claims vs primary advisories w/ owner+URL; resolve edge2 kernel question; reconcile CT106-vs-CT107 headscale location; define+announce Boise windows | None | Read-only / sign-off | Each decision recorded before Phase 1/2/3 can start | N/A | Gates Phase 1/2/3 | +| 0 | Decision gates | Close daemon-restart tolerance sign-off (Open Dec #7); verify post-cutoff CVE claims vs primary advisories w/ owner+URL; resolve edge2 kernel question (uname -r vs repo availability); define+announce Boise maintenance windows with user-facing blip list | None | Read-only / sign-off | Each decision recorded before Phase 1/2/3 can start | N/A | Gates Phase 1/2/3 | | 0 | edge2 CT108 mailcow (STOPPED) | Confirm superseded by edge1, back up, decommission | `pct snapshot 108 predecommission` + vzdump + mailcow native backup | Confirm edge1 mail live + no MX at edge2; `pct stop`/`pct destroy 108` ON APPROVAL | CT108 gone/archived; edge1 mail in+out works; vzdump restorable | `pct restore 108` + start; re-point MX | edge1 confirmed; approval (Open Dec #9) | | 0 | edge2 host kernel CVE question | Decide if DirtyFrag/copy.fail require a reboot | None (record `pveversion -v`) | Compare installed proxmox-kernel vs verified-advisory fixed versions; escalate if no-sub repo lacks fix (no repo changes w/o approval) | Decision (reboot yes/no) recorded | N/A | Gates edge2 Phase 3 row | | 1 | Procedure canary — utility CT112 cobalt (idle) or CT102 searxng | Prove apt→snapshot→security upgrade→needrestart→verify on low-stakes guest | `pct snapshot` pre-phase1 | `apt-get update`; security pocket only; `needrestart -r l` then deliberate | Security-upgradable=0; service healthy; needrestart clear | `pct rollback` | Phase 0; runs BEFORE worst-first guests | @@ -328,58 +412,24 @@ Guests first (low blast radius), then hypervisor hosts. Per target: `apt-get upd --- -## Plan Hardening (from review) +## Key Facts & Top Risks -The review surfaced corrections across six dimensions. They are merged and de-duplicated below, critical/high first, grouped by phase, with hosts/CTs named. Several issues recur because the same architectural fact (one corosync cluster; control plane runs over the tailnet; `data` is 92% full; SSO + ingress + git + tailnet are mutually entangled) drives multiple failure modes — the consolidated corrections address the root cause once. +### Cluster & control-plane facts -### Cluster & control-plane facts that the whole plan must respect -- **The five PVE nodes are ONE corosync cluster (`echo6-cluster`), not standalone hypervisors.** Confirmed in `vault/docs/hardware/environment.md` (line 17) and `vault/runbooks/proxmox-onboard-node.md`. Quorum = 3 of 5. The plan never mentions corosync/quorum. **Every Phase 3 node reboot must be gated on `pvecm status` → `Quorate: Yes`, expected votes = 5; reboot ONE node at a time; wait for full rejoin (5/5) before the next.** Losing quorum makes `/etc/pve` read-only fleet-wide and freezes all VM/CT start/stop/config on every surviving node. Also run `ha-manager status` first — if any guest is HA-managed, a reboot triggers fencing/auto-migration, not a clean local stop/start. -- **toc is the 5th cluster vote AND the GPU host for cortex (VM150) AND cortex is the documented cluster "Management host" + the Claude Code control host.** Rebooting toc removes a vote and kills the box you'd diagnose from. **toc+cortex must be the final cluster action and must never overlap any other node reboot** (toc down + a second node down = bare 3/5; one corosync flap loses quorum). -- **Reconcile the headscale location contradiction BEFORE touching any of it.** The audit places headscale at utility CT106 *and* edge2 CT107 (and calls CT106 both "meshtastic-hs" and "headscale control plane"); the vault docs (`headscale-onboard-node.md`) describe headscale as a Docker container on Contabo (`100.64.0.1`) reached over the VPN. You cannot protect a control plane you have mislocated. Pin the real instance(s) and their reachability first; this gates every headscale step below. +- **One corosync cluster.** data, utility, cloud, media, toc are `echo6-cluster` — quorum 3 of 5. Every Phase 3 reboot is gated on `pvecm status` (Quorate:Yes, expected votes=5); ONE node at a time; confirm 5/5 rejoin before next. +- **toc coupling.** toc is the 5th cluster vote + GPU host for cortex VM150 + the management/Claude Code host. toc+cortex are the FINAL Phase 3 action, alone — never overlap another node reboot. +- **Headscale topology (resolved).** Main fleet tailnet (34 nodes, ControlURL `vpn.echo6.co`) = edge2 CT107 (Headscale 0.28.0) — HIGH lockout risk; upgrade LAST, drive from out-of-band. Separate IdahoMesh sub-tailnet (`vpn.idahomesh.com`, 3 nodes) = utility CT106 — low risk; upgrade FIRST as rehearsal. No routing through old-Contabo. -### Phase 0 — preflight (promote these to hard gates; some must move to first) -- **(MOVE TO FIRST) Verify a non-Tailscale, non-Caddy break-glass path to every node** — Contabo/Proxmox console for edge2, LAN `192.168.1.x` for the home nodes, PVE noVNC into each CT — and confirm PVE/Proxmox web auth is PAM/local, NOT behind Authentik. Losing the management path mid-run is the single highest-consequence failure on this fleet. This must pass before anything mutating. -- **Measure, don't hand-wave, snapshot headroom.** Run `pvesm status` + `df -h` on data/utility/cloud/media/toc/edge2. Record the storage backend per node (LVM-thin/ZFS/dir-qcow2) and its over-provision failure mode. Set a hard numeric rule: never start a snapshot below ~20% free; require ≥2× the largest guest's expected write-delta. `data` at 92% (~73 GB free) cannot safely snapshot VM1130 recon-vm (nominatim/overture/padus), and **cloud node free space (CT120 immich, CT121 nextcloud AIO) is not measured at all** — add it. Identify where vzdump lands; if it's `data`, freeing `data` is a fleet-wide rollback prerequisite, not a data-node nicety. -- **`data` disk remediation is a blocking gate, sized from measurement.** Identify consumers (old vzdump, snapshots, ISO/template store, recon-vm dangling Docker layers — do NOT remove the pinned `nominatim:4.5` image), confirm nothing deleted is the only copy of a backup or live NAS data, and size cleanup to the largest planned snapshot on the node. Must complete before `data`'s own host step and before it is used as a canary. -- **Verify off-host restorable backups exist** (distinct from per-step snapshots) for every stateful guest: central PG/NATS, opentakserver PG + RabbitMQ definitions, forgejo PG + repos, matrix Synapse PG, nextcloud AIO (borg), edge2 livesync CouchDB. At least one copy must live off the node being changed. -- **Capture a pre-change baseline HEALTH snapshot** (per-target up/down, `docker ps`/`pct`/`qm status`, key endpoint 200 checks, and `apt list --upgradable`/security counts) so "verify after" has a comparison and known oddities are recorded: jellyseerr on the preview-OIDC dev tag, mailcow CT108 already STOPPED, pinned-stale nominatim, CT118 archivist rpcbind on `0.0.0.0:111`. -- **Close Open Decision #8 as a gate:** explicit sign-off on needrestart-driven daemon bounces for central (CT104 PG16/NATS/JetStream), opentakserver (CT109), peertube (CT110), matrix Synapse, recon-vm VM1130 PG16 + navi-backend, and the auth/tailnet daemons — or set those guests `NEEDRESTART_MODE=l` (list-only) so the operator controls the bounce. -- **Verify the load-bearing post-cutoff CVE claims against primary advisories before they drive ordering** (Authentik May-2026 waves justifying the mandatory 2025.12.6 hop; Valkey 9.1.0; Immich 2.6/2.7; RabbitMQ 4.x; edge2 DirtyFrag CVE-2026-43284/-43500 and copy.fail CVE-2026-31431). Assign an owner + primary-source URL per claim. **Resolve the edge2 kernel question fully here** (pin `uname -r`/`proxmox-boot-tool kernel list`, check repo kernel availability) so edge2 is either firmly scheduled for its own no-failover reboot window or firmly excluded — not left conditional inside Phase 3. -- **Decommission decision for mailcow CT108** (STOPPED, superseded by edge1): vzdump + mailcow native backup, confirm edge1 is handling mail and no MX points at edge2, then `pct destroy` only on approval. Removes CT108 from all later scope. -- **Define and announce maintenance windows in America/Boise**, naming user-facing blips: Authentik SSO (the upgrade "briefly breaks login to everything"), all home services behind Caddy, matrix CT106, media VM105, immich/nextcloud/peertube. +### Top risks (critical first) -### Phase 1 — guest OS security sweep (ordering + needrestart policy) -- **Prove the procedure on a low-stakes guest first** (utility CT112 cobalt idle, or CT102 searxng) before the never-patched worst-first targets (CT119 "never patched", 179 upgradable / 91 sec; CT108 meshai). Keep worst-first for security urgency only AFTER the procedure is proven; snapshot each never-patched guest immediately before its first-ever security upgrade. -- **Carve the stateful guests out of the bulk needrestart pass** (CT104 central, CT109 opentakserver, CT110 peertube, VM1130 recon-vm): set `NEEDRESTART_MODE=l`, stage the packages, take a logical DB dump (`pg_dump`/JetStream snapshot) with the app quiesced, then bounce each service deliberately in a controlled order. A JetStream restart mid-write loses messages on an at-most-once pipeline; a RabbitMQ/redis bounce mid-job is not a clean "blip." -- **Carve the lockout-critical daemons out too** (Authentik CT105, the headscale instance(s), Caddy, and sshd on any host you're connected over). Run `needrestart -r l` to see what *would* restart and bounce deliberately, so a libc/openssl bump can't auto-bounce the auth/tailnet/SSH daemons out from under the operator mid-pass. -- **Order control-plane guests LAST as singletons** — the home-ingress Caddy and the headscale instance(s) — each with DB backup + node-stays-joined verification, never inside a batch. Before touching headscale, confirm the out-of-band path (LAN SSH / PVE console via `pct exec`) to every still-queued target. -- **edge2 CT106 matrix / CT107 headscale have NO Tailscale client (Caddy-only ingress).** Drive their security-apt step via `pct exec` from the edge2 HOST (`root@184.174.35.153`), never over service ingress, so a needrestart bounce of a network/resolver daemon can't strand the only path in. Treat them as their own singleton steps, not part of the edge2 batch. -- **Gate `data`'s host step on the disk remediation being verified DONE** (its snapshot guardrail needs the space). - -### Phase 1 hosts vs Phase 3 (sequencing) -- **Fold the four cluster hosts' OS-security apt into their own Phase 3 reboot window** rather than applying host libc/openssl in Phase 1 and living mixed-state (new libc / old kernel, bounced smbd) across the entire Phase 2 app campaign. Guests still get Phase-1 security immediately; each host transitions in one clean window. (Open Decision #2 — this is the sequencing-correct answer.) - -### Phase 2 — app upgrades (sequence so a failure is isolatable, and never co-schedule recovery paths) -- **Hard rule: {Headscale, Authentik, Caddy, Forgejo} are NEVER in the same maintenance window** so at least one of {tailnet, SSO, ingress, git} is always a known-good recovery path. -- **Authentik first** among load-bearing apps (it fronts Forgejo, Vaultwarden, WordPress, Nextcloud, Matrix). Prove break-glass with evidence first: log in with the local akadmin in a private session; confirm each protected app has a working local-admin fallback (Forgejo `admin user`, Nextcloud `occ`, PDM/Vaultwarden local). The chain `2025.12.4 → 2025.12.6 → 2026.2.x → 2026.5.3` runs irreversible Django migrations at each hop — **`pg_dump` per hop (labelled), one hop at a time, verified-login gate between each, abort+restore on first migration error.** Add the 2025.12 hard pre-flight: audit/rename duplicate group names (migration "fails loudly" on dupes), and if local `/media` storage is used, stop and `mv ./media ./data/media` + rewrite the compose volume before starting the new version. Verify all dependent SSO logins BEFORE upgrading the dependents. -- **Caddy early** (utility ingress + media CT111), before home-service app verification, so backend sign-offs run through the new ingress; pair CT101/CT111 in one deliberate window. Map the Caddy↔Authentik path first and document a direct (Caddy-bypassing) Authentik admin URL as break-glass. -- **OpenTAKServer CT109 — decouple RabbitMQ from the OTS bump.** The OTS updater only upgrades OTS + webUI + schema; it does NOT migrate RabbitMQ, Erlang, or feature flags (verified against docs.opentakserver.io/installation/upgrading.html), so the most security-urgent EOL item (RabbitMQ 3.12.1, no 3.x backports) would silently go unfixed. Write RabbitMQ as its own sub-task: snapshot + `export_definitions` → Erlang ≥26 → 3.12 → 3.13.x → `rabbitmqctl enable_feature_flag all` (confirm all enabled) → only then 4.x; confirm no classic-queue-mirroring config remains. A naive 3.12→4.x jump refuses to boot. Gate the OTS step on RabbitMQ 4.x up + TAK clients reconnecting before accepting the bundled MediaMTX/Mumble bumps. -- **Forgejo CT103 14→15** is a branch migration on an EOL branch (14.x EOL 2026-04-30) with forward-only FK migrations. Run `forgejo doctor check --all [--fix]` to repair stopwatch/tracked_time inconsistencies BEFORE the backup+bump. Pre-step: `git clone --mirror` echo6-docs (and other critical repos) to an OFF-edge2 location and pause the `echo6-docs-autocommit` cron so the recovery docs survive an edge2/Forgejo failure and the cron doesn't loop-error into a half-migrated instance. -- **Headscale: upgrade the lower-stakes instance FIRST as a rehearsal, the front-door LAST**, each driven from a path that does NOT depend on the instance being upgraded (PVE console / direct edge2-host → CT internal IP). Before the front-door hop: disable/extend node key-expiry on all nodes so persistent tunnels don't drop on re-handshake mid-migration; `pct snapshot` + dump the headscale DB off-tailnet; keep a second already-authenticated SSH session open. Rollback restores the snapshot via console, not via tailnet. Verify headplane compatibility with 0.29. -- **Immich CT120 / Nextcloud AIO CT121 — method is correct, but size disk and protect the DB.** Prune old images + `docker image prune` before the snapshot; do NOT interrupt Immich's first-boot 2.5→2.7 DB migration; AIO self-gates intermediate mastercontainer versions (a stale AIO may need two cycles) — use AIO's borg backup as the rollback artifact (mastercontainer downgrades unsupported). The rollback unit for every DB-bearing app is a **quiesced logical dump + old binary together** — "redeploy previous image tag" is NOT valid after forward-only migrations, and a `pct snapshot` of a hot, separate-volume DB can restore torn. Disable Immich background regeneration until free space is confirmed. -- **jellyseerr is a data-compatibility decision, not a retag.** preview-OIDC (dev) → stable 3.3.0 is a schema DOWNGRADE; TypeORM stable may refuse to start against a dev-migrated DB. Confirm the correct destination artifact (jellyseerr stable vs the unified "Seerr" successor) and, if OIDC is in active use, confirm 3.3.0 parity — else STOP and report. Test the stable image against a COPY of the DB first. -- **cortex ollama 0.16→0.30:** reframe — old local blobs read fine; the real trap is no clean downgrade after pulling any new model. Snapshot the models volume (the only rollback), upgrade, verify existing models `ollama run` without re-pull. Verify the vault-tagger engine (localhost:11434) and TEI `related:` engine (localhost:8090) still work after. -- **SABnzbd 4.5→5.0:** pause/empty the queue, audit custom post-proc scripts (scripts now run on failed jobs), note `empty_postproc` removed and that downgrade needs a queue repair. -- **Nominatim 4→5 + Photon 1.1→1.2 is a separate project, not part of the patch campaign.** Re-import into a SEPARATE DB/instance; keep the `nominatim:4.5` image + data until 5.x is validated; size transient import disk (flat-nodes tens of GB) against real free space on a non-starved node — this cannot live on the 92%-full `data` pool; sequence Photon 1.2 AFTER nominatim 5.x is validated, never concurrently. - -### Phase 3 — platform & reboot windows -- **Per-node order: `apt update && apt dist-upgrade` (full-upgrade so proxmox-ve/qemu/lxc metapackages pull) → reboot → verify `pveversion` + all guests return.** Hard rule for the mixed-version window: **NO live migration and disable HA migration across the QEMU 10/11 (PVE 9.1/9.2) boundary** — running VMs keep QEMU 10 machine type until cold-started, and migration across the boundary fails. Optionally bump VM machine types + cold-restart after all 5 are on 9.2 (noted follow-up). -- **Canary by blast radius + snapshot headroom, not label.** `data` (92% full, hosts recon-vm overture/padus) must NOT be canary until its disk is remediated, and its near-full pool can fail the mandatory pre-reboot snapshot. Lowest-stakes is **media** (single arr VM + peertube, non-auth/non-DB-critical) — but note media carries freshly-majored apps; the reconciled order is: canary the genuinely lowest-blast node with headroom → then **utility LAST among the early nodes** (it carries Caddy ingress + the mesh/central stack — a deliberately-scheduled full home-ingress + mesh-coordination outage) → cloud (immich/nextcloud) → data → **toc+cortex last and alone**, all four other nodes confirmed quorate. -- **Pre-window: confirm every guest has `onboot=1`** so they actually return; take vzdump of guests to an EXTERNAL target (not the local pool, never `data`) so a guest that won't start under LXC 7 can be restored elsewhere. Read PVE 9.2 + LXC 7 release notes for unprivileged-CT/cgroupv2/apparmor breaking changes before the first node. -- **Add a post-Phase-3 re-verification gate:** after each node boots into 9.2/LXC7, re-run the Phase-2 health check for every guest that received a major app upgrade on that node (CT120 immich, CT121 nextcloud, CT110 peertube, VM105 arr, CT109 OTS) — the substrate changed under them. -- **toc+cortex window:** name and TEST a concrete alternate control point (a non-cortex host with keys + tooling to the fleet) BEFORE the window; stabilize the tailnet well before it (never in the same period as a headscale upgrade). Split cortex: snapshot VM150 from toc → NVIDIA 580.159→580.167 + DKMS + container-toolkit 1.19 as a discrete reversible step, confirm `dkms status` built and `nvidia-smi` BEFORE rebooting toc → reboot toc → confirm toc rejoins (5/5 votes) and cortex + GPU passthrough return. Keep the old driver package for reinstall. -- **edge2 is PVE 8.4, architecturally separate, single SPOF for ingress (no failover).** Do NOT fold it into the cluster windows. If the Phase 0 kernel verification says it needs a reboot, give it a dedicated window AFTER Authentik CT105 + headscale CT107 are upgraded and verified, announce SSO+tailnet downtime, drive from a non-tailnet path, and confirm all CTs auto-start. -- **pi-nas: split into two reboots, not one.** OMV 8.1→8.4 via OMV's own update path (NOT raw `apt full-upgrade`) → reboot → verify shares/SMB/NFS/omv-salt; THEN kernel 6.12→6.18 → reboot → verify boot + disk remount. Confirm physical/serial console access (no remote KVM on an RPi), back up `/boot`+`/boot/firmware` and keep the old kernel installed as fallback, export `config.xml` off-box, and confirm it's not mid-sync as a Syncthing/backup target before taking it down. Independent of the cluster windows. +1. **Headscale edge2 CT107 self-lockout** (CRITICAL) — upgrading the main tailnet over the tailnet itself loses the control path to 34 nodes. Mitigation: out-of-band only (edge2 host console / direct SSH 184.174.35.153), key-expiry extended, DB dumped off-tailnet, boot chain verified. +2. **Corosync quorum loss** (CRITICAL) — two nodes down simultaneously = quorum lost, `/etc/pve` read-only fleet-wide. Mitigation: one node at a time, `pvecm status` gated before each, HA disabled during window. +3. **Authentik SSO migration chain** (CRITICAL) — 4-hop irreversible Django migration sequence; a failed hop with no per-hop pg_dump can strand the entire SSO surface. Mitigation: pg_dump per hop, verified-login gate, break-glass proven before starting. +4. **data node snapshot-space exhaustion** (CRITICAL) — 92% full; a snapshot failure mid-upgrade leaves the guest in a half-upgraded, unrollbackable state. Mitigation: disk remediated to <85% before any snapshot on that node; do not use as Phase 3 canary until resolved. +5. **RabbitMQ EOL/won't-boot** (HIGH) — 3.12→4.x direct jump refuses to start; 3.x has no security backports. Mitigation: staged hops (3.12→3.13.x→feature-flags→4.x), decoupled from OTS updater. +6. **Forward-only DB migrations / false-rollback assumption** (HIGH) — "redeploy previous image tag" is NOT a valid rollback for authentik, forgejo, nextcloud, peertube, immich after a migration runs. Rollback = quiesced logical dump + old binary together. +7. **toc reboot drops cortex + management host** (HIGH) — cortex is the Claude Code host; losing it during diagnosis is double-jeopardy. Mitigation: alternate control point tested before window; cortex NVIDIA/DKMS verified before toc reboot; toc+cortex done last and alone. +8. **pi-nas non-boot after combined kernel+OMV change** (HIGH) — RPi has no remote KVM; a bad combined update requires on-site SD reflash. Mitigation: two separate reboots (OMV first, kernel second); retain 6.12; `/boot`+`/boot/firmware` backed up; physical access confirmed before each. --- diff --git a/vault/projects/nominatim-v5-reimport.md b/vault/projects/nominatim-v5-reimport.md new file mode 100644 index 0000000..97bdab5 --- /dev/null +++ b/vault/projects/nominatim-v5-reimport.md @@ -0,0 +1,42 @@ +--- +title: "Nominatim v5 Re-import (deferred project)" +type: project +tags: [recon, storage] +related: [] +status: planned +updated: 2026-06-19 +--- + +# Nominatim v5 Re-import (deferred project) + +Spun off from the [[fleet-patch-audit]] (2026-06-19). Explicitly out of scope for that patch pass — to be scheduled as its own maintenance window. + +## Context + +Nominatim on **recon-vm** (VM 1130, data node) is running **4.5.0** via Docker image `mediagis/nominatim:4.5` (~14 months old). Latest upstream is **5.3.2**. + +This is NOT a routine version bump. Nominatim v5 (released February 2025) changed the underlying data model. Moving from 4.x to 5.x requires deploying the `mediagis/nominatim:5.x` image and running a **full OpenStreetMap re-import** — there is no in-place upgrade path. + +## Coupled dependency: Photon + +**Photon must be upgraded from 1.1.0 → 1.2.0 in coordination.** Photon 1.2.0 was released specifically to read the Nominatim v5 data model. Upgrading only one side risks incompatibility; both must move together at cutover. + +## Constraints and risks + +- **Disk is the primary blocker.** The data node is at ~92% capacity (~73 GB free). A full OSM re-import is large and disk-heavy. The import must target external or provisioned storage — not the existing data pool. Disk and time budgets must be planned before starting. +- Re-imports are slow; schedule a maintenance window with adequate runway. +- The 4.5 deployment must remain live until validation is complete — do not destroy it before cutover. + +## Suggested steps (high level) + +1. Provision sufficient disk on or attached to recon-vm for the import scratch space and new data volume. +2. Stand up `nominatim:5.x` alongside the running 4.5 deployment (parallel, not replacement). +3. Run a fresh OSM import into the new deployment. +4. Upgrade Photon to 1.2.0 and point it at the new Nominatim v5 data. +5. Validate geocode and reverse-geocode parity between old and new stacks. +6. Cut over (update any consumers pointing at Nominatim/Photon endpoints). +7. Retire the 4.5 deployment and reclaim its disk. + +## Status + +Deferred. Not scheduled. Pick this up as a standalone maintenance window separate from routine fleet patching.