diff --git a/engine/changelog.md b/engine/changelog.md index 85d0756..becd4b9 100644 --- a/engine/changelog.md +++ b/engine/changelog.md @@ -169,3 +169,5 @@ - mode: incremental - docs selected: 1 - processed: 1 | written: 1 | flagged: 0 | errors: 0 + +## 2026-07-15T09:00:01Z — sweep deferred (competing GPU process: 1479040 /usr/bin/python3 /usr/local/bin/whisper-ctranslate2-real /home/zvx/.cache/peertube-runner-nodejs/default/transcoding/7fd5c3f5-460b-4727-a08c-5cc870990b69 --model medium --word_timestamps True --vad_filter true --vad_min_silence_duration_ms 5000 --output_format all --output_dir /home/zvx/.cache/peertube-runner-nodejs/default/transcription/oHfKgprXtJn5hsvZXNS1pX --model medium --device cpu --compute_type int8 --word_timestamps False --vad_min_silence_duration_ms 500) diff --git a/engine/lint-report.md b/engine/lint-report.md index c0b8ce0..2594c6e 100644 --- a/engine/lint-report.md +++ b/engine/lint-report.md @@ -1,6 +1,6 @@ # Vault Lint Report -Generated: 2026-07-15T00:00:18Z | Docs scanned: 107 | Elapsed: 0.0s +Generated: 2026-07-15T06:00:18Z | Docs scanned: 107 | Elapsed: 0.0s ## Summary diff --git a/vault/.obsidian/workspace.json b/vault/.obsidian/workspace.json index b1804f2..11dc4d2 100644 --- a/vault/.obsidian/workspace.json +++ b/vault/.obsidian/workspace.json @@ -199,19 +199,19 @@ }, "active": "17bd4a6166f789d0", "lastOpenFiles": [ + "runbooks/add-peertube-channel.md.tmp.767109.d14b6e68779c", + "runbooks/add-peertube-channel.md.tmp.767109.366b5b6ebfe2", + "runbooks/peertube-remote-runner.md.tmp.767109.9cf753daf1b2", + "runbooks/peertube-remote-runner.md.tmp.767109.f739384d8248", + "runbooks/peertube-remote-runner.md.tmp.767109.76833270c7a0", + "docs/hardware/environment.md.tmp.767109.4e9ed4494a50", + "docs/hardware/environment.md.tmp.767109.3462012af99e", + "docs/hardware/environment.md.tmp.767109.55addc2a6c2b", + "docs/hardware/environment.md.tmp.767109.5759a0dcf2a1", "runbooks/conduit-operations.md.tmp.1002651.4b545d0c7965", "runbooks/conduit-operations.md.tmp.1002651.015f0192a793", - "docs/hardware/ip-allocation.md.tmp.767109.ecc1c9977e03", - "docs/hardware/ip-allocation.md.tmp.767109.c1f2a642bed7", - "docs/hardware/ip-allocation.md.tmp.767109.d6ca8e1e1a1c", - "runbooks/conduit-operations.md.tmp.2217597.60ef73d82d1c", - "runbooks/conduit-operations.md.tmp.2217597.cf9b78291c98", - "docs/software/central.md.tmp.2217597.90c56d780840", - "docs/software/central.md.tmp.2217597.5af41a8bdfd0", "runbooks/conduit-operations.md", - "runbooks/conduit-operations.md.tmp.2217597.e825be725a27", "docs/software/conduit.md", - "docs/software/conduit.md.tmp.2217597.5f47d4e96ae8", "archive/projects/vaultwarden-plan.md", "archive/projects/meshai-native-fire-severity-audit-cc-handoff.md", "projects/meshai-native-fire-severity-audit-cc-handoff.md", diff --git a/vault/docs/hardware/environment.md b/vault/docs/hardware/environment.md index 3e5fc89..a9494f4 100644 --- a/vault/docs/hardware/environment.md +++ b/vault/docs/hardware/environment.md @@ -10,7 +10,7 @@ related: - [[proxmox-create-ubuntu-vm]] - [[headscale-onboard-node]] - [[ct-runbook]] -updated: 2026-07-13 +updated: 2026-07-15 --- # Echo6 Environment Reference @@ -35,9 +35,21 @@ Five nodes running Proxmox VE: | media | 2x Intel SSDPEKNU512GZH 512GB (NVMe) | — | | toc | 512GB NVMe | — | +### Node Hardware Identifiers + +| Node | Make/Model | Serial/Service Tag | +|------|-----------|---------------------| +| data | Lenovo ThinkCentre M75q Gen 2 (11JN002RUS) | MZ010LPV | +| utility | Lenovo ThinkCentre M75q Gen 2 (11JN002RUS) | MJ0LZNYT | +| cloud | Lenovo ThinkCentre M70q Gen 3 (11T3000RUS) | MJ0LQCGJ | +| media | Dell OptiPlex Micro 7020 | 58SM6X3 (BIOS 1.20.0, 2025-09-04) | +| toc | HP Z4 G4 Workstation | MXL2383MVK | + +**No BMC/IPMI/iDRAC on any node** (verified `dmidecode -t 38` empty on all five, 2026-07-14) — there is no remote power-cycle path for any Proxmox host. Recovery from a hung or powered-off node requires physical access. + ### Network Notes -- **media NIC:** Original Intel e1000e NIC crashes under sustained NFS load — replaced with USB Realtek RTL8153 GbE adapter on vmbr0 +- **media NIC:** Original Intel e1000e NIC (`nic0`, MAC `e8:cf:83:20:8b:cb`) crashes under sustained NFS load — present but DOWN/unused. Sole uplink is a USB Realtek RTL8153 GbE dongle (MAC `0c:37:96:0e:e8:53`) on vmbr0. - **Tailscale [[dns]] bootstrap:** All LXC containers with Tailscale have a systemd drop-in (`/etc/systemd/system/tailscaled.service.d/dns-bootstrap.conf`) that ensures fallback [[dns]] exists before tailscaled starts, preventing chicken-and-egg DNS resolution failures on reboot ### TOC Node Details @@ -120,8 +132,8 @@ Five nodes running Proxmox VE: | [[archivist]] | utility (CT 118) | 192.168.1.118 | — | [[archivist]] knowledge pipeline | | [[argus]] | utility (CT 103) | 192.168.1.103 | 100.64.0.25 | [[argus]] - OSINT intelligence gathering platform | | [[central]] | utility (CT 104) | 192.168.1.104 | 100.64.0.12 | Data-hub spine (central.echo6.mesh) — ~25 adapters, NATS/JetStream, TimescaleDB/PostGIS — see [[central]] | -| peertube | media (CT 110) | 192.168.1.170 | 100.64.0.23 | PeerTube video streaming | -| mcc | media (CT 111) | 192.168.1.111 | — | pymc console web app (Caddy + Postfix, /api+/auth+/ws → aida-nebra :8000) | +| peertube | media (CT 110) | 192.168.1.170 | 100.64.0.17 | PeerTube video streaming — Tailscale identity is `peertube-4hve9pdr` (collision-suffixed) since ~2026-06-22; `.23` is stale/gone | +| mcc | media (CT 111) | 192.168.1.111 | 100.64.0.19 | pymc console web app (Caddy + Postfix, /api+/auth+/ws → aida-nebra :8000) | | mailcow | edge1 (CT 101) | 10.10.10.2 | — | Mailcow email server (privileged LXC, mail-only host; reached via host DNAT + Caddy) | | pdm | edge2 (CT 100) | 10.10.10.10 | 100.64.0.28 | Proxmox Datacenter Manager | | wordpress | edge2 (CT 101) | 10.10.10.11 | 100.64.0.31 | WordPress for intermountainmesh.com | @@ -171,7 +183,7 @@ Current registered nodes (25 total): | arr | 100.64.0.18 | VM | | pi-nas | 100.64.0.21 | Pi | | mesh-bridge | 100.64.0.22 | LXC | -| peertube | 100.64.0.23 | LXC | +| peertube-4hve9pdr | 100.64.0.17 | LXC (collision-suffixed identity since ~2026-06-22; formerly `peertube` at 100.64.0.23, now stale/gone) | | recon | 100.64.0.24 | VM | | argus | 100.64.0.25 | LXC | | [[central]] | 100.64.0.12 | LXC (utility CT 104 — central.echo6.mesh) | diff --git a/vault/runbooks/add-peertube-channel.md b/vault/runbooks/add-peertube-channel.md index e235209..2e970c5 100644 --- a/vault/runbooks/add-peertube-channel.md +++ b/vault/runbooks/add-peertube-channel.md @@ -10,7 +10,7 @@ related: - [[recon-service-integration]] - [[proxmox-onboard-node]] - [[ct-runbook]] -updated: 2026-07-13 +updated: 2026-07-15 --- # Add PeerTube Channel @@ -181,7 +181,23 @@ If `tee` race condition empties the file: | channel-map.json empty (0 bytes) | tee race condition | Always write to temp file first, then tee. Restore from backup or edge1 | | sudo: password required | Sudoers not set up | Create `/etc/sudoers.d/recon-mgmt` via `pct exec 110` from root@192.168.1.243 | | PeerTube "actor name already exists" | Channel exists in PeerTube but not in JSON | Add entry to JSON manually with correct `peertube_channel_id` | +| pt-importer 100% upload failures after a reboot | Boot race (started before `peertube.service` ready) + permission drift on `/opt/bulk-import/transcoded//` dirs | See "Recovery: pt-importer stalled after unclean reboot" below | + +## Recovery: pt-importer stalled after unclean reboot + +After media's unclean reboot (2026-07-14), `pt-importer` on CT 110 failed 100% of uploads. Two causes, both hit at once: + +1. **Boot race:** `pt-importer` started before `peertube.service` was ready and crash-looped. Fixed 2026-07-15 by adding `After=peertube.service` / `Wants=peertube.service` to the `pt-importer` systemd unit. +2. **Permission drift + unguarded rename:** the unclean reboot left ~26 `/opt/bulk-import/transcoded//` dirs with drifted ownership/perms (`2755`, no group-write) instead of the normal `peertube:peertube 2775`. `importer.py`'s directory rename on the failure path is **not** wrapped in try/except, so a single `PermissionError` (or a truncated/corrupt "poison-pill" source file) crashes the whole process. The service then restarts from the top of the queue and hits the same file forever, blocking all other imports behind it. + +**Recovery:** +```bash +ssh zvx@192.168.1.170 "sudo chmod g+w /opt/bulk-import/transcoded/*/" +ssh zvx@192.168.1.170 "sudo systemctl restart pt-importer" +``` + +**Known latent bug:** the unguarded rename in `importer.py` means any future perm-drift or corrupt file will re-trigger the same total stall. A try/except around that rename (skip-and-continue instead of crash-and-restart-from-top) would fix this durably — not yet done. --- -*Last updated: 2026-02-18 — Initial creation* +*Last updated: 2026-07-15 — Added pt-importer boot-race + permission-drift recovery (2026-07-14 media reboot incident)* diff --git a/vault/runbooks/peertube-remote-runner.md b/vault/runbooks/peertube-remote-runner.md index ead2ea9..771e138 100644 --- a/vault/runbooks/peertube-remote-runner.md +++ b/vault/runbooks/peertube-remote-runner.md @@ -10,7 +10,7 @@ related: - [[nordvpn-lxc]] - [[proxmox-onboard-node]] - [[recon-service-integration]] -updated: 2026-07-13 +updated: 2026-07-15 --- # PeerTube Remote Runner — GPU Transcoding @@ -42,7 +42,7 @@ Prompt the user for all of these before executing: RUNNER_HOST= # SSH alias or IP for the runner machine (e.g., cortex) RUNNER_NAME= # Human-readable runner name (e.g., "cortex-nvenc") RUNNER_USER= # User to run the service as (e.g., "zvx") -PT_URL= # PeerTube instance URL reachable from runner (e.g., "http://100.64.0.23:9000") +PT_URL= # PeerTube instance URL reachable from runner (e.g., "http://100.64.0.17:9000") PT_HOST_HEADER= # PeerTube's public hostname for Host header (e.g., "stream.echo6.co") PT_ADMIN_USER= # PeerTube admin username (e.g., "root") PT_ADMIN_PASS= # PeerTube admin password @@ -426,6 +426,12 @@ Registration is tied to the PeerTube instance. After a rebuild: 3. Re-register: `peertube-runner register --url $PT_URL --registration-token --runner-name $RUNNER_NAME` 4. Restart: `sudo systemctl restart peertube-runner` +### Runner target must track CT110's current tailnet IP + +The registration URL (`registeredInstances` in `~/.config/peertube-runner-nodejs/default/config.toml` on cortex) is a static IP, not a hostname — it does **not** follow CT110 if its Tailscale identity drifts. When CT110's identity drifted (`peertube` → `peertube-4hve9pdr`, 100.64.0.23 → 100.64.0.17, see [[environment]]), the runner kept dialing the dead `.23` and ALL transcription/HLS/caption jobs silently stalled (~7,900 backlogged, zero throughput for 36h+ — unnoticed because downloads kept flowing independently). Fixed 2026-07-15 by pointing the config's url at `http://100.64.0.17:9000`. The runner was also found `disabled` (it only survived because cortex hadn't rebooted) — now `enabled`. + +**Symptom to watch for:** `Cannot connect to http://:9000/runners` timeouts in `journalctl -u peertube-runner`. + --- ## Quick Reference: Current Runners