From 05f723556e959cc8ecbddd8e74c65179a9b40f93 Mon Sep 17 00:00:00 2001 From: Sam Date: Mon, 20 Jul 2026 16:02:36 -0700 Subject: [PATCH] docs+tooling: land the deploy bundle docs, portproxy script, checkout pinning deploy.sh referenced DEPLOYMENT-TYPES.md and router-bmp-config.md, which were never committed (they lived only in a chat session's outputs). Land them under docs/ along with PORTABILITY-FINDINGS.md covering all 12 greenfield findings, including the two from today's deploy (chmod-777-vs-psql, telegraf docker API) and the open router-side gap (cml tooling still targets HOST_IP:5000). - scripts/wsl-portproxy.ps1: idempotent elevated-PS helper for the Windows portproxy + firewall rules; auto-detects the current WSL IP. Replaces the copy-paste netsh block as the primary path (raw commands kept as fallback). - deploy.sh: plan now prints the branch @ commit (+dirty marker) so every deploy records exactly what it ran from; doc references point at docs/. - README: greenfield quickstart with pinned-checkout guidance; warning on the manual chmod path that breaks existing Postgres trees (finding 10). Co-Authored-By: Claude Fable 5 --- README.md | 30 ++++++++++- deploy.sh | 21 +++++--- docs/DEPLOYMENT-TYPES.md | 67 ++++++++++++++++++++++++ docs/PORTABILITY-FINDINGS.md | 99 ++++++++++++++++++++++++++++++++++++ docs/router-bmp-config.md | 98 +++++++++++++++++++++++++++++++++++ scripts/wsl-portproxy.ps1 | 67 ++++++++++++++++++++++++ 6 files changed, 375 insertions(+), 7 deletions(-) create mode 100644 docs/DEPLOYMENT-TYPES.md create mode 100644 docs/PORTABILITY-FINDINGS.md create mode 100644 docs/router-bmp-config.md create mode 100644 scripts/wsl-portproxy.ps1 diff --git a/README.md b/README.md index 6889f31..45cbab1 100644 --- a/README.md +++ b/README.md @@ -33,9 +33,32 @@ Each docker file contains a readme file, see below: * [PSQL Consumer](psql-app/README.md) +## Greenfield deploy (recommended): deploy.sh + +Deploying onto a new host? Use `deploy.sh` — it reconciles the host-specific +`.env` values (HOST_IP, router-facing IP/port, auth mode) *before* running +setup.sh, guards against the known portability traps +([docs/PORTABILITY-FINDINGS.md](docs/PORTABILITY-FINDINGS.md)), and brings the +stack up in stages: + +``` +git clone && cd obmp-docker +git checkout # pin the deploy: note the branch AND commit +git log -1 --oneline # record what you deployed +./deploy.sh # interactive; or --wsl/--prod --scope ... --yes +``` + +A greenfield deploy is only reproducible if the checkout is pinned — "clone +and run" silently depends on whatever branch was checked out. Record the +branch + commit with the deployment. + +Deployment types (`--scope full-stack|remote|central-store`) are described in +[docs/DEPLOYMENT-TYPES.md](docs/DEPLOYMENT-TYPES.md). Router-side BMP config: +[docs/router-bmp-config.md](docs/router-bmp-config.md). + ## Using Docker Compose to run everything -> **Quick start (recommended):** copy `.env.example` to `.env`, fill it in, and +> **Quick start:** copy `.env.example` to `.env`, fill it in, and > run `./setup.sh` — it creates the data directories, syncs Grafana > provisioning, and generates Authelia secrets. Then: > ``` @@ -74,6 +97,11 @@ mkdir -p ${OBMP_DATA_ROOT}/grafana/dashboards sudo chmod -R 7777 $OBMP_DATA_ROOT ``` +> **WARNING:** on a host with an *existing* Postgres data tree, a recursive +> chmod makes `psql_server.key` group/world-accessible and Postgres will +> refuse to start. Skip `postgres/` (setup.sh does this for you — see +> [docs/PORTABILITY-FINDINGS.md](docs/PORTABILITY-FINDINGS.md) finding 10). + > In order to init the DB tables, you must create the file ```${OBMP_DATA_ROOT}/config/init_db```. This should > only be done once or whenever you want to completely wipe out the DB and start over. diff --git a/deploy.sh b/deploy.sh index 4c4baf6..98fc1a7 100755 --- a/deploy.sh +++ b/deploy.sh @@ -314,7 +314,7 @@ fi case "$HOSTTYPE" in wsl|prod) ;; *) die "host type must be 'wsl' or 'prod' (got '$HOSTTYPE')";; esac # --- 2. deployment TYPE (described menu, resolved before auth/endpoints) ----- -# Three types from DEPLOYMENT-TYPES.md. Old scope names still accepted as +# Three types from docs/DEPLOYMENT-TYPES.md. Old scope names still accepted as # aliases via --scope so existing invocations don't break. # full-stack = collocated ingest + store (the old 'feeders'/'full') # remote = collector/forwarder only (the old 'standalone') @@ -527,8 +527,13 @@ type_resources() { esac } echo; hr; log "Deployment plan"; hr +# Record what is being deployed - a greenfield deploy is only reproducible if +# the checkout is pinned (branch + commit go in the plan and the terminal log). +git_desc="$(git -C "$REPO_DIR" rev-parse --abbrev-ref HEAD 2>/dev/null || true)" +[ -n "$git_desc" ] && git_desc="$git_desc @ $(git -C "$REPO_DIR" rev-parse --short HEAD 2>/dev/null)$(git -C "$REPO_DIR" diff --quiet 2>/dev/null || echo ' (dirty)')" cat <=250 GB; + lab-only 4 vCPU / 16 GB / SSD. + +## 2. remote + +Collector + local Kafka only. Forwards parsed/raw data to a **central store's +Kafka** (`--central-kafka HOST:PORT`, written to `KAFKA_FQDN`). No local +Postgres/psql-app/Grafana. + +- **Use when:** scaling ingest out — run these near the routers so each node's + Kafka spool absorbs only its own routers' RIB dumps instead of every table + hitting one host at once. +- **Resources:** light — 2-4 vCPU / 4-8 GB / 20-40 GB fast disk (the Kafka + spool must buffer a full connect/flap burst from its local routers). + +## 3. central-store + +The store half: Kafka + Postgres + Grafana + feeders, **no local collector**. +Ingest arrives from remote collectors over Kafka. Run one of these behind N +`remote` nodes. + +- **Use when:** you've split ingest with remote collectors. +- **Resources:** same as the full-stack store tier — 16 vCPU / 48-64 GB / + NVMe >=250 GB (it carries all remotes' data). + +## Sizing: dimension for BMP burst, not steady state + +BMP is event-driven and bursty. Steady state is trivial; the sizing events are +(1) the initial RIB dump on PEER_UP, (2) all routers reconnecting at once +(collector restart / RR failover), (3) reconvergence churn. + +The multiplier that actually bites is **full table x monitored sessions**. A +full v4+v6 table is ~1.18M paths; in an RR topology the same table is +reflected to every client, so BMP ingests it once **per monitored session**. +Monitoring the RR-clients group means absorbing full-table x (RRs x clients) +simultaneously — that is what saturates a single host. The wall is Postgres +write latency + write amplification backpressuring psql-app -> Kafka -> the +BMP TCP sessions; the collector process itself is light and is *not* the +sizing driver. + +Mitigations: NVMe with real IOPS headroom (>=5000); BMP-monitor pre-policy on +the RRs only (not every client session); stagger connects +(`initial-refresh delay/spread` on the routers); raise `PSQL_MEM_LIMIT` / +Kafka memory; or split ingest with `remote` collectors. + +Storage scales with prefixes: ~1 GB per peer with a full internet table +(+~50 MB/day of timeseries); internal peers far less. + +> All figures are engineering estimates to calibrate against your own +> watermarking, not measured specs. Replace them with real numbers once a +> prod-realistic node has been load-tested (see docs/production-sizing.md). diff --git a/docs/PORTABILITY-FINDINGS.md b/docs/PORTABILITY-FINDINGS.md new file mode 100644 index 0000000..d5a6729 --- /dev/null +++ b/docs/PORTABILITY-FINDINGS.md @@ -0,0 +1,99 @@ +# Portability findings — what breaks on a greenfield deploy + +Every artifact hit while standing this stack up on a fresh host (WSL2 +`NWE-LT02`, Ubuntu 24.04, native docker-ce 29.x — 2026-07), ordered roughly as +encountered. Each one is something a clean deploy has to survive; most are now +handled automatically by `deploy.sh`/`setup.sh`. Kept as a checklist for the +next new host and as the rationale for the deploy tooling. + +## 1. Bind-mounted volumes need pre-existing source dirs +`docker-compose.yml` binds `${OBMP_DATA_ROOT}/postgres/{data,ts}` with +`type:none,o:bind`. Docker will not create these; if absent, Postgres fails +with `no such file or directory`. `setup.sh` creates them — the error only +appears when setup.sh wasn't run on the target. deploy.sh double-checks the +dirs exist after setup.sh. + +## 2. Stale HOST_IP survives a copied .env +A `.env` copied from another host carries that host's `HOST_IP`; setup.sh +validation only checks non-empty/not-"changeme", so the stale IP passes and +gets rendered into `gobgpd.conf` and the Grafana root URL. **This is the core +reason deploy.sh exists** — it reconciles HOST_IP (and warns when the .env +value isn't a local address) *before* setup.sh renders anything. + +## 3. A populated .env is not a deployed host +The same copied .env had late-stage artifacts (Authelia secrets) while +early-stage host state (Postgres dirs) was missing — proof that setup.sh +completed on a *different* host. Run setup.sh per target, always. + +## 4. WSL: HOST_IP is the NAT address; routers can't reach it +`hostname -I` in WSL returns the VM's NAT address — it changes on +`wsl --shutdown` and physical routers cannot reach it. Routers must target the +**Windows LAN IP**, forwarded into WSL via `netsh portproxy` +(`scripts/wsl-portproxy.ps1`). deploy.sh separates "router-facing IP:port" +from HOST_IP and auto-detects the Windows LAN IP via powershell.exe interop. + +## 5. Cold-start contention masquerades as component bugs +Bringing the whole stack up at once crash-looped Kafka on a preflight +"writable" check that was actually resource contention, not permissions. +deploy.sh stages the bring-up: infra (zookeeper/kafka) -> psql -> core -> +feeders. + +## 6. Size for BMP burst, not steady state +Generic docs say the collector is light — true, and misleading: the store +saturates under burst (full table x monitored sessions, Postgres write +amplification). See docs/DEPLOYMENT-TYPES.md for the full model. + +## 7. Ports 5000 and 3000 are crowded +The collector's 5000 and Grafana's 3000 are commonly already occupied. +deploy.sh runs a port-collision preflight before touching anything. The +router-facing BMP port default moved 5000 -> **1790** (the IANA BMP port); +the collector still listens on 5000 inside the container. + +## 8. Docker Desktop breaks host bind mounts on WSL (the big one) +Symptom: Postgres mount fails `no such file or directory` on a path `ls` +shows exists. Cause: Docker Desktop runs the daemon in its own WSL distro +(`docker-desktop` VM), so host bind mounts resolve against *that* filesystem, +not the distro you ran compose from. No mkdir fixes it. **Fix: native +docker-in-WSL** — disable Desktop's WSL integration, install docker-ce, +`systemctl enable --now docker`, and confirm +`docker info | grep 'Operating System'` does not say Docker Desktop. +deploy.sh detects Docker Desktop and warns/refuses. + +### 8b. Desktop residue splits the mount namespace +After installing native docker, the CLI couldn't reach `/run/docker.sock` +even though dockerd held it — a stale mount-namespace split left by Desktop. +`wsl --shutdown` and reopening the distro cleared it. + +## 9. Grafana customizations must live in the repo, not grafana.db +Dashboards hand-built in the UI exist only in that host's `grafana.db` and +vanish on a greenfield deploy. All dashboards are now committed as JSON under +`obmp-grafana/dashboards/` and setup.sh syncs both provisioning YAML *and* +dashboard JSONs to the data root. If you edit a dashboard in the UI, export +and commit the JSON. + +## 10. Blanket `chmod -R 777` on the data root breaks Postgres re-deploys +setup.sh's lab-permissive chmod over an *existing* data tree made +`psql_server.key` world-accessible; Postgres refuses to boot +(`FATAL: private key file ... has group or world access`) and the whole core +cascades down. Fresh installs never hit it (the key is generated after the +chmod) — it is strictly a re-deploy bug. setup.sh now skips `postgres/` +(the image entrypoint owns those perms) and re-tightens the key/cert to 0600 +if a previous run already clobbered them. + +## 11. docker-ce 29 rejects Docker API < 1.40 — old pinned clients break +telegraf 1.28's `[[inputs.docker]]` hardcodes Docker API version 1.24, which +docker-ce >= 29 refuses (`client version 1.24 is too old`). Every +per-container stack metric silently never reached InfluxDB; the stack +dashboards sat empty while `disk`/`postgresql_*` inputs worked fine. +`DOCKER_API_VERSION` env is ignored (version pinned in code) — the fix is +telegraf >= 1.30, which negotiates the API version (telegraf/Dockerfile now +pins 1.33). Watch for the same failure in any other tooling that talks to the +daemon with an old vendored client. + +## 12. OPEN: router-side automation still targets the old lab's values +`cml/proxmox_bmp_config.py` sends routers at `HOST_IP` port `5000`, and +`cml/xrd-node-definition.yaml` bakes in `10.40.40.202:5000` — both correct on +the old native-Linux dev lab, wrong on a WSL host (finding 4: routers must +target the router-facing address, and the port default is now 1790). Fold +`ROUTER_FACING_IP`/`ROUTER_FACING_PORT` from `.env` into that tooling when +the 5000-vs-1790 decision lands. See docs/router-bmp-config.md. diff --git a/docs/router-bmp-config.md b/docs/router-bmp-config.md new file mode 100644 index 0000000..2a68e60 --- /dev/null +++ b/docs/router-bmp-config.md @@ -0,0 +1,98 @@ +# Router BMP configuration (IOS-XR) + +How to point routers at the OpenBMP collector, and which address/port they +must target depending on where the stack runs. Companion automation: +`cml/proxmox_bmp_config.py` (applies the block below over SSH to the Proxmox +lab routers) and `cml/xrd-node-definition.yaml` (bakes it into XRd images). + +## What address do routers target? + +**Never `HOST_IP` on a WSL deployment.** `HOST_IP` is the WSL VM's NAT +address — unreachable from physical routers and changed by `wsl --shutdown`. +deploy.sh records the correct target in `.env` as +`ROUTER_FACING_IP` / `ROUTER_FACING_PORT`: + +| Stack host | Routers target | Path to the collector | +|---|---|---| +| Native Linux | `ROUTER_FACING_IP` (= host IP) : `ROUTER_FACING_PORT` | compose port map -> collector :5000 | +| WSL2 | Windows LAN IP : `ROUTER_FACING_PORT` | `netsh portproxy` (Windows) -> WSL `HOST_IP` -> collector :5000 | + +On WSL the Windows-side forwarding must exist first — run +`scripts/wsl-portproxy.ps1` in an **elevated** PowerShell, and re-run it after +every WSL restart (the WSL address changes). + +The router-facing port defaults to **1790** (the IANA-registered BMP port); +the collector always listens on **5000 inside the container**. The lab +routers were historically configured for 5000 — until they are reconfigured, +deploy with `--router-port 5000` (see the open decision in +docs/PORTABILITY-FINDINGS.md, finding 12). + +## `bmp server` block + +Flat formal form (submode-free, safe to paste line-by-line at `(config)#` — +the form `proxmox_bmp_config.py` applies): + +``` +bmp server 1 host port +bmp server 1 description OpenBMP-Collector +bmp server 1 update-source MgmtEth0/RP0/CPU0/0 +bmp server 1 initial-delay 60 +bmp server 1 stats-reporting-period 300 +bmp server 1 initial-refresh delay 60 spread 30 +``` + +The `initial-refresh delay/spread` staggering matters at scale: it spreads +the full-RIB dumps out when many routers (re)connect at once, which is the +burst that sizes the whole store tier (docs/DEPLOYMENT-TYPES.md). + +## Activating monitoring on BGP sessions + +BMP only reports sessions that are `bmp-activate`d: + +``` +router bgp + neighbor-group RR-CLIENTS + bmp-activate server 1 + ! +! +``` + +**Monitor on the RRs, pre-policy, not on every client session.** In an RR +topology the same full table is reflected to every client; activating BMP on +all reflected sessions makes the collector ingest full-table x sessions +copies simultaneously. Activating on the RR side only is the single biggest +load reduction available. + +## Verification + +Router side: + +``` +show bgp bmp server 1 ! state should be ESTAB, uptime climbing +show bgp bmp summary +``` + +Collector side: + +``` +docker logs obmp-collector --tail 20 # look for ROUTER_INIT / PEER_UP +docker exec -i obmp-psql psql -U openbmp -d openbmp \ + -c "SELECT name, ip_address, bgp_id, isconnected FROM routers ORDER BY name;" +``` + +Grafana: the OBMP-Operations / Inventory and Peer Detail dashboards populate +within a minute of the first PEER_UP; RIB contents follow as the initial dump +is consumed (large tables take longer — watch the OBMP-Telemetry / Kafka Lag +dashboard drain). + +## Troubleshooting + +- **Session cycles ESTAB -> idle:** the collector accepted TCP but the + pipeline is backpressured (Postgres write latency) — check the stack + dashboards, not the router. +- **No TCP at all from a WSL deployment:** portproxy rule missing/stale + (re-run `scripts/wsl-portproxy.ps1`), or Windows firewall lacks the inbound + allow for the router-facing port. +- **Router config rejected:** IOS-XR BMP config is not in the NETCONF YANG + schema on the lab release — apply via SSH CLI (which is why + `proxmox_bmp_config.py` drives the CLI). diff --git a/scripts/wsl-portproxy.ps1 b/scripts/wsl-portproxy.ps1 new file mode 100644 index 0000000..5bf3ba5 --- /dev/null +++ b/scripts/wsl-portproxy.ps1 @@ -0,0 +1,67 @@ +<# +.SYNOPSIS + Forward router-facing OpenBMP ports from the Windows host into WSL. + +.DESCRIPTION + Physical routers cannot reach the WSL VM's NAT address, and that address + changes on every 'wsl --shutdown' (portability finding 4). This script + (re)creates the netsh portproxy rules that forward the router-facing BMP + port and Kafka from the Windows LAN IP into the current WSL address, plus + the matching inbound firewall rules. Idempotent: existing rules for the + same listen ports are replaced, so RE-RUN IT AFTER EVERY WSL RESTART. + + Must run in an ELEVATED PowerShell. + +.PARAMETER RouterPort + Router-facing BMP listen port on Windows (default 1790, the IANA BMP port). + Forwards to the collector's in-container port 5000. + +.PARAMETER WslIp + WSL address to forward to. Default: auto-detected via 'wsl hostname -I'. + +.EXAMPLE + powershell -ExecutionPolicy Bypass -File scripts\wsl-portproxy.ps1 + powershell -ExecutionPolicy Bypass -File scripts\wsl-portproxy.ps1 -RouterPort 5000 +#> +param( + [int]$RouterPort = 1790, + [int]$CollectorPort = 5000, + [int]$KafkaPort = 9092, + [string]$WslIp = "" +) + +$ErrorActionPreference = "Stop" + +$id = [System.Security.Principal.WindowsIdentity]::GetCurrent() +if (-not (New-Object System.Security.Principal.WindowsPrincipal($id)).IsInRole( + [System.Security.Principal.WindowsBuiltInRole]::Administrator)) { + Write-Error "Run this from an ELEVATED PowerShell (netsh portproxy needs admin)." + exit 1 +} + +if (-not $WslIp) { + $WslIp = (wsl hostname -I).Trim().Split(" ")[0] + if (-not $WslIp) { Write-Error "Could not auto-detect the WSL IP; pass -WslIp."; exit 1 } +} +Write-Host "WSL address: $WslIp" + +# listenport => connectport ($RouterPort maps DOWN to the collector's 5000) +$maps = @{ $RouterPort = $CollectorPort; $KafkaPort = $KafkaPort } +foreach ($listen in $maps.Keys) { + netsh interface portproxy delete v4tov4 listenport=$listen listenaddress=0.0.0.0 2>$null | Out-Null + netsh interface portproxy add v4tov4 listenport=$listen listenaddress=0.0.0.0 ` + connectport=$($maps[$listen]) connectaddress=$WslIp | Out-Null + Write-Host "portproxy 0.0.0.0:$listen -> ${WslIp}:$($maps[$listen])" + + $ruleName = "OpenBMP $listen" + if (-not (Get-NetFirewallRule -DisplayName $ruleName -ErrorAction SilentlyContinue)) { + New-NetFirewallRule -DisplayName $ruleName -Direction Inbound ` + -LocalPort $listen -Protocol TCP -Action Allow | Out-Null + Write-Host "firewall rule added: $ruleName" + } +} + +Write-Host "" +Write-Host "Active portproxy rules:" +netsh interface portproxy show v4tov4 +Write-Host "Routers target the Windows LAN IP on port $RouterPort (see docs/router-bmp-config.md)."