feat(multi-site): auto-configure OpenLDAP replication on join/every run

Removes LDAP_SERVER_ID/LDAP_REPLICATION_HOSTS as vars an operator has
to hand-set and keep in sync across every site. bootstrap/
site-ldap-register.js (new) asks sso-manager-node's new
GET /api/site/ldap-peers (spoke) or GET /directory-admin/
ldap-replication-config (master) for this node's assigned ServerID +
current peer list, persists it to /config/ldap-replication.env, and
restarts sso-manager only when the computed config actually changed
(OpenLDAP's static slapd.conf is only read at process start). Runs on
every setup.sh invocation -- both master (peer list grows as spokes
join) and spoke.

CFG_LDAP_MMR_MANUAL=true skips the automatic step entirely, for a
topology outside this theta-suite cluster the script can't derive on
its own -- without this escape hatch, an operator's hand-set
LDAP_SERVER_ID/LDAP_REPLICATION_HOSTS would get silently overwritten
on the next run, since every fresh install starts as a master (the
automatic step always runs by default).

Bumps sso-manager-node to pick up the new endpoints + SiteSpoke.ldapServerId.
This commit is contained in:
2026-08-10 23:02:01 -04:00
parent 20e9c1dfbe
commit da60310834
5 changed files with 197 additions and 8 deletions
+139
View File
@@ -0,0 +1,139 @@
#!/usr/bin/env node
/*
* theta-suite site-ldap-register — runs inside the sso-manager container on
* every setup.sh run (both master and spoke) to keep OpenLDAP N-way
* multi-master replication config (docs/replication.md) in sync without an
* operator hand-maintaining LDAP_SERVER_ID/LDAP_REPLICATION_HOSTS.
*
* The master assigns each spoke a unique LDAP_SERVER_ID at join time (same
* mechanism as jump-host's WireGuard mesh index) and derives every site's
* LDAP URL from its already-known HTTPS endpoint -- see sso-manager-node's
* GET /api/site/ldap-peers (spoke-facing) and
* GET /directory-admin/ldap-replication-config (master-local).
*
* This script fetches whichever of those two applies to this node's role,
* and writes the result to /config/ldap-replication.env (KEY=VALUE, the
* same shape setup.env/spoke.env use) if it changed since last run. setup.sh
* sources that file before starting sso-manager on every invocation, and
* restarts the container when this script reports a change -- OpenLDAP's
* static slapd.conf is only read at process start, so a config change needs
* a restart to take effect; there's no live push, which is why this has to
* be re-run periodically (every setup.sh invocation) rather than working
* once at join time and never again, especially on the MASTER, whose peer
* list changes every time a new spoke joins.
*
* docker compose exec sso-manager node /bootstrap/site-ldap-register.js <selfUrl>
*
* Self-contained (Node built-ins + global fetch), same rule as
* bootstrap.js/site-join.js -- does NOT require the SSO's internal models.
*
* Output (stdout, KEY=VALUE for setup.sh): LDAP_CONFIG_CHANGED=<yes|no>,
* LDAP_SERVER_ID=<n>, LDAP_REPLICATION_HOSTS=<space-separated, may be empty>.
* Progress logs go to stderr.
*/
'use strict';
const fs = require('fs');
const SITE_CONFIG = '/config/site.json';
const LDAP_CONFIG_FILE = '/config/ldap-replication.env';
const SSO_INTERNAL = 'http://localhost:3001';
const selfUrl = process.argv[2];
function log(msg) { console.error('[site-ldap-register] ' + msg); }
function readPersisted() {
if (!fs.existsSync(LDAP_CONFIG_FILE)) return { LDAP_SERVER_ID: '', LDAP_REPLICATION_HOSTS: '' };
const out = { LDAP_SERVER_ID: '', LDAP_REPLICATION_HOSTS: '' };
for (const line of fs.readFileSync(LDAP_CONFIG_FILE, 'utf8').split('\n')) {
const m = line.match(/^([A-Z_]+)=(.*)$/);
if (m && m[1] in out) out[m[1]] = m[2];
}
return out;
}
async function fetchMasterConfig() {
const sso = require('/config/sso-secrets.js');
const adminUid = (sso.bootstrap && sso.bootstrap.adminUid) || 'admin';
const adminPass = (sso.bootstrap && sso.bootstrap.adminPass) || '';
const loginRes = await fetch(`${SSO_INTERNAL}/api/auth/login`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ uid: adminUid, password: adminPass }),
});
if (!loginRes.ok) throw new Error(`local admin login failed (${loginRes.status}): ${await loginRes.text().catch(() => '')}`);
const { token } = await loginRes.json();
if (!token) throw new Error('local admin login returned no token');
const cfgRes = await fetch(`${SSO_INTERNAL}/api/directory-admin/ldap-replication-config`, {
headers: { 'auth-token': token },
});
if (!cfgRes.ok) throw new Error(`ldap-replication-config failed (${cfgRes.status}): ${await cfgRes.text().catch(() => '')}`);
return cfgRes.json();
}
async function fetchSpokeConfig(site, selfUrl) {
const url = `${site.masterUrl.replace(/\/+$/, '')}/api/site/ldap-peers?endpoint=${encodeURIComponent(selfUrl)}`;
const res = await fetch(url, { headers: { Authorization: 'Bearer ' + site.masterJoinKey } });
const text = await res.text().catch(() => '');
let data = null;
try { data = JSON.parse(text); } catch (e) { /* not JSON */ }
if (!res.ok) {
if (res.status === 404) {
log('This site is not registered as a spoke on the master yet (join with selfUrl, or re-run site-relay-register.js). Skipping.');
return null;
}
throw new Error(`ldap-peers failed (${res.status}): ${(data && data.message) || text}`);
}
return data;
}
async function main() {
if (!fs.existsSync(SITE_CONFIG)) {
log('No /config/site.json yet. Skipping.');
console.log('LDAP_CONFIG_CHANGED=no');
return;
}
const site = JSON.parse(fs.readFileSync(SITE_CONFIG, 'utf8'));
let result;
if (site.isMaster) {
result = await fetchMasterConfig();
} else {
if (!site.masterUrl || !site.masterJoinKey) {
log('Spoke role but missing masterUrl/masterJoinKey. Skipping.');
console.log('LDAP_CONFIG_CHANGED=no');
return;
}
if (!selfUrl) throw new Error('usage: node /bootstrap/site-ldap-register.js <selfUrl> (required for a spoke)');
result = await fetchSpokeConfig(site, selfUrl);
if (!result) {
console.log('LDAP_CONFIG_CHANGED=no');
return;
}
}
const serverId = String(result.ldapServerId || '');
const hosts = (result.peers || []).map((p) => p.ldapHost).filter(Boolean).join(' ');
const before = readPersisted();
const changed = before.LDAP_SERVER_ID !== serverId || before.LDAP_REPLICATION_HOSTS !== hosts;
if (changed) {
fs.writeFileSync(LDAP_CONFIG_FILE, `LDAP_SERVER_ID=${serverId}\nLDAP_REPLICATION_HOSTS=${hosts}\n`);
log(`Replication config changed -- ServerID ${serverId}, ${(result.peers || []).length} peer(s). Wrote ${LDAP_CONFIG_FILE}.`);
} else {
log(`Replication config unchanged -- ServerID ${serverId}, ${(result.peers || []).length} peer(s).`);
}
console.log(`LDAP_CONFIG_CHANGED=${changed ? 'yes' : 'no'}`);
console.log(`LDAP_SERVER_ID=${serverId}`);
console.log(`LDAP_REPLICATION_HOSTS=${hosts}`);
}
main().catch((e) => {
console.error('[site-ldap-register] FAILED: ' + e.message);
process.exit(1);
});
+1
View File
@@ -259,6 +259,7 @@ See [`AGENT_LOCAL_DISCOVERY_SPEC.md`](./AGENT_LOCAL_DISCOVERY_SPEC.md) — split
| `setup.env` / `setup.sh` join wiring | **Shipped**`CFG_MASTER_DIRECTORY_URL` / `CFG_MASTER_DIRECTORY_JOIN_KEY`, `bootstrap/site-join.js` (theta-suite v2.2.0). Also readable from a dedicated `spoke.env` (`spoke.env.example`, layered on top of `setup.env`) for operators who want join-a-cluster config kept separate from the rest of first-run setup. | | `setup.env` / `setup.sh` join wiring | **Shipped**`CFG_MASTER_DIRECTORY_URL` / `CFG_MASTER_DIRECTORY_JOIN_KEY`, `bootstrap/site-join.js` (theta-suite v2.2.0). Also readable from a dedicated `spoke.env` (`spoke.env.example`, layered on top of `setup.env`) for operators who want join-a-cluster config kept separate from the rest of first-run setup. |
| Continuous/live replication (vs. one-time export-on-join) | **Shipped** (`sso-manager-node`) — a spoke registers its own endpoint at join time (`POST /api/site/spokes`), and every successful master catalog write fires a fire-and-forget push (`utils/site_replicate.js`) at every registered spoke, which re-pulls a fresh export. Verified end-to-end in `docker-compose.multisite-e2e.yml`. | | Continuous/live replication (vs. one-time export-on-join) | **Shipped** (`sso-manager-node`) — a spoke registers its own endpoint at join time (`POST /api/site/spokes`), and every successful master catalog write fires a fire-and-forget push (`utils/site_replicate.js`) at every registered spoke, which re-pulls a fresh export. Verified end-to-end in `docker-compose.multisite-e2e.yml`. |
| Identical-directory signing key | **Shipped**`POST /api/site/export` includes the master's agent-signing key; a spoke adopts it via `agent_keys.adopt()` on join and every resync. OpenBao secret replication *beyond* this one key is still not built. | | Identical-directory signing key | **Shipped**`POST /api/site/export` includes the master's agent-signing key; a spoke adopts it via `agent_keys.adopt()` on join and every resync. OpenBao secret replication *beyond* this one key is still not built. |
| OpenLDAP N-way multi-master replication auto-config | **Shipped** — the master auto-assigns each spoke a unique `LDAP_SERVER_ID` at registration (`SiteSpoke.ldapServerId`, same pattern as jump-host's mesh index) and derives every site's LDAP URL from its already-known HTTPS endpoint; `theta-suite`'s `bootstrap/site-ldap-register.js` applies it, re-checked on every `setup.sh` run since the peer list grows as spokes join. Verified against real running containers. Known gap: the master's own config only updates when ITS `setup.sh` is re-run, not live the moment a new spoke joins (see `docs/replication.md`). |
| Coordinated master promotion (demote the old master as one action) | **Shipped**`POST /api/site/demote` + `site-promote`'s handoff logic. Fixed two real pre-existing bugs while wiring this in: `site-promote`'s god_admin check read a `req.user.groups` field nothing ever populated (permanently 403'd for everyone), and the read-only write-gate 403'd `site-promote` itself before the handler could run. | | Coordinated master promotion (demote the old master as one action) | **Shipped**`POST /api/site/demote` + `site-promote`'s handoff logic. Fixed two real pre-existing bugs while wiring this in: `site-promote`'s god_admin check read a `req.user.groups` field nothing ever populated (permanently 403'd for everyone), and the read-only write-gate 403'd `site-promote` itself before the handler could run. |
| WireGuard gateway-to-gateway mesh (`theta-gateway`) | **Shipped**`POST /api/mesh/register`/`/join` (join-token bootstrap), `utils/wg_iface.js` (kernel WireGuard, falls back to userspace `wireguard-go`). Verified with a real two-container test: actual encrypted tunnel, real ICMP traffic across it, 0% loss. `wg_iface.removePeer()` also cleans up the kernel routes `setPeer()` added (verified live: routes present after `setPeer`, gone after `removePeer`, own local route untouched), and `DELETE /api/mesh/gateways/:id` exposes it from the mesh UI. | | WireGuard gateway-to-gateway mesh (`theta-gateway`) | **Shipped**`POST /api/mesh/register`/`/join` (join-token bootstrap), `utils/wg_iface.js` (kernel WireGuard, falls back to userspace `wireguard-go`). Verified with a real two-container test: actual encrypted tunnel, real ICMP traffic across it, 0% loss. `wg_iface.removePeer()` also cleans up the kernel routes `setPeer()` added (verified live: routes present after `setPeer`, gone after `removePeer`, own local route untouched), and `DELETE /api/mesh/gateways/:id` exposes it from the mesh UI. |
| Cross-component routing (replication over the mesh) | **Shipped**`utils/site_replicate.js` tries a registered spoke's `meshIp` first (falling back to its public `endpoint` on failure) when pushing resync pings; a spoke with no `meshIp` on file behaves exactly as before. | | Cross-component routing (replication over the mesh) | **Shipped**`utils/site_replicate.js` tries a registered spoke's `meshIp` first (falling back to its public `endpoint` on failure) when pushing resync pings; a spoke with no `meshIp` on file behaves exactly as before. |
+13 -7
View File
@@ -152,14 +152,20 @@ CFG_DOMAIN=example.com
#CFG_THETA_AGENT_FULL_CONTROL=1 #CFG_THETA_AGENT_FULL_CONTROL=1
# ── Geo-Location Scaling (N-Way Multi-Master LDAP) ─────────────────────────── # ── Geo-Location Scaling (N-Way Multi-Master LDAP) ───────────────────────────
# If deploying this stack across multiple physical sites to provide local HA # If you're joining a directory cluster (CFG_MASTER_DIRECTORY_URL/spoke.env
# for directory services, you can enable N-Way Multi-Master OpenLDAP replication. # above), N-Way Multi-Master OpenLDAP replication is configured for you
# This requires assigning a unique ID to each site and listing the LDAPS URLs # automatically -- setup.sh's bootstrap/site-ldap-register.js asks the master
# of all OTHER sites in the cluster. # for a unique LDAP_SERVER_ID and the current list of every other site's LDAP
# URL on every run (see docs/replication.md), restarting sso-manager only
# when that config actually changed. Nothing to set here for the common case.
# #
# Each site MUST have a unique LDAP_SERVER_ID (e.g. 1, 2, 3). # Have a manually-coordinated LDAP MMR topology this script can't derive on
# LDAP_REPLICATION_HOSTS is a space-separated list of the other sites' LDAP URLs. # its own (e.g. peers outside this theta-suite cluster)? Set
# Example for Site 1: # CFG_LDAP_MMR_MANUAL=true to skip the automatic step entirely and set
# LDAP_SERVER_ID/LDAP_REPLICATION_HOSTS directly -- without this, the
# automatic step runs on every deployment (every fresh install starts as a
# master) and will overwrite them.
#CFG_LDAP_MMR_MANUAL=true
#LDAP_SERVER_ID=1 #LDAP_SERVER_ID=1
#LDAP_REPLICATION_HOSTS="ldaps://sso.site2.com:636 ldaps://sso.site3.com:636" #LDAP_REPLICATION_HOSTS="ldaps://sso.site2.com:636 ldaps://sso.site3.com:636"
# ── Proxy HTTP/HTTPS Defaults ──────────────────────────────────────────────── # ── Proxy HTTP/HTTPS Defaults ────────────────────────────────────────────────
+43
View File
@@ -1060,6 +1060,15 @@ info "Starting bao-renewer (service-token renewal sidecar)..."
SSO_GIT_COMMIT="$(git -C sso-manager-node rev-parse --short HEAD 2>/dev/null || echo unknown)" SSO_GIT_COMMIT="$(git -C sso-manager-node rev-parse --short HEAD 2>/dev/null || echo unknown)"
export SSO_GIT_COMMIT export SSO_GIT_COMMIT
env_upsert SSO_GIT_COMMIT "$SSO_GIT_COMMIT" env_upsert SSO_GIT_COMMIT "$SSO_GIT_COMMIT"
# OpenLDAP multi-master replication (docs/replication.md, auto-configured --
# see step 7e below and bootstrap/site-ldap-register.js): pick up whatever
# LDAP_SERVER_ID/LDAP_REPLICATION_HOSTS a PRIOR run already computed, so a
# restart doesn't silently drop back to standalone (no LDAP_SERVER_ID env at
# all). A truly fresh install has no file yet -- that's fine, it just starts
# standalone until step 7e computes and applies real values. Skipped under
# CFG_LDAP_MMR_MANUAL=true so a manually hand-set LDAP_SERVER_ID/
# LDAP_REPLICATION_HOSTS in setup.env isn't clobbered by a stale auto file.
[[ "${CFG_LDAP_MMR_MANUAL:-false}" != "true" && -f "$CONFIG_DIR/ldap-replication.env" ]] && parse_kv_file "$CONFIG_DIR/ldap-replication.env"
info "Building + starting sso-manager (first run builds the image; this takes a while)..." info "Building + starting sso-manager (first run builds the image; this takes a while)..."
"${COMPOSE[@]}" up -d --build sso-manager "${COMPOSE[@]}" up -d --build sso-manager
@@ -1322,6 +1331,40 @@ if [[ "${CFG_SPOKE_NO_INBOUND:-false}" == "true" ]]; then
fi fi
fi fi
# ── 7e. OpenLDAP multi-master replication auto-config (every run) ────────────
# docs/replication.md: the master assigns each spoke a unique LDAP_SERVER_ID
# and derives every site's LDAP URL automatically (bootstrap/
# site-ldap-register.js) instead of an operator hand-maintaining
# LDAP_SERVER_ID/LDAP_REPLICATION_HOSTS. Runs on every invocation -- both
# master (its peer list grows as spokes join) and spoke -- and restarts
# sso-manager only when the computed config actually changed, since
# OpenLDAP's static slapd.conf is only read at process start.
#
# CFG_LDAP_MMR_MANUAL=true skips this entirely -- every fresh install starts
# as a master, so without this escape hatch an operator's own hand-set
# LDAP_SERVER_ID/LDAP_REPLICATION_HOSTS (a topology outside this theta-suite
# cluster this script can't derive) would get silently overwritten.
if [[ "${CFG_LDAP_MMR_MANUAL:-false}" == "true" ]]; then
info "CFG_LDAP_MMR_MANUAL=true — skipping automatic LDAP replication config."
LDAP_REG_OUT=""
else
LDAP_REG_OUT=$("${COMPOSE[@]}" exec -T sso-manager node /bootstrap/site-ldap-register.js "https://$CFG_SSO_HOST" 2>&1) || warn "LDAP replication config check failed — check: ${COMPOSE[*]} exec sso-manager node /bootstrap/site-ldap-register.js https://$CFG_SSO_HOST"
echo "$LDAP_REG_OUT" | sed 's/^/[setup] /'
fi
if echo "$LDAP_REG_OUT" | grep -q '^LDAP_CONFIG_CHANGED=yes'; then
info "LDAP replication config changed — restarting sso-manager to apply it..."
[[ -f "$CONFIG_DIR/ldap-replication.env" ]] && parse_kv_file "$CONFIG_DIR/ldap-replication.env"
"${COMPOSE[@]}" up -d --force-recreate sso-manager
info "Waiting for sso-manager to be healthy again..."
for i in $(seq 1 60); do
if docker exec sso-manager wget -q -O- http://localhost:3001/health >/dev/null 2>&1; then
info "sso-manager is healthy."; break
fi
if (( i == 60 )); then warn "sso-manager did not become healthy in 120s after the LDAP config restart. Check: ${COMPOSE[*]} logs sso-manager"; break; fi
sleep 2
done
fi
# ── 7c. Install theta-agent on the host ────────────────────────────────────── # ── 7c. Install theta-agent on the host ──────────────────────────────────────
# Controlled by CFG_THETA_AGENT_ENABLE (default: 1 = enabled) # Controlled by CFG_THETA_AGENT_ENABLE (default: 1 = enabled)
CFG_THETA_AGENT_ENABLE="${CFG_THETA_AGENT_ENABLE:-1}" CFG_THETA_AGENT_ENABLE="${CFG_THETA_AGENT_ENABLE:-1}"