build-1 guards (the coordinator, 9 October 2026, after the 09:54 OOM): the queue reader as a unit, a memory guard every minute (one fresh sync at a time by unit; free memory under 15 GB alerts), an edge check every five minutes (the workers page, the git host, /light; a failed caddy started again), Restart=always and the OOM order on caddy, cron, atd, rsyslog and the reader (last in line) and on the three sync units (first in line); install.sh runs on the box from the mirror.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
igneum-labs 2026-10-09 09:16:21 +00:00
parent dc22df3ff4
commit 27efd57303
4 changed files with 171 additions and 0 deletions

View file

@ -0,0 +1,13 @@
#!/usr/bin/env bash
# build-1's edge check (the coordinator's order, 9 October 2026): every five minutes from igneum-edge-check.timer, read
# build.igneum.network/workers.html, git.igneum.network and build.igneum.network/light/ through the edge; an empty read (no body, or
# no HTTP answer) writes a FAULT line to /srv/queue/faults.log and /srv/queue/alerts.log; a failed caddy unit is started again and said.
set -u
Q=/srv/queue; mkdir -p "$Q"; now=$(date -u +%Y-%m-%dT%H:%M:%SZ); uk=$(TZ=Europe/London date +%H:%M); red=0
for u in https://build.igneum.network/workers.html https://git.igneum.network/ https://build.igneum.network/light/; do
r=$(curl -s -o /dev/null -w '%{http_code} %{size_download}' --max-time 10 "$u" 2>/dev/null || echo "000 0"); code=${r%% *}; size=${r##* }
ok=1; case "$u" in */light/) [ "$code" != 000 ] || ok=0 ;; *) { [ "$code" != 000 ] && [ "$size" -gt 0 ]; } || ok=0 ;; esac
[ "$ok" = 1 ] || { red=1; echo "$now ($uk UK) FAULT edge: $u read $code with $size bytes" | tee -a "$Q/faults.log" >> "$Q/alerts.log"; }
done
if ! systemctl is-active --quiet caddy; then systemctl reset-failed caddy 2>/dev/null; systemctl start caddy; echo "$now ($uk UK) FAULT edge: caddy was not active; started again ($(systemctl is-active caddy))" | tee -a "$Q/faults.log" >> "$Q/alerts.log"; fi
printf '%s %s edge=%s\n' "$now" "$uk" "$( [ $red = 0 ] && echo ok || echo red)" > "$Q/edge-check.last"

View file

@ -0,0 +1,72 @@
#!/usr/bin/env bash
# Install build-1's guards as root ON the box from the mirror's master (git archive): the queue reader, the memory guard and the edge
# check as systemd units and timers; Restart=always and the OOM order (daemons last in line, node processes first) on Caddy, cron, atd,
# rsyslog, the reader and the three sync units; Forgejo's container set to restart always. Read back with systemctl show.
# bash infra/build-server/guard/install.sh (as root on igneum-build-1)
set -eu
G=/srv/igneum.git; D=/srv/guard; mkdir -p "$D"
for f in queue-reader.sh memory-guard.sh edge-check.sh; do git --git-dir=$G show "master:infra/build-server/guard/$f" > "$D/$f"; chmod 755 "$D/$f"; done
sha=$(cat "$D"/queue-reader.sh "$D"/memory-guard.sh "$D"/edge-check.sh | sha256sum | cut -c1-16)
unit() { cat > "/etc/systemd/system/$1"; }
unit igneum-queue-reader.service <<U
[Unit]
Description=Igneum build queue reader (every 30 minutes: idle and over-memory boxes as faults; build-server lane)
After=network-online.target
[Service]
User=build
Group=build
ExecStart=/bin/bash $D/queue-reader.sh
Restart=always
RestartSec=10
OOMScoreAdjust=-1000
[Install]
WantedBy=multi-user.target
U
unit igneum-memory-guard.service <<U
[Unit]
Description=Igneum build-1 memory guard (one fresh sync at a time; free memory under 15 GB alerts)
[Service]
Type=oneshot
ExecStart=/bin/bash $D/memory-guard.sh
OOMScoreAdjust=-1000
U
unit igneum-memory-guard.timer <<U
[Unit]
Description=Igneum build-1 memory guard, every minute
[Timer]
OnBootSec=1min
OnUnitActiveSec=1min
AccuracySec=5s
[Install]
WantedBy=timers.target
U
unit igneum-edge-check.service <<U
[Unit]
Description=Igneum build-1 edge check (workers page, git host, /light; caddy started again when failed)
[Service]
Type=oneshot
ExecStart=/bin/bash $D/edge-check.sh
OOMScoreAdjust=-1000
U
unit igneum-edge-check.timer <<U
[Unit]
Description=Igneum build-1 edge check, every five minutes
[Timer]
OnBootSec=2min
OnUnitActiveSec=5min
AccuracySec=10s
[Install]
WantedBy=timers.target
U
for u in caddy cron atd rsyslog; do mkdir -p "/etc/systemd/system/$u.service.d"; printf '[Service]\nRestart=always\nRestartSec=5\nOOMScoreAdjust=-1000\n' > "/etc/systemd/system/$u.service.d/guard.conf"; done
for u in igneum-dn4-seed igneum-dn4-hand igneum-light-reader; do mkdir -p "/etc/systemd/system/$u.service.d"; printf '[Service]\nOOMScoreAdjust=500\n' > "/etc/systemd/system/$u.service.d/guard.conf"; done
docker update --restart=always forgejo >/dev/null 2>&1 && echo "forgejo container: restart=always" || echo "forgejo container: docker update failed"
systemctl daemon-reload
# the old nohup reader ends by its pid file before the unit takes over
if [ -f /srv/queue/queue-reader.pid ]; then p=$(cut -d' ' -f1 /srv/queue/queue-reader.pid); kill -0 "$p" 2>/dev/null && kill -TERM "$p" && sleep 2; fi
systemctl enable --now igneum-queue-reader.service igneum-memory-guard.timer igneum-edge-check.timer >/dev/null 2>&1
systemctl start igneum-memory-guard.service igneum-edge-check.service
for u in caddy cron atd rsyslog; do systemctl restart "$u"; done
echo "GUARD sha256 (the three scripts) $sha from master $(git --git-dir=$G rev-parse --short master)"
for u in igneum-queue-reader igneum-memory-guard.timer igneum-edge-check.timer caddy cron atd rsyslog igneum-dn4-seed igneum-dn4-hand igneum-light-reader; do printf '%-28s %s\n' "$u" "$(systemctl show -p ActiveState,Restart,OOMScoreAdjust,MainPID "$u" 2>/dev/null | tr '\n' ' ')"; done
cat /srv/queue/memory-guard.last /srv/queue/edge-check.last 2>/dev/null

View file

@ -0,0 +1,24 @@
#!/usr/bin/env bash
# build-1's memory guard (the coordinator's order after the OOM of 9 October 2026, 09:54 to 10:04 UK: a fresh devnet-4 sync with a public
# listener reached 85 GB and the kernel killed nodes, Caddy, cron and rsyslogd). Runs every minute from igneum-memory-guard.timer as root.
# 1. One fresh sync at a time: of the three sync units (igneum-dn4-seed, igneum-dn4-hand, igneum-light-reader) at most one may run
# while below within-ten of the hub feed (/srv/canary/hub1-tip.txt, the reference hub's EVM height); a second one is stopped by its unit and
# the refusal is written. The sequencer (dn4-sequence.sh in the root account's home) starts the next one at the previous one's within-ten.
# 2. Free memory under 15 GB (MemAvailable) writes an ALERT line; the Mac-side relay sends it to main.
# Lines go to /srv/queue/faults.log and /srv/queue/alerts.log (build-readable). No kill by name: units only.
set -u
Q=/srv/queue; mkdir -p "$Q"; now=$(date -u +%Y-%m-%dT%H:%M:%SZ); uk=$(TZ=Europe/London date +%H:%M)
height() { curl -s --max-time 5 -X POST -H 'content-type: application/json' --data '{"jsonrpc":"2.0","id":1,"method":"eth_blockNumber","params":[]}' "http://127.0.0.1:$1" 2>/dev/null | python3 -c 'import json,sys; r=json.load(sys.stdin).get("result"); print(int(r,16) if r else -1)' 2>/dev/null || echo -1; }
hub=$(cut -d' ' -f1 /srv/canary/hub1-tip.txt 2>/dev/null); hub=${hub:-0}
syncing=(); for pair in igneum-dn4-seed:27810 igneum-dn4-hand:26870 igneum-light-reader:26881; do u=${pair%%:*}; p=${pair##*:}
systemctl is-active --quiet "$u" || continue; h=$(height "$p"); d=$(( hub - h ))
if [ "$h" -lt 0 ] || [ "$d" -gt 10 ] || [ "$d" -lt -10 ]; then syncing+=("$u"); fi
done
if [ "${#syncing[@]}" -gt 1 ]; then
# keep the earliest-started, stop the rest by unit
keep=""; keepts=0; for u in "${syncing[@]}"; do ts=$(date -d "$(systemctl show -p ActiveEnterTimestamp --value "$u")" +%s 2>/dev/null || echo 0); [ "$keepts" = 0 ] || [ "$ts" -lt "$keepts" ] && { keep=$u; keepts=$ts; }; done
for u in "${syncing[@]}"; do [ "$u" = "$keep" ] && continue; systemctl stop "$u"; echo "$now ($uk UK) REFUSED a second fresh sync: $u stopped by its unit while $keep is below within-ten of the hub feed ($hub); the box-1 rule" | tee -a "$Q/faults.log" >> "$Q/alerts.log"; done
fi
avail=$(awk '/MemAvailable/{print int($2/1024/1024)}' /proc/meminfo); used=$(free -g | awk '/Mem:/{print $3}')
if [ "$avail" -lt 15 ]; then echo "$now ($uk UK) ALERT build-1 free memory ${avail} GB (used ${used} GB; syncing: ${syncing[*]:-none}); under the 15 GB line" | tee -a "$Q/faults.log" >> "$Q/alerts.log"; fi
printf '%s %s avail=%sGB syncing=%s\n' "$now" "$uk" "$avail" "${syncing[*]:-none}" > "$Q/memory-guard.last"

View file

@ -0,0 +1,62 @@
#!/usr/bin/env bash
# The build queue reader on igneum-build-1 (9 October 2026): every 30 minutes, read /srv/queue/build-queue.md against the merged
# /srv/workers/workers.json (every box's load and running jobs, refreshed every 20 s by merge-workers.mjs) and the pid files under
# /srv/queue/pids; write /srv/queue/status.json; append one line per idle box to /srv/queue/faults.log (an idle box: load1 under 1.0,
# no running job in workers.json, no live pid file for it under /srv/queue/pids). Pid file /srv/queue/queue-reader.pid, written first;
# stopped by `kill "$(cut -d' ' -f1 /srv/queue/queue-reader.pid)"` only.
set -u
Q=/srv/queue; mkdir -p "$Q/pids"
printf '%s %s queue-reader\n' "$$" "$(date -u +%Y-%m-%dT%H:%M:%SZ)" > "$Q/queue-reader.pid"
trap 'rm -f "$Q/queue-reader.pid"; exit 0' TERM INT
while :; do
now=$(date -u +%Y-%m-%dT%H:%M:%SZ); uk=$(TZ=Europe/London date +%H:%M)
python3 - "$Q" "$now" "$uk" <<'PY' 2>>"$Q/reader.err"
import json, os, re, sys, time
Q, now, uk = sys.argv[1], sys.argv[2], sys.argv[3]
try: W = json.load(open('/srv/workers/workers.json'))
except Exception as e: W = {'boxes': []}
owners = {}
try:
for line in open(f'{Q}/build-queue.md'):
m = re.match(r'\|\s*(build-\d)\s*\|[^|]*\|\s*([^|]+?)\s*\|', line)
if m: owners[m.group(1)] = m.group(2)
except Exception: pass
live = {}
for f in os.listdir(f'{Q}/pids') if os.path.isdir(f'{Q}/pids') else []:
m = re.match(r'(build-\d)-', f)
if not m: continue
# a pid file for build-1 is checked against this box's process table; one for another box is a claim the lane keeps fresh:
# present and touched within two hours reads live, older reads stale (the lanes' jobs run on their own boxes)
try:
pid = int(open(f'{Q}/pids/{f}').read().split()[0])
if m.group(1) == 'build-1': alive = os.path.exists(f'/proc/{pid}')
else: alive = (time.time() - os.path.getmtime(f'{Q}/pids/{f}')) < 7200
except Exception: alive = False
live.setdefault(m.group(1), []).append({'file': f, 'alive': alive})
status, faults = [], []
# main's 15-minute rule: an idle box without an owner's pid file at two consecutive reads (the reader runs every 30 minutes; a 15-minute
# mark is read from the previous status) is marked unclaimed, open to any lane; the previous status gives the earlier idle time
try: prev = {b['box']: b for b in json.load(open(f'{Q}/status.json')).get('boxes', [])}
except Exception: prev = {}
for b in W.get('boxes', []):
name = (b.get('name') or '').replace('igneum-', ''); load1 = (b.get('load') or [None])[0]
running = b.get('running') or []; down = bool(b.get('down'))
pids = live.get(name, []); live_pids = [p for p in pids if p['alive']]
idle = (not down) and (load1 is not None and load1 < 1.0) and not running and not live_pids
idle_since = (prev.get(name) or {}).get('idle_since') if idle else None
if idle and not idle_since: idle_since = now
idle_min = int((time.time() - time.mktime(time.strptime(idle_since, '%Y-%m-%dT%H:%M:%SZ')) + time.timezone) / 60) if idle_since else 0
unclaimed = idle and idle_min >= 15
row = {'box': name, 'owner': owners.get(name, 'unassigned'), 'down': down, 'load1': load1, 'running': len(running), 'queue_pid_files': [p['file'] for p in live_pids], 'idle': idle, 'idle_since': idle_since, 'idle_min': idle_min, 'unclaimed': unclaimed, 'read_at': now}
status.append(row)
mem = (b.get('mem') or {}).get('used_pct'); row['mem_pct'] = mem
if mem is not None and mem >= 90: faults.append(f'{now} ({uk} UK) FAULT memory on {name}: {mem} percent used (the 90 percent line; the box-1 rule: one fresh sync at a time, one read node at most)')
if idle: faults.append(f'{now} ({uk} UK) FAULT idle box {name}: load1 {load1}, no running job, no live queue pid file; owner {row["owner"]}; idle {idle_min} min' + ('; UNCLAIMED, open to any lane (main\'s 15-minute rule)' if unclaimed else '') + ('; RED on the board (30 minutes)' if idle_min >= 30 else ''))
if down: faults.append(f'{now} ({uk} UK) FAULT box {name} down: {b.get("reason")}')
json.dump({'read_at': now, 'uk': uk, 'boxes': status, 'faults': faults}, open(f'{Q}/status.json.tmp', 'w')); os.replace(f'{Q}/status.json.tmp', f'{Q}/status.json')
with open(f'{Q}/faults.log', 'a') as fl:
for f in faults: fl.write(f + '\n')
print(f'{now} ({uk} UK) read {len(status)} boxes, {len(faults)} faults')
PY
sleep 1800 & wait $!
done