igneum/infra/build-server/capacity/run.sh
igneum-labs b6e5dffa32 Build boxes: one host file per box and a route by class (suites and benches to build-2, proving to build-3, the rest to build-1); the no-mining rule in a box README and as a guard in the capacity layer
Main's order of 7 October 2026 (igneum-build-2, AX162-1, and igneum-build-3, AX102-1, on order). lib.sh: bs_box_file N and
bs_route <class>; a class whose box has no host file yet falls back to box 1 and says so. run-from-mac.sh --box N <ip> provisions
igneum-build-N and writes build-server-N. tools/build-remote.sh --box N overrides the route; --priority gate always runs on box 1;
the proving crate routes to box 3. README.md: the kind map and the project lead's rule, no mining on any Hetzner box, ever (nodes, builds, tests,
benchmarks and CPU proving only; the pool's fast-time network mines on rented GPU pods, never on build-3). capacity/run.sh refuses a
job that would start igneum-miner mine or a GPU worker, whatever SEQUENCE says.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
2026-10-07 10:34:16 +00:00

89 lines
5.6 KiB
Bash
Executable file

#!/usr/bin/env bash
# The capacity controller on igneum-build-1 (infra/build-server/capacity). Runs the job queue in priority order, one job
# at a time, each for a bounded slice; a 5 s poll of /srv/builds/_locks SIGSTOPs the running job the instant any build
# slot or the measure hold is taken and SIGCONTs it when they clear. The layer never takes a build slot and never writes
# under /srv/builds/<worktree>. Started as user build by igneum-capacity.service (Nice 19, chrt -i 0 wrapper, CPUQuota
# leaving 8 threads free). Killed by the night battery's start and restarted after (the service's ExecStartPre/StopPost).
#
# run.sh the controller loop (systemd runs this)
# run.sh --once one pass through the weighted sequence, then exit (for a manual check)
# run.sh --dry-run call every job's --dry-run in priority order and exit
#
# The weighted sequence (CAP_SEQUENCE) gives the earlier, higher-priority jobs more turns. Default:
# pow-fuzz pow-fuzz pow-fuzz sync-fuzz sync-fuzz sim-sweeps model-sweeps clippy-audit
# Each turn runs one job for CAP_SLICE_S (default 1800 s) through run_slice, which backgrounds the job in its own
# process group and pauses/resumes it with the lock poll.
set -uo pipefail
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"; export CAP_JOB=run
. "$HERE/../capacity/lib.sh" 2>/dev/null || . "$HERE/lib.sh"
SLICE_S="${CAP_SLICE_S:-1800}"
POLL_S="${CAP_POLL_S:-5}"
SEQUENCE="${CAP_SEQUENCE:-pow-fuzz pow-fuzz pow-fuzz sync-fuzz sync-fuzz sim-sweeps model-sweeps clippy-audit}"
JOBS_DIR="$HERE/jobs"
"$NODE_BIN" "$SUMMARY" init 2>/dev/null || true
if [ "${1:-}" = --dry-run ]; then
for j in pow-fuzz sync-fuzz sim-sweeps model-sweeps clippy-audit; do
cap_say "dry-run $j"; bash "$JOBS_DIR/$j.sh" --dry-run || cap_say "$j dry-run returned $?"
done
exit 0
fi
# run_slice <job>: background the job in its own process group, poll the locks, STOP while a build or measure holds,
# CONT when clear, enforce the slice as a wall clock, reap at the end. The job writes its own summary; the controller
# updates the top-level state and current_job.
run_slice() {
local job="$1" log="$CAP_LOG/$job.slice.log"; mkdir -p "$CAP_LOG"
cap_say "slice start: $job (${SLICE_S}s)"
# own process group via setsid so STOP/CONT reach the whole cargo/python tree
setsid bash "$JOBS_DIR/$job.sh" --slice-s "$SLICE_S" > "$log" 2>&1 &
local pid=$! pgid; pgid=$(ps -o pgid= -p "$pid" 2>/dev/null | tr -d ' '); [ -n "$pgid" ] || pgid="$pid"
local paused=0 end=$(( $(date +%s) + SLICE_S + 120 )) # a 2 min grace over the slice for the job's own teardown
while kill -0 "$pid" 2>/dev/null; do
if cap_build_active; then
if [ "$paused" = 0 ]; then kill -STOP -"$pgid" 2>/dev/null && paused=1; local why; why=$(cap_hold_reason); cap_say "pause $job ($why)"; printf '{"state":"paused","current_job":%s,"paused_reason":%s,"slice_s":%s}' "$(cap_json_str "$job")" "$(cap_json_str "$why")" "$SLICE_S" | cap_controller_summary; fi
else
if [ "$paused" = 1 ]; then kill -CONT -"$pgid" 2>/dev/null; paused=0; cap_say "resume $job"; printf '{"state":"running","current_job":%s,"paused_reason":null,"slice_s":%s}' "$(cap_json_str "$job")" "$SLICE_S" | cap_controller_summary; fi
fi
# a safety stop so a wedged job cannot hold the turn forever; a paused job's clock does not advance it past end+grace
if [ "$paused" = 0 ] && [ "$(date +%s)" -ge "$end" ]; then cap_say "slice over time, stopping $job"; kill -INT -"$pgid" 2>/dev/null; sleep 3; kill -KILL -"$pgid" 2>/dev/null; break; fi
sleep "$POLL_S"
done
[ "$paused" = 1 ] && kill -CONT -"$pgid" 2>/dev/null
wait "$pid" 2>/dev/null; local rc=$?
cap_say "slice end: $job rc $rc"
}
once() {
for job in $SEQUENCE; do
# No mining on any Hetzner box, ever (the project lead through main, 7 October 2026; Hetzner's policies forbid it): a job that would start a
# miner or a GPU worker is refused here, whatever SEQUENCE says. Nodes, builds, tests, benchmarks and CPU proving only.
if grep -qE 'igneum-miner[[:space:]]+mine|igneum-worker-(cuda|opencl)|igneum-app.*--mine|cargo run.*-p[[:space:]]+igneum-miner' "$JOBS_DIR/$job.sh" 2>/dev/null; then
echo "capacity: REFUSED job $job: it would start a miner or a GPU worker; no mining on a Hetzner box (infra/build-server/README.md)" >&2; continue
fi
[ -f "$JOBS_DIR/$job.sh" ] || { cap_say "no job $job"; continue; }
# wait out a running build before starting a slice (do not even launch during a build)
while cap_build_active; do
printf '{"state":"waiting","current_job":null,"paused_reason":%s,"slice_s":%s}' "$(cap_json_str "$(cap_hold_reason)")" "$SLICE_S" | cap_controller_summary
sleep "$POLL_S"
done
printf '{"state":"running","current_job":%s,"paused_reason":null,"slice_s":%s,"host":%s}' "$(cap_json_str "$job")" "$SLICE_S" "$(cap_json_str "$(hostname)")" | cap_controller_summary
run_slice "$job"
done
}
cap_sync_checkout >/dev/null 2>&1 || cap_say "initial checkout had trouble (jobs will retry)"
if [ "${1:-}" = --once ]; then once; printf '{"state":"idle","current_job":null,"paused_reason":null}' | cap_controller_summary; exit 0; fi
trap 'cap_say "controller stopping"; printf "{\"state\":\"stopped\",\"current_job\":null}" | cap_controller_summary; exit 0' INT TERM
cap_say "controller start (sequence: $SEQUENCE; slice ${SLICE_S}s; poll ${POLL_S}s)"
cycle=0
while true; do
cycle=$((cycle + 1))
# refresh the checkout at the top of each cycle (cheap when nothing moved); yields if a build is active
while cap_build_active; do sleep "$POLL_S"; done
cap_sync_checkout >/dev/null 2>&1 || true
once
done