igneum/tools/fleet/box-standing.sh

79 lines
8 KiB
Bash
Executable file

#!/usr/bin/env bash
# The standing-fleet supervisor on a box (the project lead's ruling, 6 October 2026 19:50 UK: rented cards stay up; a standing box is
# never destroyed on a job's end). One loop, started once by lib/standing.py and left alone:
# every 60 s the live-devnet node (ROLE=live: /opt/igneum/pkg/bin/igneumd-<ver> on the live object) or the Devnet 2
# node (ROLE=dn2: --devnet-suffix=2 on dn2-override.json) is restarted if its process is gone; the miner
# loop likewise (the package's igneum-miner on the CUDA worker, --exit-on-seed-change, pack re-exported on
# rc 42); PROVER=1 keeps box-prover.py up (the floor host on /opt/igneum-floor).
# every 10 min the recovery recipe (box-exec-snapshot.sh, the shipper's) runs when the node's exec tip reads 0 "from
# snapshot" while consensus is synced above 1,000 blocks (the dead-exec class of the fleet night, row 1/5).
# every 10 min RESULT standing lines to /root/fleet/out/standing.log: uptime, node pid/version/digest, blocks, daa,
# peers, synced, exec tip, miner MH/s and accepted/rejected, prover segments; lib/standing.py reads them.
# The version follows the live manifest from the Mac (lib/standing.py update step): it stages the new binary as
# /root/fleet/in/igneumd-<sha16> and writes its path into /root/fleet/standing.node; this loop picks it up at the next
# check and restarts the node on it (data dir kept, so seconds). Patterns are anchored on paths (the pgrep rule).
set -uo pipefail
F=/root/fleet; OUT=$F/out; B=/opt/igneum/pkg/bin; mkdir -p $OUT $F/mine/packs; exec >> $OUT/standing-supervisor.log 2>&1
stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; }
ROLE="${ROLE:-live}"; LABEL="${LABEL:-standing}"; WALLET="${WALLET:-0x1919191919191919191919191919191919191919}"; PROVER="${PROVER:-0}"; MINE="${MINE:-1}"
HUB_PEER="${HUB_PEER:-}"; SEED="${SEED:-}"; STARTED=$(date -u +%s)
if [ "$ROLE" = dn2 ]; then APP=$F/dn2; LOG=$F/dn2-node.log; OVR=$F/dn2-override.json; SUFFIX="--devnet-suffix=2"; PACK=dn2; NET=igneum-devnet-2
else APP=$F/node; LOG=$F/node.log; OVR=$F/override.json; SUFFIX=""; PACK=devnet; NET=igneum-devnet; fi
echo "RESULT standing_start $(stamp) role=$ROLE label=$LABEL prover=$PROVER mine=$MINE"
NODE_RE='^/(opt/igneum/pkg/bin|opt/igneum/pkg\.prev/bin|root/fleet/in)/igneumd(-0313|-[0-9a-f]{16})? --devnet --appdir=' # the LIVE node only, never igneumd-v4 (the rehearsal node, --devnet-suffix=400) nor a Devnet 2 node: at 19:5xZ the hub's live node died and the old pattern took the rehearsal node for it
node_running_bin() { pgrep -af "$NODE_RE" | head -1 | awk '{print $2}'; }
# adopt the node that already runs (a canary or swap may have put a newer binary than the package's in place): never downgrade it
[ -s $F/standing.node ] || { rb=$(node_running_bin); [ -n "$rb" ] && echo "$rb" > $F/standing.node; }
node_bin() { [ -s $F/standing.node ] && cat $F/standing.node || { [ -x $B/igneumd-0313 ] && echo $B/igneumd-0313 || echo $B/igneumd; }; }
node_pid() { pgrep -f "$NODE_RE" | head -1; }
start_node() {
local nb; nb=$(node_bin); local peers=""; [ -n "$HUB_PEER" ] && peers="--addpeer=$HUB_PEER"; [ -n "$SEED" ] && peers="$peers --addpeer=$SEED"
[ "$ROLE" = live ] && peers="$peers --addpeer=188.245.5.161:26611"
local ver; ver="$HOST_VERIFIER"; [ -x /opt/igneum-floor/bin/igneum-prove-host ] && ver=/opt/igneum-floor/bin/igneum-prove-host
pkill -9 -f "$NODE_RE" 2>/dev/null; sleep 2
IGNEUM_PROOF_VERIFIER=${ver:-} setsid nohup $nb --devnet $SUFFIX --appdir=$APP --rpclisten=0.0.0.0:26610 --evm-rpclisten=127.0.0.1:26790 --listen=0.0.0.0:26611 $peers --override-params-file=$OVR --nodnsseed --disable-upnp --nologfiles --yes </dev/null >> $LOG 2>&1 &
sleep 8; echo "RESULT standing_node_started $(stamp) bin=$nb pid=$(node_pid) digest=$(grep -o 'digest: [0-9a-f]*' $LOG | tail -1 | awk '{print substr($2,1,16)}')"
# a standalone igneum-miner keeps a dead template subscription after its node restarts (20:24Z: templates frozen,
# fetch_errors climbing, no submits), so the miner and its worker are restarted with the node; the loop brings them back
pkill -9 -f '^/opt/igneum/pkg/bin/igneum-miner mine' 2>/dev/null; pkill -9 -f '^/opt/igneum/pkg/bin/igneum-worker-cuda --serve --device 0 --pack packs' 2>/dev/null; true
}
HOST_VERIFIER=""
miner_loop() {
cd $F/mine; [ -d packs/$PACK ] || $B/igneum-miner export-pack grpc://127.0.0.1:26610 packs/$PACK > $OUT/export-pack.log 2>&1
while :; do
$B/igneum-miner mine grpc://127.0.0.1:26610 1 100000000 "$LABEL" --worker $B/igneum-worker-cuda --worker-args "--device 0 --pack packs/$PACK" --prepare-packs packs/$PACK-prepare --exit-on-seed-change --evm-address "$WALLET" --payout-label "$LABEL" --status-secs 60 >> $OUT/miner-0.log 2>&1; rc=$?
echo "RESULT standing_miner_exit $(stamp) rc=$rc"; [ $rc = 42 ] && { rm -rf packs/$PACK; $B/igneum-miner export-pack grpc://127.0.0.1:26610 packs/$PACK >> $OUT/export-pack.log 2>&1; } || sleep 10
done
}
n=0
while :; do
cur=$(node_bin); rb=$(node_running_bin)
if [ -z "$rb" ]; then start_node
elif [ -n "$cur" ] && [ "$rb" != "$cur" ] && [ -s $F/standing.node ]; then echo "RESULT standing_node_update $(stamp) from=$rb to=$cur"; start_node; fi
if [ "$MINE" = 1 ] && ! pgrep -f '^/opt/igneum/pkg/bin/igneum-miner mine' >/dev/null && [ -z "${MINER_LOOP_PID:-}" -o ! -e "/proc/${MINER_LOOP_PID:-0}" ]; then
w="$($B/igneum-miner watch 1 grpc://127.0.0.1:26610 2>/dev/null | grep -o 'synced=[a-z]*' | tail -1)"
if [ "$w" = synced=true ]; then miner_loop & MINER_LOOP_PID=$!; echo "RESULT standing_miner_started $(stamp) loop=$MINER_LOOP_PID"; fi
fi
if [ "$PROVER" = 1 ] && [ -x /opt/igneum-floor/bin/igneum-prove-host ] && ! pgrep -f '^python3 -u /root/fleet/in/box-prover.py' >/dev/null; then
cd $F && LABEL="$LABEL" WALLET="$WALLET" EXPORT_FROM="${EXPORT_FROM:-0}" CHAIN_NAME=$NET MINER=none RUN_HOURS=240 setsid nohup python3 -u $F/in/box-prover.py </dev/null >> $OUT/prover-launch.log 2>&1 &
echo "RESULT standing_prover_started $(stamp)"
fi
n=$((n+1))
if [ $((n % 10)) = 1 ]; then
# disk: the prover's segment exports (/root/fleet/out/segs, 50 to 500 MB each) filled 60 GB boxes in nine hours (20:42Z: the
# hub's node died on "No space left on device"); keep the last 20 minutes of them and the node log under 300 MB
[ -d $OUT/segs ] && find $OUT/segs -mindepth 1 -maxdepth 1 -mmin +20 -exec rm -rf {} + 2>/dev/null
[ "$(stat -c %s $LOG 2>/dev/null || echo 0)" -gt 300000000 ] && { tail -c 100000000 $LOG > $LOG.tail && mv $LOG.tail $LOG; echo "RESULT standing_log_trimmed $(stamp)"; }
DISK="$(df -h / | tail -1 | awk '{print $4"/"$5}')"
w="$($B/igneum-miner watch 1 grpc://127.0.0.1:26610 2>/dev/null | grep -o 'blocks=[0-9]*.*synced=[a-z]*' | tail -1 | sed -E 's/difficulty=[0-9.]* sink=[0-9a-f]* //')"
ex=$(curl -s -m 5 127.0.0.1:26790 -H 'content-type: application/json' -d '{"jsonrpc":"2.0","id":1,"method":"eth_blockNumber","params":[]}' | grep -oE '"result":"0x[0-9a-f]+"' | cut -d'"' -f4)
exn=$((${ex:-0x0})); blocks=$(printf '%s' "$w" | grep -oE 'blocks=[0-9]+' | cut -d= -f2)
m="$(grep STATUS $OUT/miner-0.log 2>/dev/null | tail -1 | grep -oE 'accepted=[0-9]+ rejected=[0-9]+|\([0-9.]+ MH/s inside jobs' | tr '\n' ' ' | tr -d '(')"
echo "RESULT standing $(stamp) up=$(( $(date -u +%s) - STARTED ))s role=$ROLE node=$(node_pid) bin=$cur version=$($cur --version 2>&1 | head -1 | awk '{print $2}') $w exec=$exn miner=\"$m\" prover=$(pgrep -c -f '^python3 -u /root/fleet/in/box-prover.py') disk=$DISK gpu=$(nvidia-smi --query-gpu=utilization.gpu,power.draw --format=csv,noheader,nounits 2>/dev/null | head -1 | tr -d ' ')" >> $OUT/standing.log
# the dead-exec class: synced consensus, exec at 0 from snapshot -> the shipper's recovery recipe (live devnet only)
if [ "$ROLE" = live ] && [ "${blocks:-0}" -gt 1000 ] && [ "$exn" = 0 ] && [ -x $F/in/box-exec-snapshot.sh ] && [ $((n / 10)) -gt 2 ]; then
echo "RESULT standing_recovery $(stamp) exec=0 blocks=$blocks: running box-exec-snapshot.sh"; HUB_PEER="$HUB_PEER" bash $F/in/box-exec-snapshot.sh; sleep 30
fi
fi
sleep 60
done