From 1b79639577bbd4982f800cba72b2a3942e218d66 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Tue, 6 Oct 2026 12:01:23 +0000 Subject: [PATCH 01/72] GPU fleet: Vast client, box setup (0.3.12 node + patched SP1 server + cuda host), the phase-1 matrix, the Ember ladder, the Linux segment prover, the fleet page builder --- tools/fleet/box-ember.sh | 86 ++++++++++ tools/fleet/box-matrix.sh | 124 ++++++++++++++ tools/fleet/box-prover.py | 246 +++++++++++++++++++++++++++ tools/fleet/box-setup.sh | 111 ++++++++++++ tools/fleet/fleet.py | 153 +++++++++++++++++ tools/fleet/floor.patch | 345 ++++++++++++++++++++++++++++++++++++++ tools/fleet/override.json | 1 + tools/fleet/page.py | 61 +++++++ tools/fleet/vast.py | 122 ++++++++++++++ 9 files changed, 1249 insertions(+) create mode 100755 tools/fleet/box-ember.sh create mode 100755 tools/fleet/box-matrix.sh create mode 100644 tools/fleet/box-prover.py create mode 100755 tools/fleet/box-setup.sh create mode 100755 tools/fleet/fleet.py create mode 100644 tools/fleet/floor.patch create mode 100644 tools/fleet/override.json create mode 100755 tools/fleet/page.py create mode 100755 tools/fleet/vast.py diff --git a/tools/fleet/box-ember.sh b/tools/fleet/box-ember.sh new file mode 100755 index 000000000..60a5f8969 --- /dev/null +++ b/tools/fleet/box-ember.sh @@ -0,0 +1,86 @@ +#!/usr/bin/env bash +# Ember Tune's two-knob ladder on a rented NVIDIA card (docs/plans/ember-tune.md, branch ember-tune): the miner runs +# throughout; the power ladder 100, 90, 80, 70, 60, 50% of the default limit at the unlocked clock (clamped at the +# card's reported minimum), then the clock ladder 90, 80, 70, 60% of the maximum graphics clock at the power the +# first ladder chose; 15 s settle and 60 s hold per step; the choice is the best MH/W among the steps whose rate is +# within 1% of the fastest. Linux root: nvidia-smi -pl and -lgc, no prompt. Every step prints a RESULT line; the +# ladder and the choice go to /root/fleet/out/ember.json as a TUNE-shaped record (relay/lib/ember.mjs parseRecords). +set -uo pipefail +F=/root/fleet; OUT=$F/out; LOG=$OUT/ember.log; B=/opt/igneum/pkg/bin +LABEL="${LABEL:-box}"; WALLET="${WALLET:-0x1919191919191919191919191919191919191919}" +SETTLE="${SETTLE:-15}"; HOLD="${HOLD:-60}" +mkdir -p $OUT $F/mine/packs +exec > >(tee -a $LOG) 2>&1 +stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; } +say() { echo "$(stamp) $*"; } +q() { nvidia-smi --query-gpu="$1" --format=csv,noheader,nounits -i 0 2>/dev/null | head -1 | tr -d ' '; } +NAME="$(q name)"; DRV="$(q driver_version)"; PDEF="$(q power.default_limit)"; PMIN="$(q power.min_limit)"; PMAX="$(q power.max_limit)"; CMAX="$(q clocks.max.graphics)" +pkill -f sp1-gpu-server 2>/dev/null; rm -f /tmp/sp1-cuda-*.sock +echo "RESULT start $(stamp) card=$NAME driver=$DRV power_default_w=$PDEF min_w=$PMIN max_w=$PMAX clock_max_mhz=$CMAX" +# can we set anything? +nvidia-smi -i 0 -pl "$PDEF" >/dev/null 2>&1 && PL_OK=1 || PL_OK=0 +nvidia-smi -i 0 -lgc 0,"$CMAX" >/dev/null 2>&1 && LGC_OK=1 || LGC_OK=0 +nvidia-smi -i 0 -rgc >/dev/null 2>&1 +echo "RESULT knobs power_limit_settable=$PL_OK clock_cap_settable=$LGC_OK" +cd $F/mine +rm -rf packs/devnet; $B/igneum-miner export-pack grpc://127.0.0.1:26610 packs/devnet > $OUT/ember-export.log 2>&1 +nohup $B/igneum-miner mine grpc://127.0.0.1:26610 1 100000000 "$LABEL" --worker $B/igneum-worker-cuda --worker-args "--device 0 --pack packs/devnet" \ + --prepare-packs packs/prepare --exit-on-seed-change --evm-address "$WALLET" --payout-label "$LABEL" --status-secs 10 > $OUT/ember-miner.log 2>&1 & +MPID=$!; cd $F +cleanup() { nvidia-smi -i 0 -rgc >/dev/null 2>&1; [ "$PL_OK" = 1 ] && nvidia-smi -i 0 -pl "$PDEF" >/dev/null 2>&1; kill $MPID 2>/dev/null; pkill -f igneum-worker-cuda 2>/dev/null; } +trap cleanup EXIT +say "miner warming 90 s"; sleep 90 +STEPS=$OUT/ember-steps.jsonl; : > $STEPS +step() { #