GPU fleet: the orchestrator loop, a killable per-second GPU sampler, the STATUS-line rate parser, an atomic locked registry, the p2p hub peer

This commit is contained in:
igneum-josh 2026-10-06 13:34:25 +01:00
parent 78e93d4c50
commit 83b71e8c65
9 changed files with 117 additions and 29 deletions

59
tools/fleet/autorun.py Normal file
View file

@ -0,0 +1,59 @@
#!/usr/bin/env python3
"""The fleet's orchestrator loop (runs on the Mac in the background, one pass a minute): advances every live box
through its stages without a human, pulls the results of each finished stage into ~/Desktop/fleet/<instance>/, runs the
collector after a matrix or an Ember ladder lands, and refreshes the fleet page every 5 minutes.
phase 1: setup_done -> box-matrix.sh -> matrix_done -> box-ember.sh -> ember_done -> box-prover.sh (joins phase 2)
phase 2: setup_done -> box-prover.sh
setup_failed: retried once (the toolchain CDN class), then marked failed and left for a human
Writes ~/Desktop/fleet/autorun.log; every change is a line there and in the page log.
"""
import json, os, sys, time, subprocess, datetime
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import fleet
ROOT = fleet.ROOT; LOG = os.path.join(ROOT, "autorun.log")
def now(): return datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
def log(s, page=False):
line = f"{now()} {s}"; open(LOG, "a").write(line + "\n"); print(line, flush=True)
if page: subprocess.run([sys.executable, os.path.join(fleet.HERE, "page.py"), "log", s])
def probe(b):
rc, out, err = fleet.ssh(b, "tail -1 /root/fleet/setup.log 2>/dev/null | grep -o '^RESULT setup_[a-z]*'; grep -c . /root/fleet/out/matrix.log 2>/dev/null; grep -o '^RESULT matrix_[a-z]*' /root/fleet/out/matrix.log 2>/dev/null | tail -1; grep -c . /root/fleet/out/ember.log 2>/dev/null; grep -o '^RESULT ember_done' /root/fleet/out/ember.log 2>/dev/null | tail -1; pgrep -c -f '^(bash in/box-|python3 -u /root/fleet/in/box-prover)'; grep -E '^RESULT (point|miner|step|choice|claim|submitted|paid|segment_refused|seg [0-9]+ (chain|shards))' /root/fleet/out/matrix.log /root/fleet/out/ember.log /root/fleet/out/prover.log 2>/dev/null | tail -1 | cut -c1-200", timeout=40)
if rc != 0 and not out.strip(): return None
l = out.split("\n")
g = lambda i: l[i].strip() if i < len(l) else ""
return {"setup": g(0), "matrix_lines": g(1), "matrix": g(2), "ember_lines": g(3), "ember": g(4), "running": g(5), "last": g(6)}
def start(b, script, env=""):
fleet.run(script, [b["label"]]) if not env else fleet.run_env(script, [b["label"]], env)
last_publish = 0; retried = set()
while True:
reg = fleet.load(); changed = False
for iid, b in list(reg.items()):
if b.get("state") in ("destroyed", "failed") or not b.get("ssh_ok") or b.get("phase") == "5": continue
p = probe(b)
if p is None: continue
stage = b.get("stage", "setup")
if p["last"]: fleet.patch(iid, last_line=p["last"])
if p["running"] not in ("0", ""): # a stage script is running on the box: never start another (the 5070 ran two matrices at once, 12:23Z)
if stage == "setup" and p["setup"] == "RESULT setup_done": fleet.patch(iid, stage="matrix" if b["phase"] == "1" else "prover", state="running")
continue
if stage == "setup":
if p["setup"] == "RESULT setup_done":
nxt = "matrix" if b["phase"] == "1" else "prover"
fleet.patch(iid, stage=nxt, state="running", doing=("phase 1 matrix: idle, miner, stock server, patched server alone and beside the miner" if nxt == "matrix" else "phase 2: node + miner + segment prover on the devnet"))
fleet.run("box-matrix.sh" if nxt == "matrix" else "box-prover.sh", [b["label"]]); log(f"{b['label']}: setup done, {nxt} started", page=True); changed = True
elif p["setup"] == "RESULT setup_failed":
if iid not in retried: retried.add(iid); fleet.setup([b["label"]]); log(f"{b['label']}: setup failed once, retried")
else: fleet.patch(iid, state="failed", doing="setup failed twice; see setup.log"); log(f"{b['label']}: setup failed twice", page=True)
elif stage == "matrix":
if p["matrix"] == "RESULT matrix_done":
fleet.pull([b["label"]]); subprocess.run([sys.executable, os.path.join(fleet.HERE, "collect.py")], capture_output=True)
fleet.patch(iid, stage="ember", doing="Ember two-knob ladder: 6 power steps, 4 clock caps, 75 s each"); fleet.run("box-ember.sh", [b["label"]]); log(f"{b['label']}: matrix done ({p['last'][:80]}), Ember ladder started", page=True); changed = True
elif p["matrix"] == "RESULT matrix_failed":
fleet.pull([b["label"]]); fleet.patch(iid, stage="ember", doing="matrix failed (see matrix.log); Ember ladder started"); fleet.run("box-ember.sh", [b["label"]]); log(f"{b['label']}: matrix FAILED ({p['last'][:80]}), Ember started", page=True)
elif stage == "ember":
if p["ember"] == "RESULT ember_done" or (p["ember_lines"].isdigit() and int(p["ember_lines"]) > 0 and "ember_failed" in p["last"]):
fleet.pull([b["label"]]); subprocess.run([sys.executable, os.path.join(fleet.HERE, "collect.py")], capture_output=True)
fleet.patch(iid, stage="prover", phase="2", doing="phase 2: node + miner + segment prover on the devnet"); fleet.run("box-prover.sh", [b["label"]]); log(f"{b['label']}: Ember done ({p['last'][:80]}), prover started", page=True); changed = True
if time.time() - last_publish > 300 or changed:
subprocess.run([sys.executable, os.path.join(fleet.HERE, "page.py"), "publish"], capture_output=True); last_publish = time.time()
time.sleep(60)

View file

@ -36,11 +36,11 @@ step() { # <label> <power_pct> <limit_w> <clock_cap_mhz or 0>
if [ "$PL_OK" = 1 ]; then nvidia-smi -i 0 -pl "$lim" >/dev/null 2>&1 || say "could not set -pl $lim"; fi
if [ "$cap" != 0 ]; then nvidia-smi -i 0 -lgc 0,"$cap" >/dev/null 2>&1 || say "could not set -lgc $cap"; else nvidia-smi -i 0 -rgc >/dev/null 2>&1; fi
sleep "$SETTLE"
local n0; n0="$(grep -c 'now=' $OUT/ember-miner.log)"
nvidia-smi --query-gpu=power.draw,clocks.sm,clocks.mem,temperature.gpu --format=csv,noheader,nounits -i 0 -l 1 > $OUT/ember-samp-$lab.csv 2>/dev/null & local sp=$!
sleep "$HOLD"; kill $sp 2>/dev/null; wait $sp 2>/dev/null
local n1; n1="$(grep -c 'now=' $OUT/ember-miner.log)"
local mhs; mhs="$(grep -o 'now=[0-9.]*' $OUT/ember-miner.log | tail -n $((n1 - n0 > 0 ? n1 - n0 : 1)) | cut -d= -f2 | awk '{s+=$1; n++} END {if (n) printf "%.2f", s/n; else print 0}')"
local n0; n0="$(grep -c 'STATUS' $OUT/ember-miner.log)"
( while :; do nvidia-smi --query-gpu=power.draw,clocks.sm,clocks.mem,temperature.gpu --format=csv,noheader,nounits -i 0 2>/dev/null; sleep 1; done ) > $OUT/ember-samp-$lab.csv & local sp=$!
sleep "$HOLD"; pkill -P $sp 2>/dev/null; kill $sp 2>/dev/null; sleep 1; kill -9 $sp 2>/dev/null
local n1; n1="$(grep -c 'STATUS' $OUT/ember-miner.log)"
local mhs; mhs="$(grep STATUS $OUT/ember-miner.log | grep -o ' now=[0-9.]*' | tail -n $((n1 - n0 > 0 ? n1 - n0 : 1)) | cut -d= -f2 | awk '{s+=$1; n++} END {if (n) printf "%.2f", s/n; else print 0}')"
local w gclk mclk tmax; read -r w gclk mclk tmax <<< "$(awk -F', *' '{w+=$1; g+=$2; m+=$3; if ($4+0 > t) t=$4+0; n++} END {if (n) printf "%.1f %.0f %.0f %d", w/n, g/n, m/n, t; else print "0 0 0 0"}' $OUT/ember-samp-$lab.csv)"
local eff; eff="$(awk -v a="$mhs" -v b="$w" 'BEGIN {if (b > 0) printf "%.4f", a / b; else print 0}')"
echo "RESULT step $lab power_pct=$pct limit_w=$lim clock_cap_mhz=$cap watts=$w mhs=$mhs mhw=$eff gclk=$gclk mclk=$mclk tmax=$tmax"

View file

@ -8,6 +8,7 @@
# line is "RESULT matrix_done" or "RESULT matrix_failed <why>".
set -uo pipefail
F=/root/fleet; OUT=$F/out; LOG=$OUT/matrix.log; B=/opt/igneum/pkg/bin; FLOOR=/opt/igneum-floor
for b in igneum-prove-host igneum-prove-export; do for d in $FLOOR/target/release $FLOOR/prove/proving/igneum-prove/target/release; do [ -x $d/$b ] && ln -sfn $d/$b $FLOOR/bin/$b; done; done
HOST=$FLOOR/bin/igneum-prove-host; FIX=$FLOOR/prove/proving/fixtures
V1=$FIX/fees-v1-shards2.json; EMPTY=$FIX/block-72854-empty-block-first.json
LABEL="${LABEL:-box}"; WALLET="${WALLET:-0x1919191919191919191919191919191919191919}"
@ -25,8 +26,10 @@ ROWS=$OUT/rows.jsonl; : > $ROWS
# the sampler: one csv per point, "ts,mem_mib,power_w,util_pct"
SAMP=""
sampler_start() { nvidia-smi --query-gpu=timestamp,memory.used,power.draw,utilization.gpu --format=csv,noheader,nounits -i 0 -l 1 > "$1" 2>/dev/null & SAMP=$!; }
sampler_stop() { [ -n "$SAMP" ] && kill $SAMP 2>/dev/null; wait $SAMP 2>/dev/null; SAMP=""; }
# one nvidia-smi call a second in a loop: `nvidia-smi -l 1` buffers its file output in 4 KB chunks and ignored SIGTERM
# on the 3080 box (12:19Z), which hung the first matrix in sampler_stop
sampler_start() { ( while :; do nvidia-smi --query-gpu=timestamp,memory.used,power.draw,utilization.gpu --format=csv,noheader,nounits -i 0 2>/dev/null; sleep 1; done ) > "$1" & SAMP=$!; }
sampler_stop() { [ -n "$SAMP" ] && { pkill -P $SAMP 2>/dev/null; kill $SAMP 2>/dev/null; sleep 1; kill -9 $SAMP 2>/dev/null; }; SAMP=""; }
peak_of() { awk -F', *' 'NR>0 {if ($2+0 > m) m=$2+0} END {print m+0}' "$1"; }
mean_col() { awk -F', *' -v c="$2" '{s+=$c; n++} END {if (n) printf "%.1f", s/n; else print 0}' "$1"; }
kill_server() { pkill -f sp1-gpu-server 2>/dev/null; sleep 2; pkill -9 -f sp1-gpu-server 2>/dev/null; rm -f /tmp/sp1-cuda-*.sock; }
@ -54,7 +57,7 @@ miner_start() {
}
miner_stop() { [ -n "$MPID" ] && { kill $MPID 2>/dev/null; sleep 2; kill -9 $MPID 2>/dev/null; }; pkill -f igneum-worker-cuda 2>/dev/null; pkill -f "igneum-miner mine" 2>/dev/null; MPID=""; sleep 3; }
miner_rate() { # mean of the STATUS now= values in the last N lines
grep -o 'now=[0-9.]*' $OUT/miner.log | tail -n "${1:-10}" | cut -d= -f2 | awk '{s+=$1; n++} END {if (n) printf "%.2f", s/n; else print 0}'
grep STATUS $OUT/miner.log | grep -o ' now=[0-9.]*' | tail -n "${1:-10}" | cut -d= -f2 | awk '{s+=$1; n++} END {if (n) printf "%.2f", s/n; else print 0}'
}
miner_start
say "miner warming 60 s"; sleep 60

View file

@ -60,7 +60,7 @@ def miner_stop():
subprocess.run("pkill -f igneum-worker-cuda", shell=True)
def miner_rate(n=6):
try:
vals = [float(l.split("now=")[1].split()[0]) for l in open(f"{OUT}/prover-miner.log").read().split("\n") if "now=" in l][-n:]
vals = [float(l.split(" now=")[1].split()[0]) for l in open(f"{OUT}/prover-miner.log").read().split("\n") if "STATUS" in l and " now=" in l][-n:]
return round(sum(vals) / len(vals), 2) if vals else 0
except Exception: return 0
# the key
@ -187,11 +187,11 @@ while (time.time() - t_run0) / 3600 < RUN_HOURS:
if THRESHOLD: env["SP1_GPU_ELEMENT_THRESHOLD"] = THRESHOLD
args = [HOST, "--mode", "chain", "--chain", ",".join(fixtures), "--prover", WALLET, "--save-shards", "--out", f"{d}/chain-results.json"]
if prev_file: args += ["--prev", prev_file]
samp = subprocess.Popen("nvidia-smi --query-gpu=memory.used,utilization.gpu,power.draw --format=csv,noheader,nounits -l 1", shell=True, stdout=open(f"{d}/smi.csv", "w"), stderr=subprocess.DEVNULL)
samp = subprocess.Popen("while :; do nvidia-smi --query-gpu=memory.used,utilization.gpu,power.draw --format=csv,noheader,nounits; sleep 1; done", shell=True, stdout=open(f"{d}/smi.csv", "w"), stderr=subprocess.DEVNULL, start_new_session=True)
t = time.time()
try: rr = subprocess.run(args, env=env, capture_output=True, text=True, timeout=3600)
except subprocess.TimeoutExpired: rr = None
seg["chain_s"] = round(time.time() - t, 1); samp.terminate(); kill_server()
seg["chain_s"] = round(time.time() - t, 1); os.killpg(samp.pid, signal.SIGKILL); kill_server()
if MINER == "pause": miner_start()
try: peak = max(float(l.split(",")[0]) for l in open(f"{d}/smi.csv") if l.strip())
except Exception: peak = 0

View file

@ -3,7 +3,8 @@
# carry them, the proving agent's note of 6 October 2026), waits for it to be synced again (the data dir is kept, so
# seconds), then runs box-prover.py with the card's profile: THRESHOLD and MINER from the card's memory unless given.
set -uo pipefail
F=/root/fleet; OUT=$F/out; B=/opt/igneum/pkg/bin; FLOOR=/opt/igneum-floor; HOST=$FLOOR/bin/igneum-prove-host
F=/root/fleet; OUT=$F/out; B=/opt/igneum/pkg/bin; FLOOR=/opt/igneum-floor; for b in igneum-prove-host igneum-prove-export; do for d in $FLOOR/target/release $FLOOR/prove/proving/igneum-prove/target/release; do [ -x $d/$b ] && ln -sfn $d/$b $FLOOR/bin/$b; done; done
HOST=$FLOOR/bin/igneum-prove-host
mkdir -p $OUT; exec >> $OUT/prover-launch.log 2>&1
stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; }
TOTAL="$(nvidia-smi --query-gpu=memory.total --format=csv,noheader,nounits -i 0 | tr -d ' ')"
@ -14,8 +15,9 @@ if [ -z "${MINER:-}" ]; then if [ "$TOTAL" -ge 15000 ]; then MINER=keep; else MI
echo "RESULT launch $(stamp) total_mib=$TOTAL threshold=${THRESHOLD:-default} miner=$MINER"
pkill -f "igneum-miner mine" 2>/dev/null; pkill -f igneum-worker-cuda 2>/dev/null; pkill -f sp1-gpu-server 2>/dev/null; rm -f /tmp/sp1-cuda-*.sock
pkill -x igneumd; sleep 4; pkill -9 -x igneumd 2>/dev/null; sleep 1
HUBARG=""; [ -n "${HUB_PEER:-}" ] && HUBARG="--addpeer=$HUB_PEER" # the fleet's hub box (a second peer beside the seed, 12:35Z)
IGNEUM_PROOF_VERIFIER=$HOST nohup $B/igneumd --devnet --appdir=$F/node --rpclisten=127.0.0.1:26610 --evm-rpclisten=127.0.0.1:26790 --listen=0.0.0.0:26611 \
--addpeer=188.245.5.161:26611 --override-params-file=$F/override.json --nodnsseed --disable-upnp --nologfiles --yes >> $F/node.log 2>&1 &
--addpeer=188.245.5.161:26611 $HUBARG --override-params-file=$F/override.json --nodnsseed --disable-upnp --nologfiles --yes >> $F/node.log 2>&1 &
sleep 10
echo "RESULT node_restarted $(stamp) pid=$(pgrep -x igneumd | head -1) verifier=$(grep -c 'IGNEUM_PROOF_VERIFIER\|proof verifier' $F/node.log)"
export THRESHOLD MINER

View file

@ -92,7 +92,8 @@ echo "RESULT server bytes=$(stat -c %s /opt/igneum-floor/bin/sp1-gpu-server) sha
# 5. the host and the exporter (cuda feature), from the prover-floor bundle
H=/opt/igneum-floor/prove
if [ ! -x $H/proving/igneum-prove/target/release/igneum-prove-host ]; then
unset CARGO_TARGET_DIR # the server build's target dir must not catch the host (the 3080, 12:09Z: the host landed under /opt/igneum-floor/target)
if [ ! -x $H/proving/igneum-prove/target/release/igneum-prove-host ] && [ ! -x /opt/igneum-floor/target/release/igneum-prove-host ]; then
say "host"
rm -rf $H && mkdir -p $H && unzip -q -o $IN/igneum-prove-wsl2-floor.zip -d $F/unz && cp -r $F/unz/igneum-prove-wsl2/package/. $H/ && find $H -type f -exec touch {} +
cd $H/proving/igneum-prove || fail host_src
@ -104,6 +105,7 @@ if [ ! -x $H/proving/igneum-prove/target/release/igneum-prove-host ]; then
[ $rc -eq 0 ] || { grep -n -A8 '^error' /opt/igneum-floor/logs/build-host.log | head -60; fail host_build; }
fi
HOST=$H/proving/igneum-prove/target/release/igneum-prove-host; EXPORT=$H/proving/igneum-prove/target/release/igneum-prove-export
[ -x "$HOST" ] || { HOST=/opt/igneum-floor/target/release/igneum-prove-host; EXPORT=/opt/igneum-floor/target/release/igneum-prove-export; }
ln -sfn $HOST /opt/igneum-floor/bin/igneum-prove-host; ln -sfn $EXPORT /opt/igneum-floor/bin/igneum-prove-export
echo "RESULT host sha256=$(sha256sum $HOST | cut -c1-16) export=$(sha256sum $EXPORT | cut -c1-16) ids=$($HOST --mode id 2>&1 | grep -o '0x[0-9a-f]*' | tr '\n' ' ')"
echo "RESULT fixtures $(ls $H/proving/fixtures | wc -l) files, v1=$(sha256sum $H/proving/fixtures/fees-v1-shards2.json | cut -c1-16) empty=$(sha256sum $H/proving/fixtures/block-72854-empty-block-first.json | cut -c1-16)"

View file

@ -20,6 +20,12 @@ for iid, b in reg.items():
if not os.path.exists(mp): continue
m = json.load(open(mp)); pts = {r["name"]: r for r in m["rows"] if r.get("row") == "point"}; miner = next((r for r in m["rows"] if r.get("row") == "miner"), {})
total = m["total_mib"]; idle = m["idle_mib"]
# the miner's rate from the raw STATUS lines (the first matrices parsed a second now= field): the mean of the last
# 12 STATUS lines before the miner was stopped for the stock point, i.e. the sampled window
ml = f"{ROOT}/{iid}/miner.log"
if os.path.exists(ml):
vals = [float(l.split(" now=")[1].split()[0]) for l in open(ml, errors="replace") if "STATUS" in l and " now=" in l]
if vals: miner = dict(miner, mhs=round(sum(vals[-12:]) / len(vals[-12:]), 2), mhs_raw_lines=len(vals))
def ok(n): p = pts.get(n); return p and p.get("verified") == "yes"
def own(n): p = pts.get(n); return p.get("own_mib") if p else None
def secs(n): p = pts.get(n); return p.get("prove_s") if p else None

View file

@ -23,8 +23,22 @@ SSH_OPTS = ["-i", os.path.expanduser("~/.ssh/igneum-fleet"), "-o", "StrictHostKe
INPUTS = [os.path.expanduser("~/Desktop/igneum-prove-wsl2-floor.zip"), os.path.join(HERE, "floor.patch"), os.path.join(HERE, "override.json")]
def now(): return datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
def load(): return json.load(open(REG)) if os.path.exists(REG) else {}
def save(reg): os.makedirs(ROOT, exist_ok=True); json.dump(reg, open(REG, "w"), indent=1)
import fcntl
def load():
if not os.path.exists(REG): return {}
for attempt in range(6): # a reader can meet a half-written file only if a writer bypasses save(); retry anyway
try: return json.load(open(REG))
except json.JSONDecodeError:
time.sleep(0.2 * (attempt + 1))
raise
def save(reg):
os.makedirs(ROOT, exist_ok=True); tmp = REG + f".tmp.{os.getpid()}"
json.dump(reg, open(tmp, "w"), indent=1); os.replace(tmp, REG)
def patch(iid, **fields):
with open(REG + ".lock", "w") as lk:
fcntl.flock(lk, fcntl.LOCK_EX)
reg = load(); reg.setdefault(str(iid), {}).update(fields); save(reg); return reg[str(iid)]
def boxes(labels, reg=None):
reg = reg or load()
sel = {k: v for k, v in reg.items() if v.get("state") != "destroyed" and (not labels or v["label"] in labels)}
@ -35,8 +49,8 @@ def refresh_ssh(reg):
live = {str(i.get("id")): i for i in vast.instances()}
for iid, b in reg.items():
if b.get("provider", "vast") == "vast" and iid in live:
i = live[iid]; b["ssh_host"] = i.get("ssh_host"); b["ssh_port"] = i.get("ssh_port"); b["actual_status"] = i.get("actual_status"); b["status_msg"] = (i.get("status_msg") or "")[:120]
save(reg); return reg
i = live[iid]; patch(iid, ssh_host=i.get("ssh_host"), ssh_port=i.get("ssh_port"), actual_status=i.get("actual_status"), status_msg=(i.get("status_msg") or "")[:120])
return load()
def ssh(b, cmd, timeout=120, capture=True):
try:
@ -72,9 +86,8 @@ def wait(labels, limit=1200):
if not b.get("ssh_host") or not b.get("ssh_port"): pending.append((b["label"], b.get("actual_status"), b.get("status_msg"))); continue
try: rc, out, err = ssh(b, "nvidia-smi --query-gpu=name,memory.total --format=csv,noheader", timeout=40)
except subprocess.TimeoutExpired: rc, out, err = 1, "", "timeout"
if rc == 0 and out.strip(): b["ssh_ok"] = True; b["state"] = "installing"; b["gpu_seen"] = out.strip().replace("\n", ";"); print(f"{b['label']}: ssh ok, {b['gpu_seen']}")
if rc == 0 and out.strip(): patch(iid, ssh_ok=True, state="installing", gpu_seen=out.strip().replace("\n", ";")); print(f"{b['label']}: ssh ok, {out.strip()[:60]}")
else: pending.append((b["label"], b.get("actual_status"), (err or out).strip()[:80]))
save(reg)
if not pending: print("all up"); return
if time.time() - t0 > limit: print("still pending:", pending); return
print(f"waiting ({int(time.time()-t0)} s): " + "; ".join(f"{l} {s} {m}" for l, s, m in pending)); time.sleep(20)
@ -86,8 +99,7 @@ def setup(labels):
if scp(b, INPUTS + [os.path.join(HERE, "box-setup.sh")], "/root/fleet/in/") != 0: print(b["label"], "scp failed"); continue
rc, out, err = ssh(b, f"cd /root/fleet && chmod +x in/box-setup.sh && if pgrep -f '^bash in/box-setup.sh' >/dev/null; then echo already; else RUSTUP_TOOLCHAIN=stable ARCHS={b['archs']} LABEL={b['label']} WALLET={b['wallet']} setsid nohup in/box-setup.sh </dev/null >/dev/null 2>&1 & echo started; fi", timeout=60)
print(b["label"], out.strip()[:40])
b["setup_started"] = now(); print(b["label"], "setup started")
save(load() | {k: v for k, v in boxes(labels).items()})
patch(iid, setup_started=now())
def status(labels):
reg = load()
@ -104,10 +116,11 @@ def status(labels):
with ThreadPoolExecutor(max_workers=16) as ex: results = list(ex.map(one, items))
for (iid, b), (line, done) in zip(items, results):
print(line)
if done and b.get("state") == "installing": b["state"] = "running"
save(reg)
if done and b.get("state") == "installing": patch(iid, state="running")
def run(script, labels, env=""):
hub = next((v for v in load().values() if v.get("hub") and v.get("hub_peer")), None)
if hub and "HUB_PEER" not in env: env = f"HUB_PEER={hub['hub_peer']} " + env
for iid, b in boxes(labels).items():
if scp(b, [os.path.join(HERE, script)], "/root/fleet/in/") != 0: print(b["label"], "scp failed"); continue
name = os.path.basename(script)
@ -126,10 +139,9 @@ def destroy(labels):
if b.get("ssh_ok"):
try: pull([b["label"]])
except Exception as e: print("pull failed", e)
vast.destroy(iid); b["state"] = "destroyed"; b["destroyed_at"] = now()
t0 = datetime.datetime.strptime(b["rented_at"], "%Y-%m-%dT%H:%M:%SZ"); h = (datetime.datetime.strptime(b["destroyed_at"], "%Y-%m-%dT%H:%M:%SZ") - t0).total_seconds() / 3600
b["hours"] = round(h, 2); b["cost_usd"] = round(h * b["dph"], 3); print(b["label"], f"destroyed after {h:.2f} h, USD {b['cost_usd']:.2f}")
save(reg)
vast.destroy(iid); t1 = now()
t0 = datetime.datetime.strptime(b["rented_at"], "%Y-%m-%dT%H:%M:%SZ"); h = (datetime.datetime.strptime(t1, "%Y-%m-%dT%H:%M:%SZ") - t0).total_seconds() / 3600
patch(iid, state="destroyed", destroyed_at=t1, hours=round(h, 2), cost_usd=round(h * b["dph"], 3)); print(b["label"], f"destroyed after {h:.2f} h, USD {h * b['dph']:.2f}")
if __name__ == "__main__":
a = sys.argv[1:]

View file

@ -24,7 +24,11 @@ def call(method, path, body=None, timeout=90):
for attempt in range(4):
try:
with urllib.request.urlopen(req, timeout=timeout) as r:
return json.loads(r.read() or b"{}")
raw = r.read() or b"{}"
try: return json.loads(raw)
except json.JSONDecodeError:
if attempt < 3: time.sleep(3 * (attempt + 1)); continue
raise SystemExit(f"{method} {path}: non-JSON reply: {raw[:120]!r}")
except urllib.error.HTTPError as e:
txt = e.read().decode(errors="replace")
if e.code in (429, 502, 503, 504) and attempt < 3: