GPU fleet: box-kill.sh (anchored kills as a file), log rotation before every stage start, parallel orchestrator probes, the listed-after-rent check
This commit is contained in:
parent
215384565d
commit
eab97643ee
3 changed files with 23 additions and 3 deletions
|
|
@ -27,9 +27,11 @@ def start(b, script, env=""):
|
|||
last_publish = 0; retried = set()
|
||||
while True:
|
||||
reg = fleet.load(); changed = False
|
||||
for iid, b in list(reg.items()):
|
||||
if b.get("state") in ("destroyed", "failed") or not b.get("ssh_ok") or b.get("phase") == "5": continue
|
||||
p = probe(b)
|
||||
live = [(iid, b) for iid, b in reg.items() if b.get("state") not in ("destroyed", "failed") and b.get("ssh_ok") and b.get("phase") != "5"]
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
with ThreadPoolExecutor(max_workers=16) as ex: probes = dict(zip([i for i, _ in live], ex.map(lambda x: probe(x[1]), live)))
|
||||
for iid, b in live:
|
||||
p = probes.get(iid)
|
||||
if p is None: continue
|
||||
stage = b.get("stage", "setup")
|
||||
if p["last"]: fleet.patch(iid, last_line=p["last"])
|
||||
|
|
|
|||
15
tools/fleet/box-kill.sh
Executable file
15
tools/fleet/box-kill.sh
Executable file
|
|
@ -0,0 +1,15 @@
|
|||
#!/usr/bin/env bash
|
||||
# Stops every stage process on a box (the matrix, Ember, the prover loop, their miners, workers, samplers and SP1
|
||||
# servers); never the node. Run as a FILE (bash in/box-kill.sh) so that no pattern below can match the shell that runs
|
||||
# it (an inline `pkill -f 'igneum-prove-host'` killed its own ssh shell on 6 October 2026, 12:38Z).
|
||||
for i in 1 2; do
|
||||
pkill -9 -f '^bash in/box-matrix.sh' ; pkill -9 -f '^bash in/box-ember.sh'; pkill -9 -f '^bash in/box-prover.sh'; pkill -9 -f '^python3 -u /root/fleet/in/box-prover.py'
|
||||
pkill -9 -x igneum-miner; pkill -9 -f '^/opt/igneum/pkg/bin/igneum-worker-cuda'; pkill -9 -x sp1-gpu-server; pkill -9 -f '^/opt/igneum-floor/(bin|target|prove)/.*igneum-prove-(host|export)' # pkill -x cannot match a name over 15 characters (igneum-worker-cuda, igneum-prove-host)
|
||||
pkill -9 -f '^nvidia-smi --query'
|
||||
sleep 2
|
||||
done
|
||||
rm -f /tmp/sp1-cuda-*.sock
|
||||
nvidia-smi -i 0 -rgc >/dev/null 2>&1
|
||||
d="$(nvidia-smi --query-gpu=power.default_limit --format=csv,noheader,nounits -i 0 | tr -d ' ' | cut -d. -f1)"; [ -n "$d" ] && nvidia-smi -i 0 -pl "$d" >/dev/null 2>&1
|
||||
cd /root/fleet/out 2>/dev/null && for f in matrix.log ember.log prover.log prover-launch.log rows.jsonl; do [ -f "$f" ] && mv "$f" "$f.$(date +%s).old"; done
|
||||
echo "count=$(pgrep -c -f '^bash in/box-|^python3 -u /root/fleet/in/box-prover')+$(pgrep -c -x igneum-miner)+$(pgrep -c -f '^/opt/igneum/pkg/bin/igneum-worker-cuda')+$(pgrep -c -x sp1-gpu-server)+$(pgrep -c -f '^/opt/igneum-floor/(bin|target|prove)/.*igneum-prove-host') mem=$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits -i 0 | tr -d ' ')"
|
||||
|
|
@ -124,6 +124,9 @@ def run(script, labels, env=""):
|
|||
for iid, b in boxes(labels).items():
|
||||
if scp(b, [os.path.join(HERE, script)], "/root/fleet/in/") != 0: print(b["label"], "scp failed"); continue
|
||||
name = os.path.basename(script)
|
||||
logname = {"box-matrix.sh": "matrix", "box-ember.sh": "ember", "box-prover.sh": "prover"}.get(name, name)
|
||||
# rotate the stage's log before the start: an old run's end line must never be read as this run's (the 5070, 12:33Z)
|
||||
ssh(b, f"cd /root/fleet/out 2>/dev/null && for f in {logname}.log {logname}-launch.log rows.jsonl; do [ -f $f ] && mv $f $f.$(date +%s).old; done; true", timeout=30)
|
||||
ssh(b, f"cd /root/fleet && chmod +x in/{name} && LABEL={b['label']} WALLET={b['wallet']} ARCHS={b['archs']} {env} setsid nohup in/{name} </dev/null >/dev/null 2>&1 & echo ok", timeout=60)
|
||||
print(b["label"], name, "started")
|
||||
|
||||
|
|
|
|||
Loading…
Reference in a new issue