infra/cloud-devnet: hcloud (doctl variant) create, builder-VM provision from a git-archive source tarball, systemd units for igneumd --devnet-suffix with a sparse --addpeer mesh and a CPU trickle miner per node, stdlib wRPC client, experiments (latency, partition, hop, collect, observer hookup), README with the command sequence and the Hetzner API prices of 3 Oct 2026. infra/gpu-bench: RunPod image recipes (CUDA 12.8, ROCm), bundle, run.sh (vectors gate, 10-min raw, sweep, inline shortcut ratio, nvcc/NVRTC/OpenCL recompile timings, results row, intake upload), bench-log template. infra/seed-nodes: create-seed (persistent IPv4, firewall), provision on the VM, health check, addPeer from the Mac over grpcurl, seeds.txt; igneum-seed-1 created at 188.245.5.161 (Hetzner cx23, fsn1). docs/plans/cloud-devnet.md and docs/plans/seed-nodes.md. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
141 lines
10 KiB
Bash
Executable file
141 lines
10 KiB
Bash
Executable file
#!/usr/bin/env bash
|
|
# Igneum GPU benchmark on a rented card. Runs on the pod, inside the unpacked bundle (make-bundle.sh). Clones nothing:
|
|
# the sources are in the bundle. Needs nvcc (CUDA 12.8 image) for NVIDIA, or an OpenCL ICD plus a C compiler for AMD.
|
|
#
|
|
# ./run.sh everything below, results in results-<stamp>/ and one row appended to results.md
|
|
# MINUTES=10 PACK=igneum-genesis-mh PACK2=igneum-hourly LABEL=rtx4090 UPLOAD=1 ./run.sh
|
|
#
|
|
# Steps (NVIDIA lane): build the memory-hard pack and the closed-form second pack with --serve support (host.cu has
|
|
# it), the inline-shortcut variant (make-inline.sh), the NVRTC timer; gate on the vectors (96/96 PASS or stop); raw
|
|
# bench for MINUTES at 1 GiB; sweep 64, 128, 256, 512, 1024 MiB; inline run; recompile timings (nvcc -cubin as the
|
|
# worker's prepare path does it, NVRTC in process, OpenCL build if an ICD is present); results row; upload.
|
|
# AMD lane (ROCm image): the same through proto-opencl/host.c, minus NVRTC and the inline variant (OpenCL packs have
|
|
# no inline kernel yet; the sed of make-inline.sh is CUDA-only).
|
|
set -uo pipefail
|
|
cd "$(dirname "$0")"
|
|
MINUTES="${MINUTES:-10}"; PACK="${PACK:-igneum-genesis-mh}"; PACK2="${PACK2:-igneum-hourly}"
|
|
LABEL="${LABEL:-}"; UPLOAD="${UPLOAD:-1}"; ITER="${ITER:-3}"
|
|
STAMP=$(date -u +%Y%m%d-%H%M%S); OUT="results-$STAMP"; mkdir -p "$OUT"
|
|
exec > >(tee "$OUT/run.log") 2>&1
|
|
log() { printf '%s %s\n' "$(date -u +%H:%M:%S)" "$*"; }
|
|
rate_of() { grep -E '^\s*(GPU|device)\b.*Mhash/s|^\s*rate' "$1" | grep -oE '[0-9]+\.[0-9]+ Mhash/s' | head -1 | cut -d' ' -f1; }
|
|
rate_wall() { grep -oE '[0-9]+\.[0-9]+ Mhash/s' "$1" | tail -1 | cut -d' ' -f1; }
|
|
now_ms() { python3 -c 'import time; print(int(time.time()*1000))'; }
|
|
|
|
PACKDIR="proto-cuda/packs/$PACK"; PACK2DIR="proto-cuda/packs/$PACK2"
|
|
[ -d "$PACKDIR" ] || { log "no $PACKDIR in the bundle"; exit 2; }
|
|
|
|
if command -v nvidia-smi >/dev/null 2>&1 && nvidia-smi -L >/dev/null 2>&1; then LANE=cuda; else LANE=opencl; fi
|
|
log "lane: $LANE; pack $PACK (memory-hard), second pack $PACK2; $MINUTES min raw bench"
|
|
|
|
# ---- device facts --------------------------------------------------------------------------------------------------
|
|
if [ "$LANE" = cuda ]; then
|
|
nvidia-smi --query-gpu=name,driver_version,memory.total,clocks.max.sm,clocks.max.mem --format=csv | tee "$OUT/device.txt"
|
|
nvcc --version | tail -2 | tee -a "$OUT/device.txt"
|
|
CARD=$(nvidia-smi --query-gpu=name --format=csv,noheader | head -1 | sed 's/NVIDIA //; s/GeForce //')
|
|
DRIVER=$(nvidia-smi --query-gpu=driver_version --format=csv,noheader | head -1)
|
|
CUDAV=$(nvcc --version | grep -oE 'release [0-9.]+' | cut -d' ' -f2)
|
|
else
|
|
clinfo 2>/dev/null | grep -E 'Device Name|Driver Version|Device Version|Global memory size|Max compute units' | head -12 | tee "$OUT/device.txt"
|
|
CARD=$(clinfo 2>/dev/null | grep -m1 'Device Name' | sed 's/.*Device Name *//'); DRIVER=$(clinfo 2>/dev/null | grep -m1 'Driver Version' | sed 's/.*Version *//'); CUDAV="opencl"
|
|
rocminfo 2>/dev/null | grep -m1 -E 'gfx[0-9]+' | tee -a "$OUT/device.txt" || true
|
|
fi
|
|
[ -n "$LABEL" ] || LABEL=$(printf '%s' "$CARD" | tr 'A-Z ' 'a-z-' | tr -cd 'a-z0-9-')
|
|
log "card: $CARD, driver $DRIVER, toolkit $CUDAV, label $LABEL"
|
|
|
|
# ---- build ---------------------------------------------------------------------------------------------------------
|
|
t0=$(now_ms)
|
|
if [ "$LANE" = cuda ]; then
|
|
ARCH="${CUDA_ARCH:-native}"
|
|
log "nvcc -arch=$ARCH: bench-$PACK, bench-$PACK2, bench-inline, nvrtc-time"
|
|
nvcc -O3 -std=c++17 -arch="$ARCH" -I "$PACKDIR" -o "bench-$PACK" proto-cuda/host.cu "$PACKDIR/kernel.cu" 2>&1 | tail -5 || { log "build of $PACK failed"; exit 2; }
|
|
nvcc -O3 -std=c++17 -arch="$ARCH" -I "$PACK2DIR" -o "bench-$PACK2" proto-cuda/host.cu "$PACK2DIR/kernel.cu" 2>&1 | tail -5 || log "build of $PACK2 failed (continuing)"
|
|
./make-inline.sh "$PACKDIR" "$OUT/inline-pack" && nvcc -O3 -std=c++17 -arch="$ARCH" -I "$OUT/inline-pack" -o bench-inline proto-cuda/host.cu "$OUT/inline-pack/kernel.cu" 2>&1 | tail -5 || log "inline variant did not build (continuing)"
|
|
nvcc -O2 -std=c++17 -o nvrtc-time nvrtc-time.cu -lnvrtc -lcuda 2>&1 | tail -3 || log "nvrtc-time did not build (continuing)"
|
|
WORKER="./bench-$PACK"
|
|
else
|
|
log "cc: bench-cl-$PACK (OpenCL at runtime), clbuild-time"
|
|
cc -std=c99 -O2 -I "$PACKDIR" -DIGNEUM_KERNEL_PATH="\"$PACKDIR/kernel.cl\"" -o "bench-cl-$PACK" proto-opencl/host.c -lOpenCL -ldl 2>&1 | tail -5 || { log "build failed"; exit 2; }
|
|
cc -std=c99 -O2 -I "$PACK2DIR" -DIGNEUM_KERNEL_PATH="\"$PACK2DIR/kernel.cl\"" -o "bench-cl-$PACK2" proto-opencl/host.c -lOpenCL -ldl 2>&1 | tail -5 || true
|
|
cc -std=c99 -O2 -o clbuild-time clbuild-time.c -lOpenCL 2>&1 | tail -3 || true
|
|
WORKER="./bench-cl-$PACK"
|
|
fi
|
|
BUILD_MS=$(( $(now_ms) - t0 )); log "build took $BUILD_MS ms"
|
|
# --serve support is in host.cu and host.c (ready line "ready cuda|opencl ... prepare N"); prove it answers:
|
|
printf 'quit\n' | timeout 120 "$WORKER" --serve 2>&1 | head -3 | tee "$OUT/serve-ready.txt" || true
|
|
|
|
# ---- gate: vectors -------------------------------------------------------------------------------------------------
|
|
log "gate: vectors on $PACK (3 batches)"
|
|
"$WORKER" --batches 3 > "$OUT/gate.txt" 2>&1
|
|
grep -E 'verify warp|cache check|dataset self-test|OVERALL' "$OUT/gate.txt"
|
|
if ! grep -q 'OVERALL: PASS' "$OUT/gate.txt"; then log "GATE FAIL: the card does not reproduce the Mac's vectors; stopping (send $OUT/gate.txt)"; VECTORS=FAIL; else VECTORS="96/96 PASS"; fi
|
|
[ "$VECTORS" = FAIL ] && exit 1
|
|
|
|
# ---- raw bench, MINUTES at 1 GiB -------------------------------------------------------------------------------------
|
|
r0=$(rate_of "$OUT/gate.txt"); [ -n "$r0" ] || r0=$(rate_wall "$OUT/gate.txt")
|
|
batches=$(python3 -c "import math; r=float('${r0:-50}'); print(max(5, min(100000, int(math.ceil($MINUTES*60*r*1e6/2**24)))))")
|
|
log "raw: $batches batches of 2^24 at about $r0 Mhash/s (about $MINUTES min)"
|
|
t0=$(now_ms); "$WORKER" --batches "$batches" > "$OUT/raw.txt" 2>&1; RAW_S=$(( ($(now_ms) - t0) / 1000 ))
|
|
RAW=$(rate_of "$OUT/raw.txt"); [ -n "$RAW" ] || RAW=$(rate_wall "$OUT/raw.txt")
|
|
grep -E 'Mhash/s|OVERALL' "$OUT/raw.txt" | head -4
|
|
log "raw: $RAW Mhash/s over $RAW_S s"
|
|
|
|
# ---- sweep ---------------------------------------------------------------------------------------------------------
|
|
SWEEP=""
|
|
for mib in 64 128 256 512 1024; do
|
|
"$WORKER" --dataset-mib "$mib" --batches 20 > "$OUT/sweep-$mib.txt" 2>&1
|
|
r=$(rate_of "$OUT/sweep-$mib.txt"); [ -n "$r" ] || r=$(rate_wall "$OUT/sweep-$mib.txt")
|
|
SWEEP="$SWEEP $mib:${r:-n/a}"; log "sweep $mib MiB: ${r:-n/a} Mhash/s"
|
|
done
|
|
R64=$(printf '%s' "$SWEEP" | grep -oE ' 64:[0-9.n/a]+' | cut -d: -f2); R1024=$(printf '%s' "$SWEEP" | grep -oE '1024:[0-9.n/a]+' | cut -d: -f2)
|
|
CLIFF=$(python3 -c "
|
|
try: print('%.1f' % (float('$R64') / float('$R1024')))
|
|
except Exception: print('n/a')")
|
|
|
|
# ---- second pack (closed form, for comparison with the Mac's and the 5090's tables) -----------------------------------
|
|
R2="n/a"
|
|
if [ -x "./bench-$PACK2" ] || [ -x "./bench-cl-$PACK2" ]; then
|
|
W2="./bench-$PACK2"; [ -x "$W2" ] || W2="./bench-cl-$PACK2"
|
|
"$W2" --batches 20 > "$OUT/pack2.txt" 2>&1; R2=$(rate_of "$OUT/pack2.txt"); [ -n "$R2" ] || R2=$(rate_wall "$OUT/pack2.txt")
|
|
log "$PACK2: $R2 Mhash/s, $(grep -c 'PASS' "$OUT/pack2.txt") PASS lines"
|
|
fi
|
|
|
|
# ---- inline shortcut (CUDA only) -------------------------------------------------------------------------------------
|
|
INLINE="n/a"; RATIO="n/a"
|
|
if [ -x ./bench-inline ]; then
|
|
# By construction the inline binary FAILS the dataset self-test (the dataset buffer holds the cache copy) and must
|
|
# PASS the vectors (mh_word recomputes the true words). Only the rate line and the vector lines count.
|
|
./bench-inline --batches 20 > "$OUT/inline.txt" 2>&1 || true
|
|
INLINE=$(rate_of "$OUT/inline.txt"); [ -n "$INLINE" ] || INLINE=$(rate_wall "$OUT/inline.txt")
|
|
IV=$(grep -c 'verify warp.*: PASS' "$OUT/inline.txt"); log "inline: ${INLINE:-n/a} Mhash/s, $IV vector PASS lines (6 expected)"
|
|
[ "$IV" -ge 3 ] || { log "inline vectors did not pass: ratio discarded"; INLINE="n/a(vectors)"; }
|
|
RATIO=$(python3 -c "
|
|
try: print('%.3f' % (float('$INLINE') / float('$RAW')))
|
|
except Exception: print('n/a')")
|
|
fi
|
|
|
|
# ---- recompile timings ---------------------------------------------------------------------------------------------
|
|
NVCC_MS="n/a"; NVRTC_MS="n/a"; CL_MS="n/a"
|
|
if [ "$LANE" = cuda ]; then
|
|
ARCHSM=$(nvidia-smi --query-gpu=compute_cap --format=csv,noheader | head -1 | tr -d '.')
|
|
xs=""; for i in $(seq 1 "$ITER"); do t0=$(now_ms); nvcc -cubin -O3 -std=c++17 -arch="sm_$ARCHSM" -allow-unsupported-compiler -I "$PACKDIR" -o "$OUT/kernel.cubin" "$PACKDIR/kernel.cu" >/dev/null 2>&1; xs="$xs $(( $(now_ms) - t0 ))"; done
|
|
NVCC_MS=$(python3 -c "xs=sorted(int(x) for x in '$xs'.split()); print(xs[len(xs)//2] if xs else 'n/a')"); log "nvcc -cubin (the worker's prepare path): $xs ms, median $NVCC_MS"
|
|
if [ -x ./nvrtc-time ]; then ./nvrtc-time "$PACKDIR" "$ITER" | tee "$OUT/nvrtc.txt"; NVRTC_MS=$(grep -oE 'median [0-9.]+' "$OUT/nvrtc.txt" | cut -d' ' -f2); fi
|
|
fi
|
|
if command -v clinfo >/dev/null 2>&1 && clinfo 2>/dev/null | grep -q 'Device Name'; then
|
|
[ -x ./clbuild-time ] || cc -std=c99 -O2 -o clbuild-time clbuild-time.c -lOpenCL 2>/dev/null || true
|
|
[ -x ./clbuild-time ] && { ./clbuild-time "$PACKDIR/kernel.cl" "$ITER" | tee "$OUT/clbuild.txt"; CL_MS=$(grep -oE 'median [0-9.]+' "$OUT/clbuild.txt" | cut -d' ' -f2); }
|
|
fi
|
|
|
|
# ---- results row -----------------------------------------------------------------------------------------------------
|
|
L2=$(nvidia-smi --query-gpu=name --format=csv,noheader 2>/dev/null | head -1 | grep -qE '5090' && echo 96 || echo "?")
|
|
ROW="| $CARD | $DRIVER / $CUDAV | $(date -u +%Y-%m-%d) | $RAW | $RAW_S | $R2 | $(printf '%s' "$SWEEP" | sed 's/^ //; s/ / \/ /g') | $CLIFF | $INLINE | $RATIO | $NVCC_MS | $NVRTC_MS | $CL_MS | $VECTORS | $BUILD_MS |"
|
|
HEAD="| card | driver / toolkit | date | Mhash/s at 1 GiB ($MINUTES min) | s | $PACK2 Mhash/s | sweep MiB:Mhash/s 64 / 128 / 256 / 512 / 1024 | 64 MiB over 1 GiB | inline Mhash/s | inline / honest | nvcc cubin ms | NVRTC ms | OpenCL build ms | vectors | build ms |"
|
|
[ -f results.md ] || { printf '%s\n|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|\n' "$HEAD" > results.md; }
|
|
printf '%s\n' "$ROW" >> results.md
|
|
printf '%s\n%s\n' "$HEAD" "$ROW" > "$OUT/row.md"
|
|
log "RESULT $ROW"
|
|
|
|
# ---- upload ----------------------------------------------------------------------------------------------------------
|
|
if [ "$UPLOAD" = 1 ]; then ./upload.sh "$OUT/run.log" "gpubench-$LABEL" "gpubench-$LABEL-$STAMP" || log "upload failed (the row is in results.md and $OUT/row.md)"; fi
|
|
log "done: $OUT/ (run.log, gate.txt, raw.txt, sweep-*.txt, inline.txt, nvrtc.txt, clbuild.txt, row.md)"
|