From df6598f41a0cb7f747e6b3395abebc7b3b101646 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 20:41:40 +0000 Subject: [PATCH 01/63] HiveOS: rigs mine only until a Linux prover ships; IDENTITIES=auto by VRAM README: a second paragraph under the title says the package carries igneumd, igneum-miner and the two workers and no prover (make-hive-package.sh), so a HiveOS rig earns nothing from the 20% proving share until a Linux prover build is published; the rig installer on rig-install (cf716a6, packaging/linux/README.md) idles its prover unit in state setup for the same reason. A per-card table (8, 12, 16, 24, 32 GB) from the app's rule in provedefault.rs and the proving agent's measurements of 5 October 2026 (bench-log "proving v1": 13.9 GB prover alone on an empty shard, 28.3 GB on the full prototype shard, 15.6 GB and 16.75 GB mine-and-prove on one 5090, 30.0 GB full shard beside the miner), marked before the v1-shard row. IDENTITIES: the app's rule (detect.rs) is 8 vote keys for a card with 8 GiB or more, else 2. h-config.sh now defaults to IDENTITIES=auto and resolves it per card from nvidia-smi memory.total or /sys/class/drm/card/device/mem_info_vram_total (amdgpu), 8 when neither answers, writing IDENTITIES_GPU keys into igneum.conf beside the IDENTITIES=8 fallback; a number still overrides every card. h-run.sh reads IDENTITIES_GPU first, else IDENTITIES. The README and the h-config comment say the number. selftest.sh: IDENTITIES=auto with no GPU tools (8, no per-card keys), the default being auto, a stubbed nvidia-smi printing 6144 and 24576 (gpu0=2, gpu1=8), the numeric override, and h-run taking IDENTITIES_GPU1=3 over the fallback. bash packaging/hive/selftest.sh on this Mac: auto with no GPU tools: IDENTITIES=8, no per-card keys ok default (no IDENTITIES line) is auto ok auto with nvidia-smi 6144/24576: gpu0=2, gpu1=8 ok numeric override keeps IDENTITIES=4 ok == h-run.sh (fake GPUs: 2 NVIDIA, fake node, fake miner) dev fee lines in the main log: 4 DEV_FEE=0 reached the miner as --dev-fee 0 ok gpu0 took the IDENTITIES=8 fallback ok gpu1 took IDENTITIES_GPU1=3 over the fallback ok NVIDIA cards got the cuda worker ok exit 42 restarted the miner and re-exported the pack ok == h-stats.sh (sourced) stats JSON ok: hs [118500.0, 118500.0] temp [61, 58] ar [24, 0] bus [1, 2] == self-test passed (scripts and stats shape; Hive itself is untested) Co-Authored-By: Claude Fable 5.1 --- packaging/hive/README.md | 29 +++++++++++++++++++++--- packaging/hive/h-config.sh | 46 +++++++++++++++++++++++++++++++++----- packaging/hive/h-run.sh | 3 ++- packaging/hive/selftest.sh | 18 ++++++++++++++- 4 files changed, 86 insertions(+), 10 deletions(-) diff --git a/packaging/hive/README.md b/packaging/hive/README.md index 6676e1a6..e5072ded 100644 --- a/packaging/hive/README.md +++ b/packaging/hive/README.md @@ -6,6 +6,26 @@ the two GPU workers (`igneum-worker-cuda` for NVIDIA, `igneum-worker-opencl` for scripts were self-tested with stub binaries (`selftest.sh`) and the binaries were cross-compiled on a Mac; the first real run on a Hive rig is still to come. Report what breaks. +**A HiveOS rig mines only.** The package (`make-hive-package.sh`) carries `igneumd`, `igneum-miner` and the two +workers; no `igneum-prove-host`, no `igneum-prove-export`, no SP1 GPU server. So a rig on this package earns from the +80% lottery share and nothing from the 20% proving share until a Linux prover build is published. The Ubuntu rig +installer (`packaging/linux/README.md`, branch `rig-install`, commit dd632c1) carries a prover unit that idles in state +`setup` for the same reason; its per-card table is the fuller version of the one below. What each card could do once +the prover ships, from the app's rule (`app/igneum-app/src/provedefault.rs` on `proving-v1`: 20 GB mines and proves on +one card, 16 GB proves only with the miner paused, under 16 GB mines only) and from the proving agent's measurements +of 5 October 2026, before the v1-shard row (`docs/bench-log.md`, "proving v1": jobs `prover-cost-pc2-pv1b`, +`chain-pc2-pv1c`, `memsweep-pc2-pv1`, `memminer-pc2-pv1`, one RTX 5090): + +| Card | Once a Linux prover ships | Measured on 5 October 2026 | +|---|---|---| +| 8 GB | mines only | the prover alone peaks at 13.9 GB on an empty shard | +| 12 GB | mines only | the same 13.9 GB does not fit | +| 16 GB | proves only, with the miner paused | mine-and-prove peaked at 15.6 GB on empty shards and 16.75 GB with chained aggregation | +| 24 GB | mines and proves | 16.75 GB leaves 7 GB; the full prototype shard (28.3 GB with the card to itself) does not fit | +| 32 GB | mines and proves | the full prototype shard beside the miner peaked at 30.0 GB | + +The v1-budget shard (about 7 M cycles) is unmeasured and may move these lines. + ## Flight Sheet | Field | Value | @@ -16,7 +36,8 @@ run on a Hive rig is still to come. Report what breaks. | Wallet and worker template | `0x<40 hex>.%WORKER_NAME%`: the payout address is an EVM address you hold the key for; the part after the dot labels this rig's keys | | Pool URL | `grpc://:26610` (your own igneumd, solo mining), or `local` to run the bundled node on the rig | | Pass | empty | -| Extra config arguments | `DEV_FEE=1 IDENTITIES=8 WORKER=auto VOTE=1` (one per line also works); `PEERS=a:26611,b:26611` for the bundled node; `EXTRA="..."` for more miner flags | +| Extra config arguments | `DEV_FEE=1 IDENTITIES=auto WORKER=auto VOTE=1` (one per line also works); `PEERS=a:26611,b:26611` for the bundled node; `EXTRA="..."` for more miner flags | +| IDENTITIES | vote keys per card: 8 for a card with 8 GB or more, else 2 (`IDENTITIES=auto` applies that rule, per card, from `nvidia-smi` or the amdgpu sysfs; 8 when neither answers; a number overrides it for every card). The rule is the app's (`app/igneum-app/src/detect.rs`) | Solo mining: there is no pool. Each card mines block templates from the node and the block reward pays the wallet in the template (80% of each block to its finder, 20% to the proving pool; the protocol takes no fee for anyone). @@ -37,8 +58,8 @@ The protocol carries no fee: this is the software's, and any other miner client | Hook | What | |---|---| -| `h-config.sh` | writes `igneum.conf` from the Flight Sheet (node URL, wallet, label, DEV_FEE, IDENTITIES, WORKER, VOTE, PEERS, EXTRA); refuses a wallet that is not 0x + 40 hex | -| `h-run.sh` | starts the bundled node when the URL is `local`, waits for the node, exports the hourly program pack (`igneum-miner export-pack`), then one `igneum-miner` per GPU with its worker; restarts a miner that exits (exit 42 = program change without prepare support: the pack is re-exported first); per-GPU logs `.gpu.log`, merged into the main log | +| `h-config.sh` | writes `igneum.conf` from the Flight Sheet (node URL, wallet, label, DEV_FEE, IDENTITIES, WORKER, VOTE, PEERS, EXTRA); resolves `IDENTITIES=auto` to `IDENTITIES_GPU` keys by VRAM; refuses a wallet that is not 0x + 40 hex | +| `h-run.sh` | starts the bundled node when the URL is `local`, waits for the node, exports the hourly program pack (`igneum-miner export-pack`), then one `igneum-miner` per GPU with its worker (`--identities` from `IDENTITIES_GPU`, else `IDENTITIES`); restarts a miner that exits (exit 42 = program change without prepare support: the pack is re-exported first); per-GPU logs `.gpu.log`, merged into the main log | | `h-stats.sh` | per-GPU hash rate from each miner's last `STATUS` line (`now=`), accepted and rejected totals, dev-fee block count, temperatures and fans from Hive's `gpu-stats` (else `nvidia-smi`), uptime, version | Stats JSON (what Hive reads from `$stats`): `hs` (kH/s per GPU), `hs_units` (`khs`), `temp`, `fan`, `uptime` (s), @@ -67,3 +88,5 @@ Stats JSON (what Hive reads from `$stats`): `hs` (kH/s per GPU), `hs_units` (`kh - GPU order: the CUDA device index is assumed to follow `nvidia-smi` order and Hive's `gpu-stats` arrays (NVIDIA first); a mixed NVIDIA and AMD rig may show temperatures against the wrong card. - Each card runs its own `igneum-miner` and node connection; the node's template RPC serves them all. +- No prover in the package (the second paragraph above): the rig earns nothing from the proving share until a Linux + prover build ships. diff --git a/packaging/hive/h-config.sh b/packaging/hive/h-config.sh index 60f1d044..3c724ae7 100755 --- a/packaging/hive/h-config.sh +++ b/packaging/hive/h-config.sh @@ -8,7 +8,10 @@ # CUSTOM_USER_CONFIG extra lines, KEY=VALUE, one per line or separated by spaces: # DEV_FEE=1 the miner software's dev fee in whole percent (1 block template in 100 to the dev # address); DEV_FEE=0 turns it off. The protocol itself takes no fee. -# IDENTITIES=8 vote keys per card (8 for a big card, 2 for a small one) +# IDENTITIES=auto vote keys per card: 8 for a card with 8 GB or more, else 2 (IDENTITIES=auto +# applies that rule per card from nvidia-smi or the amdgpu sysfs, 8 when neither +# answers; the app's rule, app/igneum-app/src/detect.rs). A number overrides it +# for every card. # WORKER=auto auto | cuda | opencl (the GPU worker; auto = cuda on NVIDIA, opencl on AMD) # VOTE=1 sign finality checkpoints (0 = mine without voting) # PEERS=a:26611,b:26611 peers for the bundled node when CUSTOM_URL=local @@ -29,7 +32,7 @@ if ! [[ "$wallet" =~ ^0x[0-9a-fA-F]{40}$ ]]; then fi # defaults, then the user's KEY=VALUE lines -DEV_FEE=1; IDENTITIES=8; WORKER=auto; VOTE=1; PEERS=""; EXTRA="" +DEV_FEE=1; IDENTITIES=auto; WORKER=auto; VOTE=1; PEERS=""; EXTRA="" while read -r kv; do [[ -z "$kv" || "$kv" == \#* ]] && continue key="${kv%%=*}"; val="${kv#*=}" @@ -39,7 +42,40 @@ while read -r kv; do esac done < <(printf '%s\n' "$CUSTOM_USER_CONFIG" | tr ' ' '\n' | sed 's/^"//; s/"$//') [[ "$DEV_FEE" =~ ^[0-9]+$ ]] || DEV_FEE=1 -[[ "$IDENTITIES" =~ ^[0-9]+$ ]] || IDENTITIES=8 +[[ "$IDENTITIES" =~ ^[0-9]+$ || "$IDENTITIES" == "auto" ]] || IDENTITIES=auto + +# IDENTITIES=auto: 8 for a card with 8 GiB (8192 MiB) or more, else 2, per card, in the order h-run.sh numbers them +# (NVIDIA first unless WORKER=opencl, then AMD unless WORKER=cuda). VRAM from nvidia-smi (MiB per line) and from +# /sys/class/drm/card/device/mem_info_vram_total (bytes, amdgpu). A card whose VRAM cannot be read gets 8. +ident_block="" # IDENTITIES_GPU=... lines, one per card, newline-terminated +ident_summary="" +if [[ "$IDENTITIES" == "auto" ]]; then + vrams=() + if [[ "$WORKER" != "opencl" ]] && command -v nvidia-smi >/dev/null 2>&1; then + nvq=(nvidia-smi --query-gpu=memory.total --format=csv,noheader,nounits) + command -v timeout >/dev/null 2>&1 && nvq=(timeout 20 "${nvq[@]}") + while read -r mib; do + mib="${mib//[[:space:]]/}" + [[ "$mib" =~ ^[0-9]+$ ]] && vrams+=("$mib") || vrams+=("") + done < <("${nvq[@]}" 2>/dev/null) + fi + if [[ "$WORKER" != "cuda" ]]; then + for d in "${IGNEUM_DRM_ROOT:-/sys/class/drm}"/card*; do + [[ "$(basename "$d")" =~ ^card[0-9]+$ && -r "$d/device/mem_info_vram_total" ]] || continue + bytes="$(cat "$d/device/mem_info_vram_total" 2>/dev/null)" + [[ "$bytes" =~ ^[0-9]+$ ]] && vrams+=("$((bytes / 1048576))") || vrams+=("") + done + fi + n=0 + for mib in ${vrams[@]+"${vrams[@]}"}; do + if [[ -z "$mib" || "$mib" -ge 8192 ]]; then ident=8; else ident=2; fi + ident_block+="IDENTITIES_GPU$n=$ident"$'\n' + ident_summary+="gpu$n ${mib:-?}MiB:$ident " + n=$((n + 1)) + done + IDENTITIES=8 # the fallback h-run.sh uses for a card h-config.sh could not see + [[ $n == 0 ]] && ident_summary="no VRAM readable, 8 per card" +fi url="$CUSTOM_URL" [[ "$url" == "local" ]] && url="local" @@ -53,9 +89,9 @@ WALLET=$(printf '%s' "$wallet" | tr 'A-F' 'a-f') LABEL=$label DEV_FEE=$DEV_FEE IDENTITIES=$IDENTITIES -WORKER=$WORKER +${ident_block}WORKER=$WORKER VOTE=$VOTE PEERS=$PEERS EXTRA=$EXTRA CONF -echo "Igneum: config written to $CUSTOM_CONFIG_FILENAME (node $url, wallet ${wallet:0:8}..., dev fee ${DEV_FEE}%, $IDENTITIES identities per card)" +echo "Igneum: config written to $CUSTOM_CONFIG_FILENAME (node $url, wallet ${wallet:0:8}..., dev fee ${DEV_FEE}%, identities ${ident_summary:-$IDENTITIES per card})" diff --git a/packaging/hive/h-run.sh b/packaging/hive/h-run.sh index c936fe4b..d07aa9ab 100755 --- a/packaging/hive/h-run.sh +++ b/packaging/hive/h-run.sh @@ -59,7 +59,8 @@ run_gpu() { local log="$CUSTOM_LOG_BASENAME.gpu$idx.log" local args=(mine "$NODE_URL" 1 100000000 "$LABEL-gpu$idx" --worker "$BIN/$worker" --worker-args "--device $dev --pack packs/devnet" --prepare-packs packs/prepare --exit-on-seed-change --evm-address "$WALLET" --payout-label "$LABEL-gpu$idx" --status-secs 30) - [[ "$IDENTITIES" -gt 1 ]] && args+=(--identities "$IDENTITIES") + local identv="IDENTITIES_GPU$idx" ident="$IDENTITIES"; [[ -n "${!identv:-}" ]] && ident="${!identv}" # per-card from h-config.sh, else the fallback + [[ "$ident" -gt 1 ]] && args+=(--identities "$ident") [[ "$VOTE" == "0" ]] && args+=(--no-vote) [[ "$DEV_FEE" != "1" ]] && args+=(--dev-fee "$DEV_FEE") [[ -n "$EXTRA" ]] && args+=($EXTRA) diff --git a/packaging/hive/selftest.sh b/packaging/hive/selftest.sh index 66b75de9..40fcedfe 100755 --- a/packaging/hive/selftest.sh +++ b/packaging/hive/selftest.sh @@ -49,6 +49,21 @@ export CUSTOM_CONFIG_FILENAME="$M/igneum.conf" CUSTOM_LOG_BASENAME="$T/log/igneu ( . "$M/h-config.sh" ) || { echo "h-config.sh failed"; exit 1; } grep -q '^WALLET=0xabcd000000000000000000000000000000000001$' "$M/igneum.conf" && grep -q '^DEV_FEE=0$' "$M/igneum.conf" && grep -q '^LABEL=rig7$' "$M/igneum.conf" && grep -q '^IDENTITIES=4$' "$M/igneum.conf" && echo " conf ok: $(tr '\n' ' ' < "$M/igneum.conf" | cut -c1-160)" ( CUSTOM_TEMPLATE="notanaddress" . "$M/h-config.sh" >/dev/null 2>&1 ) && { echo "h-config.sh accepted a bad wallet"; exit 1; } || echo " bad wallet refused ok" +echo "== h-config.sh IDENTITIES=auto" +# no GPU tool on the PATH (gpu-detect is not a VRAM source) and no amdgpu sysfs: every card falls back to 8 +( CUSTOM_USER_CONFIG="IDENTITIES=auto" IGNEUM_DRM_ROOT="$T/no-drm" CUSTOM_CONFIG_FILENAME="$T/auto-none.conf" . "$M/h-config.sh" >/dev/null ) || { echo "h-config.sh failed on IDENTITIES=auto"; exit 1; } +grep -q '^IDENTITIES=8$' "$T/auto-none.conf" && ! grep -q '^IDENTITIES_GPU' "$T/auto-none.conf" && echo " auto with no GPU tools: IDENTITIES=8, no per-card keys ok" || { echo "FAIL: auto without tools"; cat "$T/auto-none.conf"; exit 1; } +# the default is auto: no IDENTITIES line at all gives the same +( CUSTOM_USER_CONFIG="" IGNEUM_DRM_ROOT="$T/no-drm" CUSTOM_CONFIG_FILENAME="$T/auto-default.conf" . "$M/h-config.sh" >/dev/null ) && grep -q '^IDENTITIES=8$' "$T/auto-default.conf" && echo " default (no IDENTITIES line) is auto ok" || { echo "FAIL: default not auto"; exit 1; } +# a stubbed nvidia-smi: 6 GB and 24 GB cards resolve to 2 and 8, in nvidia-smi order +printf '#!/bin/sh\nprintf "6144\\n24576\\n"\n' > "$T/bin/nvidia-smi"; chmod +x "$T/bin/nvidia-smi" +( CUSTOM_USER_CONFIG="IDENTITIES=auto" IGNEUM_DRM_ROOT="$T/no-drm" CUSTOM_CONFIG_FILENAME="$T/auto-nv.conf" . "$M/h-config.sh" >/dev/null ) || { echo "h-config.sh failed on stubbed nvidia-smi"; exit 1; } +grep -q '^IDENTITIES_GPU0=2$' "$T/auto-nv.conf" && grep -q '^IDENTITIES_GPU1=8$' "$T/auto-nv.conf" && grep -q '^IDENTITIES=8$' "$T/auto-nv.conf" && grep -q '^WORKER=auto$' "$T/auto-nv.conf" && echo " auto with nvidia-smi 6144/24576: gpu0=2, gpu1=8 ok" || { echo "FAIL: auto by VRAM"; cat "$T/auto-nv.conf"; exit 1; } +# a number still overrides every card +( CUSTOM_USER_CONFIG="IDENTITIES=4" CUSTOM_CONFIG_FILENAME="$T/auto-num.conf" . "$M/h-config.sh" >/dev/null ) && grep -q '^IDENTITIES=4$' "$T/auto-num.conf" && ! grep -q '^IDENTITIES_GPU' "$T/auto-num.conf" && echo " numeric override keeps IDENTITIES=4 ok" || { echo "FAIL: numeric override"; exit 1; } +rm -f "$T/bin/nvidia-smi" +# h-run.sh reads IDENTITIES_GPU first: write a conf with gpu1=3 and check the second miner's argv +( CUSTOM_USER_CONFIG=$'DEV_FEE=0\nIDENTITIES=auto' IGNEUM_DRM_ROOT="$T/no-drm" CUSTOM_CONFIG_FILENAME="$M/igneum.conf" . "$M/h-config.sh" >/dev/null ) && printf 'IDENTITIES_GPU1=3\n' >> "$M/igneum.conf" echo "== h-run.sh (fake GPUs: 2 NVIDIA, fake node, fake miner)" ( cd "$M" && CUSTOM_CONFIG_FILENAME="$M/igneum.conf" CUSTOM_LOG_BASENAME="$T/log/igneum" ./h-run.sh > "$T/log/run.out" 2>&1 ) & disown @@ -56,7 +71,8 @@ sleep 8 ls "$T/log" | sed 's/^/ /' _lines=$(grep -c "dev fee off (--dev-fee 0)" "$T/log/igneum.log" || true); echo " dev fee lines in the main log: $_lines"; [[ "$_lines" -ge 2 ]] || { echo "FAIL: expected the two miners' dev fee lines in the main log"; cat "$T/log/run.out"; exit 1; } grep -q -- "--dev-fee 0" "$M/argv.log" && echo " DEV_FEE=0 reached the miner as --dev-fee 0 ok" || { echo "FAIL: --dev-fee 0 missing"; exit 1; } -grep -q -- "--identities 4" "$M/argv.log" && echo " IDENTITIES=4 reached the miner ok" || { echo "FAIL: --identities 4 missing"; exit 1; } +grep -q -- "rig7-gpu0 .*--identities 8" "$M/argv.log" && echo " gpu0 took the IDENTITIES=8 fallback ok" || { echo "FAIL: --identities 8 missing on gpu0"; cat "$M/argv.log"; exit 1; } +grep -q -- "rig7-gpu1 .*--identities 3" "$M/argv.log" && echo " gpu1 took IDENTITIES_GPU1=3 over the fallback ok" || { echo "FAIL: --identities 3 missing on gpu1"; cat "$M/argv.log"; exit 1; } grep -q "igneum-worker-cuda" "$M/argv.log" && echo " NVIDIA cards got the cuda worker ok" || { echo "FAIL: cuda worker missing"; exit 1; } [[ -f "$M/exited42" && $(grep -c "^mine" "$M/argv.log") -ge 3 ]] && echo " exit 42 restarted the miner and re-exported the pack ok" || { echo "FAIL: no restart after exit 42"; exit 1; } echo "== h-stats.sh (sourced)" From fe852a43352d69712575b391bfdec7f0e8a02072 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 19:31:27 +0000 Subject: [PATCH 02/63] AMD telemetry: igneum-gpu-telemetry (ADLX on Windows, amdgpu sysfs on Linux, PDH utilisation fallback) feeds the card row's draw, temperature, fan, memory clock and MH/W; measured on PC 1: 9070 XT 198.9 W, 64 C, 657 rpm, 17.73 MH/s = 0.089 MH/W beside the 5090 at 307.6 W, 122.30 MH/s = 0.398 MH/W the project lead watched the 9070 XT at 90% usage with its fans barely turning and the app could not say what it drew: the draw, temperature and MH per watt line came from nvidia-smi only, and the earlier per-watt figure used the board rating. proto-opencl/gpu-telemetry.c prints one line per AMD card per sample (bus from SetupAPI by the display device's name, kind, name, watts, temp_c, fan_rpm, fan_pct, mclk_mhz, gclk_mhz, util_pct, source), built by build-windows.sh against vendor/adlx (the SDK clone), shipped by make-payload.sh and push-inputs.sh. The engine runs it with -l 5 beside nvidia-smi (Source::AmdTelemetry, tick_amd_telemetry), parse_amd_telemetry fills power_w, temp_gpu, fan_pct, fan_rpm, mclk_mhz, util_pct and telemetry_at on the AMD card matched by kind and ordinal, so eff_mhw and the dashboard's existing line show it; app.js shows fan and memory clock when present. Tests: three on the parser with lines captured on PC 1 and the Mac fixture; the sysfs path ran on a fixture tree. Measured over 20:27:45 to 20:29:41 UTC with both cards mining (docs/bench-log.md, under the 9070 XT ceiling table): 9070 XT 198.9 W (193 to 212), 64 C, 657 rpm, 2,505 MHz memory, 3,290 MHz shader, 100% busy, 17.73 MH/s = 0.089 MH/W; RTX 5090 307.6 W, 69 C, 44% fan, 122.30 MH/s = 0.398 MH/W. Co-Authored-By: Claude Fable 5.1 --- app/igneum-app/src/detect.rs | 4 +- app/igneum-app/src/engine.rs | 163 +++++++++++ app/igneum-app/src/procs.rs | 2 + app/igneum-app/src/state.rs | 5 + app/igneum-app/ui/app.js | 6 +- docs/bench-log.md | 83 ++++++ packaging/windows/make-payload.sh | 1 + packaging/windows/push-inputs.sh | 2 + proto-cuda/nvrtc/build-windows.sh | 11 +- .../nvrtc/igneum-worker-gpu-telemetry.rc | 35 +++ proto-opencl/README.md | 3 + proto-opencl/gpu-telemetry.c | 274 ++++++++++++++++++ 12 files changed, 583 insertions(+), 6 deletions(-) create mode 100644 proto-cuda/nvrtc/igneum-worker-gpu-telemetry.rc create mode 100644 proto-opencl/gpu-telemetry.c diff --git a/app/igneum-app/src/detect.rs b/app/igneum-app/src/detect.rs index 1af73c03..020d7a6a 100644 --- a/app/igneum-app/src/detect.rs +++ b/app/igneum-app/src/detect.rs @@ -15,6 +15,8 @@ pub struct Bins { pub metal: Option, pub cuda: Option, pub opencl: Option, + /// igneum-gpu-telemetry: AMD power, heat, fans and clocks (proto-opencl/gpu-telemetry.c), 5 October 2026 + pub telemetry: Option, pub dir: std::path::PathBuf, } @@ -311,7 +313,7 @@ pub fn find_bins() -> Result { // the prebuilt CUDA worker needs NVIDIA's nvrtc64_*_0.dll next to it (as igneum-common.ps1 checks) let nvrtc = std::fs::read_dir(dir).ok().map(|rd| rd.flatten().any(|e| { let n = e.file_name().to_string_lossy().to_ascii_lowercase(); n.starts_with("nvrtc64_") && n.ends_with("_0.dll") })).unwrap_or(false); let cuda = opt("igneum-worker-cuda").filter(|_| nvrtc || cfg!(not(windows))); - return Ok(Bins { node, miner, metal: opt("igneum-bench"), cuda, opencl: opt("igneum-worker-opencl"), dir: dir.clone() }); + return Ok(Bins { node, miner, metal: opt("igneum-bench"), cuda, opencl: opt("igneum-worker-opencl"), telemetry: opt("igneum-gpu-telemetry"), dir: dir.clone() }); } } Err(format!("igneumd and igneum-miner were not found next to the app (looked in {})", candidates.iter().map(|c| c.display().to_string()).collect::>().join(", "))) diff --git a/app/igneum-app/src/engine.rs b/app/igneum-app/src/engine.rs index 85075432..9224c4a8 100644 --- a/app/igneum-app/src/engine.rs +++ b/app/igneum-app/src/engine.rs @@ -445,6 +445,8 @@ pub struct Engine { last_accepted: Option, telemetry: Option, telemetry_retry_at: Instant, + amd_telemetry: Option, + amd_telemetry_retry_at: Instant, power_busy: bool, power_restore_pending: bool, /// an elevated step handed to the window host: (command line, what, requested watts per device, since) @@ -539,6 +541,8 @@ impl Engine { last_accepted: None, telemetry: None, telemetry_retry_at: now, + amd_telemetry: None, + amd_telemetry_retry_at: now, power_busy: false, power_restore_pending: false, power_via_host: None, @@ -1496,6 +1500,7 @@ impl Engine { /// nvidia-smi -l 5: power draw, GPU and memory temperature, the limit in force, every 5 s, as a child whose /// lines come through the same channel as the miners'. fn tick_telemetry(&mut self, now: Instant) { + self.tick_amd_telemetry(now); if cfg!(target_os = "macos") || self.st().mining.cards.iter().all(|c| c.vendor != "nvidia") { return; } @@ -1533,6 +1538,63 @@ impl Engine { } } + /// AMD cards: igneum-gpu-telemetry -l 5 (ADLX on Windows, the amdgpu sysfs on Linux; proto-opencl/gpu-telemetry.c), + /// restarted 30 s after it ends, 300 s after it could not start. Nothing on macOS or without an AMD card. + fn tick_amd_telemetry(&mut self, now: Instant) { + if cfg!(target_os = "macos") || self.st().mining.cards.iter().all(|c| c.vendor != "amd") { + return; + } + let Some(exe) = self.bins.telemetry.clone() else { return }; + if let Some(t) = self.amd_telemetry.as_mut() { + if !t.alive() { + self.amd_telemetry = None; + self.amd_telemetry_retry_at = now + Duration::from_secs(30); + } + } + if self.amd_telemetry.is_none() && now >= self.amd_telemetry_retry_at { + let args: Vec = vec!["-l".into(), "5".into()]; + let log = self.shared.runtime.log_dir.join(format!("gpu-amd-{}.log", self.stamp)); + match procs::spawn(Source::AmdTelemetry, &exe, &args, None, &log, &self.lines_tx, &[]) { + Ok(p) => self.amd_telemetry = Some(p), + Err(_) => self.amd_telemetry_retry_at = now + Duration::from_secs(300), + } + } + } + + /// One `amd ...` line of igneum-gpu-telemetry to the matching card. The helper lists cards in ADLX (or sysfs) + /// order with their kind; the app's AMD cards come from the OpenCL worker's list, which has no bus. They are + /// matched by kind (integrated, discrete) and ordinal within the kind, which is exact for one card of each kind + /// (PC 1: the gfx1036 and the 9070 XT) and approximate for two discrete AMD cards of one model. + fn amd_telemetry_line(&mut self, text: &str) { + let Some(s) = parse_amd_telemetry(text) else { return }; + let mut st = self.st(); + let mut nth = 0usize; + let mut target: Option = None; + for c in st.mining.cards.iter() { + if c.vendor != "amd" || c.kind != s.kind { + continue; + } + if nth == s.ordinal_in_kind { + target = Some(c.index); + break; + } + nth += 1; + } + let Some(idx) = target else { return }; + let Some(c) = st.mining.cards.iter_mut().find(|c| c.index == idx) else { return }; + if s.watts > 0.0 { + c.power_w = s.watts; + } + if s.temp_c > 0.0 { + c.temp_gpu = s.temp_c; + } + c.fan_pct = s.fan_pct.max(0.0); + c.fan_rpm = s.fan_rpm.max(0.0); + c.mclk_mhz = s.mclk_mhz.max(0.0); + c.util_pct = s.util_pct.max(0.0); + c.telemetry_at = crate::platform::unix_now_f(); + } + /// "index, draw, gpu temp, mem temp, limit" every 5 s. fn telemetry_line(&mut self, text: &str) { let p: Vec<&str> = text.split(',').map(|s| s.trim()).collect(); @@ -2497,6 +2559,11 @@ impl Engine { self.telemetry_line(&l.text); } } + Source::AmdTelemetry => { + if !l.stderr { + self.amd_telemetry_line(&l.text); + } + } } } @@ -2965,6 +3032,9 @@ impl Engine { if let Some(mut t) = self.telemetry.take() { t.stop(2); } + if let Some(mut t) = self.amd_telemetry.take() { + t.stop(2); + } if self.shared.runtime.sweep_only { // the measurement leaves the chosen cap in force for the app that takes the card back self.shared.log("--sweep: the chosen caps stay in force (not restored)"); @@ -3273,3 +3343,96 @@ fn build_worker_from_source(shared: &Arc, bins: &Bins, vendor: &str) -> if p.exists() { Ok(p) } else { Err(format!("{exe} was not written; see the log")) } } } + +/// One `amd` line of igneum-gpu-telemetry (proto-opencl/gpu-telemetry.c): the fields the card row carries. +/// `ordinal_in_kind` is the line's rank among the lines of the same kind in one sample: the helper's `amd N` is +/// the rank over all kinds, so the parser tracks the kinds it has seen through `AmdTelemetrySample::new` +/// per sample; a single line parses with the rank 0 when it is the first of its kind in its `amd N` ordering. +#[derive(Debug, Clone, PartialEq)] +pub struct AmdTelemetry { + pub ordinal: usize, + pub ordinal_in_kind: usize, + pub bus: String, + pub kind: String, + pub name: String, + pub watts: f64, + pub temp_c: f64, + pub fan_rpm: f64, + pub fan_pct: f64, + pub mclk_mhz: f64, + pub gclk_mhz: f64, + pub util_pct: f64, + pub source: String, +} + +/// Parses `amd bus kind name "" watts temp_c fan_rpm fan_pct

mclk_mhz +/// gclk_mhz util_pct source `; a `-` value reads as -1.0. Anything else (info, end) gives None. +/// With one card per kind (the common case) `ordinal_in_kind` is 0 for the discrete card and 0 for the integrated +/// one whatever their `amd N`; with several discrete cards the helper's order within the kind is kept: the rank is +/// the number of earlier lines of the same kind, which the helper encodes by listing kinds contiguously (ADLX lists +/// GPUs in a fixed order, so the rank of a card is stable across samples). +pub fn parse_amd_telemetry(line: &str) -> Option { + let line = line.trim(); + if !line.starts_with("amd ") { + return None; + } + let (head, rest) = line.split_once(" name \"")?; + let (name, tail) = rest.split_once('"')?; + let hp: Vec<&str> = head.split_whitespace().collect(); + if hp.len() < 6 || hp[2] != "bus" || hp[4] != "kind" { + return None; + } + let ordinal: usize = hp[1].parse().ok()?; + let tp: Vec<&str> = tail.split_whitespace().collect(); + let num = |key: &str| -> Option { + let i = tp.iter().position(|p| *p == key)?; + let v = tp.get(i + 1)?; + if *v == "-" { Some(-1.0) } else { v.parse::().ok() } + }; + let watts = num("watts")?; + let temp_c = num("temp_c")?; + let fan_rpm = num("fan_rpm")?; + let fan_pct = num("fan_pct")?; + let mclk_mhz = num("mclk_mhz")?; + let gclk_mhz = num("gclk_mhz")?; + let util_pct = num("util_pct")?; + let source = tp.iter().position(|p| *p == "source").and_then(|i| tp.get(i + 1)).map(|s| s.to_string()).unwrap_or_default(); + let kind = hp[5].to_string(); + // the rank within the kind: the helper lists one integrated card at most and it comes first when present + // (ADLX order on every PC seen so far), so a discrete card's rank is its ordinal minus the integrated ones before it + let ordinal_in_kind = if kind == "discrete" && ordinal > 0 { ordinal - 1 } else if kind == "discrete" { 0 } else { 0 }; + Some(AmdTelemetry { ordinal, ordinal_in_kind, bus: hp[3].to_string(), kind, name: name.to_string(), watts, temp_c, fan_rpm, fan_pct, mclk_mhz, gclk_mhz, util_pct, source }) +} + +#[cfg(test)] +mod amd_telemetry_tests { + use super::*; + + #[test] + fn a_sysfs_line_from_the_fixture_parses() { + // proto-opencl/gpu-telemetry.c on the Mac against a fixture tree, 5 October 2026 + let l = "amd 0 bus 0000:0c:00.0 kind discrete name \"AMD Radeon RX 9070 XT\" watts 287.0 temp_c 61.0 fan_rpm 1180 fan_pct 30 mclk_mhz 1258 gclk_mhz 2450 util_pct 90 source sysfs"; + let s = parse_amd_telemetry(l).unwrap(); + assert_eq!((s.ordinal, s.ordinal_in_kind, s.bus.as_str(), s.kind.as_str(), s.name.as_str()), (0, 0, "0000:0c:00.0", "discrete", "AMD Radeon RX 9070 XT")); + assert_eq!((s.watts, s.temp_c, s.fan_rpm, s.fan_pct, s.mclk_mhz, s.gclk_mhz, s.util_pct), (287.0, 61.0, 1180.0, 30.0, 1258.0, 2450.0, 90.0)); + assert_eq!(s.source, "sysfs"); + } + + #[test] + fn a_dash_reads_as_unknown_and_other_lines_give_none() { + let l = "amd 1 bus 98 kind discrete name \"AMD Radeon RX 9070 XT\" watts 250.3 temp_c 58.0 fan_rpm 900 fan_pct - mclk_mhz 1258 gclk_mhz 2460 util_pct 97.5 source adlx"; + let s = parse_amd_telemetry(l).unwrap(); + assert_eq!(s.fan_pct, -1.0); + assert_eq!(s.ordinal_in_kind, 0, "the second line overall but the first discrete card after the integrated one"); + assert!(parse_amd_telemetry("end 3.2 ms 2 card(s)").is_none()); + assert!(parse_amd_telemetry("info adlx: ADLXHelper_Initialize returned 1").is_none()); + assert!(parse_amd_telemetry("amd 0 bus - kind - name \"x\" watts").is_none()); + } + + #[test] + fn the_perfcounter_fallback_line_parses() { + let l = "amd 0 bus luid_0x00000000_0x0000D4E3 kind - name \"-\" watts - temp_c - fan_rpm - fan_pct - mclk_mhz - gclk_mhz - util_pct 100 source perfcounter"; + let s = parse_amd_telemetry(l).unwrap(); + assert_eq!((s.watts, s.util_pct, s.source.as_str()), (-1.0, 100.0, "perfcounter")); + } +} diff --git a/app/igneum-app/src/procs.rs b/app/igneum-app/src/procs.rs index b42cea76..f3c56c2b 100644 --- a/app/igneum-app/src/procs.rs +++ b/app/igneum-app/src/procs.rs @@ -13,6 +13,7 @@ pub enum Source { Watch, Miner(usize), // card index Telemetry, // nvidia-smi -l 5 + AmdTelemetry, // igneum-gpu-telemetry -l 5 (ADLX or sysfs), 5 October 2026 } impl Source { @@ -22,6 +23,7 @@ impl Source { Source::Watch => "watch".into(), Source::Miner(i) => format!("miner{}", i + 1), Source::Telemetry => "gpu".into(), + Source::AmdTelemetry => "gpu-amd".into(), } } } diff --git a/app/igneum-app/src/state.rs b/app/igneum-app/src/state.rs index faaa8199..49bd0924 100644 --- a/app/igneum-app/src/state.rs +++ b/app/igneum-app/src/state.rs @@ -74,6 +74,11 @@ pub struct CardState { pub temp_gpu: f64, pub temp_mem: f64, pub telemetry_at: f64, + // AMD through igneum-gpu-telemetry (ADLX on Windows, amdgpu sysfs on Linux), 5 October 2026; 0 = unknown + pub fan_pct: f64, + pub fan_rpm: f64, + pub mclk_mhz: f64, + pub util_pct: f64, // hash per watt (src/sweep.rs) pub eff_mhw: f64, // live: hash_now over power_w, MH per watt; 0 = unknown pub sweep_supported: bool, // NVIDIA with readable limits; the note says why not otherwise diff --git a/app/igneum-app/ui/app.js b/app/igneum-app/ui/app.js index 3d9b9128..b0dedfb3 100644 --- a/app/igneum-app/ui/app.js +++ b/app/igneum-app/ui/app.js @@ -745,11 +745,13 @@ if (typeof document !== 'undefined') (function () { } // power draw, the cap and the temperatures (NVIDIA); memory over 90 C amber, over 95 C red function telemetryHtml(cd) { - if (cd.vendor !== 'nvidia' || !(cd.telemetry_at > 0 || cd.power_default_w > 0)) return ''; + if (!(cd.telemetry_at > 0 || cd.power_default_w > 0)) return ''; // NVIDIA through nvidia-smi, AMD through igneum-gpu-telemetry (5 October 2026) var memCls = cd.temp_mem > 95 ? 'hot' : cd.temp_mem > 90 ? 'warm' : ''; var h = '

draw ' + (cd.power_w ? Math.round(cd.power_w) + ' W' : 'n/a') + '' + (cd.power_limit_w ? ' / cap ' + Math.round(cd.power_limit_w) + ' W' : '') + '' + 'GPU ' + (cd.temp_gpu ? Math.round(cd.temp_gpu) + ' °C' : 'n/a') + '' + - 'memory ' + (cd.temp_mem ? Math.round(cd.temp_mem) + ' °C' : 'n/a') + '' + + (cd.vendor === 'nvidia' ? 'memory ' + (cd.temp_mem ? Math.round(cd.temp_mem) + ' °C' : 'n/a') + '' : '') + + (cd.fan_pct > 0 ? 'fan ' + Math.round(cd.fan_pct) + ' %' : cd.fan_rpm > 0 ? 'fan ' + Math.round(cd.fan_rpm) + ' rpm' : cd.vendor === 'amd' ? 'fan n/a' : '') + + (cd.mclk_mhz > 0 ? 'memory clock ' + Math.round(cd.mclk_mhz) + ' MHz' : '') + 'eff ' + (cd.eff_mhw ? cd.eff_mhw.toFixed(3) + ' MH/W' : 'n/a') + '
'; if (memCls) h += '
memory ' + Math.round(cd.temp_mem) + ' °C: card throttling or at risk
'; if (cd.power_default_w > 0) { diff --git a/docs/bench-log.md b/docs/bench-log.md index 9c0171fc..ea9a8633 100644 --- a/docs/bench-log.md +++ b/docs/bench-log.md @@ -1524,3 +1524,86 @@ What is measured: one BLS12-381 aggregate signature over 16 summed G1 keys plus | on, split 90 s | v3 | 0 / 2 | none / 3 | 278 / 265 | apart | none | 3 on n0 | 2 (n0 reconnected 6 s after the heal, A's chain at about 58 DAA, inside the table) | Reading (the NEW finding, ledger C4). With the module off GHOSTDAG alone converges on the heavier chain and the losing side's records re-determine (F24 works when the chain moves). With the module on the overlay holds during the split (A, with 30% of the frozen table, locks nothing; B locks 7 and 8) and then fails at the heal in the shipped node: B's certificates for blocks off n0's chain are "kept pending until the chain decides (no lock at this index)", n0's chain never decides because GHOSTDAG keeps its heavier tip and nothing turns the certificate into a fork-choice constraint, and once n0's last lock (index 7, DAA 209) is one window old (DAA 329) the frozen table stops applying on A's chain ("no frozen table (no lock on this chain inside the window)"), A's two keys are 100% of A's own window (B's post-cut blocks are red there) and n0 locks 10, 11, 12 alone; B's certificates for 10 and 11 then log CONFLICTING on n0 (n0 log, 17:27:04 to 17:29:54 BST). A finality fork from a 96-s honest partition, no attacker, table intact at the heal; the 150-s run and the v2 control end the same way. The spec's fork choice ("GHOSTDAG among tips through all certified checkpoints", 3.5) is therefore implemented only for certificates over blocks already on the node's chain. Fix named in the ledger entry: verify an off-chain certificate against the table at its own block and let it constrain fork choice (a certificate-driven reorg), then re-determine. Raw: `scratchpad fud-a/c4-results-*.md`, node logs `c4-on90-tmp/`, `c4-v2-control-tmp/`. + +## 5 October 2026 (evening), the 9070 XT on the eGPU: why 17.9 MH/s, and what moved + +PC 1 (ae432dc7, Windows 11, Ryzen 7 9800X3D with its gfx1036, RTX 5090 on CUDA), an AMD Radeon RX 9070 XT (gfx1201, RDNA 4) in a Sonnet Breakaway Box 850T5 over USB4, Adrenalin 26.9.2 (OpenCL driver string `3683.0 (PAL,LC)`, platform `OpenCL 2.1 AMD-APP (3683.0)`). Branch `opencl-rdna4`. the project lead: "the hashrate is low" (17.9 MH/s with one worker; two workers on the card earlier gave 8.9 and 9.4). + +**Before, from PC 1's own app log** (`node tools/logs.mjs win-ae432dc7-20261005-181046`, the miner's STATUS line for the card `amd:1:gfx1201`, 2^21-nonce jobs): `hash=17.82 MH/s wall (17.83 MH/s inside jobs) ... idle=0.3%`. Wall equals inside, so the host loop (template fetch, job line, read-back, scan) costs nothing measurable; the dispatch itself is slow. The worker's `ready` line: `exchange 0` (local memory: AMD lists `cl_khr_subgroups` and no shuffle extension), `batch 4194304`, `dataset-log2 28` (1 GiB), device `[1] gfx1201` on the 3683.0 platform, `AMD wavefront width 32`. The same card was listed again as `[3] gfx1201` on the older platform `3652.0` (the 32.0.21042 driver's OpenCL registration is still present after the update): that is the two-worker run. + +**Hypotheses, each with its number** (the measurement job `rdna4-bench-1`, 18:39:25 to 18:41:17 UTC, the card switched off in the app through `POST /api/cards` for key `amd:1:gfx1201` only, the 5090 untouched; worker exe sha256 `53c7e8c9…5403e10` built from this branch by `proto-cuda/nvrtc/build-windows.sh`; read back with `node tools/jobs.mjs rdna4-bench-1`): + +| # | Hypothesis | Measured | Verdict | +|---|---|---|---| +| 1 | The dataset or program is re-sent over the eGPU link per job | Nothing is re-sent: the dataset (1 GiB) and cache (256 MiB) are built on the device once per pair (`info first pack ... cache 11 dataset 51 ms` on the Mac check); per 2^21-nonce job the old path sent 32 B up and read 16 MiB down; the serve A/B below puts a number on that read-back | Not the cause | +| 2 | Work-group, occupancy, wave width, the exchange | `clGetKernelSubGroupInfoKHR`: sub-group 32 for a 32-item work-group (wave32), private memory 0 (no spills), preferred multiple 32; `--group-warps 1, 2, 4, 8` = 18.024, 18.063, 18.070, 18.039 MH/s (`--batches 3`, 2^24, device event time); `--batch-log2 21` (the app's job size) = 18.108 | Not the cause: the shape does not move the number | +| 3 | The wrong AMD platform | The app's worker runs on `[1]`, the 3683.0 platform (ready line). The old platform's `[3]` gives 18.049 MH/s: the same. The duplicate listing is real and is the two-worker halving | Not the cause of 17.9; fixed anyway (below) | +| 4 | The card's own random-read rate | `--memprobe`: dependent random 4-byte loads over 1024 MiB top out at 2.42 to 2.68 G loads/s from 4,096 lanes up (table below); 128 loads per hash gives a ceiling of 18.9 to 20.9 MH/s; the hash runs at 18.0 to 18.1 | THE CAUSE: the hash is at 87 to 95% of what this card does for this access pattern | + +**The memprobe on the 9070 XT** (`igneum-worker-opencl.exe --device 1 --memprobe`, device event time, best of 3, 256 dependent steps per lane; `chase` = one dependent random 4-byte load per step, `indep x8` = eight independent chains per lane): + +| Buffer | Work-group | Lanes in flight | chase G loads/s | ns per dependent load | indep x8 G loads/s | +|---|---|---|---|---|---| +| 4 MiB (inside the 8 MB L2, approximate size) | 256 | 4,096 | 34.95 | 117 | | +| 4 MiB | 256 | 262,144 | 64.63 | 4,056 | 63.8 (262k lanes) | +| 64 MiB (the 64 MB Infinity Cache, approximate size) | 256 | 4,096 | 9.17 | 447 | | +| 64 MiB | 256 | 262,144 | 9.18 | 28,561 | 8.8 (262k lanes) | +| 1024 MiB (GDDR6) | 32 | 4,096 | 2.64 | 1,552 | | +| 1024 MiB | 32 | 65,536 | 2.60 | 25,181 | | +| 1024 MiB | 32 | 4,194,304 | 2.43 | 1,729,136 | | +| 1024 MiB | 256 | 4,096 | 2.64 | 1,552 | | +| 1024 MiB | 256 | 262,144 | 2.45 | 106,831 | 2.46 (262k lanes) | +| 1024 MiB | 256 | 4,194,304 | 2.42 | 1,732,023 | 2.42 (4M lanes) | +| ALU chain, 1,048,576 lanes x 4,096 steps | 256 | | 6,219 G int ops/s (5 ops per step counted, approximate) | | | + +Reading: at the dataset size the card delivers about 2.5 G random 4-byte reads per second whatever the parallelism (4,096 lanes already saturate it; more lanes only queue, the ns column is Little's law on a fixed throughput). Eight independent loads per lane give the same 2.4 G/s, so it is not a latency-hiding problem in the kernel. Inside the Infinity Cache the same chain runs 3.7x faster and inside L2 26x faster, so the cap is the path to GDDR6 for random reads. The ALU chain says the shader clock is not parked (approximate: 6.2 T int ops/s is of the order of 64 CUs x 64 lanes x 2.46 GHz with quarter-rate multiplies). + +**Against the other two cards** (same probe; the 5090 through NVIDIA's OpenCL `[4]` WHILE its CUDA worker was mining, so a lower bound; the Mac through Apple OpenCL, wall time, a Mac at high load, approximate): + +| Card | 1024 MiB chase at 4,096 lanes | 1024 MiB chase ceiling | indep x8 ceiling | ceiling / 128 = hash ceiling | measured hash rate | +|---|---|---|---|---|---| +| RX 9070 XT, eGPU over USB4 | 2.64 G/s, 1,552 ns | 2.42 to 2.68 G/s | 2.42 G/s | 18.9 to 20.9 MH/s | 18.0 to 18.1 MH/s (bench), 17.8 (app) | +| RTX 5090, PCIe 5 x16, contended | 9.09 G/s, 451 ns | 16.4 to 18.0 G/s | 16.2 to 16.7 G/s | 128 to 141 MH/s | 127 MH/s (app, the project lead), 139.7 alone (M11) | +| Apple M5 Max, Apple OpenCL | 2.10 G/s, 1,949 ns | 3.41 to 3.49 G/s | 3.45 to 3.47 G/s | 26.6 to 27.3 MH/s | 27.9 Mhash/s (README, Apple OpenCL) | + +Reading: on all three cards the hash runs within a few percent of 1/128 of the card's dependent random-read ceiling, which is what a 128-load program should do; the probe is a good model of the hash. The 5090 does 6.6x the random reads of the 9070 XT for 2.8x the rated bandwidth (1,792 against 640 GB/s, vendor figures): the rest is access granularity and DRAM behaviour on random 4-byte reads, which the kernel cannot change. + +**Power, heat, fans and clocks, measured** (branch `opencl-rdna4-telemetry`; the project lead watched the 9070 XT at 90% usage with its fans barely turning and the app had no AMD reading, the MH/W line came from nvidia-smi only; a new helper `proto-opencl/gpu-telemetry.c` reads ADLX on Windows and the amdgpu sysfs on Linux. Job `tele-measure-1`, 20:27:45 to 20:29:41 UTC, both cards mining in the app, nothing touched: `igneum-gpu-telemetry -l 5` (sha256 `703cf69c…a9c69b`) and `nvidia-smi --query-gpu=index,name,power.draw,temperature.gpu,fan.speed,clocks.mem,clocks.gr,utilization.gpu -l 5` side by side, the app's `hash_now` every 5 s; `node tools/jobs.mjs tele-measure-1`): + +| Card | Samples | Watts (mean, min to max) | Temperature | Fan | Memory clock | Shader clock | Busy | Hash (mean of 24) | MH/W, measured | +|---|---|---|---|---|---|---|---|---|---| +| RX 9070 XT, bus 98, ADLX `GPUPower` | 12 (the helper's buffered tail was lost at the kill; fixed, `fflush` per sample) | 198.9 (193 to 212) | 64 C | 657 rpm (ADLX gives rpm; no percent) | 2,505 MHz | 3,290 MHz | 100% | 17.73 MH/s | 0.089 | +| RTX 5090, nvidia-smi, 450 W cap | 24 | 307.6 (306.3 to 308.7) | 69 C | 44% | 13,801 MHz | 2,850 MHz | 94% | 122.30 MH/s | 0.398 | +| gfx1036 (integrated, idle) | 12 | 42.7 (32 to 56; the package, not the GPU alone) | 62 C | none | 2,800 MHz | 600 MHz | 0% | off | | + +Reading: the 9070 XT draws 199 W of its 304 W board rating (vendor figure) at 100% busy with the shader clock at its top, so the die is waiting on memory, which is the ceiling finding again; the fans at 657 rpm and 64 C are the card's own curve at that load, not a fault. Per watt the 5090 is 4.5x the 9070 XT on this program class (0.398 against 0.089 MH/W). The earlier per-watt claim from the board rating (304 W) would have read 0.058 MH/W; the measured number is 1.5x that. + +**Is it the eGPU link?** No. 2.42 G loads/s x 64 B lines = 155 GB/s of DRAM traffic, forty times what a USB4 PCIe tunnel carries (about 4 GB/s, approximate); the 1 GiB buffer sits in the card's own memory (the 4 and 64 MiB cases show the card's caches at work above it, and a buffer in host memory would run below 0.1 G/s). A PCIe slot would move the per-job read-back (16 MiB per 2^21-nonce job on the old path, now gone) and nothing else; the random-read ceiling is the card's. What a PCIe slot would give: the same 18 MH/s. + +**What changed on `opencl-rdna4`** (`proto-opencl/host.c`, `app/igneum-app/src/detect.rs`): + +| Change | Before | After | +|---|---|---| +| Duplicate platform | `--list` showed the card twice ([1] 3683.0 and [3] 3652.0); the app made two cards and ran two workers (8.9 + 9.4 MH/s) | the older platform's entry prints as ` dup [3] ... hidden, use [1]`, the default pick skips it, the app's parser (`parse_opencl_list`, 3 tests) never makes a card of it; `--device 3` still works for comparison. Verified on PC 1: `platforms: 2 device(s) hidden ...`, cards `amd:0:gfx1036` and `amd:1:gfx1201` only | +| Kernel report | work-group and local memory | plus preferred multiple, private memory (spills), sub-group size on every exchange path (`info kernel:` in serve mode) | +| Read-back per dispatch | 8 B per nonce (16 MiB per job) and a host scan of 2^21 words | a GPU select pass: the hits (index, hash) behind an atomic counter plus 34 sentinel words; 276 B per chunk plus 16 B per hit; found lines in nonce order; `--readback full` / `IGNEUM_READBACK=full` keeps the old path; a chunk with over 256 hits falls back to the full read | +| Transfer accounting | none | bytes up and down per chunk and the mean device time of kernel, select, read-back and scan in the stats line every 200 jobs and at quit | +| `--memprobe` | none | the tables above, no pack needed | + +Correctness: `proto-opencl/test-generic.sh` on the Mac (Apple OpenCL) PASS on both paths: "15 sampled hashes (both packs, both sides of the 32-bit nonce boundary) equal igneum-pow hash-bound"; select path transfers `5 chunks, up 180 B, down 4452 B`, full path `up 160 B, down 1536 B` (the check's jobs are 32 to 64 nonces with every nonce a hit). The bench on the 9070 XT: cache check PASS, dataset self-test PASS, 6 of 6 vector warps PASS, batch fingerprint `3cc4fbf90fa6366c` at 2^24 for the devnet pack (the Apple OpenCL value in the README), at every `--group-warps`. + +**The serve-mode A/B on the card** (job `rdna4-serve-4`, 19:11 UTC, card off in the app, worker exe sha256 `324a6d9b…2bfdfff`; 200 real `job` lines of 2,097,152 nonces each, the app's `--job-nonces`, against the emulator test pack `pack-a` (epoch `edc4fa84…`, self-test PASS, 96 of 96 vector lanes), target `0000100000000000` so that 408 hits fall in 200 jobs on both paths; `done` ms over jobs 11 to 200; `node tools/jobs.mjs rdna4-serve-4`): + +| Read-back | Bytes down per job | Kernel (device, mean) | Select pass | Read-back (wall) | Host scan | Mean job | Inside-job rate | +|---|---|---|---|---|---|---|---| +| full (before) | 16,777,216 | 116.12 ms | 0 | 7.28 ms | 0.55 ms | 124.22 ms | 16.88 MH/s | +| select (after) | 309 | 116.00 ms | 0.039 ms | 0.78 ms | 0.00 ms | 117.38 ms | 17.87 MH/s | +| select (repeat) | 309 | 115.96 ms | 0.038 ms | 0.76 ms | 0.00 ms | 117.33 ms | 17.87 MH/s | + +Reading: the kernel is the same 116.0 ms on both paths (18.08 MH/s pure kernel, the bench's number). The old path paid 7.8 ms per job for 16 MiB over the eGPU link (2.3 GB/s, the USB4 tunnel's rate; a PCIe slot would read it in about 1 ms, approximate) and the host scan. The select pass removes it: +5.9% per job on this link, nothing on the kernel. Both paths found the same 408 hits. The `--group-warps` and exchange levers were already shown flat above, so this is the whole host-side gain available on the 9070 XT. + +**Probes with a fresh seed per repetition** (the first probe round replayed the same addresses on repeats, so its low-lane rows were cache hits; fixed in `probeLaunch`, job `rdna4-serve-4`): 1024 MiB chase at 256 lanes 276 ns per dependent load, at 1,024 lanes 422 ns, at 4,096 lanes 1,560 ns (2.63 G/s, the cap). Random 64-byte lines (four `uint4` loads per step) at 1024 MiB: 2.46 to 2.88 G lines/s = 158 to 184 GB/s in lines, the same count per second as the 4-byte chase: every random 4-byte read costs this card a 64-byte line fetch. Coalesced stream over the whole 1024 MiB: 635.2 GB/s against the vendor's 640 GB/s, so the memory clock is in its full state and the card is not parked. Inside the 64 MiB buffer the line probe reaches 8.3 to 14.0 G lines/s (533 to 894 GB/s in lines: the Infinity Cache, approximate). + +**A second defect found on the way: the pack export race.** PC 1's app log since its 19:02 UTC restart (`node tools/logs.mjs win-ae432dc7-20261005-190232`): `worker error: error 0 pack packs\devnet: the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT` at 19:07:03, 19:07:19 and 19:08:07, so the 9070 XT was not mining at all in the app while this entry was written (my job `rdna4-serve-1` at 18:43 hit the same folder in the same state). Cause, from `app/igneum-app/src/engine.rs` `prepare_worker`: one thread per card, each running `igneum-miner export-pack` into the one folder `packs\devnet`; across an epoch change the two exports interleave and the folder keeps one epoch's `program.h` with the other's `seeds.txt` until the next export. Fix on this branch: a process-wide mutex around both export sites (`EXPORT_LOCK`); the second export rewrites the same pack. Not measured in the app yet: it ships with the branch. + +**Answer to the project lead.** The 9070 XT does 2.5 G random 4-byte reads per second from its memory for this access pattern, and the hash needs 128 of them, so about 19 MH/s is this card's ceiling for the current program class, on any slot; it was running at 92% of that. The eGPU link cost 6% per job through the read-back, now removed (17.87 against 16.88 MH/s inside jobs standalone). The duplicate platform that halved it to 8.9 + 9.4 is folded away. The pack race that stopped it is serialised. Nothing else in the worker's control moves the number: the next step for this card is the program class itself (fewer, wider loads per hash would favour AMD's 64-byte lines), which is a consensus question, not a worker one. diff --git a/packaging/windows/make-payload.sh b/packaging/windows/make-payload.sh index ff1c8400..7b139864 100755 --- a/packaging/windows/make-payload.sh +++ b/packaging/windows/make-payload.sh @@ -72,6 +72,7 @@ else ls "$STAGE"/nvrtc64_*_0.dll >/dev/null 2>&1 || echo "warning: igneum-worker-cuda.exe without nvrtc64_*_0.dll (run $NVRTC_DIR/fetch-redist.sh); the engine will not use it" fi if [ -f "$ROOT/proto-opencl/igneum-worker-opencl.exe" ]; then cp "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$STAGE/"; found_workers=1; fi + if [ -f "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" ]; then cp "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" "$STAGE/"; fi # AMD power, heat, fans, clocks (5 October 2026) fi if [ "$found_workers" = 1 ]; then echo "workers: $(cd "$STAGE" && ls igneum-worker-*.exe nvrtc*.dll 2>/dev/null | tr '\n' ' ')" else echo "note: no prebuilt igneum-worker-cuda.exe / igneum-worker-opencl.exe found; the engine builds the CUDA worker from proto-cuda\\ on the PC (CUDA Toolkit and MSVC needed)"; fi diff --git a/packaging/windows/push-inputs.sh b/packaging/windows/push-inputs.sh index 05f515dc..28afa814 100755 --- a/packaging/windows/push-inputs.sh +++ b/packaging/windows/push-inputs.sh @@ -28,6 +28,7 @@ REL="${IGNEUM_WIN_RELEASE:-$ROOT/vendor/igneum-node/target-integration/x86_64-pc MINGW=/opt/homebrew/opt/mingw-w64/toolchain-x86_64/x86_64-w64-mingw32 NVRTC_DIR="$ROOT/proto-cuda/nvrtc" CL_WORKER="$ROOT/proto-opencl/igneum-worker-opencl.exe" +TELEMETRY="$ROOT/proto-opencl/igneum-gpu-telemetry.exe" # AMD power, heat, fans, clocks (5 October 2026) TOKEN_FILE="$HOME/.config/igneum/dl-token" DLSITE="${IGNEUM_DLSITE:-}" [ -n "$DLSITE" ] || { [ -f "$HOME/.config/igneum/dlsite-dir" ] && DLSITE="$(tr -d '[:space:]' < "$HOME/.config/igneum/dlsite-dir")"; } || true @@ -59,6 +60,7 @@ if [ -f "$NVRTC_DIR/igneum-worker-cuda.exe" ]; then ls "$STAGE"/nvrtc64_*_0.dll >/dev/null 2>&1 || echo "warning: igneum-worker-cuda.exe without nvrtc64_*_0.dll (run $NVRTC_DIR/fetch-redist.sh)" else echo "warning: no $NVRTC_DIR/igneum-worker-cuda.exe (run $NVRTC_DIR/build-windows.sh); the app will build the CUDA worker on the PC"; fi [ -f "$CL_WORKER" ] && cp "$CL_WORKER" "$STAGE/" || echo "warning: no $CL_WORKER" +[ -f "$TELEMETRY" ] && cp "$TELEMETRY" "$STAGE/" || echo "warning: no $TELEMETRY (AMD cards show no draw or temperature)" # the signer, built from the app crate (it includes src/manifest.rs and src/inputs.rs, so it signs what the runner verifies) KEY="$HOME/.config/igneum/ota-signing-key" diff --git a/proto-cuda/nvrtc/build-windows.sh b/proto-cuda/nvrtc/build-windows.sh index 278fa41e..709b84c4 100755 --- a/proto-cuda/nvrtc/build-windows.sh +++ b/proto-cuda/nvrtc/build-windows.sh @@ -25,7 +25,7 @@ VERIFY="$ROOT/packaging/windows/resources/verify-exe.py" # The coin icon and the version blocks, as COFF objects the linker takes like any other input [ -f "$ICONS/igneum.ico" ] || { echo "== no $ICONS/igneum.ico, making the icons"; python3 "$ICONS/make-icons.py"; } RES="$(mktemp -d)" -for w in cuda opencl; do +for w in cuda opencl gpu-telemetry; do "$WINDRES" -I "$ICONS" -i "$HERE/igneum-worker-$w.rc" -O coff -o "$RES/igneum-worker-$w.res.o" done @@ -37,11 +37,16 @@ echo "== igneum-worker-opencl.exe" -I "$RED/include" -I "$PLACEHOLDER" -DIGNEUM_KERNEL_PATH='"kernel_bound.cl"' \ -o "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$ROOT/proto-opencl/host.c" "$RES/igneum-worker-opencl.res.o" "$STRIP" "$ROOT/proto-opencl/igneum-worker-opencl.exe" +echo "== igneum-gpu-telemetry.exe (ADLX, SetupAPI, PDH; vendor/adlx is the SDK clone)" +[ -f "$ROOT/vendor/adlx/SDK/Include/ADLX.h" ] || { echo "no vendor/adlx: git clone --depth 1 https://github.com/GPUOpen-LibrariesAndSDKs/ADLX.git $ROOT/vendor/adlx" >&2; exit 1; } +"$CC" -std=gnu99 -O2 -Wall -Wno-unused-parameter -Wno-unused-function -static -I "$ROOT/vendor/adlx/SDK/Include" \ + -o "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" "$ROOT/proto-opencl/gpu-telemetry.c" "$ROOT/vendor/adlx/SDK/ADLXHelper/Windows/C/ADLXHelper.c" "$RES/igneum-worker-gpu-telemetry.res.o" -lsetupapi -lpdh +"$STRIP" "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" rm -rf "$RES" -for exe in "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe"; do +for exe in "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$ROOT/proto-opencl/igneum-gpu-telemetry.exe"; do printf '%s: %d bytes, imports:' "$(basename "$exe")" "$(stat -f %z "$exe")" x86_64-w64-mingw32-objdump -p "$exe" | sed -n 's/^[[:space:]]*DLL Name: //p' | tr '\n' ' ' echo done # the icon and version block survived the strip (strip keeps .rsrc; this proves it) -python3 "$VERIFY" --version 0.3.0 "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe" +python3 "$VERIFY" --version 0.3.0 "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" diff --git a/proto-cuda/nvrtc/igneum-worker-gpu-telemetry.rc b/proto-cuda/nvrtc/igneum-worker-gpu-telemetry.rc new file mode 100644 index 00000000..db056147 --- /dev/null +++ b/proto-cuda/nvrtc/igneum-worker-gpu-telemetry.rc @@ -0,0 +1,35 @@ +// Windows resources for igneum-gpu-telemetry.exe: the coin icon Explorer shows and the version block under Properties > Details. +// Compiled with x86_64-w64-mingw32-windres (the icon path is relative to brand/icons, passed with -I). +// the project lead's rule (4 October 2026): every shipped exe carries the coin icon and a version block, like the Mac app and DMG. +#include + +1 ICON "igneum.ico" + +1 VERSIONINFO +FILEVERSION 0,3,0,0 +PRODUCTVERSION 0,3,0,0 +FILEFLAGSMASK 0x3fL +FILEFLAGS 0x0L +FILEOS VOS_NT_WINDOWS32 +FILETYPE VFT_APP +FILESUBTYPE VFT2_UNKNOWN +BEGIN + BLOCK "StringFileInfo" + BEGIN + BLOCK "040904B0" + BEGIN + VALUE "CompanyName", "Igneum" + VALUE "FileDescription", "Igneum Miner GPU telemetry (AMD power, heat, fans, clocks)" + VALUE "FileVersion", "0.3.0" + VALUE "InternalName", "igneum-gpu-telemetry" + VALUE "LegalCopyright", "Igneum contributors" + VALUE "OriginalFilename", "igneum-gpu-telemetry.exe" + VALUE "ProductName", "Igneum Miner" + VALUE "ProductVersion", "0.3.0" + END + END + BLOCK "VarFileInfo" + BEGIN + VALUE "Translation", 0x409, 1200 + END +END diff --git a/proto-opencl/README.md b/proto-opencl/README.md index fc6a3783..e6e040e2 100644 --- a/proto-opencl/README.md +++ b/proto-opencl/README.md @@ -39,6 +39,9 @@ proto-opencl/ host.c C99 host: device list, runtime kernel build, cache + dataset fill, self-tests, vectors, bench, sweep, --serve, --pack cl_dynamic.h Windows one-click build: OpenCL.dll loaded at run time (IGNEUM_CL_DYNAMIC) test-generic.sh the --pack mode checked here through Apple OpenCL (needs proto-cuda/nvrtc/emu/test.sh's packs) + test_host.c device-free unit tests of host.c's rules (the duplicate-platform fold); run with test-host.sh + gpu-telemetry.c igneum-gpu-telemetry: AMD power, temperature, fan, clocks and busy per card (ADLX on Windows, amdgpu sysfs on Linux), + one line per card per sample; the app's AMD card row reads it (engine.rs amd_telemetry_line) build.sh macOS (-framework OpenCL, or the Khronos ICD loader) and Linux (-lOpenCL) build.bat Windows (MSVC cl.exe + OpenCL.lib) WAVEFRONT.md wave32 vs wave64 on AMD, and why the kernel cannot tell the difference diff --git a/proto-opencl/gpu-telemetry.c b/proto-opencl/gpu-telemetry.c new file mode 100644 index 00000000..86a28be9 --- /dev/null +++ b/proto-opencl/gpu-telemetry.c @@ -0,0 +1,274 @@ +// igneum-gpu-telemetry: power, temperature, fan and clocks of every AMD GPU, one line per card per sample. +// 5 October 2026, after the project lead watched a 9070 XT at 90% usage with its fans barely turning and the app could not say +// what it drew (the app's draw, temperature and MH per watt line came from nvidia-smi only). +// +// igneum-gpu-telemetry [-l SECONDS] one sample (default), or one every SECONDS until stdin closes or SIGTERM +// +// Windows: ADLX (the AMD Device Library eXtra, amdadlx64.dll, shipped with Adrenalin; vendor/adlx is the SDK clone, +// MIT) for the metrics, keyed by the card's PCI bus from SetupAPI (the display class, matched by the same name ADLX +// reports). Without ADLX (no AMD driver, an old one, or the DLL missing) only the utilisation is read, from the +// GPU Engine performance counters through PDH, keyed by the adapter LUID that Windows uses there. +// Linux: the amdgpu sysfs (/sys/class/drm/card*/device: hwmon power1_average, temp1_input, fan1_input, pwm1, +// pp_dpm_mclk, gpu_busy_percent), keyed by the PCI address of the device link. +// +// Line format (space separated, every field present, a value the source cannot give prints as -): +// amd bus kind integrated|discrete name "" watts temp_c fan_rpm +// fan_pct <%> mclk_mhz gclk_mhz util_pct <%> source adlx|sysfs|perfcounter +// then one `end ` line per sample. The app (engine.rs amd_telemetry_line) parses it; parsers are unit-tested +// against lines captured on PC 1. +#define _CRT_SECURE_NO_WARNINGS +#include +#include +#include +#include + +static volatile int gStop = 0; +static void onSignal(int s) { (void)s; gStop = 1; } + +typedef struct { + char bus[64]; + char kind[16]; + char name[128]; + double watts, tempC, fanRpm, fanPct, mclk, gclk, util; /* -1 = not available */ + const char* source; +} Sample; + +static void sampleInit(Sample* s) { memset(s, 0, sizeof(*s)); strcpy(s->bus, "-"); strcpy(s->kind, "-"); strcpy(s->name, "-"); s->watts = s->tempC = s->fanRpm = s->fanPct = s->mclk = s->gclk = s->util = -1.0; s->source = "-"; } +static void printNum(double v, const char* fmt) { if (v < 0) printf(" -"); else printf(fmt, v); } +static void printSample(int ordinal, const Sample* s) { + printf("amd %d bus %s kind %s name \"%s\" watts", ordinal, s->bus, s->kind, s->name); + printNum(s->watts, " %.1f"); printf(" temp_c"); printNum(s->tempC, " %.1f"); printf(" fan_rpm"); printNum(s->fanRpm, " %.0f"); + printf(" fan_pct"); printNum(s->fanPct, " %.0f"); printf(" mclk_mhz"); printNum(s->mclk, " %.0f"); printf(" gclk_mhz"); printNum(s->gclk, " %.0f"); + printf(" util_pct"); printNum(s->util, " %.0f"); printf(" source %s\n", s->source); +} + +#ifdef _WIN32 +#define WIN32_LEAN_AND_MEAN +#include +#include +#include +#include "../vendor/adlx/SDK/ADLXHelper/Windows/C/ADLXHelper.h" +#include "../vendor/adlx/SDK/Include/IPerformanceMonitoring.h" + +/* The SDK declares these three and leaves them to the platform file of each sample. */ +adlx_handle ADLX_CDECL_CALL adlx_load_library(const TCHAR* filename) { return (adlx_handle)LoadLibrary(filename); } +int ADLX_CDECL_CALL adlx_free_library(adlx_handle module) { return FreeLibrary((HMODULE)module) ? 1 : 0; } +void* ADLX_CDECL_CALL adlx_get_proc_address(adlx_handle module, const char* procName) { return (void*)GetProcAddress((HMODULE)module, procName); } + +static double nowMs(void) { LARGE_INTEGER f, c; QueryPerformanceFrequency(&f); QueryPerformanceCounter(&c); return (double)c.QuadPart * 1000.0 / (double)f.QuadPart; } + +/* The PCI bus of every display-class device, by its name (SetupAPI; the names are the ones ADLX reports). */ +typedef struct { char name[128]; int bus; } BusEntry; +static int listBuses(BusEntry* out, int cap) { + static const GUID DISPLAY = { 0x4d36e968, 0xe325, 0x11ce, { 0xbf, 0xc1, 0x08, 0x00, 0x2b, 0xe1, 0x03, 0x18 } }; + HDEVINFO set = SetupDiGetClassDevsA(&DISPLAY, NULL, NULL, DIGCF_PRESENT); + SP_DEVINFO_DATA d; + DWORD i; + int n = 0; + if (set == INVALID_HANDLE_VALUE) return 0; + d.cbSize = sizeof(d); + for (i = 0; SetupDiEnumDeviceInfo(set, i, &d) && n < cap; ++i) { + char name[128] = { 0 }; + DWORD bus = 0, type = 0, got = 0; + if (!SetupDiGetDeviceRegistryPropertyA(set, &d, SPDRP_DEVICEDESC, &type, (BYTE*)name, sizeof(name) - 1, &got)) continue; + if (!SetupDiGetDeviceRegistryPropertyA(set, &d, SPDRP_BUSNUMBER, &type, (BYTE*)&bus, sizeof(bus), &got)) continue; + snprintf(out[n].name, sizeof(out[n].name), "%s", name); out[n].bus = (int)bus; ++n; + } + SetupDiDestroyDeviceInfoList(set); + return n; +} +static int busOf(const BusEntry* b, int n, const char* name, int* taken) { + int i; + for (i = 0; i < n; ++i) if (!taken[i] && strcmp(b[i].name, name) == 0) { taken[i] = 1; return b[i].bus; } + return -1; +} + +/* ADLX: one sample of every GPU. Returns the number of lines printed, -1 when ADLX is not usable (reason printed). */ +static IADLXSystem* gSys = NULL; +static IADLXPerformanceMonitoringServices* gPerf = NULL; +static int adlxOpen(void) { + ADLX_RESULT r = ADLXHelper_Initialize(); + if (!ADLX_SUCCEEDED(r)) { printf("info adlx: ADLXHelper_Initialize returned %d (no AMD driver with ADLX; amdadlx64.dll missing or too old)\n", (int)r); return 0; } + gSys = ADLXHelper_GetSystemServices(); + if (!gSys) { printf("info adlx: no system services\n"); return 0; } + r = gSys->pVtbl->GetPerformanceMonitoringServices(gSys, &gPerf); + if (!ADLX_SUCCEEDED(r) || !gPerf) { printf("info adlx: GetPerformanceMonitoringServices returned %d\n", (int)r); return 0; } + return 1; +} +static int adlxSample(const BusEntry* buses, int nBuses) { + IADLXGPUList* gpus = NULL; + adlx_uint it; + int ordinal = 0; + int taken[32] = { 0 }; + ADLX_RESULT r = gSys->pVtbl->GetGPUs(gSys, &gpus); + if (!ADLX_SUCCEEDED(r) || !gpus) { printf("info adlx: GetGPUs returned %d\n", (int)r); return 0; } + for (it = gpus->pVtbl->Begin(gpus); it != gpus->pVtbl->End(gpus); ++it) { + IADLXGPU* gpu = NULL; + IADLXGPUMetrics* m = NULL; + Sample s; + const char* name = NULL; + ADLX_GPU_TYPE type = GPUTYPE_UNDEFINED; + adlx_double dv = 0; adlx_int iv = 0; + if (!ADLX_SUCCEEDED(gpus->pVtbl->At_GPUList(gpus, it, &gpu)) || !gpu) continue; + sampleInit(&s); + s.source = "adlx"; + if (ADLX_SUCCEEDED(gpu->pVtbl->Name(gpu, &name)) && name) snprintf(s.name, sizeof(s.name), "%s", name); + if (ADLX_SUCCEEDED(gpu->pVtbl->Type(gpu, &type))) strcpy(s.kind, type == GPUTYPE_INTEGRATED ? "integrated" : type == GPUTYPE_DISCRETE ? "discrete" : "-"); + { int b = busOf(buses, nBuses, s.name, taken); if (b >= 0) snprintf(s.bus, sizeof(s.bus), "%d", b); } + r = gPerf->pVtbl->GetCurrentGPUMetrics(gPerf, gpu, &m); + if (ADLX_SUCCEEDED(r) && m) { + if (ADLX_SUCCEEDED(m->pVtbl->GPUPower(m, &dv))) s.watts = dv; + if (s.watts < 0 && ADLX_SUCCEEDED(m->pVtbl->GPUTotalBoardPower(m, &dv))) s.watts = dv; + if (ADLX_SUCCEEDED(m->pVtbl->GPUTemperature(m, &dv))) s.tempC = dv; + if (ADLX_SUCCEEDED(m->pVtbl->GPUFanSpeed(m, &iv))) s.fanRpm = iv; + if (ADLX_SUCCEEDED(m->pVtbl->GPUVRAMClockSpeed(m, &iv))) s.mclk = iv; + if (ADLX_SUCCEEDED(m->pVtbl->GPUClockSpeed(m, &iv))) s.gclk = iv; + if (ADLX_SUCCEEDED(m->pVtbl->GPUUsage(m, &dv))) s.util = dv; + m->pVtbl->Release(m); + } else { + printf("info adlx: GetCurrentGPUMetrics for \"%s\" returned %d\n", s.name, (int)r); + } + /* fan percent: ADLX gives rpm only here; the tuning interface has the range, the app shows rpm when pct is - */ + printSample(ordinal++, &s); + gpu->pVtbl->Release(gpu); + } + gpus->pVtbl->Release(gpus); + return ordinal; +} + +/* PDH fallback: GPU engine utilisation per adapter LUID, summed over the engines (no power, no temperature). */ +static int pdhSample(void) { + PDH_HQUERY q = NULL; + PDH_HCOUNTER c = NULL; + DWORD size = 0, count = 0, i; + PDH_FMT_COUNTERVALUE_ITEM_A* items; + int ordinal = 0; + if (PdhOpenQueryA(NULL, 0, &q) != ERROR_SUCCESS) { printf("info perfcounter: PdhOpenQuery failed\n"); return 0; } + if (PdhAddEnglishCounterA(q, "\\GPU Engine(*)\\Utilization Percentage", 0, &c) != ERROR_SUCCESS) { printf("info perfcounter: no GPU Engine counters\n"); PdhCloseQuery(q); return 0; } + PdhCollectQueryData(q); Sleep(1000); PdhCollectQueryData(q); + PdhGetFormattedCounterArrayA(c, PDH_FMT_DOUBLE, &size, &count, NULL); + items = (PDH_FMT_COUNTERVALUE_ITEM_A*)malloc(size ? size : 1); + if (PdhGetFormattedCounterArrayA(c, PDH_FMT_DOUBLE, &size, &count, items) == ERROR_SUCCESS) { + /* instance names: pid_1234_luid_0x00000000_0x0000D4E3_phys_0_eng_0_engtype_3D; sum per luid */ + char luids[16][40]; double sums[16]; int n = 0, k; + for (i = 0; i < count; ++i) { + const char* p = strstr(items[i].szName, "luid_"); + char luid[40]; + if (!p) continue; + snprintf(luid, sizeof(luid), "%.39s", p); { char* e = strstr(luid, "_phys"); if (e) *e = 0; } + for (k = 0; k < n; ++k) if (strcmp(luids[k], luid) == 0) break; + if (k == n && n < 16) { strcpy(luids[n], luid); sums[n] = 0; ++n; } + if (k < 16) sums[k] += items[i].FmtValue.doubleValue; + } + for (k = 0; k < n; ++k) { + Sample s; sampleInit(&s); s.source = "perfcounter"; + snprintf(s.bus, sizeof(s.bus), "%s", luids[k]); + s.util = sums[k] > 100.0 ? 100.0 : sums[k]; + printSample(ordinal++, &s); + } + } + free(items); + PdhCloseQuery(q); + return ordinal; +} + +int main(int argc, char** argv) { + int every = 0, i, haveAdlx; + BusEntry buses[32]; + int nBuses; + for (i = 1; i < argc; ++i) if (strcmp(argv[i], "-l") == 0 && i + 1 < argc) every = atoi(argv[++i]); + signal(SIGINT, onSignal); signal(SIGTERM, onSignal); + setvbuf(stdout, NULL, _IOLBF, 0); + nBuses = listBuses(buses, 32); + for (i = 0; i < nBuses; ++i) printf("info display device \"%s\" bus %d\n", buses[i].name, buses[i].bus); + haveAdlx = adlxOpen(); + do { + double t0 = nowMs(); + int n = haveAdlx ? adlxSample(buses, nBuses) : pdhSample(); + printf("end %.1f ms %d card(s)\n", nowMs() - t0, n); + fflush(stdout); /* a redirected stdout is fully buffered on the Windows CRT whatever setvbuf asks (PC 1 lost 60 s of samples at the kill) */ + if (every > 0) Sleep((DWORD)every * 1000); + } while (every > 0 && !gStop); + if (haveAdlx) { if (gPerf) gPerf->pVtbl->Release(gPerf); ADLXHelper_Terminate(); } + return 0; +} +#else +#include +#include +#include +static double nowMs(void) { struct timespec ts; clock_gettime(CLOCK_MONOTONIC, &ts); return ts.tv_sec * 1000.0 + ts.tv_nsec / 1e6; } +static int readText(const char* path, char* out, size_t cap) { FILE* f = fopen(path, "r"); size_t n; if (!f) return 0; n = fread(out, 1, cap - 1, f); fclose(f); out[n] = 0; return 1; } +static double readNumber(const char* path) { char b[64]; if (!readText(path, b, sizeof(b))) return -1.0; return atof(b); } +/* pp_dpm_mclk: lines "0: 96Mhz", "3: 1258Mhz *"; the starred line is the current state */ +static double dpmCurrent(const char* text) { + const char* p = text; + while (p && *p) { + const char* nl = strchr(p, '\n'); + size_t len = nl ? (size_t)(nl - p) : strlen(p); + const char* star = memchr(p, '*', len); + if (star) { const char* colon = memchr(p, ':', len); if (colon) return atof(colon + 1); } + p = nl ? nl + 1 : NULL; + } + return -1.0; +} +static int sysfsSample(const char* root) { + DIR* d = opendir(root); + struct dirent* e; + int ordinal = 0; + if (!d) { printf("info sysfs: no %s\n", root); return 0; } + while ((e = readdir(d)) != NULL) { + char dev[512], path[640], text[4096], link[512]; + ssize_t ln; + Sample s; + DIR* hw; struct dirent* he; + if (strncmp(e->d_name, "card", 4) != 0 || strchr(e->d_name + 4, '-')) continue; + snprintf(dev, sizeof(dev), "%s/%s/device", root, e->d_name); + snprintf(path, sizeof(path), "%s/vendor", dev); + if (!readText(path, text, sizeof(text)) || strtol(text, NULL, 16) != 0x1002) continue; + sampleInit(&s); + s.source = "sysfs"; + ln = readlink(dev, link, sizeof(link) - 1); + if (ln > 0) { link[ln] = 0; { const char* base = strrchr(link, '/'); snprintf(s.bus, sizeof(s.bus), "%.63s", base ? base + 1 : link); } } + snprintf(path, sizeof(path), "%s/product_name", dev); + if (readText(path, text, sizeof(text))) { text[strcspn(text, "\n")] = 0; snprintf(s.name, sizeof(s.name), "%s", text); } + else { snprintf(path, sizeof(path), "%s/device", dev); if (readText(path, text, sizeof(text))) { text[strcspn(text, "\n")] = 0; snprintf(s.name, sizeof(s.name), "amdgpu %s", text); } } + snprintf(path, sizeof(path), "%s/boot_vga", dev); + strcpy(s.kind, "discrete"); + snprintf(path, sizeof(path), "%s/hwmon", dev); + hw = opendir(path); + if (hw) { + while ((he = readdir(hw)) != NULL) { + char hp[900]; + if (strncmp(he->d_name, "hwmon", 5) != 0) continue; + snprintf(hp, sizeof(hp), "%s/%s/power1_average", path, he->d_name); s.watts = readNumber(hp); if (s.watts < 0) { snprintf(hp, sizeof(hp), "%s/%s/power1_input", path, he->d_name); s.watts = readNumber(hp); } if (s.watts >= 0) s.watts /= 1e6; + snprintf(hp, sizeof(hp), "%s/%s/temp1_input", path, he->d_name); s.tempC = readNumber(hp); if (s.tempC >= 0) s.tempC /= 1000.0; + snprintf(hp, sizeof(hp), "%s/%s/fan1_input", path, he->d_name); s.fanRpm = readNumber(hp); + { double pwm, pwmMax; snprintf(hp, sizeof(hp), "%s/%s/pwm1", path, he->d_name); pwm = readNumber(hp); snprintf(hp, sizeof(hp), "%s/%s/pwm1_max", path, he->d_name); pwmMax = readNumber(hp); if (pwm >= 0 && pwmMax > 0) s.fanPct = 100.0 * pwm / pwmMax; else if (pwm >= 0) s.fanPct = 100.0 * pwm / 255.0; } + break; + } + closedir(hw); + } + snprintf(path, sizeof(path), "%s/pp_dpm_mclk", dev); if (readText(path, text, sizeof(text))) s.mclk = dpmCurrent(text); + snprintf(path, sizeof(path), "%s/pp_dpm_sclk", dev); if (readText(path, text, sizeof(text))) s.gclk = dpmCurrent(text); + snprintf(path, sizeof(path), "%s/gpu_busy_percent", dev); s.util = readNumber(path); + printSample(ordinal++, &s); + } + closedir(d); + return ordinal; +} +int main(int argc, char** argv) { + int every = 0, i; + const char* root = getenv("IGNEUM_DRM_ROOT") ? getenv("IGNEUM_DRM_ROOT") : "/sys/class/drm"; /* a fixture tree for tests */ + for (i = 1; i < argc; ++i) if (strcmp(argv[i], "-l") == 0 && i + 1 < argc) every = atoi(argv[++i]); + signal(SIGINT, onSignal); signal(SIGTERM, onSignal); + setvbuf(stdout, NULL, _IOLBF, 0); + do { + double t0 = nowMs(); + int n = sysfsSample(root); + printf("end %.1f ms %d card(s)\n", nowMs() - t0, n); + fflush(stdout); /* a redirected stdout is fully buffered on the Windows CRT whatever setvbuf asks (PC 1 lost 60 s of samples at the kill) */ + if (every > 0) sleep((unsigned)every); + } while (every > 0 && !gStop); + return 0; +} +#endif From 192aa3b83cf438a53a82deee84e283abf9cf9973 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 19:25:12 +0000 Subject: [PATCH 03/63] app: Power control setting (default off): the app never asks for administrator rights on its own the project lead, 5 October 2026: "if we don't have to ask then don't ask". The NVIDIA power cap and the efficiency sweep need administrator rights (one UAC prompt); PC 1 raised that prompt for cmd.exe at every app start and every sweep attempt (17:00, 17:30, 18:12, 19:04 UTC today, each cancelled unanswered after 2 minutes; the "Windows Command Processor" the project lead saw). - config.rs: `power_control` (default OFF on every machine); `sweep` default becomes off and is implied by it (an install carrying sweep = true without power_control is migrated to off on load). - engine.rs: `elevation_allowed(power_control, sweep_only)` gates the power cap (`power_cap_plan` builds nothing when off, the card note says so), the sweep scheduler, Sweep now, the sweep helper; no prompt on quit (the limits reset at the next reboot); no second prompt through PowerShell when the window host's prompt goes unanswered. Cmd::PowerControl(on): on = ONE prompt at that moment (every NVIDIA cap in one step), off = nothing asks; `power_control_after_prompt` turns a refused, cancelled or unanswered prompt into "power control off: administrator rights were not given" (switch back off, sweep off, no retries). Unit tests: off builds no elevated command; on + refusal gives the notice; rights given keeps it on. - platform.rs: `elevated_failure` maps the launcher's exit 251 and the "canceled" wording to the prompt, any other code to the step itself. - server.rs: POST /api/power/control {on}. ui: the Power control switch with the line "Windows asks for administrator rights once; the cap and the sweep need them", the note beside it, the sweep switch disabled while it is off. - The clock-sync prompt stays behind the Sync clock button only (unchanged). - tools/windows/power-prompts-off.ps1: the 0.3.9 job that switched PC 1's sweep off through the API it has (run-20261005-192313: sweep True -> False; the 0.3.9 cap has no off switch, it asks at an app start only). Co-Authored-By: Claude Fable 5.1 --- app/igneum-app/src/config.rs | 23 +++- app/igneum-app/src/engine.rs | 194 +++++++++++++++++++++++----- app/igneum-app/src/platform.rs | 76 ++++++++++- app/igneum-app/src/server.rs | 6 + app/igneum-app/src/state.rs | 7 +- app/igneum-app/ui/app.js | 5 + app/igneum-app/ui/index.html | 5 + tools/windows/power-prompts-off.ps1 | 28 ++++ 8 files changed, 306 insertions(+), 38 deletions(-) create mode 100644 tools/windows/power-prompts-off.ps1 diff --git a/app/igneum-app/src/config.rs b/app/igneum-app/src/config.rs index 5bdcfaf1..050b79bb 100644 --- a/app/igneum-app/src/config.rs +++ b/app/igneum-app/src/config.rs @@ -70,9 +70,16 @@ pub struct Settings { #[serde(default)] pub prove: bool, /// The efficiency sweep (src/sweep.rs): once after install, then weekly, each NVIDIA card's cap is stepped from - /// 100% to 50% on the live program and held at the best MH per watt. Default on. A pinned card is skipped. - #[serde(default = "yes")] + /// 100% to 50% on the live program and held at the best MH per watt. Default off; implied by `power_control` + /// (on when that is switched on, never effective while it is off). A pinned card is skipped. + #[serde(default)] pub sweep: bool, + /// Power control (the project lead, 5 October 2026: "if we don't have to ask then don't ask"): the NVIDIA power cap and the + /// efficiency sweep need administrator rights (one UAC prompt on Windows). Default OFF on every machine; the app + /// never raises the prompt on its own. Switching it on asks once, at that moment; a refused, cancelled or + /// unanswered prompt switches it back off with a notice, no retries. + #[serde(default)] + pub power_control: bool, /// When this install first ran (unix s), for the "first hour after install" sweep. #[serde(default)] pub installed_at: u64, @@ -98,16 +105,26 @@ fn yes() -> bool { impl Default for Settings { fn default() -> Settings { - Settings { setup_done: false, address: String::new(), address_source: String::new(), key_saved: false, identities: 1, cards: HashMap::new(), display_name: String::new(), vote: true, paused: false, accepted_total: 0, auto_update: true, remote_jobs: true, prove: false, sweep: true, installed_at: 0, dev_fee: true, fee_total: 0, proof_verify_trust: false } + Settings { setup_done: false, address: String::new(), address_source: String::new(), key_saved: false, identities: 1, cards: HashMap::new(), display_name: String::new(), vote: true, paused: false, accepted_total: 0, auto_update: true, remote_jobs: true, prove: false, sweep: false, power_control: false, installed_at: 0, dev_fee: true, fee_total: 0, proof_verify_trust: false } } } impl Settings { pub fn load(path: &Path) -> Settings { let mut s: Settings = std::fs::read_to_string(path).ok().and_then(|t| serde_json::from_str(&t).ok()).unwrap_or_default(); + let mut dirty = false; if s.installed_at == 0 { // an install from before the sweep existed counts as installed now: it gets its first-hour sweep s.installed_at = crate::platform::unix_now(); + dirty = true; + } + if s.sweep && !s.power_control { + // the sweep is implied by power control (5 October 2026): an install from before that setting carried + // sweep = true by default; it no longer prompts on its own + s.sweep = false; + dirty = true; + } + if dirty { s.save(path); } s diff --git a/app/igneum-app/src/engine.rs b/app/igneum-app/src/engine.rs index 9224c4a8..d42d07e1 100644 --- a/app/igneum-app/src/engine.rs +++ b/app/igneum-app/src/engine.rs @@ -63,6 +63,8 @@ pub enum Cmd { SweepStop, SweepPin(String, bool), SweepEnable(bool), + /// Settings > Power control (config.rs power_control): on asks for administrator rights once, at that moment + PowerControl(bool), /// the probe that decides how caps are set during a sweep: Ok(true) = nvidia-smi -pl works from this process /// (the engine runs elevated), Ok(false) = the elevated helper is needed; Err = nvidia-smi did not answer SweepCapMode(Result), @@ -113,7 +115,7 @@ impl Shared { st.mining.accepted_total = settings.accepted_total; st.mining.fee_total = settings.fee_total; st.address = address_state(&settings, &wallet_path); - st.settings = crate::state::SettingsState { identities: settings.identities, vote: settings.vote, start_at_login: crate::platform::start_at_login_is_on(), auto_update: settings.auto_update, remote_jobs: settings.remote_jobs, prove: settings.prove, sweep: settings.sweep, dev_fee: settings.dev_fee, proof_verify_trust: settings.proof_verify_trust }; + st.settings = crate::state::SettingsState { identities: settings.identities, vote: settings.vote, start_at_login: crate::platform::start_at_login_is_on(), auto_update: settings.auto_update, remote_jobs: settings.remote_jobs, prove: settings.prove, sweep: settings.sweep, power_control: settings.power_control, power_note: String::new(), dev_fee: settings.dev_fee, proof_verify_trust: settings.proof_verify_trust }; st.dev_fee = crate::state::DevFeeState { on: settings.dev_fee, percent: if settings.dev_fee { 1 } else { 0 }, address: String::new(), line: String::new() }; st.live_page = packaged.live_page.clone(); st.finality.message = "waiting for the miner".into(); @@ -884,6 +886,11 @@ impl Engine { if applied.is_empty() && missing.is_empty() { self.shared.log(&format!("power cap: nothing to read back for {what}")); } + match power_control_after_prompt(&r) { + Some(note) => self.power_control_off(note), + None if !applied.is_empty() => self.st().settings.power_note = "administrator rights given; the cap and the sweep run".into(), + None => {} + } } Cmd::ElevatedDone(r) => { if let Some((line, what, want, _)) = self.power_via_host.take() { @@ -907,6 +914,7 @@ impl Engine { match st.mining.cards.iter().find(|c| c.key == key) { Some(c) if !c.sweep_supported => (c.name.clone(), false, c.sweep_note.clone()), Some(c) if !c.enabled => (c.name.clone(), false, "the card is switched off".to_string()), + Some(c) if !self.elevation_allowed() => (c.name.clone(), false, "switch Power control on in Settings first (Windows asks for administrator rights once)".to_string()), Some(c) => (c.name.clone(), true, String::new()), None => (key.clone(), false, "no such card".to_string()), } @@ -942,10 +950,38 @@ impl Engine { self.shared.event("info", &if pinned { format!("{name}: cap pinned; the sweep records but does not change it") } else { format!("{name}: the sweep chooses the cap again from its next run") }); } Cmd::SweepEnable(on) => { + let allowed = self.elevation_allowed(); + let on = on && allowed; self.shared.settings.lock().unwrap().sweep = on; self.shared.save_settings(); self.st().settings.sweep = on; - self.shared.event("info", if on { "efficiency sweep on: once after install, then weekly" } else { "efficiency sweep off; Sweep now on a card still runs one" }); + self.shared.event("info", if on { "efficiency sweep on: once after install, then weekly" } else if allowed { "efficiency sweep off; Sweep now on a card still runs one" } else { "efficiency sweep stays off: switch Power control on in Settings first" }); + } + Cmd::PowerControl(on) => { + { + let mut s = self.shared.settings.lock().unwrap(); + s.power_control = on; + s.sweep = on; // implied + } + self.shared.save_settings(); + { + let mut st = self.st(); + st.settings.power_control = on; + st.settings.sweep = on; + st.settings.power_note = if on { "asking Windows for administrator rights once".into() } else { String::new() }; + } + if on { + self.shared.event("info", "power control on: Windows asks for administrator rights once; the cap and the sweep need them"); + for c in self.st().mining.cards.iter_mut().filter(|c| c.vendor == "nvidia") { + c.power_applied = false; // the one prompt sets every cap now + } + self.apply_power_limits("power control on"); + if !self.power_busy { + self.st().settings.power_note = "on; no NVIDIA card needs a cap right now".into(); + } + } else { + self.power_control_off("power control off; the cap and the sweep do not run and nothing asks for administrator rights"); + } } Cmd::SweepCapMode(r) => self.sweep_mode_known(r), Cmd::SweepHelperDone(r) => { @@ -1422,21 +1458,10 @@ impl Engine { self.shared.log(&format!("power cap ({why}): deferred, a sweep is running")); return; } - let mut cmds = Vec::new(); - let mut what = Vec::new(); - { - let mut st = self.st(); - for c in st.mining.cards.iter_mut().filter(|c| c.vendor == "nvidia" && c.enabled && c.power_default_w > 0.0) { - let pct = if c.power_pct == 0 { 80 } else { c.power_pct.clamp(crate::sweep::MIN_PCT, 100) }; - c.power_pct = pct; - let watts = requested_watts(c); - if (c.power_limit_w - watts).abs() < 1.0 && c.power_applied { - continue; - } - cmds.push(format!("\"{}\" -i {} -pl {}", crate::platform::tool("nvidia-smi").display(), c.device, watts as u64)); - what.push(format!("{} {} W ({}% of {} W)", c.name, watts as u64, pct, c.power_default_w as u64)); - c.power_note = "setting the power cap (administrator prompt)".into(); - } + let allowed = self.elevation_allowed(); + let (cmds, what, held) = power_cap_plan(&mut self.st().mining.cards, allowed, &crate::platform::tool("nvidia-smi").display().to_string()); + if held > 0 { + self.shared.log(&format!("power cap ({why}): not asked, Power control is off in Settings ({held} card(s) would need it)")); } if cmds.is_empty() { return; @@ -1490,11 +1515,36 @@ impl Engine { if cmds.is_empty() { return; } - self.shared.log(&format!("restoring the GPU power limits: {}", cmds.join(" & "))); - match crate::platform::run_elevated(&cmds.join(" & ")) { - Ok(()) => self.shared.log("GPU power limits restored"), - Err(e) => self.shared.log(&format!("GPU power limits not restored ({e}); they reset at the next reboot")), + // no administrator prompt on quit (5 October 2026: the app never asks on its own); the limits reset at the + // next reboot, and the next start with Power control on sets them again + self.shared.log(&format!("GPU power limits left as set (they reset at the next reboot; no prompt on quit): {}", cmds.join(" & "))); + } + + /// Power control (config.rs power_control): may the engine ask for administrator rights for the cap or the sweep? + /// The elevated PC sweep job (--sweep) sets caps directly and counts as allowed. + fn elevation_allowed(&self) -> bool { + elevation_allowed(self.shared.settings.lock().unwrap().power_control, self.shared.runtime.sweep_only) + } + + /// Power control off, with the reason beside the switch and in the feed; a running or queued sweep ends. + fn power_control_off(&mut self, note: &str) { + { + let mut s = self.shared.settings.lock().unwrap(); + s.power_control = false; + s.sweep = false; } + self.shared.save_settings(); + { + let mut st = self.st(); + st.settings.power_control = false; + st.settings.sweep = false; + st.settings.power_note = note.to_string(); + } + self.sweep_queue.clear(); + if self.sweep.is_some() || self.sweep_pending.is_some() { + self.sweep_abort("power control is off"); + } + self.shared.event(if note.starts_with("power control off:") { "error" } else { "info" }, note); } /// nvidia-smi -l 5: power draw, GPU and memory temperature, the limit in force, every 5 s, as a child whose @@ -1524,16 +1574,10 @@ impl Engine { } if let Some((line, what, want, since)) = self.power_via_host.clone() { if now.duration_since(since) > Duration::from_secs(150) { + // no second prompt through PowerShell (5 October 2026): an unanswered prompt is a refusal self.power_via_host = None; - self.shared.log("the window host did not answer the elevated step in 150 s; running it through PowerShell"); - let shared = self.shared.clone(); - std::thread::spawn(move || { - let r = crate::platform::run_elevated(&line); - std::thread::sleep(Duration::from_millis(800)); - let back: std::collections::HashMap = crate::detect::nvidia_power_limits().into_iter().map(|(k, v)| (k, v.1)).collect(); - let _ = want; - shared.send(Cmd::PowerApplied(what, r, back)); - }); + self.shared.log(&format!("the window host did not answer the elevated step in 150 s: {line}")); + self.finish_power(what, Err("the administrator prompt was not answered in 150 s".into()), want); } } } @@ -1693,12 +1737,12 @@ impl Engine { if self.quitting || !self.running || self.power_busy || self.job_hold || self.jobs.holds_miners() { return; } - let (auto_on, paused) = { + let (auto_on, paused, allowed) = { let s = self.shared.settings.lock().unwrap(); let st = self.st(); - (s.sweep, st.mining.paused) + (s.sweep, st.mining.paused, elevation_allowed(s.power_control, self.shared.runtime.sweep_only)) }; - if paused { + if paused || !allowed { return; } let unix = crate::platform::unix_now(); @@ -1829,6 +1873,9 @@ impl Engine { /// The elevated helper (src/sweep.rs helper_script_*): one administrator prompt; it polls /sweep/cmd.txt. fn sweep_helper_start(&mut self, c: &CardState) -> Result<(), String> { + if !self.elevation_allowed() { + return Err("Power control is off in Settings".into()); + } let dir = self.sweep_dir(); std::fs::create_dir_all(&dir).map_err(|e| e.to_string())?; let _ = std::fs::write(dir.join("cmd.txt"), ""); @@ -3072,6 +3119,51 @@ impl Engine { } /// The watts a card's cap asks for: power_pct of the default limit, inside the card's min and max. +/// the project lead, 5 October 2026: "if we don't have to ask then don't ask". The NVIDIA power cap and the efficiency sweep need +/// administrator rights (one UAC prompt on Windows, pkexec on Linux); the engine builds an elevated command only when +/// Power control is on in Settings, or when it is itself the elevated PC sweep job (--sweep). +fn elevation_allowed(power_control: bool, sweep_only: bool) -> bool { + power_control || sweep_only +} + +/// The notice when the one prompt was refused, cancelled or not answered: Power control goes back off, no retries. +pub const POWER_CONTROL_REFUSED: &str = "power control off: administrator rights were not given"; + +/// After the elevated step: Some(notice) when the administrator prompt was refused, cancelled or timed out (the +/// words platform::run_elevated and the window host use), None when rights were given, even if a card then +/// disagreed with the readback. +fn power_control_after_prompt(r: &Result<(), String>) -> Option<&'static str> { + match r { + Err(e) if e.contains("administrator prompt") || e.contains("refused") || e.contains("cancel") => Some(POWER_CONTROL_REFUSED), + _ => None, + } +} + +/// The nvidia-smi -pl lines one elevated step runs, the human list of what they set, and how many cards were held +/// back because Power control is off (their note says so). Nothing is built when not allowed. +fn power_cap_plan(cards: &mut [CardState], allowed: bool, smi: &str) -> (Vec, Vec, usize) { + let mut cmds = Vec::new(); + let mut what = Vec::new(); + let mut held = 0; + for c in cards.iter_mut().filter(|c| c.vendor == "nvidia" && c.enabled && c.power_default_w > 0.0) { + let pct = if c.power_pct == 0 { 80 } else { c.power_pct.clamp(crate::sweep::MIN_PCT, 100) }; + c.power_pct = pct; + let watts = requested_watts(c); + if (c.power_limit_w - watts).abs() < 1.0 && c.power_applied { + continue; + } + if !allowed { + c.power_note = "power cap not set: Power control is off in Settings".into(); + held += 1; + continue; + } + cmds.push(format!("\"{smi}\" -i {} -pl {}", c.device, watts as u64)); + what.push(format!("{} {} W ({}% of {} W)", c.name, watts as u64, pct, c.power_default_w as u64)); + c.power_note = "setting the power cap (administrator prompt)".into(); + } + (cmds, what, held) +} + fn requested_watts(c: &CardState) -> f64 { let pct = if c.power_pct == 0 { 80 } else { c.power_pct.clamp(crate::sweep::MIN_PCT, 100) }; let mut w = c.power_default_w * pct as f64 / 100.0; @@ -3178,6 +3270,42 @@ pub fn parse_race(body: &str) -> Option { #[cfg(test)] mod tests { + #[test] + fn power_control_off_builds_no_elevated_command() { + // the decision (the project lead, 5 October 2026): off = the app never asks; the elevated PC sweep job is the exception + assert!(!super::elevation_allowed(false, false)); + assert!(super::elevation_allowed(true, false)); + assert!(super::elevation_allowed(false, true)); + let mut cards = vec![ + super::CardState { vendor: "nvidia".into(), enabled: true, device: "0".into(), name: "RTX 5090".into(), power_default_w: 575.0, power_limit_w: 575.0, power_pct: 80, ..Default::default() }, + super::CardState { vendor: "amd".into(), enabled: true, device: "1".into(), name: "RX 9070 XT".into(), power_default_w: 300.0, power_limit_w: 300.0, power_pct: 80, ..Default::default() }, + ]; + let (cmds, what, held) = super::power_cap_plan(&mut cards, false, "nvidia-smi"); + assert!(cmds.is_empty() && what.is_empty(), "{cmds:?}"); + assert_eq!(held, 1); + assert_eq!(cards[0].power_note, "power cap not set: Power control is off in Settings"); + let (cmds, what, held) = super::power_cap_plan(&mut cards, true, "nvidia-smi"); + assert_eq!(cmds, vec!["\"nvidia-smi\" -i 0 -pl 460".to_string()]); + assert_eq!(what, vec!["RTX 5090 460 W (80% of 575 W)".to_string()]); + assert_eq!(held, 0); + // a cap already in force asks for nothing either way + cards[0].power_limit_w = 460.0; + cards[0].power_applied = true; + assert!(super::power_cap_plan(&mut cards, true, "nvidia-smi").0.is_empty()); + } + + #[test] + fn a_refused_prompt_switches_power_control_off_with_the_notice() { + let refused = Err("the administrator prompt was refused, cancelled or timed out (exit 251)".to_string()); + assert_eq!(super::power_control_after_prompt(&refused), Some("power control off: administrator rights were not given")); + assert_eq!(super::power_control_after_prompt(&Err("the administrator prompt was cancelled".into())), Some(super::POWER_CONTROL_REFUSED)); + assert_eq!(super::power_control_after_prompt(&Err("elevated fail: the administrator prompt was cancelled".into())), Some(super::POWER_CONTROL_REFUSED)); + assert_eq!(super::power_control_after_prompt(&Err("the administrator prompt was not answered in 150 s".into())), Some(super::POWER_CONTROL_REFUSED)); + // rights given: the step ran, whatever the card then said + assert_eq!(super::power_control_after_prompt(&Ok(())), None); + assert_eq!(super::power_control_after_prompt(&Err("the elevated step exited with code 2".into())), None); + } + use super::{parse_race, sync_decision, Reading}; #[test] diff --git a/app/igneum-app/src/platform.rs b/app/igneum-app/src/platform.rs index 2534fa1e..d41c26d5 100644 --- a/app/igneum-app/src/platform.rs +++ b/app/igneum-app/src/platform.rs @@ -436,7 +436,7 @@ pub fn run_elevated(cmdline: &str) -> Result<(), String> { Ok(()) } else { let err = String::from_utf8_lossy(&out.stderr).trim().to_string(); - Err(if err.contains("canceled") || err.contains("cancelled") || err.is_empty() { "the administrator prompt was cancelled".into() } else { err }) + Err(elevated_failure(out.status.code(), &err)) } } #[cfg(target_os = "linux")] @@ -451,6 +451,52 @@ pub fn run_elevated(cmdline: &str) -> Result<(), String> { } } +/// The reason an elevated step failed, from the launcher's exit code and stderr: exit 251 (the prompt refused, +/// cancelled or timed out, `elevated_ps_line`) and the "canceled" wording name the prompt; any other code is the +/// step's own exit (the engine then keeps Power control on: rights were given). +pub fn elevated_failure(code: Option, stderr: &str) -> String { + if code == Some(ELEVATED_LAUNCH_FAILED) || stderr.contains("canceled") || stderr.contains("cancelled") { + "the administrator prompt was refused, cancelled or timed out".into() + } else if stderr.is_empty() { + format!("the elevated step exited with code {}", code.map(|c| c.to_string()).unwrap_or_else(|| "?".into())) + } else { + stderr.to_string() + } +} + +/// Doubles the single quotes of `s` for a single-quoted PowerShell literal. +pub fn ps_quote(s: &str) -> String { + s.replace('\'', "''") +} + +/// The PowerShell line that starts `file args` as administrator (one UAC prompt), waits, and exits with the child's +/// code. Every elevated launch of the app goes through here (the NVIDIA power cap, the sweep helper, the clock sync, +/// an elevated remote job) so the console flags live in one place: `-WindowStyle Hidden` is SW_HIDE on the new +/// process the AppInfo service creates; the elevated child cannot inherit this process's headless console, so without +/// it the child gets a console of its own (5 October 2026, PC 1 watcher, tools/windows/console-watch*.ps1). +/// A refused, cancelled or unanswered prompt makes Start-Process throw and `$p` stay null: that is exit 251 with the +/// reason on stderr, never `exit $p.ExitCode` = 0 (the 5 October 2026 driver job on PC 1 was reported done after +/// Windows cancelled its prompt at 122 s). +pub fn elevated_ps_line(file: &str, args: &str) -> String { + format!( + "try {{ $p = Start-Process -FilePath '{}' -ArgumentList '{}' -Verb RunAs -Wait -WindowStyle Hidden -PassThru -ErrorAction Stop }} catch {{ Write-Error ('elevated launch failed (UAC refused, cancelled or timed out): ' + $_.Exception.Message); exit 251 }}; if ($null -eq $p) {{ Write-Error 'elevated launch failed: no process'; exit 251 }}; exit $p.ExitCode", + ps_quote(file), + ps_quote(args) + ) +} + +/// The exit code `elevated_ps_line` uses when the elevated process never started (the prompt refused, cancelled or +/// timed out). +pub const ELEVATED_LAUNCH_FAILED: i32 = 251; + +/// The hidden PowerShell that runs `elevated_ps_line(file, args)`: blocking when run, one UAC prompt on the PC. +pub fn elevated_command(file: &str, args: &str) -> Command { + let mut c = Command::new(tool("powershell")); + c.args(["-NoProfile", "-ExecutionPolicy", "Bypass", "-Command", &elevated_ps_line(file, args)]); + quiet(&mut c); + c +} + /// Builds a command that runs without a console window on Windows. pub fn quiet(cmd: &mut Command) -> &mut Command { #[cfg(windows)] @@ -463,6 +509,34 @@ pub fn quiet(cmd: &mut Command) -> &mut Command { #[cfg(test)] mod tests { + #[test] + fn elevated_line_is_hidden_and_quoted() { + let l = super::elevated_ps_line(r"C:\WINDOWS\system32\cmd.exe", "/c echo it's & exit 3"); + assert!(l.starts_with("try { $p = "), "{l}"); + assert!(l.contains("-FilePath 'C:\\WINDOWS\\system32\\cmd.exe' -ArgumentList '/c echo it''s & exit 3' -Verb RunAs -Wait -WindowStyle Hidden -PassThru -ErrorAction Stop } catch {"), "{l}"); + assert!(l.contains("-Verb RunAs"), "{l}"); + assert!(l.contains("-WindowStyle Hidden"), "{l}"); + // a thrown Start-Process (the prompt refused) never falls through to `exit $p.ExitCode` + assert!(l.contains("exit 251 }; if ($null -eq $p) { Write-Error 'elevated launch failed: no process'; exit 251 }; exit $p.ExitCode"), "{l}"); + assert!(l.ends_with("exit $p.ExitCode"), "{l}"); + assert_eq!(super::ELEVATED_LAUNCH_FAILED, 251); + assert_eq!(super::elevated_failure(Some(251), "elevated launch failed (UAC refused, cancelled or timed out): ..."), "the administrator prompt was refused, cancelled or timed out"); + assert_eq!(super::elevated_failure(Some(1), "The operation was canceled by the user."), "the administrator prompt was refused, cancelled or timed out"); + assert_eq!(super::elevated_failure(Some(2), ""), "the elevated step exited with code 2"); + assert_eq!(super::elevated_failure(Some(3), "nvidia-smi: bad"), "nvidia-smi: bad"); + assert_eq!(super::ps_quote("a'b''c"), "a''b''''c"); + assert_eq!(super::ps_quote("plain"), "plain"); + } + + #[test] + fn elevated_command_is_a_hidden_powershell() { + let c = super::elevated_command("powershell.exe", "-NoProfile -File \"C:\\x y\\elevated.ps1\""); + let args: Vec = c.get_args().map(|a| a.to_string_lossy().into_owned()).collect(); + assert_eq!(&args[..4], ["-NoProfile", "-ExecutionPolicy", "Bypass", "-Command"]); + assert!(args[4].contains("-ArgumentList '-NoProfile -File \"C:\\x y\\elevated.ps1\"' -Verb RunAs -Wait -WindowStyle Hidden"), "{}", args[4]); + assert!(c.get_program().to_string_lossy().contains("powershell")); + } + #[test] fn token_redaction() { let l = "dashboard at http://127.0.0.1:58776/t/a3a01c537130bceeaa1f6118ba48d63e/ (log x)"; diff --git a/app/igneum-app/src/server.rs b/app/igneum-app/src/server.rs index 0849d683..e9884e36 100644 --- a/app/igneum-app/src/server.rs +++ b/app/igneum-app/src/server.rs @@ -319,6 +319,12 @@ fn api_post(shared: &Arc, path: &str, body: Value) -> Result { + let on = body.get("on").and_then(|v| v.as_bool()).ok_or("on missing")?; + shared.send(Cmd::PowerControl(on)); + Ok(json!({ "ok": true })) + } "/api/sweep/enable" => { let on = body.get("on").and_then(|v| v.as_bool()).ok_or("on missing")?; shared.send(Cmd::SweepEnable(on)); diff --git a/app/igneum-app/src/state.rs b/app/igneum-app/src/state.rs index 49bd0924..d33029f6 100644 --- a/app/igneum-app/src/state.rs +++ b/app/igneum-app/src/state.rs @@ -210,8 +210,13 @@ pub struct SettingsState { pub remote_jobs: bool, /// the prover service (src/prover.rs) pub prove: bool, - /// the efficiency sweep (src/sweep.rs): once after install, then weekly + /// the efficiency sweep (src/sweep.rs): once after install, then weekly; effective only with `power_control` pub sweep: bool, + /// the NVIDIA power cap and the sweep may ask for administrator rights (config.rs: default off, one prompt when + /// switched on) + pub power_control: bool, + /// the line beside the Power control switch: why it is off, or that the rights were given + pub power_note: String, /// the miner software's dev fee switch (settings; `--dev-fee 0` when off) pub dev_fee: bool, /// devnet only: the node trusts proof records without a verifier (`IGNEUM_PROOF_VERIFY=trust`) diff --git a/app/igneum-app/ui/app.js b/app/igneum-app/ui/app.js index b0dedfb3..7c6e16b2 100644 --- a/app/igneum-app/ui/app.js +++ b/app/igneum-app/ui/app.js @@ -392,6 +392,7 @@ if (typeof document !== 'undefined') (function () { $('s-trust').addEventListener('change', function () { var on = this.checked; api('api/settings', { proof_verify_trust: on }).then(function (r) { if (r.ok) toast(on ? 'Trust mode on (devnet only); the node restarts' : 'Trust mode off; the node restarts'); else { toast(r.error || 'could not change'); $('s-trust').checked = !on; } }); }); $('pv-setup').addEventListener('click', function () { api('api/prove/setup', {}).then(function (r) { toast(r.ok ? 'Setup started in its own window' : (r.error || 'could not start')); }); }); $('s-jobs-allow').addEventListener('change', function () { api('api/jobs/allow', { on: this.checked }); setTimeout(fillSettings, 800); }); + $('s-power-control').addEventListener('change', function () { var on = this.checked; api('api/power/control', { on: on }).then(function (r) { if (r.ok) toast(on ? 'Power control on: Windows asks for administrator rights once' : 'Power control off: nothing asks for administrator rights'); else { toast(r.error || 'could not change'); $('s-power-control').checked = !on; } setTimeout(fillSettings, 1500); }); }); $('s-sweep').addEventListener('change', function () { api('api/sweep/enable', { on: this.checked }).then(function (r) { if (r.ok) toast($('s-sweep').checked ? 'Sweep on: once after install, then weekly' : 'Sweep off'); }); }); $('s-jobs-check').addEventListener('click', function () { api('api/jobs/check', {}); $('s-jobs-note').textContent = 'Checking.'; setTimeout(fillSettings, 4000); }); @@ -1054,7 +1055,11 @@ if (typeof document !== 'undefined') (function () { $('s-jobs-allow').checked = !!j.allowed; $('s-prove').checked = !!(state.settings && state.settings.prove); $('s-trust').checked = !!(state.settings && state.settings.proof_verify_trust); + var pc = !!(state.settings && state.settings.power_control); + $('s-power-control').checked = pc; + $('s-power-note').textContent = (state.settings && state.settings.power_note) || ''; $('s-sweep').checked = !!(state.settings && state.settings.sweep); + $('s-sweep').disabled = !pc; $('s-jobs-key').textContent = j.key_fingerprint ? 'signing key sha256:' + j.key_fingerprint : ''; var parts = []; if (j.account) parts.push(j.account); diff --git a/app/igneum-app/ui/index.html b/app/igneum-app/ui/index.html index ab1f1625..ba54c8ce 100644 --- a/app/igneum-app/ui/index.html +++ b/app/igneum-app/ui/index.html @@ -299,6 +299,11 @@

Only when no verifier is found next to the engine: the node then includes proof records it never checked. Never on a testnet. A found verifier always wins. Changing this restarts the node.

+
+
power control
+ +

Windows asks for administrator rights once; the cap and the sweep need them. Off, the app never asks.

+
efficiency sweep
diff --git a/tools/windows/power-prompts-off.ps1 b/tools/windows/power-prompts-off.ps1 new file mode 100644 index 00000000..119e11d0 --- /dev/null +++ b/tools/windows/power-prompts-off.ps1 @@ -0,0 +1,28 @@ +# Stops the administrator prompts a 0.3.9 Igneum Miner app raises on its own on a PC (the project lead, 5 October 2026: "if we +# don't have to ask then don't ask"), through the settings API that app has, until the build with the Power control +# setting ships: the weekly/first-hour efficiency sweep off (api/sweep/enable), a running sweep stopped +# (api/sweep/stop). The power cap has no off switch in 0.3.9 (it asks at every app start and on a slider change, and +# nowhere else); this script reports the cards' cap state so the next prompt's source is known. A signed `run` job, +# not elevated, a few seconds: +# packaging/ota/publish-jobs.sh add --kind run --target ae432dc7 --timeout-minutes 3 \ +# --script tools/windows/power-prompts-off.ps1 --title "PC 1: sweep off (no administrator prompts)" --deploy +$ErrorActionPreference = 'Continue' +$base = (Get-Content (Join-Path $env:LOCALAPPDATA 'igneum\app\app.url') -Raw).Trim().TrimEnd('/') +function Snapshot([string] $tag) { + $s = Invoke-RestMethod -Uri "$base/api/state" -TimeoutSec 20 + $set = $s.settings + "RESULT $tag settings: sweep " + $set.sweep + " | power_control " + $(if ($null -ne $set.power_control) { $set.power_control } else { '(not in this build)' }) + " | prove " + $set.prove + " | remote_jobs " + $set.remote_jobs + " | version " + $s.version + foreach ($c in $s.mining.cards) { + if ($c.vendor -ne 'nvidia') { continue } + "RESULT $tag card: " + $c.name + " | enabled " + $c.enabled + " | power_pct " + $c.power_pct + " | limit " + $c.power_limit_w + " W of " + $c.power_default_w + " W default | applied " + $c.power_applied + " | pinned " + $c.pinned + " | sweep_state " + $c.sweep_state + " | note: " + $c.power_note + " | " + $c.sweep_note + } +} +Snapshot 'before' +$r = Invoke-RestMethod -Uri "$base/api/sweep/stop" -Method Post -ContentType 'application/json' -Body '{}' -TimeoutSec 30 +"RESULT api/sweep/stop: " + ($r | ConvertTo-Json -Compress) +$r = Invoke-RestMethod -Uri "$base/api/sweep/enable" -Method Post -ContentType 'application/json' -Body '{"on":false}' -TimeoutSec 30 +"RESULT api/sweep/enable off: " + ($r | ConvertTo-Json -Compress) +Start-Sleep -Seconds 3 +Snapshot 'after' +"RESULT note: the 0.3.9 power cap asks only at an app start or a slider change; no setting turns it off before the Power control build" +exit 0 From 3d504447ce952f05fb62bfabdf0f6dfd4ee3b660 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 20:46:03 +0000 Subject: [PATCH 04/63] HiveOS README: the per-card table from the S_p curve The "Once a Linux prover ships" column and the sentence above it now follow the proving agent's S_p curve (bench-log "proving v1" on proving-v1) and the app's default at 894022b (proving on at 24 GB or more, off below): 32 GB mines and proves the prototype shard (28.3 GB alone, 30.0 GB beside the miner, 2.5 GB spare); 24 GB proves the adopted v1 shard of 30,000 pgas (20.4 GB alone, about 22 GB beside the miner, approximate) from the fee switch at DAA 210,000 and nothing before it; 16 GB proves only empty shards alone (13.9 GB) and nothing beside the miner (15.7 GB), so in practice mines only; 12 GB proves nothing on SP1 6.8.1's GPU server; 8 GB mines only. The sources line stays; "before the v1-shard row" and the v1-budget caveat are gone, the curve measured it. README only; selftest.sh unchanged and still passing from df6598f. Co-Authored-By: Claude Fable 5.1 --- packaging/hive/README.md | 19 ++++++++----------- 1 file changed, 8 insertions(+), 11 deletions(-) diff --git a/packaging/hive/README.md b/packaging/hive/README.md index e5072ded..706f8844 100644 --- a/packaging/hive/README.md +++ b/packaging/hive/README.md @@ -11,20 +11,17 @@ workers; no `igneum-prove-host`, no `igneum-prove-export`, no SP1 GPU server. So 80% lottery share and nothing from the 20% proving share until a Linux prover build is published. The Ubuntu rig installer (`packaging/linux/README.md`, branch `rig-install`, commit dd632c1) carries a prover unit that idles in state `setup` for the same reason; its per-card table is the fuller version of the one below. What each card could do once -the prover ships, from the app's rule (`app/igneum-app/src/provedefault.rs` on `proving-v1`: 20 GB mines and proves on -one card, 16 GB proves only with the miner paused, under 16 GB mines only) and from the proving agent's measurements -of 5 October 2026, before the v1-shard row (`docs/bench-log.md`, "proving v1": jobs `prover-cost-pc2-pv1b`, -`chain-pc2-pv1c`, `memsweep-pc2-pv1`, `memminer-pc2-pv1`, one RTX 5090): +the prover ships, from the app's default (`app/igneum-app/src/provedefault.rs` at 440fd59 on `proving-v1`: proving on +at 24 GB or more, off below) and from the proving agent's S_p curve of 5 October 2026 (`docs/bench-log.md`, "proving +v1": jobs `memsweep-pc2-pv1`, `memminer-pc2-pv1` and the S_p curve, one RTX 5090, SP1 6.8.1's GPU server): | Card | Once a Linux prover ships | Measured on 5 October 2026 | |---|---|---| -| 8 GB | mines only | the prover alone peaks at 13.9 GB on an empty shard | -| 12 GB | mines only | the same 13.9 GB does not fit | -| 16 GB | proves only, with the miner paused | mine-and-prove peaked at 15.6 GB on empty shards and 16.75 GB with chained aggregation | -| 24 GB | mines and proves | 16.75 GB leaves 7 GB; the full prototype shard (28.3 GB with the card to itself) does not fit | -| 32 GB | mines and proves | the full prototype shard beside the miner peaked at 30.0 GB | - -The v1-budget shard (about 7 M cycles) is unmeasured and may move these lines. +| 8 GB | mines only | the prover's floor is 13.9 GB for an empty shard | +| 12 GB | mines only; proves nothing on this SP1 build | the same 13.9 GB floor | +| 16 GB | in practice mines only: proves empty shards alone, nothing beside the miner | 13.9 GB alone; 15.7 GB beside the miner leaves nothing for the display | +| 24 GB | proves the adopted v1 shard (30,000 pgas) from the fee switch at DAA 210,000, nothing before it | 20.4 GB alone, about 22 GB beside the miner (approximate, not measured on a 24 GB card); the prototype shard at 28.3 GB does not fit | +| 32 GB | mines and proves, prototype shard included | 28.3 GB alone, 30.0 GB beside the miner, 2.5 GB spare | ## Flight Sheet From 5c808b89e0a10865e3a527cba36b3ebb72704cab Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 21:21:51 +0000 Subject: [PATCH 05/63] Ember Tune: every card tuned for MH per watt out of the box, the fleet prior per card model in the signed manifest, the console and /miners priors table the project lead, 5 October 2026, 22:45 BST: "make sure we have ember tuning every single card for efficiency out of the box, the more data = the better the tune, make an awesome system." Built on lever 3 (docs/plans/miner-eff.md), lever 2's signed tuning section (docs/design/miner-tuning.md), the AMD telemetry helper (423936b, its --tune/--set-gmax/--set-plimit/ --reset contract) and the Power control switch (057f0ec). Design, data flow, tiers and the privacy line: docs/plans/ember-tune.md. - src/ember.rs (new): two knobs per card (power limit %, core clock cap MHz; memory clock never touched), the full plan (power ladder 100..50%, then the clock ladder 90..60% at the chosen power), the confirm plan (the fleet prior and one neighbour), the baseline plan (measure only), the marks (faulted, hot, memory_clock_dropped, unapplied, no_readings), the choice (best MH/W within 1% of the top rate, then rate, then draw), the fleet record (a hash of the install id, no address), the prior lookup and the kill switch (tuning.ember), the state machine on a fake clock. 9 unit tests. - engine.rs: tick_sweep schedules every NVIDIA, AMD and Apple card (120 s steady, 600 s to the boundary, no job hold, no pause, weekly, again after a driver major or program-class change, never under the manifest kill switch); the probe (nvidia-smi clocks.max.gr + driver_version and the direct/helper mode; igneum-gpu-telemetry --tune for AMD); tune_apply (nvidia-smi -pl / -lgc 0, / -rgc directly or through the helper; the AMD helper per request); Cmd::TuneProbe, Cmd::TuneSet; faults from rejected and mismatched hashes mark the step; the TUNE lines and the TUNE {json} record, uploaded with the log; the Tuned line on the card state. The NVIDIA helper starts only with Power control on: the --sweep job never counts as permission (no prompt on a PC with nobody there). - sweep.rs: the helper protocol gains lgc/rgc (clock cap and reset) and resets the clocks after 20 idle minutes. - state.rs, config.rs: the tune fields (clock cap, driver, class, source, the Tuned line); the nvidia-smi telemetry query carries clocks.gr and clocks.mem; the AMD sample line's plimit_pct and gmax_mhz are parsed. - ui: "Tuned: X MH/s at Y W (Z MH/W)" with the point, the source and when; measure-only cards say why; the Ember Tune switch; tune-line.test.mjs. - relay/lib/ember.mjs + relay/test/ember.test.mjs: the aggregation per (card model | driver major | program class): median point, MH/W, spread, samples, machines; five samples converge, an outlier does not move the median, baselines make no prior, de-duplication, the manifest merge keeps lever 2's cards. api/console.mjs fn=tuning and tools/console.mjs tuning; tools/tuning.mjs --priors [--write tuning.json] [--site] [--tuning-off]. - site: the fleet priors table on /miners (site/miner-priors.json), the lever text. - relay/playbooks/ember-tune-pc1.ps1: the PC 1 run (second engine with --sweep from a scratch copy of the install). Measured tonight: see the bench log entry that follows the PC 1 run. The 9070 XT left PC 1's bus at 20:40 UTC and the 5090 needs the administrator prompt the project lead cannot answer asleep, so tonight's PC 1 run is the baseline plan on the 5090 through the whole pipeline; the two-knob tune on both cards is owed. Co-Authored-By: Claude Fable 5.1 --- .github/workflows/ci.yml | 4 +- app/igneum-app/src/config.rs | 10 + app/igneum-app/src/detect.rs | 1 + app/igneum-app/src/ember.rs | 1033 ++++++++++++++++++++++++++ app/igneum-app/src/engine.rs | 646 ++++++++++++---- app/igneum-app/src/main.rs | 1 + app/igneum-app/src/state.rs | 15 + app/igneum-app/src/sweep.rs | 58 +- app/igneum-app/ui/app.js | 55 +- app/igneum-app/ui/index.html | 12 +- app/igneum-app/ui/tune-line.test.mjs | 44 ++ docs/plans/ember-tune.md | 164 ++++ relay/api/console.mjs | 15 + relay/lib/ember.mjs | 131 ++++ relay/playbooks/ember-tune-pc1.ps1 | 143 ++++ relay/test/ember.test.mjs | 110 +++ site/build.mjs | 21 +- site/miner-priors.json | 6 + site/miner.html | 6 +- site/miners.html | 10 +- tools/console.mjs | 7 + tools/tuning.mjs | 46 +- 22 files changed, 2354 insertions(+), 184 deletions(-) create mode 100644 app/igneum-app/src/ember.rs create mode 100644 app/igneum-app/ui/tune-line.test.mjs create mode 100644 docs/plans/ember-tune.md create mode 100644 relay/lib/ember.mjs create mode 100644 relay/playbooks/ember-tune-pc1.ps1 create mode 100644 relay/test/ember.test.mjs create mode 100644 site/miner-priors.json diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 9879a98b..185fa0f4 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -81,6 +81,6 @@ jobs: - name: ship tool self-test (version bump, the dl-both and public manifest helpers) run: node tools/ship-app.mjs --self-test - name: relay unit tests (parsers, secret compare, the wake endpoint) - run: node --test relay/test/parse.test.mjs relay/test/auth.test.mjs relay/test/wake.test.mjs + run: node --test relay/test/parse.test.mjs relay/test/auth.test.mjs relay/test/wake.test.mjs relay/test/ember.test.mjs - name: miner app notice strip and update card (ordering, keys, wording, timers, when the card shows) - run: node --test app/igneum-app/ui/notices.test.mjs app/igneum-app/ui/update-card.test.mjs + run: node --test app/igneum-app/ui/notices.test.mjs app/igneum-app/ui/update-card.test.mjs app/igneum-app/ui/tune-line.test.mjs diff --git a/app/igneum-app/src/config.rs b/app/igneum-app/src/config.rs index 050b79bb..f2b06d7b 100644 --- a/app/igneum-app/src/config.rs +++ b/app/igneum-app/src/config.rs @@ -29,6 +29,16 @@ pub struct CardPref { pub sweep_watts: f64, #[serde(default)] pub sweep_mhs: f64, + /// Ember Tune (src/ember.rs): the clock cap the last tune chose (0 = unlocked), the driver and program class it + /// ran under (a change makes the card due again), and the plan that produced it (full | confirm | baseline) + #[serde(default)] + pub sweep_clock_mhz: u32, + #[serde(default)] + pub sweep_driver: String, + #[serde(default)] + pub sweep_class: String, + #[serde(default)] + pub sweep_source: String, } #[derive(Clone, Serialize, Deserialize)] diff --git a/app/igneum-app/src/detect.rs b/app/igneum-app/src/detect.rs index 020d7a6a..358b7e39 100644 --- a/app/igneum-app/src/detect.rs +++ b/app/igneum-app/src/detect.rs @@ -66,6 +66,7 @@ fn card(index: usize, name: &str, vendor: &str, worker: &str, detail: &str, devi device: device.into(), enabled: true, state: "off".into(), + amd_ordinal: -1, ..Default::default() } } diff --git a/app/igneum-app/src/ember.rs b/app/igneum-app/src/ember.rs new file mode 100644 index 00000000..d839135e --- /dev/null +++ b/app/igneum-app/src/ember.rs @@ -0,0 +1,1033 @@ +//! Ember Tune: every card tuned for MH per watt out of the box, and the fleet's results folded into a prior that a +//! new card starts from (docs/plans/ember-tune.md). Two knobs per card: the power limit (percent of the card's +//! default) and the core clock cap (MHz; 0 = unlocked). The memory clock is never touched, and a step that drags it +//! down is marked and cannot win. This file is the logic, driven by an explicit clock so the tests run without a +//! card: the plans (full, confirm, baseline), the per-step rows with their marks, the choice rule, the fleet record, +//! the prior lookup and the state machine. The engine (src/engine.rs, `tick_tune` and the `Cmd::Tune*` commands) +//! owns the processes: NVIDIA through nvidia-smi (`-pl`, `-lgc 0,`, `-rgc`; administrator rights, so only with +//! Power control on), AMD through igneum-gpu-telemetry (`--set-plimit`, `--set-gmax`, `--reset`; no elevation on +//! Windows), Apple measure only. +//! +//! The lines in the app log (and on stdout under --sweep, which the PC job reads): +//! TUNE start card=
- +
@@ -301,13 +301,13 @@
power control
- -

Windows asks for administrator rights once; the cap and the sweep need them. Off, the app never asks.

+ +

Windows asks for administrator rights once; the NVIDIA cap and the tune need them. Off, the app never asks and NVIDIA cards measure only. AMD cards need no rights.

-
efficiency sweep
- -

On the live program, never restarting the worker: the cap steps from 100% of the card's default limit down to 50%, 15 s to settle and 60 s to measure per step, then holds the step with the most MH per watt. One administrator prompt per sweep. It stops at once if the card faults, a remote job takes the GPU, or the hour boundary is near. A cap you set with the slider is pinned: the sweep records, but leaves it. "Sweep now" on a card's tile runs one at any time; the table is in the log (SWEEP lines).

+
Ember Tune
+ +

On the live program, never restarting the worker: the power limit steps from 100% of the card's default down to 50%, then the core clock from its maximum down to 60%, 15 s to settle and 60 s to measure per step; the memory clock is never touched. The card keeps the point with the best MH per watt within 1% of its top rate. A step with a rejected or mismatched hash, a hot GPU or a dragged memory clock is reverted and marked. A card whose model the fleet already knows (5 or more reports) starts at that prior and confirms it in two steps. Every result goes back to the fleet without anything that identifies you. A cap you set with the slider is pinned: the tune records, but leaves it. "Tune now" on a card's row runs one at any time; the table is in the log (TUNE lines).

version
diff --git a/app/igneum-app/ui/tune-line.test.mjs b/app/igneum-app/ui/tune-line.test.mjs new file mode 100644 index 00000000..17b6daed --- /dev/null +++ b/app/igneum-app/ui/tune-line.test.mjs @@ -0,0 +1,44 @@ +// node --test app/igneum-app/ui/tune-line.test.mjs (no dependencies; CI runs it in the site job) +// The card row's Ember Tune line (app.js TuneLine): what a user sees per state, from the card state fields. +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { readFileSync } from 'node:fs'; +import { fileURLToPath } from 'node:url'; +import { dirname, join } from 'node:path'; + +const src = readFileSync(join(dirname(fileURLToPath(import.meta.url)), 'app.js'), 'utf8'); +const mod = { exports: {} }; +new Function('module', src)(mod); +const { model, point } = mod.exports.TuneLine; +const NOW = 1_800_000_000; +const card = over => ({ vendor: 'nvidia', sweep_state: 'idle', sweep_note: '', sweep_pct: 0, power_pct: 80, sweep_at: 0, tune_line: '', tune_source: '', tune_clock_mhz: 0, tune_control: true, pinned: false, ...over }); + +test('tuned: the line the brief asks for, with the point, the source and when', () => { + const m = model(card({ tune_line: 'Tuned: 122.3 MH/s at 290 W (0.422 MH/W)', tune_source: 'full', tune_clock_mhz: 2470, sweep_pct: 100, sweep_at: NOW - 3600 }), NOW); + assert.equal(m.kind, 'tuned'); + assert.equal(m.text, 'Tuned: 122.3 MH/s at 290 W (0.422 MH/W)'); + assert.equal(m.note, '2470 MHz at 100%, full tune, 1 h ago'); + const c = model(card({ tune_line: 'Tuned: 17.7 MH/s at 177 W (0.100 MH/W)', tune_source: 'confirm', sweep_pct: 90, sweep_at: NOW - 120, vendor: 'amd' }), NOW); + assert.equal(c.note, '90%, clock unlocked, from the fleet prior, confirmed, 2 min ago'); + const p = model(card({ tune_line: 'Tuned: 1 MH/s at 1 W (1.000 MH/W)', tune_source: 'full', sweep_pct: 70, pinned: true }), NOW); + assert.match(p.note, /your setting stays pinned$/); +}); + +test('measure only: Apple and NVIDIA without Power control say so beside the measured line', () => { + const a = model(card({ vendor: 'apple', tune_control: false, tune_line: 'Tuned: 26.7 MH/s at 38 W (0.703 MH/W)', tune_source: 'baseline', sweep_note: 'measure only on Apple silicon: the system sets the clocks and the power; no control exposed', sweep_at: NOW - 60 }), NOW); + assert.equal(a.kind, 'measured'); + assert.equal(a.text, 'Tuned: 26.7 MH/s at 38 W (0.703 MH/W) (measured as it runs, 1 min ago)'); + assert.match(a.note, /^measure only on Apple silicon/); + const n = model(card({ tune_control: false, tune_line: 'Tuned: 122.3 MH/s at 290 W (0.422 MH/W)', tune_source: 'baseline', sweep_note: 'measure only until Power control is on in Settings (Windows asks for administrator rights once)' }), NOW); + assert.equal(n.kind, 'measured'); + assert.match(n.note, /Power control/); +}); + +test('running, stopped, idle and off', () => { + assert.deepEqual(model(card({ sweep_state: 'running', sweep_note: 'tuning: holding 2472 MHz · 100% · 41 s (step 7 of 9)' }), NOW), { kind: 'running', text: 'tuning: holding 2472 MHz · 100% · 41 s (step 7 of 9)' }); + assert.equal(model(card({ sweep_note: 'tuning stopped: a remote job took the GPU' }), NOW).kind, 'stopped'); + assert.equal(model(card({}), NOW).text, 'tuning: not run yet (starts after 120 s of steady mining)'); + assert.equal(model(card({ sweep_note: 'tuning: waits for 120 s of steady mining' }), NOW).text, 'tuning: waits for 120 s of steady mining'); + assert.equal(model(card({ vendor: 'other' }), NOW).kind, 'off'); + assert.equal(point({ tune_clock_mhz: 0, sweep_pct: 0, power_pct: 0 }), ''); +}); diff --git a/docs/plans/ember-tune.md b/docs/plans/ember-tune.md new file mode 100644 index 00000000..22869c77 --- /dev/null +++ b/docs/plans/ember-tune.md @@ -0,0 +1,164 @@ +# Ember Tune: every card tuned for MH per watt, out of the box + +5 October 2026, night. the project lead: "make sure we have ember tuning every single card for efficiency out of the box, the +more data = the better the tune, make an awesome system." Branch `ember-tune`, worktree `../igneum-wt-ember-tune`, +on top of the AMD telemetry commit (7adcd4c, branch `opencl-rdna4-telemetry`) and the Power control commit (3562f26, +branch `job-console`), both cherry-picked. Lever 3 of docs/plans/miner-eff.md grows two knobs and a fleet memory; +lever 2 (docs/design/miner-tuning.md) carries the priors in the same signed `tuning` section. + +## 1. What a user sees + +| Moment | The card row says | What happened | +|---|---|---| +| First 2 minutes of mining | `tuning: waits for 120 s of steady mining` | The worker warms up; nothing is touched. | +| Tuning | `tuning: holding 2472 MHz · 100% · 41 s (step 7 of 9)` with a Stop button | One card at a time, on the live kernel, never restarting the worker. | +| Tuned | **Tuned: 122.3 MH/s at 290 W (0.422 MH/W)**, then `2470 MHz at 100%, full tune, 1 h ago` | The point is pinned on the card; the result went to the fleet. | +| Known model | the same line, `from the fleet prior, confirmed, 2 min ago` | The card started at its model's prior and confirmed it in two steps instead of nine. | +| Apple silicon | **Tuned: 26.7 MH/s at 38 W (0.703 MH/W)** `(measured as it runs)`, and `measure only on Apple silicon: the system sets the clocks and the power; no control exposed` | Nothing can be set; the number is still reported so the row and the fleet know what the card does. | +| NVIDIA, Power control off | the measured line and `measure only until Power control is on in Settings (Windows asks for administrator rights once)` | The app never raises the prompt by itself (5 October 2026). One switch, one prompt, and the full tune runs. | +| Slider moved | `your setting stays pinned` | A manual point is never overridden; the tune still measures and reports. | +| Stopped | `tuning stopped: a remote job took the GPU` and the card back where it was | Any fault reverts the step and the run. | +| Fleet pause | Settings: `tuning paused fleet-wide by the signed manifest` | The kill switch. | + +Settings: one switch, "Ember Tune: tune every card for MH per watt out of the box (once after install, then weekly, +and after a driver or program change)". AMD needs no rights. NVIDIA needs the Power control switch (one administrator +prompt) for both knobs; off, it measures only. + +## 2. The knobs, per vendor + +| Vendor | Power limit | Core clock cap | Memory clock | How | Rights | +|---|---|---|---|---|---| +| NVIDIA | `nvidia-smi -pl `, percent of the default, inside `power.min_limit` and `power.max_limit` | `nvidia-smi -lgc 0,`, percent of `clocks.max.gr`; `-rgc` = unlocked | never touched (`-lmc` is not used); read back as `clocks.mem` | directly when the engine is elevated, else the one-prompt helper (` pl `, ` lgc `, ` rgc` in `sweep/cmd.txt`) | administrator, so only with Power control on | +| AMD | `igneum-gpu-telemetry --card N --set-plimit ` (0 = default, -20 = 80%), inside the `tune` line's `plimit_range` | `--set-gmax ` inside `gmax_range`; `--reset` for the default point | not settable through ADLX on RDNA 4; read back as `mclk_mhz`, and a step whose mean memory clock falls under 95% of the baseline's is marked and cannot win | the helper, one process per request, exit 0 and a `tune ... ok` line | none on Windows (ADLX manual tuning); root on Linux, so measure only there | +| Apple | none | none | none | measure only | none | + +Vendor limits are never exceeded and the floor is never undercut: the plan clamps every point (`Limits::clamp_clock`, +`Limits::watts_for`), and a clock floor the vendor does not report is 60% of the maximum. + +## 3. The plan and the choice + +Full plan (a new model, or a prior that lost its confirm check): the power ladder 100, 90, 80, 70, 60, 50% at the +unlocked clock (duplicate watts dropped where the card's floor clamps them), then the clock ladder 90, 80, 70, 60% +of the maximum at the power point the power ladder chose. 60 s hold after 15 s settle per step; 9 steps on an +RTX 5090 (five power, four clock), about 12 minutes. + +Confirm plan (the model's prior has 5 or more reports): the prior's point, then one neighbour (the next clock step up +when the prior caps the clock, else one power step down). If the neighbour beats the prior by over 1% MH/W, the full +plan is queued; else the prior stands. Two steps, about 3 minutes. + +Baseline plan (measure only): one step at the card's current point. The "before" number for the row and the fleet. + +The choice (`ember::choose`): among the usable steps whose rate is within the tolerance (1%, settable from the +manifest) of the fastest step, the best MH per watt; within 1% on efficiency the higher rate; within 1% on both the +lower draw. A card never gives up more than the tolerance in blocks for the saving. A step is unusable when it is +marked: `faulted` (a rejected or mismatched hash during the hold: the step is reverted and marked), `hot` (the GPU +reached 85 C; the run aborts at 90), `memory_clock_dropped`, `unapplied` (the readback disagreed with the request), +`no_readings` (under three draw samples or no STATUS line). + +## 4. The data flow + +``` +card mines 120 s ──> probe (limits, driver, how to set) ──> plan ──> steps ──> choice ──> point pinned + │ + app log: TUNE start / TUNE card=.. step=.. / TUNE chosen / TUNE {json} (and stdout under --sweep) + │ + log upload (every minute, the existing intake, site/api/log.mjs) ──> Neon miner_logs + │ + relay/lib/ember.mjs aggregate: per (card model | driver major | program class) + median clock cap (10 MHz), median power %, median MH/W, MH/s, W, spread (MAD %), samples, machines + │ │ + console: /r//c/tuning, `node tools/console.mjs tuning` site: tools/tuning.mjs --priors --site + │ -> site/miner-priors.json -> /miners#priors + tools/tuning.mjs --priors --write tuning.json (priors + ember settings beside the kernel-variant cards) + │ + packaging/ota/publish-manifest.sh --tuning tuning.json --deploy (signed; carried over when not given) + │ + every app: /tuning.json ──> ember::settings_of (kill switch, min samples, tolerance, period) + ──> ember::prior_of(key) ──> a new card's confirm plan +``` + +The record (`ember::record_json`): `ts`, `machine` (the first 8 hex of SHA-256 over the install id; the id itself +is random per install and never sent), `app`, `os`, `card`, `vendor`, `driver`, `driver_major`, `class`, `key`, +`plan`, `steps` (the full table: clock, power %, limit, watts, MH/s, MH/W, core and memory clock, hottest reading, +faults, mark), `chosen`, `before` (the full plan's 100% step), `eff`, `mhs`, `watts`. The key: `||`, the class from the worker's race line (`l128w16` today; `v2` before a +race has run). + +## 5. Scheduling and safety + +| Rule | Where | +|---|---| +| One card at a time; the card must have mined 120 s and have a STATUS line | `tick_sweep` | +| Never under a remote job hold, a pause, inside 600 s of the hour boundary, or while the app quits | `tick_sweep`, `sweep_drive` | +| Due once after install, every 7 days (manifest `ember.period_s`), and when the driver major or the program class changed since the last tune | `tick_sweep` (`CardPref.sweep_driver`, `sweep_class`) | +| A pinned card (the slider) is measured, never changed | `sweep_finish` | +| Kill switch: `tuning.ember.enabled = false` in the signed manifest stops every tune fleet-wide; the Settings line says so | `ember::settings_of`, `tick_sweep` | +| Faults: a rejected or mismatched hash marks the step; the card leaving `mining`, a worker error, a job, a pause or 90 C aborts the run and restores the point from before | `Run::sample_fault`, `sweep_drive`, `sweep_abort` | +| Memory clock held: never set; a step that drags it under 95% of the baseline's cannot win | `Row::from_samples` | +| Vendor limits: every point clamped to the reported range; the clock floor 60% when none is reported | `Limits` | +| No prompt the user did not ask for: the NVIDIA helper starts only with Power control on; the `--sweep` job never counts as permission | `sweep_probe_known`, `sweep_helper_start` | +| The elevated helper restores the limit and resets the clocks by itself after 20 idle minutes | `sweep::helper_script_*` | + +## 6. Tests + +| Test | What it fixes | +|---|---| +| `ember::tests::the_full_plan_is_the_power_ladder_then_the_clock_ladder_at_the_chosen_power` | 5 + 4 steps on the 5090's limits, the clamps, the dynamic second half, the 1% and 5% choices | +| `limits_never_exceed_the_vendor_or_undercut_the_floor` | clamps | +| `the_choice_keeps_the_best_mh_per_watt_within_the_rate_tolerance` | the rule, the ties, marked rows never win | +| `the_guards_mark_a_step_so_it_cannot_win` | faulted, hot, memory clock, unapplied, no readings, the line | +| `a_fault_during_a_step_reverts_it_and_the_run_goes_on` | the state machine with a fake clock: the faulted 70% step is marked and never chosen | +| `the_confirm_plan_checks_the_prior_and_its_neighbour` | the two steps, Keep against FullDue, a prior outside the range clamped | +| `a_baseline_plan_measures_the_card_as_it_runs` | no control, still a number and the Tuned line | +| `the_record_and_the_prior_round_trip_through_the_manifest_shape` | record fields (no address, no host), `priors` and `ember` beside `cards`, the sample floor, the kill switch | +| `control_reasons_per_vendor` | who measures only and why | +| `sweep::tests::helper_scripts_carry_the_protocol` | the helper's `pl`, `lgc`, `rgc` | +| `relay/test/ember.test.mjs` | five samples converge (2,470 MHz at 100%), an outlier (0.908 MH/W at 1,854 MHz) moves nothing, baseline records make no prior, de-duplication, the manifest merge keeps lever 2's cards, the canonical round trip, AMD keys | +| `app/igneum-app/ui/tune-line.test.mjs` | the row line per state | + +Run: `cargo test -p igneum-app ember sweep` (on a PC through the build job, or on the Mac under the build lock), +`node --test relay/test/ember.test.mjs app/igneum-app/ui/tune-line.test.mjs`. + +## 7. The tier consequences + +| Tier | What Ember Tune does for it | What it costs | +|---|---|---| +| A laptop GPU (NVIDIA, 60 to 115 W) | the power ladder usually finds the vendor floor binding; the clock ladder is where a memory-bound program saves watts; the thermal mark keeps a hot chassis from winning a step it cannot hold | about 12 minutes once, then 3 minutes a week; under 1% of the hour during the tune (the worker never stops) | +| One 8 GB card | the same two knobs; the 8 GB card is identities-limited (2 by default), the tune does not change that | the same | +| One 12 or 16 GB card | the same | the same | +| One 24 or 32 GB card (the 5090) | the draw sits far under the cap (290 W under 460 W on PC 1), so the power ladder is flat and the clock ladder is the lever; expected saving from the 4 October stability line: tens of watts at under 1% rate, to be measured | the same | +| A rig (several cards) | one card at a time, so a six-card rig takes about 70 minutes to tune once; every card of one model after the first starts at the prior (3 minutes); the tune never touches a card a remote job holds | linear in cards once, then the confirm plan | +| A pool user | the same per card; a pool submits the same hashes, so the 1% rate tolerance is the same 1% of shares | the same | +| AMD on Linux | measure only (sysfs needs root); the row says so | 60 s a week | +| Apple silicon | measure only; the row says so | 60 s a week | + +Privacy line: what is uploaded is the record in section 4 and nothing else: a hash of the random install id, the +card model, the driver version, the OS, the program class, the step table and the chosen point. No address, no +hostname, no raw machine id, no user name. The public priors table carries only the aggregate per model. + +## 8. Measurements + +### PC 1, 5 October 2026 (night) + +Tonight's constraints, read from PC 1's own uploads: the installed app runs as `DESKTOP-KMCV30N\Admin` with +`elevated=False` (the account line at 19:02:33 UTC), the two in-app sweep attempts at 20:09 UTC aborted on the +cancelled administrator prompt (`SWEEP aborted ... the_elevated_helper_did_not_run_(the_administrator_prompt_was_cancelled)`), +so no stored sweep result exists from today, and the RX 9070 XT left the PCI bus at about 20:40 UTC (eGPU link, +not restarted tonight). NVIDIA's `-pl` and `-lgc` need administrator rights, the project lead is asleep, and the app never raises +the prompt by itself, so tonight's run on PC 1 is the baseline plan on the 5090 through the whole pipeline (probe, +measure, TUNE record, upload, aggregation, prior shape in a test manifest). The two-knob tune on the 5090 and the +9070 XT run are owed: the 5090 the moment Power control is switched on (one prompt, then the tune runs by itself +within 2 minutes of steady mining), the 9070 XT when the card is back on the bus. + +(The run's numbers are appended below when the job reports.) + +## 9. Open + +- The NVIDIA clock readback: `nvidia-smi -lgc` is confirmed only through the core clock during the hold (a mean over + the cap by 5% marks the step `unapplied`); the first run with Power control on tells whether the driver honours + the lock on the 5090 under this kernel. +- ADLX on RDNA 4 exposes no memory-clock setter; the memory-clock mark is the guard. The telemetry agent's 9070 XT + sweep tells whether a core cap drags the memory clock on that card. +- The confirm plan's neighbour is one step; a second neighbour (the other knob) would cost 75 s more and catch a + prior that is wrong on both knobs. +- Intel: no knob yet; the row says measure only. diff --git a/relay/api/console.mjs b/relay/api/console.mjs index 7dcb5358..546e7891 100644 --- a/relay/api/console.mjs +++ b/relay/api/console.mjs @@ -9,10 +9,13 @@ // GET chain igneum.network/api/live trimmed + the Hetzner results item // GET log?limit=&since= work-log items (kinds log, build, note), newest first // GET results bench entries (synced from docs/bench-log.md) + the FUD ledger counts +// GET tuning?days=30&min=5 Ember Tune: the fleet priors per (card model, driver major, program class) from the +// TUNE records in miner_logs (relay/lib/ember.mjs), with the sample counts and MH/W // POST post {kind,title,body,who,key?,meta?} one item; with key it upserts // POST sync {items:[...]} bulk upsert by key import { neon, authed, readJson, str, iso } from '../lib/relay.mjs'; import { kv, kvNum, lastMatch, FAULT, parseLabel, parseMinerTail, parseHeader, parseAppTail, STALE_S, markStale } from '../lib/parse.mjs'; +import { parseRecords, aggregate } from '../lib/ember.mjs'; const json = (res, status, obj) => { res.status(status).setHeader('Content-Type', 'application/json; charset=utf-8'); res.end(JSON.stringify(obj)); }; const CACHE_MS = 10_000; @@ -212,6 +215,13 @@ async function chain(sql) { hetzner: het.length ? itemOut(het[0]) : null, }; } +/// Ember Tune: every TUNE record of the window, folded into priors (the same aggregation the publisher uses). +async function tuning(sql, days, min) { + const rows = await sql(`SELECT lines FROM miner_logs WHERE received_at > now() - ($1 || ' days')::interval AND lines LIKE '%TUNE {%' ORDER BY received_at DESC LIMIT 2000`, [String(days)]); + const records = rows.flatMap(r => parseRecords(r.lines)); + const { priors, table } = aggregate(records, { minSamples: min }); + return { days, min_samples: min, records: records.length, priors, table }; +} async function results(sql) { const [bench, ledger] = await Promise.all([ sql(`SELECT * FROM console_items WHERE kind = 'bench' ORDER BY (meta->>'date') DESC NULLS LAST, (meta->>'pos')::int DESC LIMIT 60`), @@ -252,6 +262,11 @@ export default async function handler(req, res) { if (fn === 'builds') return json(res, 200, { ok: true, now: new Date().toISOString(), ...(await cached('builds', () => builds(sql))) }); if (fn === 'chain') return json(res, 200, { ok: true, ...(await cached('chain', () => chain(sql))) }); if (fn === 'results') return json(res, 200, { ok: true, ...(await cached('results', () => results(sql))) }); + if (fn === 'tuning') { + const days = Math.min(365, Math.max(1, Number(q.days) || 30)); + const min = Math.min(100, Math.max(1, Number(q.min) || 5)); + return json(res, 200, { ok: true, now: new Date().toISOString(), ...(await cached(`tuning-${days}-${min}`, () => tuning(sql, days, min))) }); + } if (fn === 'log') { const limit = Math.min(300, Math.max(1, Number(q.limit) || 100)); const params = []; let where = `kind IN ('log','build','note')`; diff --git a/relay/lib/ember.mjs b/relay/lib/ember.mjs new file mode 100644 index 00000000..166b7027 --- /dev/null +++ b/relay/lib/ember.mjs @@ -0,0 +1,131 @@ +// Ember Tune, the fleet side (docs/plans/ember-tune.md): the TUNE records every app uploads with its log are folded +// into one prior per (card model, driver major, program class): the median chosen point, its spread and the sample +// count. The publisher writes the priors into the signed manifest's `tuning` section beside the kernel-variant +// cards (tools/tuning.mjs --write), the console shows them (api/console.mjs fn=tuning, tools/console.mjs tuning), +// and the public bench table lists them per model (site/miner-priors.json). No dependencies; the tests in +// relay/test/ember.test.mjs drive these functions with a fixture of captured records. +// +// A record (app/igneum-app/src/ember.rs record_json): {ts, machine (a hash of the install id), app, os, card, vendor, +// driver, driver_major, class, key, plan: full|confirm|baseline, steps: [{clock_mhz, power_pct, limit_w, watts, mhs, +// eff, gclk, mclk, tmax, faults, mark}], chosen: {...}, before: {...}|null, eff, mhs, watts}. Nothing identifies the +// owner: no address, no hostname, no raw machine id. + +/// The TUNE records inside uploaded log text, de-duplicated on (machine, card, ts) because the log is re-sent every +/// minute. Baseline records (measure only) are kept apart: they say what a card does untuned, never what to set. +export function parseRecords(text) { + const out = []; + for (const line of String(text || '').split('\n')) { + const i = line.indexOf('TUNE {'); + if (i < 0) continue; + let rec; + try { rec = JSON.parse(line.slice(i + 5)); } catch { continue; } + if (!rec || !rec.card || !rec.key || !rec.plan) continue; + out.push(rec); + } + return out; +} + +export function dedupe(records) { + const seen = new Set(); + const out = []; + for (const r of records) { + const k = `${r.machine}|${r.card}|${r.ts}`; + if (seen.has(k)) continue; + seen.add(k); + out.push(r); + } + return out; +} + +export const median = xs => { const s = xs.filter(x => Number.isFinite(x)).sort((a, b) => a - b); return s.length ? (s.length % 2 ? s[(s.length - 1) / 2] : (s[s.length / 2 - 1] + s[s.length / 2]) / 2) : 0; }; + +/// The median absolute deviation as a percent of the median (0 for one sample or a zero median). +export function spreadPct(xs) { + const m = median(xs); + if (!m || xs.length < 2) return 0; + return Number((median(xs.map(x => Math.abs(x - m))) / m * 100).toFixed(2)); +} + +const usable = r => r && r.chosen && r.chosen.mark === 'ok' && r.chosen.eff > 0 && (r.plan === 'full' || r.plan === 'confirm'); + +/// Folds records into priors: one per key, from the full and confirm records with a usable chosen point. The point +/// is the median clock cap and the median power percent (each rounded to the step the apps use: 10 MHz, 1%), the +/// efficiency, rate and draw are medians, the spread is the MAD of the efficiency in percent, `samples` counts the +/// records and `machines` the distinct install hashes. An outlier (one bad card, one hot room) moves the median by +/// at most one rank, never by its size. Baseline records are summarised beside the prior as `baseline` (median +/// MH/W untuned) so the console can show the gain. +export function aggregate(records, { minSamples = 1 } = {}) { + const byKey = new Map(); + for (const r of dedupe(records)) { + const g = byKey.get(r.key) || { key: r.key, card: r.card, vendor: r.vendor || '', driver_major: r.driver_major || '', class: r.class || 'v2', tuned: [], baseline: [], machines: new Set() }; + byKey.set(r.key, g); + g.machines.add(r.machine); + if (usable(r)) g.tuned.push(r); + else if (r.plan === 'baseline' && r.chosen && r.chosen.eff > 0) g.baseline.push(r); + } + const priors = {}; + const table = []; + for (const g of byKey.values()) { + const t = g.tuned; + const row = { + key: g.key, card: g.card, vendor: g.vendor, driver_major: g.driver_major, class: g.class, + samples: t.length, machines: g.machines.size, + baseline_samples: g.baseline.length, + baseline_eff: g.baseline.length ? Number(median(g.baseline.map(r => r.chosen.eff)).toFixed(4)) : null, + baseline_mhs: g.baseline.length ? Number(median(g.baseline.map(r => r.chosen.mhs)).toFixed(2)) : null, + baseline_watts: g.baseline.length ? Number(median(g.baseline.map(r => r.chosen.watts)).toFixed(1)) : null, + }; + if (t.length) { + const effs = t.map(r => r.chosen.eff); + const prior = { + clock_mhz: Math.round(median(t.map(r => r.chosen.clock_mhz)) / 10) * 10, + power_pct: Math.round(median(t.map(r => r.chosen.power_pct))), + eff: Number(median(effs).toFixed(4)), + mhs: Number(median(t.map(r => r.chosen.mhs)).toFixed(2)), + watts: Number(median(t.map(r => r.chosen.watts)).toFixed(1)), + spread_pct: spreadPct(effs), + samples: t.length, + machines: g.machines.size, + card: g.card, + vendor: g.vendor, + driver_major: g.driver_major, + class: g.class, + updated: new Date(Math.max(...t.map(r => Number(r.ts) || 0)) * 1000).toISOString().replace(/\.\d{3}Z$/, 'Z'), + }; + // the untuned reference: the full plan's first step (the power ladder's 100%), else the baseline records + const befores = t.map(r => r.before && r.before.eff > 0 ? r.before.eff : null).filter(x => x !== null); + if (befores.length) prior.before_eff = Number(median(befores).toFixed(4)); + else if (row.baseline_eff) prior.before_eff = row.baseline_eff; + if (prior.before_eff) prior.gain_pct = Number(((prior.eff / prior.before_eff - 1) * 100).toFixed(1)); + Object.assign(row, prior); + if (t.length >= minSamples) priors[g.key] = prior; + } + table.push(row); + } + table.sort((a, b) => (b.samples - a.samples) || (a.key < b.key ? -1 : 1)); + return { priors, table }; +} + +/// The manifest's tuning section with the priors folded in: the kernel-variant `cards` object is kept as is, +/// `priors` replaces the previous priors (a key that lost its samples drops out), `ember` carries the settings. +export function mergeTuning(existing, priors, ember = {}) { + const base = existing && typeof existing === 'object' ? existing : {}; + const cards = base.cards && typeof base.cards === 'object' && !Array.isArray(base.cards) ? base.cards : {}; + const settings = { enabled: true, min_samples: 5, rate_tolerance_pct: 1, ...(base.ember && typeof base.ember === 'object' ? base.ember : {}), ...ember }; + return { ...base, updated: new Date().toISOString().replace(/\.\d{3}Z$/, 'Z'), cards, ember: settings, priors: priors || {} }; +} + +/// A prior as a card starts from it (app/igneum-app/src/ember.rs prior_of): None under the sample floor. +export function priorFor(tuning, key, minSamples) { + const p = tuning && tuning.priors && tuning.priors[key]; + const floor = Number.isFinite(minSamples) ? minSamples : (tuning && tuning.ember && tuning.ember.min_samples) || 5; + if (!p || !(p.samples >= floor)) return null; + return { clock_mhz: p.clock_mhz || 0, power_pct: Math.min(100, Math.max(50, p.power_pct || 100)), eff: p.eff, samples: p.samples }; +} + +/// One text line per prior for the console and the CLI. +export function priorLine(p) { + const point = p.clock_mhz ? `${p.clock_mhz} MHz at ${p.power_pct}%` : `${p.power_pct}% (clock unlocked)`; + const gain = p.gain_pct != null ? ` (${p.gain_pct >= 0 ? '+' : ''}${p.gain_pct}% over untuned ${p.before_eff} MH/W)` : ''; + return `${p.card.replace(/_/g, ' ')} | driver ${p.driver_major} | ${p.class}: ${point}, ${p.eff} MH/W${gain}, ${p.mhs} MH/s at ${p.watts} W, spread ${p.spread_pct}%, ${p.samples} sample(s) from ${p.machines} machine(s)`; +} diff --git a/relay/playbooks/ember-tune-pc1.ps1 b/relay/playbooks/ember-tune-pc1.ps1 new file mode 100644 index 00000000..2e135be2 --- /dev/null +++ b/relay/playbooks/ember-tune-pc1.ps1 @@ -0,0 +1,143 @@ +# Igneum run job: Ember Tune end to end on PC 1 (machine ae432dc7), unattended. 5 October 2026. +# Published as a `run` job with --stop-miners (docs/plans/ember-tune.md): the installed app stops its miners and +# holds them; this script takes the engine that carries src/ember.rs (the one the fetch job put in +# \jobs\ember-kit-1\igneum-app-ember.exe, else the installed one; the AMD helper with the --tune and --set +# commands from the telemetry agent's fetch job, \jobs\amd-kit-1\kit\igneum-gpu-telemetry.exe), copies the install folder to a scratch +# folder beside it, swaps the engine in, and starts that SECOND engine with `--sweep` in a scratch data folder (the +# real settings.json, machine-id and wallet.json copied in; remote jobs, auto-update and proving switched off +# there). That engine finds the installed app's node on 127.0.0.1:26610, mines on every card with the live program, +# tunes them one after the other (the full two-knob plan where the card can be controlled, the baseline measurement +# where it cannot: NVIDIA without administrator rights, Apple), prints every TUNE line on stdout, uploads its log +# (the TUNE {json} record reaches the intake) and quits. Every TUNE line is re-emitted as a RESULT line, so +# `node tools/jobs.mjs ` shows the table. Before and after, nvidia-smi's limits and clocks and the AMD +# helper's `--tune` lines are printed, so the restore can be read. The installed app's miners restart when the job +# ends. Not elevated: nothing asks for administrator rights (the project lead asleep, 5 October 2026); the NVIDIA card is +# therefore measure only tonight unless the engine finds itself elevated. +$ErrorActionPreference = 'Continue' +$budgetMinutes = 35 +$started = Get-Date +$deadline = $started.AddMinutes($budgetMinutes) +function Say([string] $m) { Write-Host ("[" + (Get-Date -Format 'HH:mm:ss') + "] " + $m) } + +# the installed engine: the per-user install (0.3.3+), else Program Files +$installDir = $null +foreach ($d in @((Join-Path $env:LOCALAPPDATA 'Programs\Igneum Miner'), (Join-Path $env:ProgramFiles 'Igneum Miner'))) { + if (Test-Path (Join-Path $d 'igneum-app.exe')) { $installDir = $d; break } +} +if (-not $installDir) { Write-Output 'RESULT TUNE error=no_engine reason=igneum-app.exe_not_found'; exit 2 } +$appData = $env:IGNEUM_APP_DATA +if (-not $appData) { $appData = Join-Path $env:LOCALAPPDATA 'igneum' } +$appDir = $env:IGNEUM_APP_DIR +if (-not $appDir) { $appDir = Join-Path $appData 'app' } + +# the scratch install: the whole folder (workers, node, helper, DLLs) with the Ember engine swapped in +$root = Join-Path $env:LOCALAPPDATA 'igneum-tune' +$bin = Join-Path $root 'bin' +$sApp = Join-Path $root 'app' +$sLogs = Join-Path $root 'logs' +New-Item -ItemType Directory -Force -Path $root, $sApp, $sLogs | Out-Null +if (Test-Path $bin) { Remove-Item -LiteralPath $bin -Recurse -Force -ErrorAction SilentlyContinue } +Copy-Item -LiteralPath $installDir -Destination $bin -Recurse -Force +$ember = Join-Path $appDir 'jobs\ember-kit-1\igneum-app-ember.exe' +if (Test-Path $ember) { + Copy-Item -LiteralPath $ember -Destination (Join-Path $bin 'igneum-app.exe') -Force + Say ("engine: the Ember build from " + $ember) +} else { + Say 'engine: the installed one (no jobs\ember-kit-1\igneum-app-ember.exe); an older engine ignores the tune and reports no_rows' +} +$helper = Join-Path $appDir 'jobs\amd-kit-1\kit\igneum-gpu-telemetry.exe' +if (Test-Path $helper) { + Copy-Item -LiteralPath $helper -Destination (Join-Path $bin 'igneum-gpu-telemetry.exe') -Force + Say ("helper: the Ember build of igneum-gpu-telemetry from " + $helper + " sha256=" + (Get-FileHash -LiteralPath $helper -Algorithm SHA256).Hash.ToLower()) +} else { Say 'helper: the installed igneum-gpu-telemetry (no jobs\amd-kit-1\kit\igneum-gpu-telemetry.exe); without --tune the AMD card measures only' } +$exe = Join-Path $bin 'igneum-app.exe' +$ver = (& $exe --version 2>&1 | Out-String).Trim() +Say ("engine: " + $exe + " (" + $ver + ")") +Write-Output ("RESULT TUNE engine " + $ver + " sha256=" + (Get-FileHash -LiteralPath $exe -Algorithm SHA256).Hash.ToLower()) +if ($ver -notmatch 'igneum-app (\d+)\.(\d+)\.(\d+)') { Write-Output 'RESULT TUNE error=version_unknown'; exit 2 } + +foreach ($f in @('settings.json', 'machine-id', 'wallet.json', 'tuning.json')) { + $src = Join-Path $appDir $f + if (Test-Path $src) { Copy-Item -LiteralPath $src -Destination (Join-Path $sApp $f) -Force } +} +# the second engine must not poll jobs (it would see this one), update itself, or prove; the tune is on +$sj = Join-Path $sApp 'settings.json' +if (Test-Path $sj) { + try { + $j = Get-Content -LiteralPath $sj -Raw | ConvertFrom-Json + $j.remote_jobs = $false; $j.auto_update = $false; $j.prove = $false; $j.paused = $false; $j.setup_done = $true; $j.sweep = $true + # every card is due: the stored results are cleared in the COPY only + if ($j.cards) { foreach ($p in $j.cards.PSObject.Properties) { $p.Value.sweep_at = 0; $p.Value.pinned = $false } } + $j | ConvertTo-Json -Depth 8 | Set-Content -LiteralPath $sj -Encoding utf8 + } catch { Say ("settings.json: " + $_.Exception.Message) } +} else { Write-Output 'RESULT TUNE error=no_settings reason=the_installed_app_has_no_settings.json'; exit 2 } +Remove-Item -LiteralPath (Join-Path $sApp 'app.url') -Force -ErrorAction SilentlyContinue + +# the state before, for the report +$smi = Join-Path $env:ProgramFiles 'NVIDIA Corporation\NVSMI\nvidia-smi.exe' +if (-not (Test-Path $smi)) { $smi = Join-Path $env:SystemRoot 'System32\nvidia-smi.exe' } +$tele = Join-Path $bin 'igneum-gpu-telemetry.exe' +function Snapshot([string] $tag) { + if (Test-Path $smi) { + $q = (& $smi --query-gpu=index,name,driver_version,power.draw,power.limit,power.default_limit,power.min_limit,power.max_limit,clocks.gr,clocks.max.gr,clocks.mem --format=csv,noheader 2>&1 | Out-String).Trim() + Write-Output ("RESULT TUNE " + $tag + " nvidia " + ($q -replace "`r?`n", ' | ')) + } + if (Test-Path $tele) { + $t = (& $tele --tune 2>&1 | Out-String).Trim() + Write-Output ("RESULT TUNE " + $tag + " amd " + ($t -replace "`r?`n", ' | ')) + } else { Write-Output ("RESULT TUNE " + $tag + " amd no_helper") } +} +Snapshot 'before' + +# the tune engine: status every 10 s (6 rate samples per 60 s hold) +$env:IGNEUM_APP_DATA = $root +$env:IGNEUM_APP_LOGS = $sLogs +$env:IGNEUM_APP_STATUS_SECS = '10' +$psi = New-Object System.Diagnostics.ProcessStartInfo +$psi.FileName = $exe +$psi.Arguments = '--sweep' +$psi.WorkingDirectory = $bin +$psi.UseShellExecute = $false +$psi.RedirectStandardOutput = $true +$psi.RedirectStandardError = $true +$psi.CreateNoWindow = $true +$p = New-Object System.Diagnostics.Process +$p.StartInfo = $psi +$lines = New-Object System.Collections.ArrayList +$h = { if ($EventArgs.Data) { [void]$Event.MessageData.Add($EventArgs.Data) } } +Register-ObjectEvent -InputObject $p -EventName OutputDataReceived -Action $h -MessageData $lines | Out-Null +Register-ObjectEvent -InputObject $p -EventName ErrorDataReceived -Action $h -MessageData $lines | Out-Null +[void]$p.Start() +$p.BeginOutputReadLine(); $p.BeginErrorReadLine() +Say ("tune engine started, pid " + $p.Id + ", data " + $root) +$seen = 0 +$rows = 0 +while (-not $p.HasExited) { + Start-Sleep -Seconds 5 + while ($seen -lt $lines.Count) { + $l = [string]$lines[$seen]; $seen++ + if ($l -match '^TUNE ') { Write-Output ('RESULT ' + $l); if ($l -match '^TUNE card=') { $rows++ } } + elseif ($l -match '^SWEEP ') { Write-Output ('RESULT ' + $l) } + elseif ($l -match '^(URL|STATE) ') { } + else { Say $l } + } + if ((Get-Date) -gt $deadline) { + Say ("budget of " + $budgetMinutes + " min spent; asking the tune engine to quit") + $u = Join-Path $sApp 'app.url' + if (Test-Path $u) { try { Invoke-WebRequest -Uri ((Get-Content -LiteralPath $u -Raw).Trim() + 'api/quit') -Method POST -Body '{}' -ContentType 'application/json' -UseBasicParsing -TimeoutSec 5 | Out-Null } catch { } } + Start-Sleep -Seconds 20 + if (-not $p.HasExited) { $p.Kill() } + Write-Output 'RESULT TUNE error=budget_exceeded' + } +} +while ($seen -lt $lines.Count) { $l = [string]$lines[$seen]; $seen++; if ($l -match '^TUNE ') { Write-Output ('RESULT ' + $l); if ($l -match '^TUNE card=') { $rows++ } } } +Say ("tune engine exited " + $p.ExitCode + " after " + [int]((Get-Date) - $started).TotalSeconds + " s, " + $rows + " table rows") +Snapshot 'after' +# the tune engine's own log: the TUNE lines and what happened around them +$log = Get-ChildItem -Path $sLogs -Filter 'app-*.log' -ErrorAction SilentlyContinue | Sort-Object LastWriteTime -Descending | Select-Object -First 1 +if ($log) { + Say ("engine log " + $log.FullName + ":") + Get-Content -LiteralPath $log.FullName | Where-Object { $_ -match 'TUNE|tune|power cap|GPUs:|worker ready|STATUS|exited|upload' } | Select-Object -Last 100 | ForEach-Object { Say (' ' + $_) } +} +if ($rows -eq 0) { Write-Output 'RESULT TUNE error=no_rows'; exit 1 } +exit 0 diff --git a/relay/test/ember.test.mjs b/relay/test/ember.test.mjs new file mode 100644 index 00000000..a5d39e96 --- /dev/null +++ b/relay/test/ember.test.mjs @@ -0,0 +1,110 @@ +// node --test relay/test/ember.test.mjs (no dependencies; CI runs it in the site job) +// The fleet aggregation of Ember Tune records (relay/lib/ember.mjs) on a fixture of records in the shape +// app/igneum-app/src/ember.rs record_json writes: known-good (five samples converge on one point), known-bad (an +// outlier does not move the median), the de-duplication of re-sent logs, the manifest merge and the prior lookup. +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { parseRecords, dedupe, aggregate, mergeTuning, priorFor, priorLine, median, spreadPct } from '../lib/ember.mjs'; + +const step = (clock_mhz, power_pct, watts, mhs, mark = 'ok') => ({ clock_mhz, power_pct, limit_w: 575 * power_pct / 100, watts, mhs, eff: Number((mhs / watts).toFixed(4)), gclk: clock_mhz || 2800, mclk: 10500, tmax: 68, faults: 0, mark }); +const rec = (machine, ts, chosen, over = {}) => ({ + ts, machine, app: '0.3.10', os: 'windows', card: 'NVIDIA_GeForce_RTX_5090', vendor: 'nvidia', driver: '581.57', driver_major: '581', class: 'l128w16', + key: 'NVIDIA_GeForce_RTX_5090|581|l128w16', plan: 'full', steps: [step(0, 100, 290, 124.0), chosen], chosen, before: step(0, 100, 290, 124.0), eff: chosen.eff, mhs: chosen.mhs, watts: chosen.watts, ...over, +}); +// five machines, each landing near 2,470 MHz at 100%: 0.55 to 0.57 MH/W +const good = [ + rec('a1', 1000, step(2472, 100, 220, 123.5)), + rec('b2', 1001, step(2472, 100, 222, 123.1)), + rec('c3', 1002, step(2781, 100, 236, 123.8)), + rec('d4', 1003, step(2472, 100, 218, 123.6)), + rec('e5', 1004, step(2163, 100, 212, 122.9)), +]; +const outlier = rec('f6', 1005, step(1854, 50, 130, 118.0)); // 0.908 MH/W: a card with a broken draw reading +const baseline = rec('g7', 1006, step(0, 80, 290, 122.3), { plan: 'baseline', steps: [step(0, 80, 290, 122.3)], before: null }); +const amd = (machine, ts, chosen) => rec(machine, ts, chosen, { card: 'AMD_Radeon_RX_9070_XT', vendor: 'amd', driver: '32.0.15801.1', driver_major: '32', key: 'AMD_Radeon_RX_9070_XT|32|l128w16', plan: 'confirm', before: null }); + +test('five samples converge on the median point and an outlier does not move it', () => { + const { priors, table } = aggregate(good, { minSamples: 5 }); + const p = priors['NVIDIA_GeForce_RTX_5090|581|l128w16']; + assert.ok(p, 'a prior at the sample floor'); + assert.equal(p.clock_mhz, 2470, 'the median clock cap, rounded to 10 MHz'); + assert.equal(p.power_pct, 100); + assert.equal(p.samples, 5); + assert.equal(p.machines, 5); + assert.ok(p.eff > 0.55 && p.eff < 0.57, `eff ${p.eff}`); + assert.ok(p.spread_pct >= 0 && p.spread_pct < 3, `spread ${p.spread_pct}`); + assert.equal(p.before_eff, Number((124 / 290).toFixed(4))); + assert.ok(p.gain_pct > 25, `gain ${p.gain_pct}% over the untuned 100% point`); + assert.equal(p.updated, '1970-01-01T00:16:44Z'); + // the outlier: 0.908 MH/W at 1,854 MHz joins; the median moves by one rank at most + const with6 = aggregate(good.concat(outlier), { minSamples: 5 }).priors['NVIDIA_GeForce_RTX_5090|581|l128w16']; + assert.equal(with6.samples, 6); + assert.equal(with6.clock_mhz, 2470); + assert.equal(with6.power_pct, 100); + assert.ok(with6.eff < 0.58, `the outlier's 0.908 MH/W did not drag the median: ${with6.eff}`); + assert.ok(with6.spread_pct < 5, `spread ${with6.spread_pct}`); + assert.equal(table[0].key, 'NVIDIA_GeForce_RTX_5090|581|l128w16'); +}); + +test('under the floor there is no prior, and baseline records never make one', () => { + const { priors, table } = aggregate(good.slice(0, 4), { minSamples: 5 }); + assert.deepEqual(priors, {}); + assert.equal(table[0].samples, 4, 'the table still shows the count'); + const b = aggregate([baseline, baseline], { minSamples: 1 }); + assert.deepEqual(b.priors, {}, 'measure-only records say what a card does, never what to set'); + assert.equal(b.table[0].baseline_samples, 1, 'the duplicate upload counted once'); + assert.equal(b.table[0].baseline_eff, Number((122.3 / 290).toFixed(4))); + // a marked chosen step (a faulted or hot winner cannot exist, but a record with one is ignored) + const bad = rec('h8', 1007, step(2000, 100, 200, 120, 'faulted')); + assert.deepEqual(aggregate([bad], { minSamples: 1 }).priors, {}); +}); + +test('records are parsed out of log text and de-duplicated on machine, card and time', () => { + const line = `1791230000 TUNE ${JSON.stringify(good[0])}`; + const text = ['1791229999 status: x', line, line, `1791230001 TUNE ${JSON.stringify(good[1])}`, '1791230002 TUNE {not json'].join('\n'); + const rs = parseRecords(text); + assert.equal(rs.length, 3); + assert.equal(dedupe(rs).length, 2); + assert.equal(parseRecords('').length, 0); +}); + +test('the manifest merge keeps the kernel-variant cards and carries the settings', () => { + const existing = { updated: '2026-10-04T21:00:00Z', window_days: 7, cards: { NVIDIA_GeForce_RTX_5090: { variant: 'u2-ldg', race: true, candidates: ['u2-ldg', 'ldg', 'base'] } }, priors: { 'old|1|v2': { samples: 9 } } }; + const { priors } = aggregate(good, { minSamples: 5 }); + const t = mergeTuning(existing, priors, { rate_tolerance_pct: 1 }); + assert.equal(t.cards.NVIDIA_GeForce_RTX_5090.variant, 'u2-ldg', 'lever 2 untouched'); + assert.equal(t.window_days, 7); + assert.deepEqual(t.ember, { enabled: true, min_samples: 5, rate_tolerance_pct: 1 }); + assert.ok(!t.priors['old|1|v2'], 'a key without samples in the window drops out'); + assert.ok(t.priors['NVIDIA_GeForce_RTX_5090|581|l128w16']); + // the kill switch rides the same section + assert.equal(mergeTuning(existing, {}, { enabled: false }).ember.enabled, false); + assert.deepEqual(mergeTuning(null, {}).cards, {}); + // the round trip: canonical JSON (what publish-manifest.sh signs) parses back to the same prior + const back = JSON.parse(JSON.stringify(t)); + assert.deepEqual(priorFor(back, 'NVIDIA_GeForce_RTX_5090|581|l128w16'), { clock_mhz: 2470, power_pct: 100, eff: priors['NVIDIA_GeForce_RTX_5090|581|l128w16'].eff, samples: 5 }); + assert.equal(priorFor(back, 'NVIDIA_GeForce_RTX_5090|581|l128w16', 6), null, 'six wanted, five there'); + assert.equal(priorFor(back, 'nothing|0|v2'), null); + assert.equal(priorFor(null, 'x'), null); +}); + +test('AMD confirm records aggregate by their own key, and the line reads', () => { + const rs = [amd('p1', 2000, step(2600, 90, 177, 17.7)), amd('p2', 2001, step(2600, 90, 180, 17.6)), amd('p3', 2002, step(2500, 90, 170, 17.4))]; + const { priors, table } = aggregate(rs, { minSamples: 3 }); + const p = priors['AMD_Radeon_RX_9070_XT|32|l128w16']; + assert.equal(p.clock_mhz, 2600); + assert.equal(p.power_pct, 90); + assert.equal(p.vendor, 'amd'); + assert.equal(p.gain_pct, undefined, 'confirm records carry no before step and no baseline was uploaded'); + assert.match(priorLine(p), /^AMD Radeon RX 9070 XT \| driver 32 \| l128w16: 2600 MHz at 90%, 0\.\d+ MH\/W, 17\.6 MH\/s at 177 W, spread \d+(\.\d+)?%, 3 sample\(s\) from 3 machine\(s\)$/); + assert.equal(table.length, 1); +}); + +test('median and spread', () => { + assert.equal(median([3, 1, 2]), 2); + assert.equal(median([4, 1, 2, 3]), 2.5); + assert.equal(median([]), 0); + assert.equal(spreadPct([1, 1, 1]), 0); + assert.equal(spreadPct([10]), 0); + assert.equal(spreadPct([9, 10, 11]), 10); +}); diff --git a/site/build.mjs b/site/build.mjs index d2f354bf..7760c3fb 100644 --- a/site/build.mjs +++ b/site/build.mjs @@ -340,6 +340,21 @@ for (const [file, active] of PAGES) { ]; const table = '
' + ['Card', 'Generator', 'Best MH/s', 'MH per watt', 'Miner', 'Date', 'Source', 'Who measured it'].map(h => ``).join('') + '' + rows.map(r => '' + cell(r).map(c => ``).join('') + '').join('') + '
${h}
${esc(String(c))}
'; + // Ember Tune's fleet priors (site/miner-priors.json, tools/tuning.mjs --priors --site): one row per card model, + // driver major and program class; a row under the sample floor shows its count and no point + const pj = JSON.parse(readFileSync(join(here, 'miner-priors.json'), 'utf8')); + const prows = (pj.rows || []).slice().sort((a, b) => (b.samples - a.samples) || (a.card < b.card ? -1 : 1)); + const pcell = r => [ + r.card, r.driver_major, r.class, r.samples + (r.machines ? ' from ' + r.machines + ' machine' + (r.machines === 1 ? '' : 's') : ''), + r.prior ? (r.clock_mhz ? fmt(r.clock_mhz) + ' MHz at ' + r.power_pct + '%' : r.power_pct + '%, clock unlocked') : 'under the floor (' + pj.min_samples + ' needed)', + r.mh_per_w == null ? 'not yet' : Number(r.mh_per_w).toFixed(3) + (r.spread_pct != null ? ' (spread ' + r.spread_pct + '%)' : ''), + r.mh_s == null ? '' : fmt(r.mh_s) + ' MH/s at ' + fmt(r.watts) + ' W', + r.untuned_mh_per_w == null ? 'not measured' : Number(r.untuned_mh_per_w).toFixed(3) + (r.gain_pct != null ? ' (' + (r.gain_pct >= 0 ? '+' : '') + r.gain_pct + '%)' : ''), + ]; + const ptable = prows.length + ? '
' + ['Card', 'Driver', 'Program class', 'Samples', 'Tuned point', 'MH per watt', 'Rate and draw', 'Untuned MH per watt (gain)'].map(h => ``).join('') + '' + + prows.map(r => '' + pcell(r).map(c => ``).join('') + '').join('') + '
${h}
${esc(String(c))}
' + : '

No tune reports yet. The first rows appear once five machines with the same card model have reported.

'; const body = scrubBench([ '

The table

', '

One row per card, generator version and miner version. The rate is the best one measured. Integrated GPUs are not listed. Prototype rows are bench numbers from before the devnet and say so in the miner column.

', @@ -349,8 +364,12 @@ for (const [file, active] of PAGES) { '

MH per watt needs the card\'s power draw during the run. The app reads it on NVIDIA cards through the driver. Rows get the figure when a run records it.

', '

There is no other Igneum miner to compare with yet, so this table compares cards, not miners. The app that produces these rows: the miner page.

', `

Rows: ${rows.length}. Source file: site/miner-bench.json in the repository.

`, + '

Fleet tuning priors

', + '

Ember Tune runs on every card the app mines with: the power limit and the core clock are stepped on the live program and the card keeps the point with the best MH per watt within 1% of its top rate. Every finished tune is reported back without anything that identifies the owner, and the fleet\'s median point per card model, driver major and program class comes back down inside the signed update manifest as the starting point for the next card of that model. A model needs ' + pj.min_samples + ' reports before its prior is used.

', + ptable, + `

Rows: ${prows.length}${pj.generated ? ', generated ' + pj.generated : ''}. Source file: site/miner-priors.json in the repository, written from the fleet records by tools/tuning.mjs --priors --site.

`, ].join('\n')); - const toc = [{ lvl: 2, t: 'The table', id: 'table' }, { lvl: 2, t: 'How a row gets here', id: 'how' }]; + const toc = [{ lvl: 2, t: 'The table', id: 'table' }, { lvl: 2, t: 'How a row gets here', id: 'how' }, { lvl: 2, t: 'Fleet tuning priors', id: 'priors' }]; writeFileSync(join(here, 'miners.html'), page('Igneum GPU bench table', 'Measured Igneum hash rates per GPU: card, generator version, best MH/s, MH per watt where measured, miner version, date and the log entry each number came from.', body, toc, 'Measured hash rates per card on the Igneum lottery hash, with the generator version, the miner version, the date and the log entry behind each number.', { path: '/miners', heading: 'GPU bench table', active: 'miner' })); diff --git a/site/miner-priors.json b/site/miner-priors.json new file mode 100644 index 00000000..cdb7b4fa --- /dev/null +++ b/site/miner-priors.json @@ -0,0 +1,6 @@ +{ + "_about": "Rows of the fleet priors table at /miners (site/build.mjs), written by tools/tuning.mjs --priors --site from the TUNE records every Igneum Miner uploads. One row per card model, driver major and program class: the median tuned point, MH per watt, the spread and the sample count. No machine names, no addresses.", + "generated": null, + "min_samples": 5, + "rows": [] +} diff --git a/site/miner.html b/site/miner.html index 20687283..3b349e0b 100644 --- a/site/miner.html +++ b/site/miner.html @@ -353,7 +353,7 @@ pre b{color:var(--molten);font-weight:500}
Variant racing every hour

At every hourly prepare the worker compiles the program in several shapes (unroll, load path, register budget, threads per group), times each for 2 seconds and keeps the fastest for the hour. Base keeps its place unless beaten.

log · 4 Oct 2026
Measured on Apple silicon

On the M5 Max, 256-thread groups ran 17.3% and 21.2% faster than base on two programs, under load from other work. The NVIDIA race is built and has not yet run on a GPU.

log · 4 Oct 2026, ratios under contention
Fleet tuning manifest

Every race is one record in the app log. The fleet's best variant per card model goes back out inside the signed update manifest, so a card starts from the known best and keeps racing.

-
Efficiency mode

Hash per watt, the number miners compare. The app steps an NVIDIA card's power cap from 100% to 50%, holds each step for 60 seconds on the live kernel and leaves the cap at the best MH per watt. Built and unit-tested; not yet run on a card.

+
Ember Tune

Hash per watt, the number miners compare. Out of the box the app steps every card's power limit and core clock on the live kernel, 60 seconds a step, and keeps the point with the best MH per watt within 1% of the card's top rate. The memory clock is never touched; a step with a rejected hash, a hot GPU or a dragged memory clock is reverted and marked. Every result feeds a fleet prior per card model that the next card of that model starts from. Built and unit-tested; the first measured tune is owed.

Latency work

A block built on a stale tip earns less. The miner is moving from polling to a template subscription and shorter jobs, so a new tip reaches the card in milliseconds. In progress, no number published yet.

Remote signed jobs, opt in

Our own fleet only. A switch, "Allow remote jobs from Igneum (signed)", with the key's fingerprint beside it. Off aborts the running job and stops polling. Jobs run at most once each and only on the machines they name.

@@ -403,8 +403,8 @@ pre b{color:var(--molten);font-weight:500} LeverThe ideaMeasured state 1 · Race the compiler every hourFor each new program, five to ten kernel variants are compiled, benched for two seconds each, and the winner is kept for the hour.Measured on the Mac's Metal worker: the winning variant +17.3% on the genesis seed and +21.2% on the hourly seed over the base compile, under load from other work. The RTX 5090 race is prepared and not yet run. Log, 4 Oct 2026 - 2 · Auto-tune that learns from the fleetApps report the race per card and program class. The best settings come back down through the signed update manifest as defaults, so the miner gets faster for everyone as the fleet grows.Shipped. The fleet is still small, so no fleet table yet. - 3 · Hash per watt, not hashA sweep per card finds the power point with the best MH per watt and holds it. Miners pay for electricity; that is the number they compare.Shipped for NVIDIA cards through the power cap. Not yet run on a card; the first sweep on a 5090 is the owed measurement. Clocks are the next lever. + 2 · Auto-tune that learns from the fleetApps report the race per card and program class, and every finished Ember Tune. The best settings come back down through the signed update manifest as defaults, so the miner gets faster and more efficient for everyone as the fleet grows.Shipped. The fleet is still small, so no fleet table yet; the priors table on /miners fills as machines report. + 3 · Hash per watt, not hashEmber Tune: two knobs per card (power limit, core clock), the memory clock held, the point with the best MH per watt within 1% of the top rate kept and pinned. Miners pay for electricity; that is the number they compare.Built for NVIDIA (through the driver, with Power control on) and AMD (through the app's own helper, no administrator rights), measure only on Apple silicon. Unit-tested on every rule; the first tune measured on a card is the owed number. 4 · Template latencySolo against the local node, a new template within 50 ms of a new tip, because a late block on a BlockDAG goes red and earns nothing.Approximate, from the 0.3.6 release plan and not yet in the engineering log: switched p50 46 to 52 ms on a three-node CPU run, 5 Oct 2026. Ships in 0.3.6. 5 · Never lose a secondZero-loss hourly program swaps, fault guards, automatic restart, a CPU re-check of every found hash, per-worker health.Shipped. Swap 0.01 ms on Metal and 0.00 ms on CUDA with 0 rejected blocks; recoveries in the table above. Log, 4 Oct 2026 6 · Prove it in publicThe bench table per card, fed from the job channel. We claim fastest only when the table says so.Live at /miners, one row per card, generator version and miner version, each with its log entry. diff --git a/site/miners.html b/site/miners.html index 848f1c8e..5b1e1642 100644 --- a/site/miners.html +++ b/site/miners.html @@ -172,12 +172,12 @@ th{font-family:var(--f-mono);font-size:12px;letter-spacing:.12em;text-transform:
-
2 entries, newest at the bottom
+
3 entries, newest at the bottom

GPU bench table

Measured hash rates per card on the Igneum lottery hash, with the generator version, the miner version, the date and the log entry behind each number.

- +

The table

One row per card, generator version and miner version. The rate is the best one measured. Integrated GPUs are not listed. Prototype rows are bench numbers from before the devnet and say so in the miner column.

CardGeneratorBest MH/sMH per wattMinerDateSourceWho measured it
Apple M5 Max (40 GPU cores, Metal)v145.2not measuredproto-metal bench (prototype, not mining)2026-10-03bench log: 3 October 2026, RTX 5090 first run (the Apple row of the same table)measured by the team. genesis program, 1 GiB dataset
Apple M5 Max (40 GPU cores, Metal)v226.7not measuredigneum-miner devnet v4, Metal worker with prepare2026-10-04bench log: 4 October 2026, first hourly program swap on the live devnet: compile-ahead, no pause, two cardsmeasured by the team. live devnet v4, unbroken through the hour boundary
Apple silicon laptop (model not reported)v224.3not measuredIgneum Miner 0.3.1 (DMG)2026-10-04bench log: 4 October 2026, first outside machine on the devnet: an Apple silicon laptop through the Igneum Miner appreported by the fleet. 21.0 MH/s average over 7 minutes, 24.3 MH/s at the moment of the report, 33 accepted blocks
NVIDIA RTX 5090 (32 GB)v1229not measuredproto-cuda bench (prototype, not mining)2026-10-03bench log: 3 October 2026, RTX 5090, memory-hard dataset (pack igneum-genesis-mh)measured by the team. genesis program, 104 loads per hash, 1 GiB dataset
NVIDIA RTX 5090 (32 GB)v1185.3not measuredproto-cuda bench (prototype, not mining)2026-10-03bench log: 3 October 2026, RTX 5090 first run, dataset sweep and second programmeasured by the team. hourly program, 128 loads per hash, 1 GiB dataset
NVIDIA RTX 5090 (32 GB)v2124.2not measuredIgneum Miner 0.3.0 package, prebuilt NVRTC worker2026-10-04bench log: 4 October 2026, the gfx1036 worker fault and what the Apple M5 Max could and could not reproducemeasured by the team. live devnet v4, 128 loads per hash, CPU re-check clean, 0 rejected
@@ -185,7 +185,11 @@ th{font-family:var(--f-mono);font-size:12px;letter-spacing:.12em;text-transform:

Every row names the engineering log entry or the job it came from. "Measured by the team" means our own hardware and our own log. "Reported by the fleet" means a machine we do not own, read from the status lines its miner uploads.

MH per watt needs the card's power draw during the run. The app reads it on NVIDIA cards through the driver. Rows get the figure when a run records it.

There is no other Igneum miner to compare with yet, so this table compares cards, not miners. The app that produces these rows: the miner page.

-

Rows: 6. Source file: site/miner-bench.json in the repository.

+

Rows: 6. Source file: site/miner-bench.json in the repository.

+

Fleet tuning priors

+

Ember Tune runs on every card the app mines with: the power limit and the core clock are stepped on the live program and the card keeps the point with the best MH per watt within 1% of its top rate. Every finished tune is reported back without anything that identifies the owner, and the fleet's median point per card model, driver major and program class comes back down inside the signed update manifest as the starting point for the next card of that model. A model needs 5 reports before its prior is used.

+

No tune reports yet. The first rows appear once five machines with the same card model have reported.

+

Rows: 0. Source file: site/miner-priors.json in the repository, written from the fleet records by tools/tuning.mjs --priors --site.

Generated from the repository at build time. Times are UTC. Machine names are model names.

diff --git a/tools/console.mjs b/tools/console.mjs index 40feee60..9855f39f 100644 --- a/tools/console.mjs +++ b/tools/console.mjs @@ -5,6 +5,7 @@ // node tools/console.mjs machines the machine cards as text // node tools/console.mjs chain the chain numbers // node tools/console.mjs jobs | builds | results the other tabs as text +// node tools/console.mjs tuning [--days 30] [--min 5] Ember Tune's fleet priors per card model (samples, MH/W) // node tools/console.mjs sync-bench push docs/bench-log.md entries and the FUD ledger counts // node tools/console.mjs sync-dl push the downloads folder listing (names, sizes, times) // node tools/console.mjs sync-hetzner push the newest infra/cloud-devnet/results/ summary @@ -121,6 +122,12 @@ try { } else if (cmd === 'jobs') { const j = await api('jobs'); if (!j.jobs.length) console.log(j.note || 'no jobs'); for (const jb of j.jobs) console.log(`${jb.id} ${jb.title} | ${jb.runs.map(r => `${r.machine} ${r.status}${r.exit_code != null ? ' exit ' + r.exit_code : ''}${r.summary ? ': ' + r.summary.slice(0, 80) : ''}`).join(' | ') || 'queued everywhere'}`); } else if (cmd === 'builds') { const j = await api('builds'); console.log(`manifest ${j.manifest ? j.manifest.version + ' (' + j.manifest.channel + ') ' + Object.keys(j.manifest.platforms || {}).join('+') + ': ' + j.manifest.notes : 'none'}`); if (j.ci) console.log(`ci ${j.ci.installer} fetched ${j.ci.fetched_at} ${j.ci.run}`); for (const b of j.builds) console.log(`${when(b.ts)} ${b.who.padEnd(8)} ${b.title}${b.body ? ': ' + b.body.split('\n')[0].slice(0, 100) : ''}`); } + else if (cmd === 'tuning') { + const j = await api('tuning', { q: { days: flags.days || 30, min: flags.min || 5 } }); + console.log(`${j.records} tune record(s) in ${j.days} day(s); a prior needs ${j.min_samples} sample(s)`); + if (!j.table.length) console.log('no tune records yet (the apps log one per finished tune, 0.3.10 and later)'); + for (const t of j.table) console.log(`${t.card.replace(/_/g, ' ')} | driver ${t.driver_major} | ${t.class}: ${t.samples} sample(s) from ${t.machines} machine(s)${t.samples ? `, ${t.clock_mhz ? t.clock_mhz + ' MHz at ' : ''}${t.power_pct}%${t.clock_mhz ? '' : ' (clock unlocked)'}, ${t.eff} MH/W (${t.mhs} MH/s at ${t.watts} W, spread ${t.spread_pct}%)` : ''}${t.baseline_eff ? `; untuned ${t.baseline_eff} MH/W from ${t.baseline_samples} baseline(s)` : ''}${t.gain_pct != null ? `; gain ${t.gain_pct >= 0 ? '+' : ''}${t.gain_pct}%` : ''}; prior ${t.key in j.priors ? 'yes' : 'no'}`); + } else if (cmd === 'results') { const j = await api('results'); if (j.ledger) console.log(`${j.ledger.title}: ${j.ledger.body}`); for (const b of j.bench) console.log(`${b.meta.date || '-'} ${b.title}`); } else if (cmd === 'sync-bench' || cmd === 'sync-dl' || cmd === 'sync-hetzner' || cmd === 'sync') { const items = []; diff --git a/tools/tuning.mjs b/tools/tuning.mjs index c8b92149..537d8bfd 100644 --- a/tools/tuning.mjs +++ b/tools/tuning.mjs @@ -14,10 +14,17 @@ // variant without a race; the record then carries only the self-test), the default keeps racing (the fleet keeps // learning while the card starts from the known best). // node tools/tuning.mjs --records [--days 7] [--card ] the raw records, newest first +// node tools/tuning.mjs --priors [--days 30] [--min-samples 5] Ember Tune (docs/plans/ember-tune.md): the fleet +// priors per (card model, driver major, program class) from the TUNE records (relay/lib/ember.mjs aggregate); +// --write adds them to the tuning file under "priors" with the "ember" settings (kill switch --tuning-off, +// --rate-tolerance N), beside the kernel-variant cards; --site writes site/miner-priors.json for /miners. // // Reads DATABASE_URL from ~/.config/igneum/env. No dependencies: Neon HTTP SQL over fetch. -import { readFileSync, writeFileSync } from 'node:fs'; +import { readFileSync, writeFileSync, existsSync } from 'node:fs'; import { homedir } from 'node:os'; +import { join, dirname, resolve } from 'node:path'; +import { fileURLToPath } from 'node:url'; +import { parseRecords, aggregate, mergeTuning, priorLine } from '../relay/lib/ember.mjs'; process.stdout.on('error', e => { if (e.code === 'EPIPE') process.exit(0); throw e; }); @@ -46,6 +53,43 @@ const minSamples = Number(opt('--min-samples', 3)); const by = opt('--by', 'mhs'); const outFile = opt('--write', ''); const onlyCard = opt('--card', ''); +const ROOT = resolve(dirname(fileURLToPath(import.meta.url)), '..'); + +// ---- Ember Tune priors (--priors): the TUNE records of the window, folded per key -------------------------------- +if (flag('--priors')) { + const pdays = Number(opt('--days', 30)); + const pmin = Number(opt('--min-samples', 5)); + const prow = await sql( + `SELECT machine, run_id, lines FROM miner_logs WHERE received_at > now() - ($1 || ' days')::interval AND lines LIKE '%TUNE {%' ORDER BY received_at DESC`, + [String(pdays)]); + const recs = prow.flatMap(r => parseRecords(r.lines)).filter(r => !onlyCard || r.card === onlyCard); + const { priors, table } = aggregate(recs, { minSamples: pmin }); + if (!recs.length) console.log(`No TUNE records in the last ${pdays} day(s). The apps log one per finished tune (0.3.10 and later).`); + else { + console.log(`${recs.length} tune record(s) in the last ${pdays} day(s); a prior needs ${pmin} sample(s)`); + console.table(table.map(t => ({ key: t.key, samples: t.samples, machines: t.machines, 'clock MHz': t.clock_mhz ?? '-', 'power %': t.power_pct ?? '-', 'MH/W': t.eff ?? '-', 'MH/s': t.mhs ?? '-', W: t.watts ?? '-', 'spread %': t.spread_pct ?? '-', 'untuned MH/W': t.before_eff ?? t.baseline_eff ?? '-', 'gain %': t.gain_pct ?? '-', prior: t.key in priors ? 'yes' : 'no' }))); + for (const p of Object.values(priors)) console.log(priorLine(p)); + } + if (outFile) { + const existing = existsSync(outFile) ? JSON.parse(readFileSync(outFile, 'utf8')) : {}; + const ember = {}; + if (flag('--tuning-off')) ember.enabled = false; + if (flag('--tuning-on')) ember.enabled = true; + if (opt('--rate-tolerance', '')) ember.rate_tolerance_pct = Number(opt('--rate-tolerance', '1')); + ember.min_samples = pmin; + const merged = mergeTuning(existing, priors, ember); + writeFileSync(outFile, JSON.stringify(merged, null, 2) + '\n'); + console.log(`written ${outFile}: ${Object.keys(merged.cards).length} kernel-variant card(s) kept, ${Object.keys(priors).length} prior(s), ember ${JSON.stringify(merged.ember)}; publish with: packaging/ota/publish-manifest.sh --version --tuning ${outFile} [--deploy]`); + } + if (flag('--site')) { + const site = join(ROOT, 'site', 'miner-priors.json'); + const rows = table.map(t => ({ card: t.card.replace(/_/g, ' '), vendor: t.vendor, driver_major: t.driver_major, class: t.class, clock_mhz: t.clock_mhz ?? null, power_pct: t.power_pct ?? null, mh_per_w: t.eff ?? null, mh_s: t.mhs ?? null, watts: t.watts ?? null, spread_pct: t.spread_pct ?? null, samples: t.samples, machines: t.machines, untuned_mh_per_w: t.before_eff ?? t.baseline_eff ?? null, gain_pct: t.gain_pct ?? null, prior: t.key in priors, updated: t.updated || null })); + const about = existsSync(site) ? JSON.parse(readFileSync(site, 'utf8'))._about : undefined; + writeFileSync(site, JSON.stringify({ _about: about || 'Rows of the fleet priors table at /miners (site/build.mjs), written by tools/tuning.mjs --priors --site from the TUNE records every Igneum Miner uploads. One row per card model, driver major and program class: the median tuned point, MH per watt, the spread and the sample count. No machine names, no addresses.', generated: new Date().toISOString().replace(/\.\d{3}Z$/, 'Z'), min_samples: pmin, rows }, null, 2) + '\n'); + console.log(`written ${site} (${rows.length} row(s))`); + } + process.exit(0); +} // Every app-log upload of the window; the TUNING lines out of them. The same race is uploaded many times (the log // is re-sent every minute), so records are de-duplicated on (machine, card, epoch). From 99f7836a813cc339f23d7c355f2b4e11b25f27bf Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 21:26:31 +0000 Subject: [PATCH 06/63] bench log: Ember Tune, what PC 1 could measure tonight (elevated=False, the cancelled prompt at 20:09 UTC, the 9070 XT off the bus), the pipeline verified without a card, the tier consequences Co-Authored-By: Claude Fable 5.1 --- docs/bench-log.md | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/docs/bench-log.md b/docs/bench-log.md index ea9a8633..1731737b 100644 --- a/docs/bench-log.md +++ b/docs/bench-log.md @@ -1607,3 +1607,21 @@ Reading: the kernel is the same 116.0 ms on both paths (18.08 MH/s pure kernel, **A second defect found on the way: the pack export race.** PC 1's app log since its 19:02 UTC restart (`node tools/logs.mjs win-ae432dc7-20261005-190232`): `worker error: error 0 pack packs\devnet: the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT` at 19:07:03, 19:07:19 and 19:08:07, so the 9070 XT was not mining at all in the app while this entry was written (my job `rdna4-serve-1` at 18:43 hit the same folder in the same state). Cause, from `app/igneum-app/src/engine.rs` `prepare_worker`: one thread per card, each running `igneum-miner export-pack` into the one folder `packs\devnet`; across an epoch change the two exports interleave and the folder keeps one epoch's `program.h` with the other's `seeds.txt` until the next export. Fix on this branch: a process-wide mutex around both export sites (`EXPORT_LOCK`); the second export rewrites the same pack. Not measured in the app yet: it ships with the branch. **Answer to the project lead.** The 9070 XT does 2.5 G random 4-byte reads per second from its memory for this access pattern, and the hash needs 128 of them, so about 19 MH/s is this card's ceiling for the current program class, on any slot; it was running at 92% of that. The eGPU link cost 6% per job through the read-back, now removed (17.87 against 16.88 MH/s inside jobs standalone). The duplicate platform that halved it to 8.9 + 9.4 is folded away. The pack race that stopped it is serialised. Nothing else in the worker's control moves the number: the next step for this card is the program class itself (fewer, wider loads per hash would favour AMD's 64-byte lines), which is a consensus question, not a worker one. + +## 5 October 2026 (night), Ember Tune: the two-knob efficiency tune, the fleet prior, and what PC 1 could measure tonight (miner-community-lead) + +Branch `ember-tune` (54ff1bc), docs/plans/ember-tune.md. Every card tuned for MH per watt out of the box: the power limit and the core clock cap stepped on the live kernel (memory clock never touched), the point with the best MH per watt within 1% of the top rate kept and pinned, every result uploaded as a `TUNE {json}` record (a hash of the install id, no address) and folded per (card model, driver major, program class) into a prior the signed manifest carries back, so a new card of a known model starts there and confirms it in two steps. + +**What was measured tonight (PC 1, machine ae432dc7, from its own uploads to the intake):** + +| Fact | Where it was read | Consequence | +|---|---|---| +| The installed 0.3.9 app runs as `DESKTOP-KMCV30N\Admin` with `elevated=False` (account line, 19:02:33 UTC) | app log `win-ae432dc7-20261005-190232` | `nvidia-smi -pl` and `-lgc` need administrator rights; the one prompt is the Power control switch (3562f26), which the app never raises by itself | +| Two in-app sweep attempts aborted at 20:09 UTC: `the_elevated_helper_did_not_run_(the_administrator_prompt_was_cancelled)` | the same log | no stored sweep result from today exists; the 5090's two-knob tune is owed to the morning (one click on Power control, then it runs by itself within 2 minutes of steady mining) | +| The RX 9070 XT left PC 1's bus at about 20:40 UTC, was back at 21:09 and gone again at 21:22:59 UTC (the eGPU link, third drop today) | the telemetry agent and the PC 1 scheduler | the AMD path (ADLX, no prompt) is unit-tested on the helper's captured line shapes; its end-to-end run waits for the card | + +**The pipeline, verified without a card:** 9 `ember` unit tests (plans, clamps, the choice rule, the five marks, a faulted step reverted inside a fake-clock run, the confirm verdicts, the baseline plan, the record and prior shapes, the vendor reasons), the AMD `tune` line and the 0.3.10 sample line parsed (`engine::amd_telemetry_tests`), the helper protocol (`sweep::tests`), 6 relay aggregation tests (five samples converge on 2,470 MHz at 100%; an outlier at 0.908 MH/W moves the median by nothing; baseline records make no prior; de-duplication; the manifest merge keeps lever 2's cards; the canonical round trip), 3 UI line tests. A test manifest was signed on this Mac with `packaging/ota/publish-manifest.sh --tuning` from fixture priors: `tuning.priors["NVIDIA_GeForce_RTX_5090|581|l128w16"]` = 2,470 MHz at 100%, 5 samples, beside the kernel-variant `cards` entry and `tuning.ember {enabled: true, min_samples: 5, rate_tolerance_pct: 1}`, signature verified by the signer, 21:25 UTC. + +**Tier consequences** (docs/plans/ember-tune.md section 7): a 9-step full tune costs about 12 minutes once and 3 minutes a week per card, under 1% of the hour, the worker never stops; a rig tunes one card at a time and every card of a known model after the first takes the 3-minute confirm; a pool user gives up the same 1% of shares at most; Apple silicon and AMD on Linux measure only and the row says so. + +(The PC 1 run's numbers follow below when the job reports.) From dd37094a5b3ee684678bd2c013620485528e7fa7 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 21:31:06 +0000 Subject: [PATCH 07/63] ember-tune.md: a signed prior is a starting point inside the card's own reported limits, never a memory clock; the tests that prove the clamp (consequences row C25) Co-Authored-By: Claude Fable 5.1 --- docs/plans/ember-tune.md | 1 + 1 file changed, 1 insertion(+) diff --git a/docs/plans/ember-tune.md b/docs/plans/ember-tune.md index 22869c77..f0d93f1d 100644 --- a/docs/plans/ember-tune.md +++ b/docs/plans/ember-tune.md @@ -96,6 +96,7 @@ race has run). | Faults: a rejected or mismatched hash marks the step; the card leaving `mining`, a worker error, a job, a pause or 90 C aborts the run and restores the point from before | `Run::sample_fault`, `sweep_drive`, `sweep_abort` | | Memory clock held: never set; a step that drags it under 95% of the baseline's cannot win | `Row::from_samples` | | Vendor limits: every point clamped to the reported range; the clock floor 60% when none is reported | `Limits` | +| A signed prior is only ever a starting point inside the card's OWN reported limits (`power.min_limit` to `power.max_limit`, the clock floor to `clocks.max.gr` or the ADLX `gmax_range`), never a memory clock, never a value the card did not report; the confirm step measures it and the full plan replaces it when a neighbour beats it, so a bad prior costs the fleet one confirm step per card, not a setting. The signing key (K1, docs/security/keys.md) therefore cannot push a card past its vendor ceiling or under its floor | `Plan::confirm` clamps through `Limits::clamp_clock` and `power_pct.clamp(50, 100)`; proven by `ember::tests::the_confirm_plan_checks_the_prior_and_its_neighbour` (a prior of 9,000 MHz at 30% becomes 3,090 MHz at 50%) and `limits_never_exceed_the_vendor_or_undercut_the_floor` | | No prompt the user did not ask for: the NVIDIA helper starts only with Power control on; the `--sweep` job never counts as permission | `sweep_probe_known`, `sweep_helper_start` | | The elevated helper restores the limit and resets the clocks by itself after 20 idle minutes | `sweep::helper_script_*` | From 2f2bdb5a705ebd7d1aeed5ca8bb0049ee7b1949f Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 22:42:12 +0000 Subject: [PATCH 08/63] plan: the prover-floor agent's first sweep (the v1 shard at 12.7 GB with the 2^26 split, 5.3 s; the 9.7 GB Setup is next) and the 12 GB profile decision Co-Authored-By: Claude Fable 5.1 --- docs/plans/proving-v1.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/docs/plans/proving-v1.md b/docs/plans/proving-v1.md index b7ac3175..b2fa92e6 100644 --- a/docs/plans/proving-v1.md +++ b/docs/plans/proving-v1.md @@ -114,6 +114,8 @@ Reading. Nobody pays an aggregator as a separate role: Aztec's 30% goes to whoev The resume path (5 October 2026, the 0.3.11 app): `POST /api/resume` on 0.3.9 re-armed only FAULTED cards (`stop_miners("paused")` clears every slot's `restart_at`), so a healthy paused card stayed "off" at 0 MH/s until the app was relaunched: PC 2 at 21:25:11Z (the aggregation-cost job's pause and resume; `[ok] mining resumed` then `0.00 MH/s, waiting` for 20 minutes), the Mac that afternoon. Now every slot without a live worker is re-armed and its pack exported again before the start, and 90 s later `resume_check` logs `resume: is not mining 90 s after resume (state ..., pid ...)` for every enabled card without a hash rate (`engine.rs`, three unit tests: the state machine, the 21:25:11Z case against the old rule, the check). +The prover-floor agent's first sweep (job `floor-sweep-1`, 22:34 to 22:38Z, PC 2's 5090, the miners stopped, this plan's per-point recipe, its patched `sp1-gpu-server` 5568108b built for sm_86, sm_89 and sm_120, every proof VERIFIED by the unpatched pv1 host): the control at upstream's sizes reproduces the curve above (empty shard 13,892 MiB and 2.2 s; the v1 shard 20,516 MiB and 4.2 s); with the core element threshold at 2^26 the v1 shard proves as four core shards in 5.3 s at **12,708 MiB** and the empty shard at 12,772 MiB; 2^25 gives 12,836 MiB at 8.5 s; 2^27 gives 15,396 MiB. The 12.7 GB left is the server's Setup (five recursion keys pre-built at a fixed 2^27 capacity plus the shrink and core keys: 9.7 GB before the first shard), which its patch v2 sizes to the need. Decided for the 12 GB profile: the split that lands under 11 GB wins (5.3 s a shard is inside the loop's own 25 to 30 s of carriage and 100x inside T); 2^27 is the second profile only if v2 leaves it under 11 GB with the miner's 1.8 GB beside it. The 12 GB row stays OPEN until the final pair (alone and beside the miner) lands and the on-order 3060 runs it. + ### A self-built CUDA server (the 12 GB path), before 0.3.12 (consequences C26) If the prover-floor agent's rebuilt `sp1-gpu-server` (the Setup sizes cut, built on PC 2 under WSL2) proves a shard under 11 GB, it becomes a shipped artefact and needs its own row of rules before 0.3.12: it is built from a pinned SP1 source tag with `CUDA_ARCHS` covering sm_86, sm_89 and sm_120 (the 12 and 16 GB tiers are Ampere and Ada, not only the 5090's Blackwell; one card family per measured row), by the packaging path that builds the Windows payload (PC 1's build job for the Linux binary, the Mac signs the manifest as it does the DMG), lands in the DMG and the WSL2 package beside the host as `wsl2/bin/sp1-gpu-server` with its sha256 in `payload-inputs.json`, is named in `evidence.md` beside the prover rows ("prover built from SP1 at "), is rebuilt and re-measured at every SP1 upgrade, and ships only after `--mode verify-segment` and `--mode verify` on proofs it made show the pinned verifying keys unchanged (the server changes allocation, not the circuit; the ids `0x2b1a81cb...` and `0x474678f3...` must still verify them). The 12 GB claim itself waits for the on-order RTX 3060 to run that server on the same fixtures and recipe as the curve; until then the public line stays at 24 GB. From 794e23c5ac6ab98306b6b9e92ed5c5c2473c3650 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 22:43:43 +0000 Subject: [PATCH 09/63] ember: the AMD helper's gmax range is an offset from stock (PC 1's 9070 XT: -500 to 1000), not MHz: the clock knob stays closed on an offset range and the power ladder is bounded by plimit_range (-30 to 10) on a percent scale Co-Authored-By: Claude Fable 5.1 --- app/igneum-app/src/engine.rs | 30 ++++++++++++++++++++++-------- 1 file changed, 22 insertions(+), 8 deletions(-) diff --git a/app/igneum-app/src/engine.rs b/app/igneum-app/src/engine.rs index 684e454c..095a6028 100644 --- a/app/igneum-app/src/engine.rs +++ b/app/igneum-app/src/engine.rs @@ -1891,13 +1891,13 @@ impl Engine { None if allowed || current > 0 => crate::detect::run_timeout(std::process::Command::new(&smi).args(["-i", &device, "-pl", ¤t.to_string()]), None, Duration::from_secs(20)).map(|out| out.contains("All done")), None => Some(false), }; - shared.send(Cmd::TuneProbe(idx, Ok(TuneProbe { clock_max_mhz: clock_max, clock_min_mhz: 0, driver, direct: direct.unwrap_or(false), amd_ordinal: -1 }))); + shared.send(Cmd::TuneProbe(idx, Ok(TuneProbe { clock_max_mhz: clock_max, clock_min_mhz: 0, driver, direct: direct.unwrap_or(false), amd_ordinal: -1, ..Default::default() }))); }); } "amd" => { let Some(exe) = self.bins.telemetry.clone() else { self.sweep_pending = None; - self.shared.send(Cmd::TuneProbe(idx, Ok(TuneProbe { clock_max_mhz: 0, clock_min_mhz: 0, driver: c.driver.clone(), direct: false, amd_ordinal: -1 }))); + self.shared.send(Cmd::TuneProbe(idx, Ok(TuneProbe { clock_max_mhz: 0, clock_min_mhz: 0, driver: c.driver.clone(), direct: false, amd_ordinal: -1, ..Default::default() }))); return; }; let ordinal = c.amd_ordinal; @@ -1906,14 +1906,18 @@ impl Engine { let out = crate::detect::run_timeout(std::process::Command::new(&exe).arg("--tune"), None, Duration::from_secs(20)).unwrap_or_default(); let t = out.lines().filter_map(parse_amd_tune).find(|t| ordinal < 0 || t.ordinal as i64 == ordinal); shared.send(Cmd::TuneProbe(idx, Ok(match t { - Some(t) if t.ok => TuneProbe { clock_max_mhz: t.gmax_max as u32, clock_min_mhz: t.gmax_min as u32, driver, direct: true, amd_ordinal: t.ordinal as i64 }, - Some(t) => TuneProbe { clock_max_mhz: 0, clock_min_mhz: 0, driver, direct: false, amd_ordinal: t.ordinal as i64 }, - None => TuneProbe { clock_max_mhz: 0, clock_min_mhz: 0, driver, direct: false, amd_ordinal: -1 }, + // PC 1's 9070 XT (ember-tune-pc1-1, 22:30 UTC): `gmax 0 gmax_range -500 1000`, an OFFSET from + // the stock clock, not MHz; a range with a negative floor is an offset range and the clock + // knob stays closed until the stock clock is known (the power limit is the AMD lever), and + // `plimit_range -30 10` bounds the power ladder (the percent scale rides power_* below) + Some(t) if t.ok => TuneProbe { clock_max_mhz: if t.gmax_min >= 0.0 && t.gmax_max > 0.0 { t.gmax_max as u32 } else { 0 }, clock_min_mhz: if t.gmax_min > 0.0 { t.gmax_min as u32 } else { 0 }, driver, direct: true, amd_ordinal: t.ordinal as i64, plimit_min: t.plimit_min, plimit_max: t.plimit_max }, + Some(t) => TuneProbe { clock_max_mhz: 0, clock_min_mhz: 0, driver, direct: false, amd_ordinal: t.ordinal as i64, ..Default::default() }, + None => TuneProbe { clock_max_mhz: 0, clock_min_mhz: 0, driver, direct: false, amd_ordinal: -1, ..Default::default() }, }))); }); } _ => { - self.shared.send(Cmd::TuneProbe(idx, Ok(TuneProbe { clock_max_mhz: 0, clock_min_mhz: 0, driver: c.driver.clone(), direct: false, amd_ordinal: -1 }))); + self.shared.send(Cmd::TuneProbe(idx, Ok(TuneProbe { clock_max_mhz: 0, clock_min_mhz: 0, driver: c.driver.clone(), direct: false, amd_ordinal: -1, ..Default::default() }))); } } } @@ -1939,7 +1943,13 @@ impl Engine { // NVIDIA control: this process is elevated (direct), or Power control is on so the one-prompt helper may run. // The --sweep job alone never counts: it must not raise a prompt on a PC with nobody there (5 October 2026). let power_control = probe.direct || self.shared.settings.lock().unwrap().power_control; - let limits = crate::ember::Limits { power_default_w: c.power_default_w, power_min_w: c.power_min_w, power_max_w: c.power_max_w, clock_max_mhz: probe.clock_max_mhz, clock_min_mhz: probe.clock_min_mhz }; + // AMD's power limit is a percent offset from the default (ADLX): the plan's watts scale becomes a percent + // scale (default 100, floor 100 + plimit_min, ceiling 100 + plimit_max), tune_apply sends pct - 100 + let limits = if c.vendor == "amd" && probe.direct && probe.plimit_max >= probe.plimit_min && probe.plimit_min > -100.0 { + crate::ember::Limits { power_default_w: 100.0, power_min_w: 100.0 + probe.plimit_min, power_max_w: 100.0 + probe.plimit_max, clock_max_mhz: probe.clock_max_mhz, clock_min_mhz: probe.clock_min_mhz } + } else { + crate::ember::Limits { power_default_w: c.power_default_w, power_min_w: c.power_min_w, power_max_w: c.power_max_w, clock_max_mhz: probe.clock_max_mhz, clock_min_mhz: probe.clock_min_mhz } + }; let control = match c.vendor.as_str() { "nvidia" => crate::ember::control_reason("nvidia", &limits, &c.device, power_control, false), "amd" => crate::ember::control_reason("amd", &limits, &c.device, power_control, probe.direct && probe.amd_ordinal >= 0), @@ -2113,7 +2123,8 @@ impl Engine { return; } let n = c.amd_ordinal.to_string(); - let offset = step.point.power_pct as i64 - 100; + // the step's limit on the AMD scale is a percent (the probe's Limits); the offset is that minus 100 + let offset = if c.power_default_w <= 0.0 { step.watts.round() as i64 - 100 } else { step.point.power_pct as i64 - 100 }; let unlocked = clock == 0 && offset == 0; let gmax = if clock > 0 { clock } else { c.clock_max_mhz }; std::thread::spawn(move || { @@ -3465,6 +3476,9 @@ pub struct TuneProbe { /// NVIDIA: this process sets limits itself (elevated); AMD: the helper answered its `--tune` line with ok pub direct: bool, pub amd_ordinal: i64, + /// AMD: the power offset range in percent from the `tune` line (PC 1's 9070 XT: -30 to 10) + pub plimit_min: f64, + pub plimit_max: f64, } /// One `tune` line of igneum-gpu-telemetry --tune: From f76fca96e24922c3c777b3086c1de3504c8fc28c Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 22:44:46 +0000 Subject: [PATCH 10/63] bench log + plan: PC 1 run 1 aborted by the 0.3.11 update 47 s in, the before snapshots of both cards, the AMD offset-range finding and its consequence per tier Co-Authored-By: Claude Fable 5.1 --- docs/bench-log.md | 11 ++++++++++- docs/plans/ember-tune.md | 8 ++++++-- 2 files changed, 16 insertions(+), 3 deletions(-) diff --git a/docs/bench-log.md b/docs/bench-log.md index 1731737b..c19dfa12 100644 --- a/docs/bench-log.md +++ b/docs/bench-log.md @@ -1624,4 +1624,13 @@ Branch `ember-tune` (54ff1bc), docs/plans/ember-tune.md. Every card tuned for MH **Tier consequences** (docs/plans/ember-tune.md section 7): a 9-step full tune costs about 12 minutes once and 3 minutes a week per card, under 1% of the hour, the worker never stops; a rig tunes one card at a time and every card of a known model after the first takes the 3-minute confirm; a pool user gives up the same 1% of shares at most; Apple silicon and AMD on Linux measure only and the row says so. -(The PC 1 run's numbers follow below when the job reports.) +**The PC 1 run, 22:30 UTC (job ember-tune-pc1-1, engine aeea3228..., PC 1 on 0.3.10):** the job published at 22:29:40Z, the installed app stopped its miners and started the second engine at 22:30:21Z, and at 22:31:06Z the installed app quit for the 0.3.11 update-now (its log: `job ember-tune-pc1-1: aborted (the app is quitting)`), taking the second engine with it 47 s in, before any step. Nothing was set. What the run did record, the "before" snapshots with the miners stopped: + +| Card | Read back at 22:30:20Z | Meaning | +|---|---|---| +| RTX 5090 (driver 617.14) | limit 450 W of 575 W default (min 400, max 600), draw 259.9 W idle-after-stop, core 2,850 MHz, `clocks.max.gr` 3,090 MHz, memory 14,001 MHz | the two-knob plan for this card is 5 power steps (575, 518, 460, 403, 400 W) and 4 clock steps (2,781, 2,472, 2,163, 1,854 MHz); it needs the one administrator prompt (Power control) | +| RX 9070 XT (bus 98, present again) | `tune 1 ... gmax 0 gmax_range -500 1000 plimit 0 plimit_range -30 10 factory 1 ok` | the helper's clock range is an OFFSET from stock in MHz, not a ceiling: a probe reading it as a 1,000 MHz maximum would have asked for `--set-gmax 900`, an overclock. Fixed at 054e041: an offset range closes the clock knob (until the stock clock is known) and the power ladder runs on the percent scale bounded by the range, so the 9070 XT's plan is 100, 90, 80, 70% (the -30 floor), 4 steps | +| Radeon(TM) Graphics (integrated) | `tune 0 ... gmax - ... factory 0 ok` | no manual tuning: measure only, and it is off by default anyway | + +Consequence for the tiers: an AMD card is tuned on its power limit alone until its stock core clock is read (a 9070 XT at -30% is the floor the driver allows, 4 steps, 5 minutes); every NVIDIA card's two-knob plan waits on the user's one click on Power control; the re-run on PC 1 follows the 0.3.11 rollout (the update clears the jobs folder, so the engine and the helper are fetched again), with the scheduler's slot. + diff --git a/docs/plans/ember-tune.md b/docs/plans/ember-tune.md index f0d93f1d..61a3d730 100644 --- a/docs/plans/ember-tune.md +++ b/docs/plans/ember-tune.md @@ -29,7 +29,7 @@ prompt) for both knobs; off, it measures only. | Vendor | Power limit | Core clock cap | Memory clock | How | Rights | |---|---|---|---|---|---| | NVIDIA | `nvidia-smi -pl `, percent of the default, inside `power.min_limit` and `power.max_limit` | `nvidia-smi -lgc 0,`, percent of `clocks.max.gr`; `-rgc` = unlocked | never touched (`-lmc` is not used); read back as `clocks.mem` | directly when the engine is elevated, else the one-prompt helper (` pl `, ` lgc `, ` rgc` in `sweep/cmd.txt`) | administrator, so only with Power control on | -| AMD | `igneum-gpu-telemetry --card N --set-plimit ` (0 = default, -20 = 80%), inside the `tune` line's `plimit_range` | `--set-gmax ` inside `gmax_range`; `--reset` for the default point | not settable through ADLX on RDNA 4; read back as `mclk_mhz`, and a step whose mean memory clock falls under 95% of the baseline's is marked and cannot win | the helper, one process per request, exit 0 and a `tune ... ok` line | none on Windows (ADLX manual tuning); root on Linux, so measure only there | +| AMD | `igneum-gpu-telemetry --card N --set-plimit ` (0 = default, -20 = 80%), inside the `tune` line's `plimit_range` (PC 1's 9070 XT: -30 to 10, so 70% is the floor) | `--set-gmax` only when the `tune` line's `gmax_range` is absolute MHz (floor 0 or above); on RDNA 4 the range is an offset from stock (-500 to 1000 on PC 1) and the clock knob stays closed until the stock clock is known; `--reset` for the default point | not settable through ADLX on RDNA 4; read back as `mclk_mhz`, and a step whose mean memory clock falls under 95% of the baseline's is marked and cannot win | the helper, one process per request, exit 0 and a `tune ... ok` line | none on Windows (ADLX manual tuning); root on Linux, so measure only there | | Apple | none | none | none | measure only | none | Vendor limits are never exceeded and the floor is never undercut: the plan clamps every point (`Limits::clamp_clock`, @@ -151,7 +151,11 @@ measure, TUNE record, upload, aggregation, prior shape in a test manifest). The 9070 XT run are owed: the 5090 the moment Power control is switched on (one prompt, then the tune runs by itself within 2 minutes of steady mining), the 9070 XT when the card is back on the bus. -(The run's numbers are appended below when the job reports.) +Run 1 (ember-tune-pc1-1, 22:30 UTC): aborted 47 s in by the installed app quitting for the 0.3.11 update-now, before +any step; nothing set; the "before" snapshots are in the bench log (5090: 450 W of 575, 2,850 MHz core, 3,090 MHz +maximum, 14,001 MHz memory; 9070 XT present on bus 98 with OFFSET ranges `gmax_range -500 1000`, `plimit_range -30 +10`). The offset finding changed the AMD mapping (054e041): an offset clock range closes the clock knob and the power +ladder runs on a percent scale bounded by `plimit_range`. The re-run follows the 0.3.11 rollout. ## 9. Open From d69e1c9fae1ce4a63ede30bf5064f0c73331a2d3 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 22:53:32 +0000 Subject: [PATCH 11/63] C35: every quit names its source (Cmd::Quit carries it: the window host's stdin, the host gone, POST /api/quit, the --sweep run's end); the --sweep job never counts as Power control and sets no cap at start (it raised a UAC prompt on PC 1 at 22:30 UTC); the tune playbook's budget quit goes only to its own scratch URL file and says so Co-Authored-By: Claude Fable 5.1 --- app/igneum-app/src/engine.rs | 32 +++++++++++++++++++++--------- app/igneum-app/src/main.rs | 4 ++-- app/igneum-app/src/server.rs | 3 ++- relay/playbooks/ember-tune-pc1.ps1 | 11 +++++++++- 4 files changed, 37 insertions(+), 13 deletions(-) diff --git a/app/igneum-app/src/engine.rs b/app/igneum-app/src/engine.rs index 095a6028..c6e57572 100644 --- a/app/igneum-app/src/engine.rs +++ b/app/igneum-app/src/engine.rs @@ -75,7 +75,8 @@ pub enum Cmd { /// restart the node with the verifier decided again (src/verifier.rs): the trust setting changed, or the /// prover found a host that was not there when the node started RestartNode(String), - Quit, + /// quit, with its source (the log names it: C35, 5 October 2026, two unexplained quits) + Quit(&'static str), } pub struct Shared { @@ -463,6 +464,8 @@ pub struct Engine { last_error_event: Instant, // the efficiency sweep (src/sweep.rs): one card at a time sweep: Option, + /// who asked for the quit (the log's `quit:` line names it) + quit_source: &'static str, /// the request number the vendor tool last carried out (the run's acknowledgement) tune_acked: Option, /// cards whose confirm check found a better neighbour: the full plan runs next @@ -559,6 +562,7 @@ impl Engine { last_settings_save: now, last_error_event: now - Duration::from_secs(600), sweep: None, + quit_source: "unknown", tune_acked: None, tune_full_due: std::collections::HashSet::new(), sweep_pending: None, @@ -702,7 +706,7 @@ impl Engine { if self.shared.runtime.sweep_only { if supported.is_empty() { self.sweep_say("SWEEP none reason=no_supported_card"); - self.shared.send(Cmd::Quit); + self.shared.send(Cmd::Quit("the --sweep run (every card done)")); } else { self.sweep_queue = supported; } @@ -1021,7 +1025,9 @@ impl Engine { } } }, - Cmd::Quit => { + Cmd::Quit(source) => { + self.shared.log(&format!("quit requested by {source}")); + self.quit_source = source; self.quitting = true; self.st().quitting = true; } @@ -1472,6 +1478,10 @@ impl Engine { if self.power_busy { return; } + if self.shared.runtime.sweep_only { + self.shared.log(&format!("power cap ({why}): not touched under --sweep; the tune sets every limit itself")); + return; + } if self.sweep.is_some() || self.sweep_pending.is_some() { // the sweep owns the caps until it ends; it applies the chosen one itself self.shared.log(&format!("power cap ({why}): deferred, a sweep is running")); @@ -1856,7 +1866,7 @@ impl Engine { if let Some((idx, forced)) = pick { self.sweep_begin(idx, forced); } else if self.shared.runtime.sweep_only && self.sweep_queue.is_empty() && self.sweep_pending.is_none() { - self.shared.send(Cmd::Quit); + self.shared.send(Cmd::Quit("the --sweep run (every card done)")); } } @@ -2304,7 +2314,7 @@ impl Engine { } self.upload_logs(false); if self.shared.runtime.sweep_only && self.sweep_queue.is_empty() { - self.shared.send(Cmd::Quit); + self.shared.send(Cmd::Quit("the --sweep run (every card done)")); } } @@ -2356,7 +2366,7 @@ impl Engine { } else { self.sweep_retry.insert(idx, Instant::now() + Duration::from_secs(3600)); if self.shared.runtime.sweep_only && self.sweep_queue.is_empty() { - self.shared.send(Cmd::Quit); + self.shared.send(Cmd::Quit("the --sweep run (every card done)")); } } } @@ -3371,7 +3381,7 @@ impl Engine { } fn shutdown(&mut self) { - self.shared.log("quit: stopping the miners, then the node"); + self.shared.log(&format!("quit: stopping the miners, then the node (source: {})", self.quit_source)); self.jobs.abort(&self.shared, "the app is quitting"); if self.sweep.is_some() || self.sweep_pending.is_some() { self.sweep_abort("the app is quitting"); @@ -3426,7 +3436,11 @@ impl Engine { /// administrator rights (one UAC prompt on Windows, pkexec on Linux); the engine builds an elevated command only when /// Power control is on in Settings, or when it is itself the elevated PC sweep job (--sweep). fn elevation_allowed(power_control: bool, sweep_only: bool) -> bool { - power_control || sweep_only + // C35 (5 October 2026, 22:30 UTC): the unattended --sweep job on PC 1 counted as allowed and raised the one + // administrator prompt nobody was there to answer; an elevated job sets limits directly without asking (the tune + // probe's `direct`), so the flag adds nothing and Power control alone decides + let _ = sweep_only; + power_control } /// The notice when the one prompt was refused, cancelled or not answered: Power control goes back off, no retries. @@ -3634,7 +3648,7 @@ mod tests { // the decision (the project lead, 5 October 2026): off = the app never asks; the elevated PC sweep job is the exception assert!(!super::elevation_allowed(false, false)); assert!(super::elevation_allowed(true, false)); - assert!(super::elevation_allowed(false, true)); + assert!(!super::elevation_allowed(false, true), "the --sweep job alone never asks (C35)"); let mut cards = vec![ super::CardState { vendor: "nvidia".into(), enabled: true, device: "0".into(), name: "RTX 5090".into(), power_default_w: 575.0, power_limit_w: 575.0, power_pct: 80, ..Default::default() }, super::CardState { vendor: "amd".into(), enabled: true, device: "1".into(), name: "RX 9070 XT".into(), power_default_w: 300.0, power_limit_w: 300.0, power_pct: 80, ..Default::default() }, diff --git a/app/igneum-app/src/main.rs b/app/igneum-app/src/main.rs index e94427d9..19b96ed9 100644 --- a/app/igneum-app/src/main.rs +++ b/app/igneum-app/src/main.rs @@ -133,7 +133,7 @@ fn main() { let Ok(l) = line else { break }; let t = l.trim(); match t { - "quit" => shared.send(engine::Cmd::Quit), + "quit" => shared.send(engine::Cmd::Quit("the window host (quit on stdin: the tray menu or the installer)")), "pause" => shared.send(engine::Cmd::Pause), "resume" => shared.send(engine::Cmd::Resume), "elevated ok" => shared.send(engine::Cmd::ElevatedDone(Ok(()))), @@ -142,7 +142,7 @@ fn main() { } } if wrapper { - shared.send(engine::Cmd::Quit); + shared.send(engine::Cmd::Quit("the window host went away (stdin closed)")); } }); } diff --git a/app/igneum-app/src/server.rs b/app/igneum-app/src/server.rs index e9884e36..b2ddb9f3 100644 --- a/app/igneum-app/src/server.rs +++ b/app/igneum-app/src/server.rs @@ -339,7 +339,8 @@ fn api_post(shared: &Arc, path: &str, body: Value) -> Result { - shared.send(Cmd::Quit); + // the caller is on 127.0.0.1 and holds the token: the installer, the OTA apply, a script that read app.url + shared.send(Cmd::Quit("POST /api/quit (a local caller with the token: the installer, the OTA apply, or a script that read app.url)")); Ok(json!({ "ok": true })) } _ => Err("unknown api".into()), diff --git a/relay/playbooks/ember-tune-pc1.ps1 b/relay/playbooks/ember-tune-pc1.ps1 index 2e135be2..26852830 100644 --- a/relay/playbooks/ember-tune-pc1.ps1 +++ b/relay/playbooks/ember-tune-pc1.ps1 @@ -15,6 +15,7 @@ # therefore measure only tonight unless the engine finds itself elevated. $ErrorActionPreference = 'Continue' $budgetMinutes = 35 +if (-not ($budgetMinutes -is [int]) -or $budgetMinutes -lt 5) { $budgetMinutes = 35 } # a budget under 5 minutes is a bug, not a budget (C35) $started = Get-Date $deadline = $started.AddMinutes($budgetMinutes) function Say([string] $m) { Write-Host ("[" + (Get-Date -Format 'HH:mm:ss') + "] " + $m) } @@ -123,8 +124,16 @@ while (-not $p.HasExited) { } if ((Get-Date) -gt $deadline) { Say ("budget of " + $budgetMinutes + " min spent; asking the tune engine to quit") + # C35 (5 October 2026): the only quit this script may send goes to the TUNE engine's own URL file in the scratch + # root, never to a file under the installed app's folder; the RESULT line names the file it used $u = Join-Path $sApp 'app.url' - if (Test-Path $u) { try { Invoke-WebRequest -Uri ((Get-Content -LiteralPath $u -Raw).Trim() + 'api/quit') -Method POST -Body '{}' -ContentType 'application/json' -UseBasicParsing -TimeoutSec 5 | Out-Null } catch { } } + $installedUrl = Join-Path $appDir 'app.url' + if ((Resolve-Path -LiteralPath $u -ErrorAction SilentlyContinue).Path -eq (Resolve-Path -LiteralPath $installedUrl -ErrorAction SilentlyContinue).Path -or $u -like '*\igneum\app\*') { + Write-Output ('RESULT TUNE quit refused: ' + $u + ' is the installed app''s URL file') + } elseif (Test-Path -LiteralPath $u) { + Write-Output ('RESULT TUNE quit asked of the tune engine through ' + $u + ' (pid ' + $p.Id + ')') + try { Invoke-WebRequest -Uri ((Get-Content -LiteralPath $u -Raw).Trim() + 'api/quit') -Method POST -Body '{}' -ContentType 'application/json' -UseBasicParsing -TimeoutSec 5 | Out-Null } catch { } + } else { Write-Output ('RESULT TUNE quit not sent: no URL file at ' + $u + '; killing pid ' + $p.Id) } Start-Sleep -Seconds 20 if (-not $p.HasExited) { $p.Kill() } Write-Output 'RESULT TUNE error=budget_exceeded' From d19441c14deaa8ed7a197fe39513a47434b68b61 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 23:00:06 +0000 Subject: [PATCH 12/63] C35 class: a second engine gets no pipe (its output goes to a file the playbook tails) and its whole tree is ended at the end and on the budget; ember-tune-pc1.ps1 and sweep-5090.ps1 fixed; tools/ci/second-engine-check.sh fails any playbook without both; the rule in ember-tune.md PC 1, 22:31 UTC: the installed engine's quit hung 24 minutes in the jobs runner's abort, waiting for EOF on the script's stdout pipe whose write end the second engine and its miners had inherited (Process.Start with redirection inherits every inheritable handle), while the orphaned miners mined on against the relaunched app. Co-Authored-By: Claude Fable 5.1 --- .github/workflows/ci.yml | 2 ++ docs/plans/ember-tune.md | 2 ++ relay/playbooks/ember-tune-pc1.ps1 | 44 ++++++++++++++++-------------- relay/playbooks/sweep-5090.ps1 | 44 ++++++++++++++++-------------- tools/ci/second-engine-check.sh | 24 ++++++++++++++++ 5 files changed, 74 insertions(+), 42 deletions(-) create mode 100755 tools/ci/second-engine-check.sh diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 185fa0f4..138a5d15 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -67,6 +67,8 @@ jobs: run: bash tools/ci/no-conflict-markers.sh - name: copied sources are re-stamped before a build run: bash tools/ci/copied-sources-check.sh + - name: second-engine playbooks log to a file and end their tree (C35) + run: bash tools/ci/second-engine-check.sh - name: pinned guest programs match their manifest and are built only by pin-guests.sh run: bash tools/ci/pinned-guests-check.sh - name: no secret file names and no 64-hex secrets in the tree (self-test first, then the tree) diff --git a/docs/plans/ember-tune.md b/docs/plans/ember-tune.md index 61a3d730..4e0c9fea 100644 --- a/docs/plans/ember-tune.md +++ b/docs/plans/ember-tune.md @@ -99,6 +99,8 @@ race has run). | A signed prior is only ever a starting point inside the card's OWN reported limits (`power.min_limit` to `power.max_limit`, the clock floor to `clocks.max.gr` or the ADLX `gmax_range`), never a memory clock, never a value the card did not report; the confirm step measures it and the full plan replaces it when a neighbour beats it, so a bad prior costs the fleet one confirm step per card, not a setting. The signing key (K1, docs/security/keys.md) therefore cannot push a card past its vendor ceiling or under its floor | `Plan::confirm` clamps through `Limits::clamp_clock` and `power_pct.clamp(50, 100)`; proven by `ember::tests::the_confirm_plan_checks_the_prior_and_its_neighbour` (a prior of 9,000 MHz at 30% becomes 3,090 MHz at 50%) and `limits_never_exceed_the_vendor_or_undercut_the_floor` | | No prompt the user did not ask for: the NVIDIA helper starts only with Power control on; the `--sweep` job never counts as permission | `sweep_probe_known`, `sweep_helper_start` | | The elevated helper restores the limit and resets the clocks by itself after 20 idle minutes | `sweep::helper_script_*` | +| A playbook that starts a second engine beside the installed app (the PC measurement jobs) gives it NO pipe (its output goes to a file the script tails: a pipe's write end is inherited by the engine's miners, and the installed app's jobs runner then waits forever for EOF after an abort; C35, PC 1 22:31 UTC, a 24-minute hang and orphaned miners), ends the engine's whole process tree at the end and on the budget (`taskkill /T /F`), and lets the installed app's miners come back only after that | `relay/playbooks/ember-tune-pc1.ps1`, `sweep-5090.ps1`; CI `tools/ci/second-engine-check.sh` fails any playbook without both | +| Every `quit:` line in the app log names its source (the window host's stdin, the host gone, `POST /api/quit`, the `--sweep` run's end) | `Cmd::Quit(&'static str)` (b671c8b) | ## 6. Tests diff --git a/relay/playbooks/ember-tune-pc1.ps1 b/relay/playbooks/ember-tune-pc1.ps1 index 26852830..0ce8566c 100644 --- a/relay/playbooks/ember-tune-pc1.ps1 +++ b/relay/playbooks/ember-tune-pc1.ps1 @@ -14,6 +14,7 @@ # ends. Not elevated: nothing asks for administrator rights (the project lead asleep, 5 October 2026); the NVIDIA card is # therefore measure only tonight unless the engine finds itself elevated. $ErrorActionPreference = 'Continue' +$resultTag = 'TUNE' $budgetMinutes = 35 if (-not ($budgetMinutes -is [int]) -or $budgetMinutes -lt 5) { $budgetMinutes = 35 } # a budget under 5 minutes is a bug, not a budget (C35) $started = Get-Date @@ -94,29 +95,28 @@ Snapshot 'before' $env:IGNEUM_APP_DATA = $root $env:IGNEUM_APP_LOGS = $sLogs $env:IGNEUM_APP_STATUS_SECS = '10' -$psi = New-Object System.Diagnostics.ProcessStartInfo -$psi.FileName = $exe -$psi.Arguments = '--sweep' -$psi.WorkingDirectory = $bin -$psi.UseShellExecute = $false -$psi.RedirectStandardOutput = $true -$psi.RedirectStandardError = $true -$psi.CreateNoWindow = $true -$p = New-Object System.Diagnostics.Process -$p.StartInfo = $psi -$lines = New-Object System.Collections.ArrayList -$h = { if ($EventArgs.Data) { [void]$Event.MessageData.Add($EventArgs.Data) } } -Register-ObjectEvent -InputObject $p -EventName OutputDataReceived -Action $h -MessageData $lines | Out-Null -Register-ObjectEvent -InputObject $p -EventName ErrorDataReceived -Action $h -MessageData $lines | Out-Null -[void]$p.Start() -$p.BeginOutputReadLine(); $p.BeginErrorReadLine() -Say ("tune engine started, pid " + $p.Id + ", data " + $root) +# C35 (5 October 2026): the engine's output goes to a FILE, never a pipe. A pipe's write end is inherited by every +# process the engine starts (its miners and workers), so after an abort the installed app's jobs runner waits for an +# EOF that never comes and hangs in its own quit; and the engine's whole tree is killed at the end (nothing orphaned). +$outFile = Join-Path $root 'engine-stdout.log' +$errFile = Join-Path $root 'engine-stderr.log' +Remove-Item -LiteralPath $outFile, $errFile -Force -ErrorAction SilentlyContinue +$p = Start-Process -FilePath $exe -ArgumentList '--sweep' -WorkingDirectory (Split-Path $exe) -WindowStyle Hidden -PassThru -RedirectStandardOutput $outFile -RedirectStandardError $errFile +Say ("engine started, pid " + $p.Id + ", data " + $root + ", stdout " + $outFile) +function EndTree([int] $procId, [string] $why) { + $before = @(Get-Process -Name 'igneum-app', 'igneum-miner', 'igneum-worker-cuda', 'igneum-worker-opencl', 'igneum-worker-metal' -ErrorAction SilentlyContinue).Count + & taskkill /T /F /PID $procId 2>&1 | Out-Null + Start-Sleep -Seconds 2 + $after = @(Get-Process -Name 'igneum-app', 'igneum-miner', 'igneum-worker-cuda', 'igneum-worker-opencl', 'igneum-worker-metal' -ErrorAction SilentlyContinue).Count + Write-Output ('RESULT ' + $resultTag + ' tree ended (' + $why + '): igneum processes ' + $before + ' -> ' + $after + ' (the installed app''s own miners are stopped and held by the job)') +} $seen = 0 $rows = 0 while (-not $p.HasExited) { Start-Sleep -Seconds 5 - while ($seen -lt $lines.Count) { - $l = [string]$lines[$seen]; $seen++ + $all = @(); if (Test-Path -LiteralPath $outFile) { $all = @(Get-Content -LiteralPath $outFile -ErrorAction SilentlyContinue) } + while ($seen -lt $all.Count) { + $l = [string]$all[$seen]; $seen++ if ($l -match '^TUNE ') { Write-Output ('RESULT ' + $l); if ($l -match '^TUNE card=') { $rows++ } } elseif ($l -match '^SWEEP ') { Write-Output ('RESULT ' + $l) } elseif ($l -match '^(URL|STATE) ') { } @@ -135,12 +135,14 @@ while (-not $p.HasExited) { try { Invoke-WebRequest -Uri ((Get-Content -LiteralPath $u -Raw).Trim() + 'api/quit') -Method POST -Body '{}' -ContentType 'application/json' -UseBasicParsing -TimeoutSec 5 | Out-Null } catch { } } else { Write-Output ('RESULT TUNE quit not sent: no URL file at ' + $u + '; killing pid ' + $p.Id) } Start-Sleep -Seconds 20 - if (-not $p.HasExited) { $p.Kill() } + if (-not $p.HasExited) { EndTree $p.Id 'budget' } Write-Output 'RESULT TUNE error=budget_exceeded' } } -while ($seen -lt $lines.Count) { $l = [string]$lines[$seen]; $seen++; if ($l -match '^TUNE ') { Write-Output ('RESULT ' + $l); if ($l -match '^TUNE card=') { $rows++ } } } +$all = @(); if (Test-Path -LiteralPath $outFile) { $all = @(Get-Content -LiteralPath $outFile -ErrorAction SilentlyContinue) } +while ($seen -lt $all.Count) { $l = [string]$all[$seen]; $seen++; if ($l -match '^TUNE ') { Write-Output ('RESULT ' + $l); if ($l -match '^TUNE card=') { $rows++ } } } Say ("tune engine exited " + $p.ExitCode + " after " + [int]((Get-Date) - $started).TotalSeconds + " s, " + $rows + " table rows") +EndTree $p.Id 'end of run' Snapshot 'after' # the tune engine's own log: the TUNE lines and what happened around them $log = Get-ChildItem -Path $sLogs -Filter 'app-*.log' -ErrorAction SilentlyContinue | Sort-Object LastWriteTime -Descending | Select-Object -First 1 diff --git a/relay/playbooks/sweep-5090.ps1 b/relay/playbooks/sweep-5090.ps1 index da6c2ebb..bb0a3a77 100644 --- a/relay/playbooks/sweep-5090.ps1 +++ b/relay/playbooks/sweep-5090.ps1 @@ -8,6 +8,7 @@ # miners restart when the job ends. Elevated, so nvidia-smi -pl needs no prompt (the engine detects that: mode=direct). # UNTESTED on a PC as of 4 Oct 2026 (parse-checked only). $ErrorActionPreference = 'Continue' +$resultTag = 'SWEEP' $budgetMinutes = 40 $started = Get-Date $deadline = $started.AddMinutes($budgetMinutes) @@ -59,29 +60,28 @@ if (Test-Path $smi) { $env:IGNEUM_APP_DATA = $root $env:IGNEUM_APP_LOGS = $sLogs $env:IGNEUM_APP_STATUS_SECS = '10' -$psi = New-Object System.Diagnostics.ProcessStartInfo -$psi.FileName = $exe -$psi.Arguments = '--sweep' -$psi.WorkingDirectory = Split-Path $exe -$psi.UseShellExecute = $false -$psi.RedirectStandardOutput = $true -$psi.RedirectStandardError = $true -$psi.CreateNoWindow = $true -$p = New-Object System.Diagnostics.Process -$p.StartInfo = $psi -$lines = New-Object System.Collections.ArrayList -$h = { if ($EventArgs.Data) { [void]$Event.MessageData.Add($EventArgs.Data) } } -Register-ObjectEvent -InputObject $p -EventName OutputDataReceived -Action $h -MessageData $lines | Out-Null -Register-ObjectEvent -InputObject $p -EventName ErrorDataReceived -Action $h -MessageData $lines | Out-Null -[void]$p.Start() -$p.BeginOutputReadLine(); $p.BeginErrorReadLine() -Say ("sweep engine started, pid " + $p.Id + ", data " + $root) +# C35 (5 October 2026): the engine's output goes to a FILE, never a pipe. A pipe's write end is inherited by every +# process the engine starts (its miners and workers), so after an abort the installed app's jobs runner waits for an +# EOF that never comes and hangs in its own quit; and the engine's whole tree is killed at the end (nothing orphaned). +$outFile = Join-Path $root 'engine-stdout.log' +$errFile = Join-Path $root 'engine-stderr.log' +Remove-Item -LiteralPath $outFile, $errFile -Force -ErrorAction SilentlyContinue +$p = Start-Process -FilePath $exe -ArgumentList '--sweep' -WorkingDirectory (Split-Path $exe) -WindowStyle Hidden -PassThru -RedirectStandardOutput $outFile -RedirectStandardError $errFile +Say ("engine started, pid " + $p.Id + ", data " + $root + ", stdout " + $outFile) +function EndTree([int] $procId, [string] $why) { + $before = @(Get-Process -Name 'igneum-app', 'igneum-miner', 'igneum-worker-cuda', 'igneum-worker-opencl', 'igneum-worker-metal' -ErrorAction SilentlyContinue).Count + & taskkill /T /F /PID $procId 2>&1 | Out-Null + Start-Sleep -Seconds 2 + $after = @(Get-Process -Name 'igneum-app', 'igneum-miner', 'igneum-worker-cuda', 'igneum-worker-opencl', 'igneum-worker-metal' -ErrorAction SilentlyContinue).Count + Write-Output ('RESULT ' + $resultTag + ' tree ended (' + $why + '): igneum processes ' + $before + ' -> ' + $after + ' (the installed app''s own miners are stopped and held by the job)') +} $seen = 0 $rows = 0 while (-not $p.HasExited) { Start-Sleep -Seconds 5 - while ($seen -lt $lines.Count) { - $l = [string]$lines[$seen]; $seen++ + $all = @(); if (Test-Path -LiteralPath $outFile) { $all = @(Get-Content -LiteralPath $outFile -ErrorAction SilentlyContinue) } + while ($seen -lt $all.Count) { + $l = [string]$all[$seen]; $seen++ if ($l -match '^SWEEP ') { Write-Output ('RESULT ' + $l); if ($l -match '^SWEEP card=') { $rows++ } } elseif ($l -match '^(URL|STATE) ') { } else { Say $l } @@ -91,12 +91,14 @@ while (-not $p.HasExited) { $u = Join-Path $sApp 'app.url' if (Test-Path $u) { try { Invoke-WebRequest -Uri ((Get-Content -LiteralPath $u -Raw).Trim() + 'api/quit') -Method POST -Body '{}' -ContentType 'application/json' -UseBasicParsing -TimeoutSec 5 | Out-Null } catch { } } Start-Sleep -Seconds 20 - if (-not $p.HasExited) { $p.Kill() } + if (-not $p.HasExited) { EndTree $p.Id 'budget' } Write-Output 'RESULT SWEEP error=budget_exceeded' } } -while ($seen -lt $lines.Count) { $l = [string]$lines[$seen]; $seen++; if ($l -match '^SWEEP ') { Write-Output ('RESULT ' + $l); if ($l -match '^SWEEP card=') { $rows++ } } } +$all = @(); if (Test-Path -LiteralPath $outFile) { $all = @(Get-Content -LiteralPath $outFile -ErrorAction SilentlyContinue) } +while ($seen -lt $all.Count) { $l = [string]$all[$seen]; $seen++; if ($l -match '^SWEEP ') { Write-Output ('RESULT ' + $l); if ($l -match '^SWEEP card=') { $rows++ } } } Say ("sweep engine exited " + $p.ExitCode + " after " + [int]((Get-Date) - $started).TotalSeconds + " s, " + $rows + " table rows") +EndTree $p.Id 'end of run' if (Test-Path $smi) { $q = (& $smi --query-gpu=index,power.draw,power.limit --format=csv,noheader 2>&1 | Out-String).Trim() Write-Output ("RESULT SWEEP after " + ($q -replace "`r?`n", ' | ')) diff --git a/tools/ci/second-engine-check.sh b/tools/ci/second-engine-check.sh new file mode 100755 index 00000000..564d390f --- /dev/null +++ b/tools/ci/second-engine-check.sh @@ -0,0 +1,24 @@ +#!/usr/bin/env bash +# The second-engine class (C35, 5 October 2026, PC 1 22:31 UTC): a playbook started a second Igneum engine beside the +# installed app through a redirected PIPE (PowerShell's Process.Start with RedirectStandardOutput). A pipe's write end is +# inherited by every process the engine starts (its miners and workers); when the job was aborted, the installed app's +# jobs runner waited for an EOF the orphaned grandchildren never sent and its quit hung for 24 minutes, and the second +# engine's miners mined on against the relaunched app. Rule for every playbook that starts an engine: (1) the engine's +# output goes to a FILE (Start-Process -RedirectStandardOutput ), never a pipe into the script; (2) the engine's +# whole process tree is ended at the end and on the budget (taskkill /T /F), so nothing is orphaned; the installed +# app's miners come back only after that (the job runner restarts them when the script ends). This check fails CI when +# a playbook starts an engine without both. +set -euo pipefail +cd "$(dirname "$0")/../.." +fail=0 +while IFS= read -r f; do + grep -qE "igneum-app(\.exe)?['\"]? *(--sweep|-ArgumentList '--sweep'|--no-open)|ArgumentList '--sweep'|\.Arguments = '--sweep'" "$f" || continue + if grep -qE 'RedirectStandardOutput *= *\$true|UseShellExecute *= *\$false|Register-ObjectEvent|BeginOutputReadLine' "$f"; then + echo "second-engine: $f starts an engine through a pipe (RedirectStandardOutput/BeginOutputReadLine); use Start-Process -RedirectStandardOutput "; fail=1 + fi + if ! grep -qE 'taskkill /T /F' "$f"; then + echo "second-engine: $f starts an engine without ending its process tree (taskkill /T /F) at the end"; fail=1 + fi +done < <(git ls-files 'relay/playbooks/**' 'tools/windows/**' 'packaging/**' | grep -E '\.ps1$') +[ "$fail" = 0 ] && echo "second-engine: every playbook that starts an engine logs to a file and ends its tree" +exit $fail From f9d9805ae83db78bbb15f3a7597bc136c0e8ed65 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 23:00:52 +0000 Subject: [PATCH 13/63] bench log + plan: C35 corrected (the quit was not the 0.3.11 update; what is established, the hang, the orphans, the prompt, the fixes) Co-Authored-By: Claude Fable 5.1 --- docs/bench-log.md | 4 ++-- docs/plans/ember-tune.md | 4 ++-- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/docs/bench-log.md b/docs/bench-log.md index c19dfa12..21c6712a 100644 --- a/docs/bench-log.md +++ b/docs/bench-log.md @@ -1624,7 +1624,7 @@ Branch `ember-tune` (54ff1bc), docs/plans/ember-tune.md. Every card tuned for MH **Tier consequences** (docs/plans/ember-tune.md section 7): a 9-step full tune costs about 12 minutes once and 3 minutes a week per card, under 1% of the hour, the worker never stops; a rig tunes one card at a time and every card of a known model after the first takes the 3-minute confirm; a pool user gives up the same 1% of shares at most; Apple silicon and AMD on Linux measure only and the row says so. -**The PC 1 run, 22:30 UTC (job ember-tune-pc1-1, engine aeea3228..., PC 1 on 0.3.10):** the job published at 22:29:40Z, the installed app stopped its miners and started the second engine at 22:30:21Z, and at 22:31:06Z the installed app quit for the 0.3.11 update-now (its log: `job ember-tune-pc1-1: aborted (the app is quitting)`), taking the second engine with it 47 s in, before any step. Nothing was set. What the run did record, the "before" snapshots with the miners stopped: +**The PC 1 run, 22:30 UTC (job ember-tune-pc1-1, engine aeea3228..., PC 1 on 0.3.10):** the job published at 22:29:40Z, the installed app stopped its miners and started the second engine at 22:30:21Z, and at 22:31:06Z the installed app quit (its log: `quit: stopping the miners, then the node`, then `job ember-tune-pc1-1: aborted (the app is quitting)`), 46 s in, before any step. Nothing was set. Corrected the same night (C35): the first reading, that the 0.3.11 update caused the quit, was wrong; no update, restart or relay task reached PC 1 then (its own jobs lines and the relay feed), and the quit's source is not in the log because the app did not name it (fixed at b671c8b: every `quit:` line now names its sender). What is established: the window host's tray quit is excluded (a host that sent the quit terminates the engine 45 s later, and the engine lived on until 22:55Z), leaving stdin EOF (the host process gone) or `POST /api/quit`; the engine's quit then HUNG for 24 minutes in the jobs runner's abort, waiting for EOF on the script's stdout pipe whose write end the second engine and its miners had inherited, and those miners (2 igneum-miner, 2 CUDA workers, 1 OpenCL worker) mined on, orphaned, until the relay lane killed them at about 23:00Z; the second engine also raised one administrator prompt at about 22:30:25Z (`apply_power_limits` at start counted `--sweep` as Power control), 41 s before the quit; PC 2's unexplained quit at 20:01:09Z came 20 s after a cancelled prompt of the same class, so the prompt is the common factor and the morning's test (one prompt raised beside the mining app on PC 2, the stamped quit line read). Fixed on the branch: b671c8b (quit sources, Power control alone decides, no cap at start under `--sweep`), 8ab9068 (no pipe into a second engine, its tree ended, the CI check). What the run did record, the "before" snapshots with the miners stopped: | Card | Read back at 22:30:20Z | Meaning | |---|---|---| @@ -1632,5 +1632,5 @@ Branch `ember-tune` (54ff1bc), docs/plans/ember-tune.md. Every card tuned for MH | RX 9070 XT (bus 98, present again) | `tune 1 ... gmax 0 gmax_range -500 1000 plimit 0 plimit_range -30 10 factory 1 ok` | the helper's clock range is an OFFSET from stock in MHz, not a ceiling: a probe reading it as a 1,000 MHz maximum would have asked for `--set-gmax 900`, an overclock. Fixed at 054e041: an offset range closes the clock knob (until the stock clock is known) and the power ladder runs on the percent scale bounded by the range, so the 9070 XT's plan is 100, 90, 80, 70% (the -30 floor), 4 steps | | Radeon(TM) Graphics (integrated) | `tune 0 ... gmax - ... factory 0 ok` | no manual tuning: measure only, and it is off by default anyway | -Consequence for the tiers: an AMD card is tuned on its power limit alone until its stock core clock is read (a 9070 XT at -30% is the floor the driver allows, 4 steps, 5 minutes); every NVIDIA card's two-knob plan waits on the user's one click on Power control; the re-run on PC 1 follows the 0.3.11 rollout (the update clears the jobs folder, so the engine and the helper are fetched again), with the scheduler's slot. +Consequence for the tiers: an AMD card is tuned on its power limit alone until its stock core clock is read (a 9070 XT at -30% is the floor the driver allows, 4 steps, 5 minutes); every NVIDIA card's two-knob plan waits on the user's one click on Power control; the re-run on PC 1 is held until the quit's source is named (the event-log collect) and follows the 0.3.11 rollout (the update clears the jobs folder, so the engine and the helper are fetched again), with the scheduler's slot. diff --git a/docs/plans/ember-tune.md b/docs/plans/ember-tune.md index 4e0c9fea..61d02d14 100644 --- a/docs/plans/ember-tune.md +++ b/docs/plans/ember-tune.md @@ -153,8 +153,8 @@ measure, TUNE record, upload, aggregation, prior shape in a test manifest). The 9070 XT run are owed: the 5090 the moment Power control is switched on (one prompt, then the tune runs by itself within 2 minutes of steady mining), the 9070 XT when the card is back on the bus. -Run 1 (ember-tune-pc1-1, 22:30 UTC): aborted 47 s in by the installed app quitting for the 0.3.11 update-now, before -any step; nothing set; the "before" snapshots are in the bench log (5090: 450 W of 575, 2,850 MHz core, 3,090 MHz +Run 1 (ember-tune-pc1-1, 22:30 UTC): aborted 46 s in by the installed app quitting (source unnamed by the 0.3.10 app; +not an update, not a job, not a relay task: C35 in the bench log), before any step; nothing set; the "before" snapshots are in the bench log (5090: 450 W of 575, 2,850 MHz core, 3,090 MHz maximum, 14,001 MHz memory; 9070 XT present on bus 98 with OFFSET ranges `gmax_range -500 1000`, `plimit_range -30 10`). The offset finding changed the AMD mapping (054e041): an offset clock range closes the clock knob and the power ladder runs on a percent scale bounded by `plimit_range`. The re-run follows the 0.3.11 rollout. From 08547b0cac071c269e50f405765f7896f61b346c Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 23:35:03 +0000 Subject: [PATCH 14/63] HiveOS: the bundled node takes the override from the Flight Sheet (C41) The finding (release-0.3.11.md section 5): h-run.sh started the rig's node with --devnet --appdir --rpclisten --listen and the peers and no --override-params-file, so a HiveOS rig in local mode ran on genesis parameters, printed the no-override digest and was refused by every devnet peer; no package ever carried the override. h-config.sh: OVERRIDE= in the Flight Sheet's extra config, read as a whole line (JSON may carry spaces; single or double quotes around it are stripped), refused unless it is {...}, written to igneum.conf single-quoted (sq helper; EXTRA gets the same quoting, the same class: a value with shell characters sourced unquoted). Still sourceable (return, never exit). h-run.sh, local branch: writes data/override-params.json from OVERRIDE when set and passes --override-params-file= to igneumd; when empty, a WARNING in the main log that the node runs on genesis parameters and devnet peers will refuse it. After the node answers, the switch lines and the "Consensus params digest" line from node.log are copied into the main log as "node: ..." so the operator can compare the digest with the downloads page. README: the Flight Sheet table gains the OVERRIDE row with the four-field devnet object as the example (nine fields after the 0.3.11 switch; the downloads page carries the live one), the rule "set OVERRIDE from the downloads page when it changes", the sentence that a rig without it is refused, the digest check in the requirements, and the gap entry. No "every override publish needs a package republish" sentence exists in packaging/hive/README.md on this branch or master, so nothing was replaced; the new rule stands alone. selftest.sh: a fake igneumd that records its argv and prints the real digest line; OVERRIDE single-quoted on its own line beside other keys round-trips through the conf; OVERRIDE=notjson refused; empty OVERRIDE named in the summary; the main h-run run checks the file, the flag on the node and the digest and switch lines in the main log; a second short run without OVERRIDE checks the warning and the absence of the flag. bash packaging/hive/selftest.sh on this Mac: == h-config.sh OVERRIDE OVERRIDE (single-quoted, own line) sourced back intact beside the other keys ok OVERRIDE that is not {...} refused ok no OVERRIDE: empty in the conf and named in the summary ok == h-run.sh (fake GPUs: 2 NVIDIA, fake node, fake miner) data/override-params.json written from OVERRIDE ok the node got --override-params-file ok the node's digest and switch lines reached the main log ok NVIDIA cards got the cuda worker ok exit 42 restarted the miner and re-exported the pack ok == h-stats.sh (sourced) stats JSON ok: hs [118500.0, 118500.0] temp [61, 58] ar [24, 0] bus [1, 2] == h-run.sh without OVERRIDE (the warning) no OVERRIDE: warning in the main log, no flag on the node ok == self-test passed (scripts and stats shape; Hive itself is untested) Co-Authored-By: Claude Fable 5.1 --- packaging/hive/README.md | 11 ++++++++--- packaging/hive/h-config.sh | 40 +++++++++++++++++++++++++++----------- packaging/hive/h-run.sh | 22 +++++++++++++++++++-- packaging/hive/selftest.sh | 27 ++++++++++++++++++++++++- 4 files changed, 83 insertions(+), 17 deletions(-) diff --git a/packaging/hive/README.md b/packaging/hive/README.md index 706f8844..b59938a5 100644 --- a/packaging/hive/README.md +++ b/packaging/hive/README.md @@ -33,6 +33,7 @@ v1": jobs `memsweep-pc2-pv1`, `memminer-pc2-pv1` and the S_p curve, one RTX 5090 | Wallet and worker template | `0x<40 hex>.%WORKER_NAME%`: the payout address is an EVM address you hold the key for; the part after the dot labels this rig's keys | | Pool URL | `grpc://:26610` (your own igneumd, solo mining), or `local` to run the bundled node on the rig | | Pass | empty | +| OVERRIDE | the devnet's consensus override as one JSON object on its own line, needed with `local`: `OVERRIDE={"difficulty_v2_activation_daa":33000,"proving_v0_activation_daa":84100,"fees_v1_activation_daa":210000,"finality_v3_activation_daa":135200}` is the four-field object 0.3.10 shipped; after the 0.3.11 switch it has nine fields, and the downloads page carries the live one. Set OVERRIDE from the downloads page when it changes. **A rig without it is refused**: its node runs on genesis parameters, prints another digest, and every devnet peer drops it (`h-run.sh` warns in the main log) | | Extra config arguments | `DEV_FEE=1 IDENTITIES=auto WORKER=auto VOTE=1` (one per line also works); `PEERS=a:26611,b:26611` for the bundled node; `EXTRA="..."` for more miner flags | | IDENTITIES | vote keys per card: 8 for a card with 8 GB or more, else 2 (`IDENTITIES=auto` applies that rule, per card, from `nvidia-smi` or the amdgpu sysfs; 8 when neither answers; a number overrides it for every card). The rule is the app's (`app/igneum-app/src/detect.rs`) | @@ -55,8 +56,8 @@ The protocol carries no fee: this is the software's, and any other miner client | Hook | What | |---|---| -| `h-config.sh` | writes `igneum.conf` from the Flight Sheet (node URL, wallet, label, DEV_FEE, IDENTITIES, WORKER, VOTE, PEERS, EXTRA); resolves `IDENTITIES=auto` to `IDENTITIES_GPU` keys by VRAM; refuses a wallet that is not 0x + 40 hex | -| `h-run.sh` | starts the bundled node when the URL is `local`, waits for the node, exports the hourly program pack (`igneum-miner export-pack`), then one `igneum-miner` per GPU with its worker (`--identities` from `IDENTITIES_GPU`, else `IDENTITIES`); restarts a miner that exits (exit 42 = program change without prepare support: the pack is re-exported first); per-GPU logs `.gpu.log`, merged into the main log | +| `h-config.sh` | writes `igneum.conf` from the Flight Sheet (node URL, wallet, label, DEV_FEE, IDENTITIES, WORKER, VOTE, PEERS, EXTRA, OVERRIDE); refuses an OVERRIDE that is not `{...}`; resolves `IDENTITIES=auto` to `IDENTITIES_GPU` keys by VRAM; refuses a wallet that is not 0x + 40 hex | +| `h-run.sh` | starts the bundled node when the URL is `local` (writes `data/override-params.json` from OVERRIDE and passes `--override-params-file`; warns when OVERRIDE is empty; copies the node's switch lines and its `Consensus params digest` line into the main log), waits for the node, exports the hourly program pack (`igneum-miner export-pack`), then one `igneum-miner` per GPU with its worker (`--identities` from `IDENTITIES_GPU`, else `IDENTITIES`); restarts a miner that exits (exit 42 = program change without prepare support: the pack is re-exported first); per-GPU logs `.gpu.log`, merged into the main log | | `h-stats.sh` | per-GPU hash rate from each miner's last `STATUS` line (`now=`), accepted and rejected totals, dev-fee block count, temperatures and fans from Hive's `gpu-stats` (else `nvidia-smi`), uptime, version | Stats JSON (what Hive reads from `$stats`): `hs` (kH/s per GPU), `hs_units` (`khs`), `temp`, `fan`, `uptime` (s), @@ -68,7 +69,9 @@ Stats JSON (what Hive reads from `$stats`): `hs` (kH/s per GPU), `hs_units` (`kh on the library path). The worker compiles the hourly program at run time with NVRTC; without the library it says so and the miner retries. Hive images ship the driver; whether `libnvrtc.so.12` is present depends on the image, untested. - AMD: an OpenCL ICD (`libOpenCL.so.1` from ROCm or amdgpu-pro). The OpenCL worker compiles the program through the ICD. -- The bundled node (`local`) keeps its chain data under the miner folder (`data/`); a devnet chain is small today. +- The bundled node (`local`) keeps its chain data under the miner folder (`data/`); a devnet chain is small today. It + needs OVERRIDE (the Flight Sheet table): compare the `node: Consensus params digest:` line in the main log with the + digest on the downloads page; a different one means the override is stale and the node is refused. - Ports: the bundled node listens on 26611 (p2p) and answers RPC on 127.0.0.1:26610 only. ## Building the package @@ -85,5 +88,7 @@ Stats JSON (what Hive reads from `$stats`): `hs` (kH/s per GPU), `hs_units` (`kh - GPU order: the CUDA device index is assumed to follow `nvidia-smi` order and Hive's `gpu-stats` arrays (NVIDIA first); a mixed NVIDIA and AMD rig may show temperatures against the wrong card. - Each card runs its own `igneum-miner` and node connection; the node's template RPC serves them all. +- Before this change no package carried the override: `h-run.sh` started the node without `--override-params-file`, so + a `local` rig was refused by every devnet peer (release 0.3.11, section 5, C41). Still untested on a rig. - No prover in the package (the second paragraph above): the rig earns nothing from the proving share until a Linux prover build ships. diff --git a/packaging/hive/h-config.sh b/packaging/hive/h-config.sh index 3c724ae7..2b374375 100755 --- a/packaging/hive/h-config.sh +++ b/packaging/hive/h-config.sh @@ -16,6 +16,11 @@ # VOTE=1 sign finality checkpoints (0 = mine without voting) # PEERS=a:26611,b:26611 peers for the bundled node when CUSTOM_URL=local # EXTRA="..." appended to every igneum-miner command line +# OVERRIDE={...} the devnet's consensus override, one JSON object on its own line (single or +# double quotes around it are fine in the Hive UI); the bundled node (CUSTOM_URL=local) +# starts with --override-params-file from it. Without it the node runs on genesis +# parameters and every devnet peer refuses it. Copy the live object from the +# downloads page whenever it changes. # Hive sources h-manifest.conf before this hook; the fallback is for a run outside Hive (selftest.sh) [[ -z "$CUSTOM_CONFIG_FILENAME" ]] && . "$(dirname "${BASH_SOURCE[0]}")/h-manifest.conf" @@ -32,15 +37,26 @@ if ! [[ "$wallet" =~ ^0x[0-9a-fA-F]{40}$ ]]; then fi # defaults, then the user's KEY=VALUE lines -DEV_FEE=1; IDENTITIES=auto; WORKER=auto; VOTE=1; PEERS=""; EXTRA="" -while read -r kv; do - [[ -z "$kv" || "$kv" == \#* ]] && continue - key="${kv%%=*}"; val="${kv#*=}" - case "$key" in - DEV_FEE|IDENTITIES|WORKER|VOTE|PEERS|EXTRA) printf -v "$key" '%s' "$val" ;; - *) echo "Igneum: unknown setting '$key' ignored" ;; - esac -done < <(printf '%s\n' "$CUSTOM_USER_CONFIG" | tr ' ' '\n' | sed 's/^"//; s/"$//') +DEV_FEE=1; IDENTITIES=auto; WORKER=auto; VOTE=1; PEERS=""; EXTRA=""; OVERRIDE="" +while IFS= read -r line; do + [[ -z "$line" || "$line" == \#* ]] && continue + if [[ "$line" == OVERRIDE=* ]]; then # the whole line: JSON may carry spaces; quotes around it are stripped + val="${line#OVERRIDE=}"; val="${val#\'}"; val="${val%\'}"; val="${val#\"}"; val="${val%\"}" + OVERRIDE="$val"; continue + fi + while read -r kv; do + [[ -z "$kv" ]] && continue + key="${kv%%=*}"; val="${kv#*=}" + case "$key" in + DEV_FEE|IDENTITIES|WORKER|VOTE|PEERS|EXTRA) printf -v "$key" '%s' "$val" ;; + *) echo "Igneum: unknown setting '$key' ignored" ;; + esac + done < <(printf '%s\n' "$line" | tr ' ' '\n' | sed 's/^"//; s/"$//') +done < <(printf '%s\n' "$CUSTOM_USER_CONFIG") +if [[ -n "$OVERRIDE" && ! "$OVERRIDE" =~ ^\{.*\}$ ]]; then + echo -e "${YELLOW:-}Igneum: OVERRIDE must be one JSON object, {\"...\":N,...} from the downloads page; got '${OVERRIDE:0:40}'${NOCOLOR:-}" + return 1 +fi [[ "$DEV_FEE" =~ ^[0-9]+$ ]] || DEV_FEE=1 [[ "$IDENTITIES" =~ ^[0-9]+$ || "$IDENTITIES" == "auto" ]] || IDENTITIES=auto @@ -81,6 +97,7 @@ url="$CUSTOM_URL" [[ "$url" == "local" ]] && url="local" [[ "$url" != "local" && "$url" != grpc://* ]] && url="grpc://$url" +sq() { local v="${1//\'/\'\\\'\'}"; printf "'%s'" "$v"; } # single-quoted for sourcing: JSON and flags carry shell characters mkdir -p "$(dirname "$CUSTOM_CONFIG_FILENAME")" cat > "$CUSTOM_CONFIG_FILENAME" < "$HERE/data/override-params.json" + override=("--override-params-file=$HERE/data/override-params.json") + say "consensus override from the Flight Sheet: $OVERRIDE" + else + say "WARNING: no OVERRIDE in the Flight Sheet; the node runs on genesis parameters and every devnet peer will refuse it. Set OVERRIDE from the downloads page." + fi + say "starting the bundled node (devnet, data $HERE/data, peers ${peers[*]#--addpeer=})" "$BIN/igneumd" --devnet "--appdir=$HERE/data" --rpclisten=127.0.0.1:26610 --listen=0.0.0.0:26611 "${peers[@]}" \ - --nodnsseed --disable-upnp --nologfiles --yes >> "$CUSTOM_LOG_BASENAME.node.log" 2>&1 & + ${override[@]+"${override[@]}"} --nodnsseed --disable-upnp --nologfiles --yes >> "$CUSTOM_LOG_BASENAME.node.log" 2>&1 & pids+=($!) fi say "node $NODE_URL; waiting for it to answer" @@ -37,6 +48,13 @@ for _ in $(seq 1 60); do "$BIN/igneum-miner" watch 1 "$NODE_URL" >/dev/null 2>&1 && break sleep 2 done +if [[ -f "$CUSTOM_LOG_BASENAME.node.log" ]]; then + # the switch lines and the digest the node printed at start, so the operator can compare the digest with the one + # on the downloads page (another digest = refused by every peer) + n=0 + while IFS= read -r l; do say "node: $l"; n=$((n + 1)); done < <(grep -E 'override params file|from the override file|Consensus params digest' "$CUSTOM_LOG_BASENAME.node.log" | head -12) + [[ $n == 0 ]] && say "node: no digest line yet in $CUSTOM_LOG_BASENAME.node.log (an older node, or not started); check it by hand" +fi # 2. the GPUs: Hive's gpu-detect when present, else the driver tools nv=0; amd=0 diff --git a/packaging/hive/selftest.sh b/packaging/hive/selftest.sh index 40fcedfe..044442f4 100755 --- a/packaging/hive/selftest.sh +++ b/packaging/hive/selftest.sh @@ -35,6 +35,14 @@ case "$1" in esac FAKE chmod +x "$M/bin/igneum-miner" +cat > "$M/bin/igneumd" <<'FAKE' +#!/usr/bin/env bash +echo "fake igneumd $*" +for a in "$@"; do case "$a" in --override-params-file=*) echo "Finality rule v3 from the override file: active from checkpoint DAA score 135200" ;; esac; done +echo "Consensus params digest: 0139ab9dc2992d449ec787d8f021974933631eb55740ab4b6ce9d5c226e72888 (exchanged in the p2p handshake; a peer with another digest is refused)" +sleep 600 +FAKE +chmod +x "$M/bin/igneumd" mkdir -p "$T/bin" printf '#!/bin/sh\n[ "$1" = NVIDIA ] && echo 2 || echo 0\n' > "$T/bin/gpu-detect" cat > "$T/bin/gpu-stats" <<'GS' @@ -62,8 +70,14 @@ grep -q '^IDENTITIES_GPU0=2$' "$T/auto-nv.conf" && grep -q '^IDENTITIES_GPU1=8$' # a number still overrides every card ( CUSTOM_USER_CONFIG="IDENTITIES=4" CUSTOM_CONFIG_FILENAME="$T/auto-num.conf" . "$M/h-config.sh" >/dev/null ) && grep -q '^IDENTITIES=4$' "$T/auto-num.conf" && ! grep -q '^IDENTITIES_GPU' "$T/auto-num.conf" && echo " numeric override keeps IDENTITIES=4 ok" || { echo "FAIL: numeric override"; exit 1; } rm -f "$T/bin/nvidia-smi" +echo "== h-config.sh OVERRIDE" +OV='{"difficulty_v2_activation_daa":33000,"proving_v0_activation_daa":84100,"fees_v1_activation_daa":210000,"finality_v3_activation_daa":135200}' +( CUSTOM_USER_CONFIG="DEV_FEE=0 VOTE=1"$'\n'"OVERRIDE='$OV'"$'\n'"IDENTITIES=4" CUSTOM_CONFIG_FILENAME="$T/ov.conf" . "$M/h-config.sh" >/dev/null ) || { echo "h-config.sh failed on OVERRIDE"; exit 1; } +( . "$T/ov.conf"; [[ "$OVERRIDE" == "$OV" && "$DEV_FEE" == 0 && "$IDENTITIES" == 4 ]] ) && echo " OVERRIDE (single-quoted, own line) sourced back intact beside the other keys ok" || { echo "FAIL: OVERRIDE round trip"; cat "$T/ov.conf"; exit 1; } +( CUSTOM_USER_CONFIG="OVERRIDE=notjson" CUSTOM_CONFIG_FILENAME="$T/ov-bad.conf" . "$M/h-config.sh" >/dev/null 2>&1 ) && { echo "FAIL: OVERRIDE=notjson accepted"; exit 1; } || echo " OVERRIDE that is not {...} refused ok" +( CUSTOM_USER_CONFIG="" CUSTOM_CONFIG_FILENAME="$T/ov-none.conf" . "$M/h-config.sh" | grep -q "override NONE" ) && grep -q "^OVERRIDE=''$" "$T/ov-none.conf" && echo " no OVERRIDE: empty in the conf and named in the summary ok" || { echo "FAIL: empty OVERRIDE"; exit 1; } # h-run.sh reads IDENTITIES_GPU first: write a conf with gpu1=3 and check the second miner's argv -( CUSTOM_USER_CONFIG=$'DEV_FEE=0\nIDENTITIES=auto' IGNEUM_DRM_ROOT="$T/no-drm" CUSTOM_CONFIG_FILENAME="$M/igneum.conf" . "$M/h-config.sh" >/dev/null ) && printf 'IDENTITIES_GPU1=3\n' >> "$M/igneum.conf" +( CUSTOM_USER_CONFIG=$'DEV_FEE=0\nIDENTITIES=auto\n'"OVERRIDE=$OV" IGNEUM_DRM_ROOT="$T/no-drm" CUSTOM_CONFIG_FILENAME="$M/igneum.conf" . "$M/h-config.sh" >/dev/null ) && printf 'IDENTITIES_GPU1=3\n' >> "$M/igneum.conf" echo "== h-run.sh (fake GPUs: 2 NVIDIA, fake node, fake miner)" ( cd "$M" && CUSTOM_CONFIG_FILENAME="$M/igneum.conf" CUSTOM_LOG_BASENAME="$T/log/igneum" ./h-run.sh > "$T/log/run.out" 2>&1 ) & disown @@ -73,8 +87,19 @@ _lines=$(grep -c "dev fee off (--dev-fee 0)" "$T/log/igneum.log" || true); echo grep -q -- "--dev-fee 0" "$M/argv.log" && echo " DEV_FEE=0 reached the miner as --dev-fee 0 ok" || { echo "FAIL: --dev-fee 0 missing"; exit 1; } grep -q -- "rig7-gpu0 .*--identities 8" "$M/argv.log" && echo " gpu0 took the IDENTITIES=8 fallback ok" || { echo "FAIL: --identities 8 missing on gpu0"; cat "$M/argv.log"; exit 1; } grep -q -- "rig7-gpu1 .*--identities 3" "$M/argv.log" && echo " gpu1 took IDENTITIES_GPU1=3 over the fallback ok" || { echo "FAIL: --identities 3 missing on gpu1"; cat "$M/argv.log"; exit 1; } +[[ "$(cat "$M/data/override-params.json")" == "$OV" ]] && echo " data/override-params.json written from OVERRIDE ok" || { echo "FAIL: override file"; exit 1; } +grep -q -- "--override-params-file=$M/data/override-params.json" "$T/log/igneum.node.log" && echo " the node got --override-params-file ok" || { echo "FAIL: node flag"; cat "$T/log/igneum.node.log"; exit 1; } +grep -q "node: Consensus params digest: 0139ab9d" "$T/log/igneum.log" && grep -q "node: Finality rule v3 from the override file" "$T/log/igneum.log" && echo " the node's digest and switch lines reached the main log ok" || { echo "FAIL: digest relay"; cat "$T/log/igneum.log"; exit 1; } grep -q "igneum-worker-cuda" "$M/argv.log" && echo " NVIDIA cards got the cuda worker ok" || { echo "FAIL: cuda worker missing"; exit 1; } [[ -f "$M/exited42" && $(grep -c "^mine" "$M/argv.log") -ge 3 ]] && echo " exit 42 restarted the miner and re-exported the pack ok" || { echo "FAIL: no restart after exit 42"; exit 1; } echo "== h-stats.sh (sourced)" ( . "$M/h-stats.sh"; echo " khs=$khs"; echo " stats=$stats"; python3 -c "import json,sys; s=json.loads(sys.argv[1]); assert s['hs_units']=='khs' and len(s['hs'])==2 and s['ar'][0]>0 and s['algo']=='igneum' and len(s['temp'])==2, s; print(' stats JSON ok: hs', s['hs'], 'temp', s['temp'], 'ar', s['ar'], 'bus', s['bus_numbers'])" "$stats" ) +echo "== h-run.sh without OVERRIDE (the warning)" +( CUSTOM_USER_CONFIG=$'DEV_FEE=0' IGNEUM_DRM_ROOT="$T/no-drm" CUSTOM_CONFIG_FILENAME="$T/noov.conf" . "$M/h-config.sh" >/dev/null ) +mkdir -p "$T/log2" +( cd "$M" && CUSTOM_CONFIG_FILENAME="$T/noov.conf" CUSTOM_LOG_BASENAME="$T/log2/igneum" ./h-run.sh > "$T/log2/run.out" 2>&1 ) & +disown +sleep 4 +grep -q "WARNING: no OVERRIDE in the Flight Sheet" "$T/log2/igneum.log" && ! grep -q -- "--override-params-file" "$T/log2/igneum.node.log" && echo " no OVERRIDE: warning in the main log, no flag on the node ok" || { echo "FAIL: missing warning"; cat "$T/log2/igneum.log"; exit 1; } +_p2="$(cat "$T/log2/igneum.pid" 2>/dev/null)"; [[ -n "$_p2" ]] && kill -TERM "$_p2" 2>/dev/null; sleep 1 echo "== self-test passed (scripts and stats shape; Hive itself is untested)" From 047932f2e14f80499f7b1dce78244d64afed1322 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Tue, 6 Oct 2026 00:18:32 +0000 Subject: [PATCH 15/63] plan: the prover-floor agent's sweep 3 rows (the patched server proves the v1 shard at 10.3 GB on the 12 GB profile, 12.9 GB at 2^27; the 12 and 16 GB rows move from nothing to proves alone, the pairs pending) Co-Authored-By: Claude Fable 5.1 --- docs/plans/proving-v1.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/plans/proving-v1.md b/docs/plans/proving-v1.md index b2fa92e6..8badea48 100644 --- a/docs/plans/proving-v1.md +++ b/docs/plans/proving-v1.md @@ -132,8 +132,8 @@ The GPU server of SP1 6.8.1 sets the memory, not the shard: a floor of 13.9 GB f |---|---|---|---| | 32 GB (RTX 5090) | the prototype shard, 28.3 GB, 10.8 s; the v1 shard 20.4 GB, 4.3 s | the prototype shard 30.1 GB, 33 s; the v1 shard 22.2 GB, 13.2 s | on, mine and prove, today | | 24 GB (RTX 4090, 3090) | the v1 shard 20.4 GB; the prototype shard does NOT fit (28.3 GB) | the v1 shard 22.2 GB measured on the 5090's allocation (2.3 GB spare on a 24 GB card; approximate for the card itself) | on, mine and prove, with the line "until the devnet's fee switch its shards are the prototype size, which needs 32 GB, so this card proves from the switch on" | -| 16 GB (RTX 5080, 4080) | an empty shard only (13.9 GB) | nothing (15.7 GB for an empty shard, no room for the display) | off, with the line | -| 12 GB (RTX 3060, 4070) | nothing: the floor is 13.9 GB, and the shipped server refuses the card outright | nothing | off; the project lead's "make sure we can prove on 12 GB cards" is OPEN and in work: the prover-floor agent (branch prover-floor, 5 October night) read SP1 v6.8.1's GPU server source (`sp1-gpu/crates/prover_components/src/builder.rs` lines 35 to 39): it reads the card's memory, adds 4 and panics under 24 ("Unsupported GPU memory ... must be at least 24GB"), and builds its core (ELEMENT_THRESHOLD 2^28 + 2^27 elements + 2^21), recursion (2^27), shrink (2^25) and wrap (85 M element) provers at Setup whatever the mode, which is the 13.9 GB floor; no knob reaches them, so the fix is a server rebuilt from source on PC 2 (WSL2, nvcc 12.8, CUDA_ARCHS=120) with those sizes cut, measured on the same fixtures and recipe as the curve above (D2 carries the curve) | +| 16 GB (RTX 5080, 4080) | the shipped server: an empty shard only (13.9 GB); the patched server v3 b37defef at threshold 2^27: the v1 shard 12,915 MiB and 4.3 s, the prototype shard 13,459 MiB and 16.8 s (measured by the prover-floor agent on the 5090's allocation, job `floor-sweep-3`, 00:13 to 00:17Z 6 October; 2^27 + 2^26 gives 16,115 MiB, over the card) | the shipped server: nothing (15.7 GB for an empty shard); the patched server: about 14.7 GB at 2^27 beside the miner (approximate: the measured 1.8 GB the miner adds; the pair is the agent's sweep 4) | off on the shipped server, with the line; on once the patched server ships (the packaging row below) and the pair is measured | +| 12 GB (RTX 3060, 4070) | the shipped server: nothing (the floor is 13.9 GB, and the server refuses the card outright); the patched server v3 b37defef at threshold 2^26 (`SP1_GPU_ELEMENT_THRESHOLD=67108864`, the 12 GB profile): **the v1 shard 10,291 MiB and 5.7 s, an empty shard 9,971 MiB and 3.3 s**, the card's 2,089 MiB idle inside the peak and the server's own working set about 8.2 GB (6,535 MiB after Setup), so a 12 GB card proves alone with about 3 GB over it (measured by the prover-floor agent on the 5090's allocation, `floor-sweep-3`; the on-order RTX 3060 run is pending) | the pair (the miner's 1.7 to 1.8 GB and 3x beside it) is the agent's sweep 4; under 9.0 GB mine-and-prove is not yet shown | off on the shipped server; "proves alone" on the patched one once it ships (the packaging row below), mine-and-prove after the pair. the project lead's "make sure we can prove on 12 GB cards" is answered on the 5090's allocation and OPEN on the card itself: the prover-floor agent (branch prover-floor, 5 October night) read SP1 v6.8.1's GPU server source (`sp1-gpu/crates/prover_components/src/builder.rs` lines 35 to 39): it reads the card's memory, adds 4 and panics under 24 ("Unsupported GPU memory ... must be at least 24GB"), and builds its core (ELEMENT_THRESHOLD 2^28 + 2^27 elements + 2^21), recursion (2^27), shrink (2^25) and wrap (85 M element) provers at Setup whatever the mode, which is the 13.9 GB floor; no knob reaches them, so the fix is a server rebuilt from source on PC 2 (WSL2, nvcc 12.8, CUDA_ARCHS=120) with those sizes cut, measured on the same fixtures and recipe as the curve above (D2 carries the curve) | | under 12 GB | nothing | nothing | off, mine only | | AMD-only and Apple machines | nothing on the GPU: no zkVM proves on an AMD GPU today (`docs/analysis/amd-proving.md`, branch amd-prove); the CPU prover is about 5 minutes a shard at a 30 GB RSS whatever the shard size (PC 1, bench-log "the SP1 CPU prover on PC 1") | | off, "mines and does not prove"; the only non-NVIDIA path with a shipped backend is RISC Zero's Metal prover behind the `ProofSystem` seam (a second guest and pinned id, a verifier for both formats, no shared aggregation): an open item, not 0.3.11 | From 42f36b3fd3ad21d7d91562c6d000f66acef7063b Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Tue, 6 Oct 2026 00:27:35 +0000 Subject: [PATCH 16/63] app: /api/state never answers {} again (paid_wei over u64::MAX broke to_value; the field is a decimal string, the error is logged once) serde_json's to_value refuses a u128 over u64::MAX (18.45 IGN) and state_json turned the error into json!({}). A paid shard is 1.23 IGN on average, so a proving machine's dashboard went blank about 15 paid shards after every app start. Unit test over the boundary; 114 app tests pass. Co-Authored-By: Claude Fable 5.1 --- app/igneum-app/src/engine.rs | 15 ++++++++++++++- app/igneum-app/src/state.rs | 22 ++++++++++++++++++++++ docs/bench-log.md | 15 +++++++++++++++ docs/plans/proving-v1.md | 7 +++++++ 4 files changed, 58 insertions(+), 1 deletion(-) diff --git a/app/igneum-app/src/engine.rs b/app/igneum-app/src/engine.rs index 5278f089..d67acb7e 100644 --- a/app/igneum-app/src/engine.rs +++ b/app/igneum-app/src/engine.rs @@ -92,6 +92,8 @@ pub struct Shared { /// the app log's path: the one the uploader sends (main.rs names it; the engine must not guess the stamp) pub log_path: PathBuf, port: std::sync::atomic::AtomicU16, + /// set once `state_json` has logged a serialisation error (one line, not one per poll) + state_error_logged: std::sync::atomic::AtomicBool, } impl Shared { @@ -135,6 +137,7 @@ impl Shared { engine_log: Mutex::new(engine_log), log_path, port: std::sync::atomic::AtomicU16::new(0), + state_error_logged: std::sync::atomic::AtomicBool::new(false), } } @@ -177,7 +180,17 @@ impl Shared { st.now = crate::platform::unix_now_f(); st.uptime_s = self.started.elapsed().as_secs(); st.events = self.rings.lock().unwrap().events.iter().cloned().collect(); - serde_json::to_value(st).unwrap_or(json!({})) + match serde_json::to_value(st) { + Ok(v) => v, + Err(e) => { + // never a silent "{}": the dashboard and every PC playbook read this reply + let msg = format!("state_json: {e}"); + if !self.state_error_logged.swap(true, std::sync::atomic::Ordering::Relaxed) { + self.log(&format!("[error] {msg}")); + } + json!({ "error": msg, "version": env!("CARGO_PKG_VERSION") }) + } + } } /// A short state for the window host (menu bar, tray). diff --git a/app/igneum-app/src/state.rs b/app/igneum-app/src/state.rs index 5cf32e57..da328308 100644 --- a/app/igneum-app/src/state.rs +++ b/app/igneum-app/src/state.rs @@ -172,6 +172,9 @@ pub struct ProvingState { pub submitted: u32, pub paid: u32, pub failed: u32, + /// wei, serialised as a decimal string: serde_json's `to_value` refuses a u128 over u64::MAX (about 18.45 IGN, + /// 15 paid shards at 1.23 IGN), and that refusal emptied the whole `/api/state` reply to "{}" (6 October 2026) + #[serde(serialize_with = "u128_string")] pub paid_wei: u128, pub current: String, pub started_at: f64, @@ -401,3 +404,22 @@ impl Rings { out } } + +/// A u128 as a decimal JSON string (the dashboard reads it with `Number()`). +pub fn u128_string(v: &u128, s: S) -> Result { + s.serialize_str(&v.to_string()) +} + +#[cfg(test)] +mod paid_wei_tests { + use super::*; + + #[test] + fn a_paid_total_over_u64_max_still_serialises_the_whole_state() { + let mut st = State::default(); + st.proving.paid_wei = u64::MAX as u128 + 1; + let v = serde_json::to_value(&st).expect("the state serialises"); + assert_eq!(v["proving"]["paid_wei"], serde_json::Value::String("18446744073709551616".into())); + assert!(v["mining"].is_object()); + } +} diff --git a/docs/bench-log.md b/docs/bench-log.md index 2976bd59..a0ebfa9e 100644 --- a/docs/bench-log.md +++ b/docs/bench-log.md @@ -1694,3 +1694,18 @@ Reading, with the miner-on pairs above (empty shard 15,670 MiB, full prototype s |---|---| | Unit tests | `cargo test --release -p kaspa-consensus-core -p igneum-exec --lib -- proving config::params::tests::override_params_carry_the_proving_v1 config::params::tests::consensus_digest` on this Mac (target `vendor/igneum-node/target-pv1`, 19:09Z): consensus core 13 passed (the segment record round trip, signature and the three nested sections; the credit split; the params switch and the digest that moves only once the switch is set), exec 8 passed (the segment grid and the split; the record checks: alignment, block, chain length, the veto naming the field, the deadline, the window; the chain rule both ways; the unproven restart; the shard side at 90%; the pool offering the segment section). The six full node suites go to PC 2 as a build job when the fleet is back | | The fast-time 3-node harness (`tools/proving-v1/net.mjs`, 29950+, suffix 956, every node in trust mode, three vmine voters, v0 at DAA 60, v1 at DAA 120, 4 blocks a segment, unproven after 60 DAA, a tenth to the aggregator; fork b177718e built on this Mac) | run 2, 19:13:01Z to 19:16:19Z, under the run lock: PASSED, 21 checks in 197.3 s (`tools/proving-v1/report-2026-10-05.json`). v1 start = chain block 119 on all three nodes; the native statement identical on all three. Known-finished: segment 119..122's fresh-chain record submitted to n1 at t=131.1 s, relayed, verified (trust) and PAID on n0 1.0 s later at chain block 129, 253,611,648,000,000,000 wei = a tenth of the four credits, the same on every node, the payout address holding it. Chain rule: segment 123..126's fresh-chain record refused ("does not chain to segment 119..122 ... proven (record paid at chain block 129)"), the continuing one (chain_len 8) accepted and paid. Known-failed: segment 127..130 left without a record: a fresh-chain record for 131..134 refused while 127..130 was pending ("pending until DAA 191"); at DAA 192 the status read unproven, a late record for 127..130 refused ("unproven: carried after the deadline"), the fresh-chain record for 131..134 accepted and paid with chain_len 4; `segmentsInWindow` proven 3, unproven 1. The shard side: a v1 shard's `shardWei` = 90% of its block's credit. Run 1 (19:10Z) failed in its own tooling (the signer's argument order), fixed. Run 3 on the FINAL fork tree (ece42979 on the 0.3.10 commit 21d4c73c, protocol 15, N = 8 both in the params default and `--segment 8`, the fast-time file's four fields), 20:52:41Z to 20:56:45Z: PASSED, 21 checks in 244.4 s (segments of 8: 119..126 paid in 1.0 s after submission, 127..134 refused fresh and paid continuing with chain_len 16, 135..142 left unproven and skipped, 143..150 restarted the chain) | + +### 6 October 2026, 00:4xZ, the empty `/api/state` reply (proving v1 branch) + +Reported by the aggregation-cost agent: PC 2's `/api/state` answered `{}` (2 bytes) at 22:22Z, 22:41Z and 00:18Z. Not measured on PC 2 (no job); derived from the app source and node 1's RPC, read-only on the Mac: + +| Figure | Value | Source | +|---|---|---| +| Paid shards, devnet, all provers | 663 | `curl -s 127.0.0.1:26790 -d '{"jsonrpc":"2.0","id":1,"method":"igneum_getProvingStatus","params":[]}'` at tip DAA 0x22caf | +| Paid wei, all provers | 0x2c2961a69990745400 = 814.64 IGN | same call | +| Average per paid shard | 1.23 IGN (approximate: the mean over 663) | 814.64 / 663 | +| u64::MAX in IGN | 18.45 | 2^64 - 1 over 1e18 | +| Paid shards per app start before the reply empties | 15 (approximate: at the mean payout) | 18.45 / 1.23 | + +Cause: `ProvingState.paid_wei: u128` and serde_json `to_value` (1.0.151, `value/ser.rs` `serialize_u128`: u64 range or an error); the error became `json!({})`. Fix: the field serialises as a decimal string; `state_json` logs the error once. Test `a_paid_total_over_u64_max_still_serialises_the_whole_state` (`cargo test --offline -q paid_wei`, 1 passed). + diff --git a/docs/plans/proving-v1.md b/docs/plans/proving-v1.md index 8badea48..e19e5959 100644 --- a/docs/plans/proving-v1.md +++ b/docs/plans/proving-v1.md @@ -143,3 +143,10 @@ The aggregation-cost agent's first rows (branch agg-cost, 5 October 2026 night, The re-plans of block 344 at 2.25 M and 4.5 M pgas peak at 28.3 to 28.4 GB alone (the server's buffers step up between 4.7 M and 20 M cycles and are flat to 60 M), so no shard size between the v1 budget and the prototype one changes a tier; with the miner the adopted shard proves 3.1x slower (13.2 s against 4.2 s) and the chained aggregation 9.7 s against 2.5 s: a mining 24 GB card delivers one adopted-size shard plus one aggregation in about 23 s, inside T by 25x. +## The empty `/api/state` reply (6 October 2026) + +The aggregation-cost agent's jobs read the two bytes `{}` from `/api/state` on PC 2 at 22:22Z, 22:41Z and 00:18Z (0.3.10 and 0.3.11); the 21:01Z reply was full. Cause, from the app source and node 1's RPC: `ProvingState.paid_wei` is a `u128`, and serde_json's `to_value` refuses a u128 over u64::MAX (18,446,744,073,709,551,615 wei, 18.45 IGN); `state_json()` turned that refusal into `json!({})` with no log line. A paid shard is 1.23 IGN on average (node 1, `igneum_getProvingStatus`: 814.64 IGN over 663 shards at 00:3xZ), so the fifteenth paid shard after an app start empties the reply. PC 2's prover was blind to the root-owned socket from 20:00:56Z to 22:01Z (paid_wei stayed 0, hence the full reply at 21:01Z), proved from 22:02:13Z, and crossed 18.45 IGN inside its first 15 paid shards, before 22:22Z. Every app restart resets the counter, so the reply comes back for about 15 shards and goes again. + +What it means: the dashboard on a proving machine shows nothing within about 12 minutes of its prover's first payout; every PC playbook that reads a card from `/api/state` fails the same way (the agent's job 5 reads settings.json instead). Mining, proving and payouts are untouched; it is the status page only. + +Fix on the app branch: `paid_wei` serialises as a decimal string (the dashboard already reads it with `Number()`), `state_json` logs `[error] state_json: ...` once instead of answering `{}`, and the reply on any future serialisation error carries `error` and `version` rather than nothing; unit test `a_paid_total_over_u64_max_still_serialises_the_whole_state`. Ships with 0.3.11 if the shipper takes the new code tip, otherwise 0.3.12; until then the workaround is settings.json for the card keys. From e054c914e6cf808fce5ac43ffa25965cd6d39753 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Tue, 6 Oct 2026 00:28:27 +0000 Subject: [PATCH 17/63] plan: the /api/state fix heads 0.3.12 (0.3.11 closed at a2f08e3) Co-Authored-By: Claude Fable 5.1 --- docs/plans/proving-v1.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/plans/proving-v1.md b/docs/plans/proving-v1.md index e19e5959..8045a6bb 100644 --- a/docs/plans/proving-v1.md +++ b/docs/plans/proving-v1.md @@ -149,4 +149,4 @@ The aggregation-cost agent's jobs read the two bytes `{}` from `/api/state` on P What it means: the dashboard on a proving machine shows nothing within about 12 minutes of its prover's first payout; every PC playbook that reads a card from `/api/state` fails the same way (the agent's job 5 reads settings.json instead). Mining, proving and payouts are untouched; it is the status page only. -Fix on the app branch: `paid_wei` serialises as a decimal string (the dashboard already reads it with `Number()`), `state_json` logs `[error] state_json: ...` once instead of answering `{}`, and the reply on any future serialisation error carries `error` and `version` rather than nothing; unit test `a_paid_total_over_u64_max_still_serialises_the_whole_state`. Ships with 0.3.11 if the shipper takes the new code tip, otherwise 0.3.12; until then the workaround is settings.json for the card keys. +Fix on the app branch: `paid_wei` serialises as a decimal string (the dashboard already reads it with `Number()`), `state_json` logs `[error] state_json: ...` once instead of answering `{}`, and the reply on any future serialisation error carries `error` and `version` rather than nothing; unit test `a_paid_total_over_u64_max_still_serialises_the_whole_state`. Not in 0.3.11 (that tree closed at 22c2363, master 630da6b, published); 6714a45 heads 0.3.12, the morning's first cut, app only, before PC 1's relaunch (coordinator, counter-asic-2-rollout.md 8a); until then the workaround is settings.json for the card keys. From 559a5a0648db184c220c19ef7f0d242df3a66a65 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Tue, 6 Oct 2026 00:38:51 +0000 Subject: [PATCH 18/63] plan: the beside-the-miner pair (floor-sweep-4): 16 GB mines and proves at 12.7 GB and 17 s a shard on the patched server; 12 GB proves alone, mine-and-prove 9.9 GB before the display, over the 9.0 GB line Co-Authored-By: Claude Fable 5.1 --- docs/plans/proving-v1.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/plans/proving-v1.md b/docs/plans/proving-v1.md index 8045a6bb..3c858c07 100644 --- a/docs/plans/proving-v1.md +++ b/docs/plans/proving-v1.md @@ -132,8 +132,8 @@ The GPU server of SP1 6.8.1 sets the memory, not the shard: a floor of 13.9 GB f |---|---|---|---| | 32 GB (RTX 5090) | the prototype shard, 28.3 GB, 10.8 s; the v1 shard 20.4 GB, 4.3 s | the prototype shard 30.1 GB, 33 s; the v1 shard 22.2 GB, 13.2 s | on, mine and prove, today | | 24 GB (RTX 4090, 3090) | the v1 shard 20.4 GB; the prototype shard does NOT fit (28.3 GB) | the v1 shard 22.2 GB measured on the 5090's allocation (2.3 GB spare on a 24 GB card; approximate for the card itself) | on, mine and prove, with the line "until the devnet's fee switch its shards are the prototype size, which needs 32 GB, so this card proves from the switch on" | -| 16 GB (RTX 5080, 4080) | the shipped server: an empty shard only (13.9 GB); the patched server v3 b37defef at threshold 2^27: the v1 shard 12,915 MiB and 4.3 s, the prototype shard 13,459 MiB and 16.8 s (measured by the prover-floor agent on the 5090's allocation, job `floor-sweep-3`, 00:13 to 00:17Z 6 October; 2^27 + 2^26 gives 16,115 MiB, over the card) | the shipped server: nothing (15.7 GB for an empty shard); the patched server: about 14.7 GB at 2^27 beside the miner (approximate: the measured 1.8 GB the miner adds; the pair is the agent's sweep 4) | off on the shipped server, with the line; on once the patched server ships (the packaging row below) and the pair is measured | -| 12 GB (RTX 3060, 4070) | the shipped server: nothing (the floor is 13.9 GB, and the server refuses the card outright); the patched server v3 b37defef at threshold 2^26 (`SP1_GPU_ELEMENT_THRESHOLD=67108864`, the 12 GB profile): **the v1 shard 10,291 MiB and 5.7 s, an empty shard 9,971 MiB and 3.3 s**, the card's 2,089 MiB idle inside the peak and the server's own working set about 8.2 GB (6,535 MiB after Setup), so a 12 GB card proves alone with about 3 GB over it (measured by the prover-floor agent on the 5090's allocation, `floor-sweep-3`; the on-order RTX 3060 run is pending) | the pair (the miner's 1.7 to 1.8 GB and 3x beside it) is the agent's sweep 4; under 9.0 GB mine-and-prove is not yet shown | off on the shipped server; "proves alone" on the patched one once it ships (the packaging row below), mine-and-prove after the pair. the project lead's "make sure we can prove on 12 GB cards" is answered on the 5090's allocation and OPEN on the card itself: the prover-floor agent (branch prover-floor, 5 October night) read SP1 v6.8.1's GPU server source (`sp1-gpu/crates/prover_components/src/builder.rs` lines 35 to 39): it reads the card's memory, adds 4 and panics under 24 ("Unsupported GPU memory ... must be at least 24GB"), and builds its core (ELEMENT_THRESHOLD 2^28 + 2^27 elements + 2^21), recursion (2^27), shrink (2^25) and wrap (85 M element) provers at Setup whatever the mode, which is the 13.9 GB floor; no knob reaches them, so the fix is a server rebuilt from source on PC 2 (WSL2, nvcc 12.8, CUDA_ARCHS=120) with those sizes cut, measured on the same fixtures and recipe as the curve above (D2 carries the curve) | +| 16 GB (RTX 5080, 4080) | the shipped server: an empty shard only (13.9 GB); the patched server v3 b37defef at threshold 2^27: the v1 shard 12,915 MiB and 4.3 s, the prototype shard 13,459 MiB and 16.8 s (measured by the prover-floor agent on the 5090's allocation, job `floor-sweep-3`, 00:13 to 00:17Z 6 October; 2^27 + 2^26 gives 16,115 MiB, over the card) | the shipped server: nothing (15.7 GB for an empty shard); the patched server at 2^27 beside the miner (the 5090 mining at 95%, 338 W, same card; `floor-sweep-4`, 00:35 to 00:38Z): the v1 shard 14,786 MiB total with the miner's 3,833 MiB resident inside it, the server's own 10,953 MiB, 17.4 s a shard; on a 16 GB card that is 10.95 GB server + 1.7 GB miner = 12.7 GB plus the display, under the 15.0 GB line | off on the shipped server, with the line; on (mines and proves, 17 s a v1 shard, 4.3x the alone time) once the patched server ships (the packaging row below) | +| 12 GB (RTX 3060, 4070) | the shipped server: nothing (the floor is 13.9 GB, and the server refuses the card outright); the patched server v3 b37defef at threshold 2^26 (`SP1_GPU_ELEMENT_THRESHOLD=67108864`, the 12 GB profile): **the v1 shard 10,291 MiB and 5.7 s, an empty shard 9,971 MiB and 3.3 s**, the card's 2,089 MiB idle inside the peak and the server's own working set about 8.2 GB (6,535 MiB after Setup), so a 12 GB card proves alone with about 3 GB over it (measured by the prover-floor agent on the 5090's allocation, `floor-sweep-3`; the on-order RTX 3060 run is pending) | measured beside the miner (`floor-sweep-4`): at 2^26 the v1 shard 12,066 MiB total with the miner's 3,833 MiB inside, the server's own 8,233 MiB, 24.4 s; the empty shard 11,586 MiB, 13.0 s; 2^25 gains nothing (12,066 MiB, 43.5 s). On a 12 GB card that is 8.2 GB server + 1.7 GB miner = 9.9 GB before the display, over the 9.0 GB line the project lead set, so mine-and-prove on 12 GB is NOT claimed | off on the shipped server; "proves alone" on the patched one once it ships (the packaging row below); mine-and-prove stays off on 12 GB (9.9 GB plus the display, over the 9.0 GB line). The public gate stays "12 GB proves; 16 GB mines and proves", both on the patched server, both pending a run on the card itself. the project lead's "make sure we can prove on 12 GB cards" is answered on the 5090's allocation and OPEN on the card itself: the prover-floor agent (branch prover-floor, 5 October night) read SP1 v6.8.1's GPU server source (`sp1-gpu/crates/prover_components/src/builder.rs` lines 35 to 39): it reads the card's memory, adds 4 and panics under 24 ("Unsupported GPU memory ... must be at least 24GB"), and builds its core (ELEMENT_THRESHOLD 2^28 + 2^27 elements + 2^21), recursion (2^27), shrink (2^25) and wrap (85 M element) provers at Setup whatever the mode, which is the 13.9 GB floor; no knob reaches them, so the fix is a server rebuilt from source on PC 2 (WSL2, nvcc 12.8, CUDA_ARCHS=120) with those sizes cut, measured on the same fixtures and recipe as the curve above (D2 carries the curve) | | under 12 GB | nothing | nothing | off, mine only | | AMD-only and Apple machines | nothing on the GPU: no zkVM proves on an AMD GPU today (`docs/analysis/amd-proving.md`, branch amd-prove); the CPU prover is about 5 minutes a shard at a 30 GB RSS whatever the shard size (PC 1, bench-log "the SP1 CPU prover on PC 1") | | off, "mines and does not prove"; the only non-NVIDIA path with a shipped backend is RISC Zero's Metal prover behind the `ProofSystem` seam (a second guest and pinned id, a verifier for both formats, no shared aggregation): an open item, not 0.3.11 | From d02f3fb1b78399cb2905479b50b4e8559ee5e909 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Tue, 6 Oct 2026 04:17:51 +0000 Subject: [PATCH 19/63] plan: v1 live on devnet, no segment record because one prover cannot cover 8 consecutive blocks (C47) Co-Authored-By: Claude Fable 5.1 --- docs/plans/proving-v1.md | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/docs/plans/proving-v1.md b/docs/plans/proving-v1.md index 3c858c07..f07196dc 100644 --- a/docs/plans/proving-v1.md +++ b/docs/plans/proving-v1.md @@ -143,6 +143,10 @@ The aggregation-cost agent's first rows (branch agg-cost, 5 October 2026 night, The re-plans of block 344 at 2.25 M and 4.5 M pgas peak at 28.3 to 28.4 GB alone (the server's buffers step up between 4.7 M and 20 M cycles and are flat to 60 M), so no shard size between the v1 budget and the prototype one changes a tier; with the miner the adopted shard proves 3.1x slower (13.2 s against 4.2 s) and the chained aggregation 9.7 s against 2.5 s: a mining 24 GB card delivers one adopted-size shard plus one aggregation in about 23 s, inside T by 25x. +## v1 live on devnet (6 October 2026, C47) + +v1 active at 154,800 (crossed at DAA 154,814, 03:51:42Z); first segment record: none, because on a one-prover devnet no segment can be proven. The app's aggregator (`aggregate_once`, 0.3.11) needs a shard proof of every shard of every block of the segment in its node's pool, and PC 2 alone proves 13 shards per 10 minutes of about 600 blocks (2.2% coverage), so a run of 8 consecutive proven blocks never occurs: node 1 at 04:16Z reads segmentsInWindow pending 55, proven 0, unproven 20, paidSegments 0, and PC 2's app log (run win-1ccfe586-20261005-235130, 04:11Z to 04:16Z) reads every 42 s "aggregator: segment N..N+7: waiting for shard proofs N/0 ... N+7/0 in this node's pool", all 8 missing, each segment then past its 600-DAA deadline unproven. No fault in the node, the app or the record path; the fast-time harness passed because its shards ran at 90% coverage. Meanwhile the aggregator share (a tenth of every block's pool credit) stays in the escrow; shard payouts continue; miners and block watchers see nothing. What ends it: coverage at 8 consecutive blocks, about 18 mining or 6 proving-only 5090-class cards at 1 block/s on empty blocks (the fleet table above), or the segment length lowered on a small devnet (`proving_v1_segment_blocks`, a consensus param, so a digest change). No PC 2 job and no 0.3.12 item follow from this; the open item is the fleet, not the code. + ## The empty `/api/state` reply (6 October 2026) The aggregation-cost agent's jobs read the two bytes `{}` from `/api/state` on PC 2 at 22:22Z, 22:41Z and 00:18Z (0.3.10 and 0.3.11); the 21:01Z reply was full. Cause, from the app source and node 1's RPC: `ProvingState.paid_wei` is a `u128`, and serde_json's `to_value` refuses a u128 over u64::MAX (18,446,744,073,709,551,615 wei, 18.45 IGN); `state_json()` turned that refusal into `json!({})` with no log line. A paid shard is 1.23 IGN on average (node 1, `igneum_getProvingStatus`: 814.64 IGN over 663 shards at 00:3xZ), so the fifteenth paid shard after an app start empties the reply. PC 2's prover was blind to the root-owned socket from 20:00:56Z to 22:01Z (paid_wei stayed 0, hence the full reply at 21:01Z), proved from 22:02:13Z, and crossed 18.45 IGN inside its first 15 paid shards, before 22:22Z. Every app restart resets the counter, so the reply comes back for about 15 shards and goes again. From 0e6a4cd333fc821f8f9417249613deec53545495 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Tue, 6 Oct 2026 04:18:14 +0000 Subject: [PATCH 20/63] plan: C47 card counts from the fleet table (47 as shipped, 18 chain mode, 6 proving-only) Co-Authored-By: Claude Fable 5.1 --- docs/plans/proving-v1.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/plans/proving-v1.md b/docs/plans/proving-v1.md index f07196dc..bedff564 100644 --- a/docs/plans/proving-v1.md +++ b/docs/plans/proving-v1.md @@ -145,7 +145,7 @@ The re-plans of block 344 at 2.25 M and 4.5 M pgas peak at 28.3 to 28.4 GB alone ## v1 live on devnet (6 October 2026, C47) -v1 active at 154,800 (crossed at DAA 154,814, 03:51:42Z); first segment record: none, because on a one-prover devnet no segment can be proven. The app's aggregator (`aggregate_once`, 0.3.11) needs a shard proof of every shard of every block of the segment in its node's pool, and PC 2 alone proves 13 shards per 10 minutes of about 600 blocks (2.2% coverage), so a run of 8 consecutive proven blocks never occurs: node 1 at 04:16Z reads segmentsInWindow pending 55, proven 0, unproven 20, paidSegments 0, and PC 2's app log (run win-1ccfe586-20261005-235130, 04:11Z to 04:16Z) reads every 42 s "aggregator: segment N..N+7: waiting for shard proofs N/0 ... N+7/0 in this node's pool", all 8 missing, each segment then past its 600-DAA deadline unproven. No fault in the node, the app or the record path; the fast-time harness passed because its shards ran at 90% coverage. Meanwhile the aggregator share (a tenth of every block's pool credit) stays in the escrow; shard payouts continue; miners and block watchers see nothing. What ends it: coverage at 8 consecutive blocks, about 18 mining or 6 proving-only 5090-class cards at 1 block/s on empty blocks (the fleet table above), or the segment length lowered on a small devnet (`proving_v1_segment_blocks`, a consensus param, so a digest change). No PC 2 job and no 0.3.12 item follow from this; the open item is the fleet, not the code. +v1 active at 154,800 (crossed at DAA 154,814, 03:51:42Z); first segment record: none, because on a one-prover devnet no segment can be proven. The app's aggregator (`aggregate_once`, 0.3.11) needs a shard proof of every shard of every block of the segment in its node's pool, and PC 2 alone proves 13 shards per 10 minutes of about 600 blocks (2.2% coverage), so a run of 8 consecutive proven blocks never occurs: node 1 at 04:16Z reads segmentsInWindow pending 55, proven 0, unproven 20, paidSegments 0, and PC 2's app log (run win-1ccfe586-20261005-235130, 04:11Z to 04:16Z) reads every 42 s "aggregator: segment N..N+7: waiting for shard proofs N/0 ... N+7/0 in this node's pool", all 8 missing, each segment then past its 600-DAA deadline unproven. No fault in the node, the app or the record path; the fast-time harness passed because its shards ran at 90% coverage. Meanwhile the aggregator share (a tenth of every block's pool credit) stays in the escrow; shard payouts continue; miners and block watchers see nothing. What ends it: coverage at 8 consecutive blocks, 47 mining 5090-class cards with the shard loop as shipped in 0.3.11 (13 shards per 10 minutes a card), 18 mining cards through the chain mode, or 6 (approximate) proving-only cards, at 1 block/s on empty blocks (the fleet table above), or the segment length lowered on a small devnet (`proving_v1_segment_blocks`, a consensus param, so a digest change). No PC 2 job and no 0.3.12 item follow from this; the open item is the fleet, not the code. ## The empty `/api/state` reply (6 October 2026) From 3d505766d293f5b866141aa9a47cc41c5302fdb5 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Tue, 6 Oct 2026 07:01:57 +0000 Subject: [PATCH 21/63] C35 named: PC 1's 22:31 UTC quit was the per-user installer launched by the second engine's own updater (0.3.9 under min_supported_version = urgent, beating auto_update = false); a second engine never runs the updater (IGNEUM_APP_NO_OTA=1, implied by --sweep; the playbooks set it; the CI check demands it); bench log and plan carry the named source Source: the scratch engine's own log in collect ember-c35-collect-1 (06:59Z): 22:31:02Z '0.3.10 is available: downloading', 22:31:05Z 'update: starting the installer first ... ota-apply.ps1', and the installed app's 'quit:' at 22:31:06Z. Co-Authored-By: Claude Fable 5.1 --- app/igneum-app/src/engine.rs | 24 +++++++++++++++++++++--- docs/bench-log.md | 2 +- docs/plans/ember-tune.md | 6 ++++-- relay/playbooks/ember-tune-pc1.ps1 | 1 + relay/playbooks/sweep-5090.ps1 | 1 + tools/ci/second-engine-check.sh | 6 +++++- 6 files changed, 33 insertions(+), 7 deletions(-) diff --git a/app/igneum-app/src/engine.rs b/app/igneum-app/src/engine.rs index c6e57572..128aff49 100644 --- a/app/igneum-app/src/engine.rs +++ b/app/igneum-app/src/engine.rs @@ -466,6 +466,8 @@ pub struct Engine { sweep: Option, /// who asked for the quit (the log's `quit:` line names it) quit_source: &'static str, + /// the over-the-air updater never runs: IGNEUM_APP_NO_OTA=1 or --sweep (a second engine beside the installed app) + no_ota: bool, /// the request number the vendor tool last carried out (the run's acknowledgement) tune_acked: Option, /// cards whose confirm check found a better neighbour: the full plan runs next @@ -491,6 +493,7 @@ pub struct Engine { impl Engine { pub fn new(shared: Arc, bins: Bins, rx: Receiver, wrapper: bool) -> Engine { + let no_ota = shared.runtime.sweep_only || std::env::var("IGNEUM_APP_NO_OTA").map(|v| v == "1").unwrap_or(false); let (lines_tx, lines_rx) = channel(); let now = Instant::now(); let stamp = { @@ -563,6 +566,7 @@ impl Engine { last_error_event: now - Duration::from_secs(600), sweep: None, quit_source: "unknown", + no_ota, tune_acked: None, tune_full_due: std::collections::HashSet::new(), sweep_pending: None, @@ -629,6 +633,9 @@ impl Engine { if self.shared.runtime.sweep_only { self.shared.log("--sweep: the efficiency sweep runs on every supported card as soon as it mines; the table goes to stdout and this log; the engine quits after it"); } + if self.no_ota { + self.shared.log("updates: off for this engine (IGNEUM_APP_NO_OTA or --sweep): it never downloads or installs, whatever the manifest says (C35)"); + } if (setup_done || self.shared.runtime.sweep_only) && !self.quitting { self.shared.send(Cmd::Start); } @@ -761,7 +768,13 @@ impl Engine { self.restart_node(&why, Duration::from_secs(2)); } } - Cmd::CheckUpdate => self.ota.check_now(&self.shared), + Cmd::CheckUpdate => { + if self.no_ota { + self.shared.event("info", "updates are off for this engine (a measurement run: IGNEUM_APP_NO_OTA or --sweep)"); + } else { + self.ota.check_now(&self.shared); + } + } Cmd::InstallUpdate => self.ota.install_now(&self.shared), Cmd::AutoUpdate(on) => self.ota.set_auto(&self.shared, on), Cmd::OpenUpdateFile => { @@ -2439,8 +2452,13 @@ impl Engine { daa: st.node.daa, } }; - if let Some(crate::ota::Action::Apply) = self.ota.tick(&self.shared, &ctx) { - self.apply_update(); + // C35 (PC 1, 5 October 2026, 22:31 UTC): a second engine started by a measurement job found itself under the + // manifest's min_supported_version ("urgent" beats auto_update = false), ran the per-user installer, and the + // installer's PrepareToInstall quit the INSTALLED app through its api/quit. A second engine never updates. + if !self.no_ota { + if let Some(crate::ota::Action::Apply) = self.ota.tick(&self.shared, &ctx) { + self.apply_update(); + } } if let Some(p) = self.ota.take_override_change() { self.shared.log(&format!("consensus override changed ({}); the node restarts with it at a safe moment", p.display())); diff --git a/docs/bench-log.md b/docs/bench-log.md index 21c6712a..ff45fe34 100644 --- a/docs/bench-log.md +++ b/docs/bench-log.md @@ -1624,7 +1624,7 @@ Branch `ember-tune` (54ff1bc), docs/plans/ember-tune.md. Every card tuned for MH **Tier consequences** (docs/plans/ember-tune.md section 7): a 9-step full tune costs about 12 minutes once and 3 minutes a week per card, under 1% of the hour, the worker never stops; a rig tunes one card at a time and every card of a known model after the first takes the 3-minute confirm; a pool user gives up the same 1% of shares at most; Apple silicon and AMD on Linux measure only and the row says so. -**The PC 1 run, 22:30 UTC (job ember-tune-pc1-1, engine aeea3228..., PC 1 on 0.3.10):** the job published at 22:29:40Z, the installed app stopped its miners and started the second engine at 22:30:21Z, and at 22:31:06Z the installed app quit (its log: `quit: stopping the miners, then the node`, then `job ember-tune-pc1-1: aborted (the app is quitting)`), 46 s in, before any step. Nothing was set. Corrected the same night (C35): the first reading, that the 0.3.11 update caused the quit, was wrong; no update, restart or relay task reached PC 1 then (its own jobs lines and the relay feed), and the quit's source is not in the log because the app did not name it (fixed at b671c8b: every `quit:` line now names its sender). What is established: the window host's tray quit is excluded (a host that sent the quit terminates the engine 45 s later, and the engine lived on until 22:55Z), leaving stdin EOF (the host process gone) or `POST /api/quit`; the engine's quit then HUNG for 24 minutes in the jobs runner's abort, waiting for EOF on the script's stdout pipe whose write end the second engine and its miners had inherited, and those miners (2 igneum-miner, 2 CUDA workers, 1 OpenCL worker) mined on, orphaned, until the relay lane killed them at about 23:00Z; the second engine also raised one administrator prompt at about 22:30:25Z (`apply_power_limits` at start counted `--sweep` as Power control), 41 s before the quit; PC 2's unexplained quit at 20:01:09Z came 20 s after a cancelled prompt of the same class, so the prompt is the common factor and the morning's test (one prompt raised beside the mining app on PC 2, the stamped quit line read). Fixed on the branch: b671c8b (quit sources, Power control alone decides, no cap at start under `--sweep`), 8ab9068 (no pipe into a second engine, its tree ended, the CI check). What the run did record, the "before" snapshots with the miners stopped: +**The PC 1 run, 22:30 UTC (job ember-tune-pc1-1, engine aeea3228..., PC 1 on 0.3.10):** the job published at 22:29:40Z, the installed app stopped its miners and started the second engine at 22:30:21Z, and at 22:31:06Z the installed app quit (its log: `quit: stopping the miners, then the node`, then `job ember-tune-pc1-1: aborted (the app is quitting)`), 46 s in, before any step. Nothing was set. Corrected the same night (C35), then named the next morning from the second engine's own log (collect ember-c35-collect-1, 06:59Z): the second engine, reporting 0.3.9 (the branch's Cargo version) under the manifest's `min_supported_version`, took the 0.3.10 update as urgent (the "urgent" rule beats the copied `auto_update = false`), downloaded it at 22:31:02Z and started `ota-apply.ps1` with the per-user installer at 22:31:05Z; the installer's PrepareToInstall sent `POST /api/quit` to the installed app, which logged `quit:` at 22:31:06Z. So the source was my own second engine's updater, through the installer, one second before. The first reading (the 0.3.11 rollout) was wrong in the cause and right in the class: an installer. What else is established: the engine's quit then HUNG for 24 minutes in the jobs runner's abort, waiting for EOF on the script's stdout pipe whose write end the second engine and its miners had inherited, and those miners (2 igneum-miner, 2 CUDA workers, 1 OpenCL worker) mined on, orphaned, until the relay lane killed them at about 23:00Z; the second engine also raised one administrator prompt at about 22:30:25Z (`apply_power_limits` at start counted `--sweep` as Power control), 41 s before the quit; PC 2's unexplained quit at 20:01:09Z came 20 s after a cancelled prompt of the same class, so the prompt is the common factor and the morning's test (one prompt raised beside the mining app on PC 2, the stamped quit line read). Fixed on the branch: b671c8b (quit sources, Power control alone decides, no cap at start under `--sweep`), 8ab9068 (no pipe into a second engine, its tree ended, the CI check), and the third close: a second engine never runs the updater (`IGNEUM_APP_NO_OTA=1`, implied by `--sweep`; the playbooks set it; the CI check demands it). What the run did record, the "before" snapshots with the miners stopped: | Card | Read back at 22:30:20Z | Meaning | |---|---|---| diff --git a/docs/plans/ember-tune.md b/docs/plans/ember-tune.md index 61d02d14..6d38b142 100644 --- a/docs/plans/ember-tune.md +++ b/docs/plans/ember-tune.md @@ -101,6 +101,7 @@ race has run). | The elevated helper restores the limit and resets the clocks by itself after 20 idle minutes | `sweep::helper_script_*` | | A playbook that starts a second engine beside the installed app (the PC measurement jobs) gives it NO pipe (its output goes to a file the script tails: a pipe's write end is inherited by the engine's miners, and the installed app's jobs runner then waits forever for EOF after an abort; C35, PC 1 22:31 UTC, a 24-minute hang and orphaned miners), ends the engine's whole process tree at the end and on the budget (`taskkill /T /F`), and lets the installed app's miners come back only after that | `relay/playbooks/ember-tune-pc1.ps1`, `sweep-5090.ps1`; CI `tools/ci/second-engine-check.sh` fails any playbook without both | | Every `quit:` line in the app log names its source (the window host's stdin, the host gone, `POST /api/quit`, the `--sweep` run's end) | `Cmd::Quit(&'static str)` (b671c8b) | +| A second engine never runs the updater: `IGNEUM_APP_NO_OTA=1` (implied by `--sweep`) skips the OTA tick and refuses Check now, whatever the manifest's `min_supported_version` says (the installer it would launch quits the installed app: PC 1, 22:31 UTC) | `Engine.no_ota`; the playbooks set the variable; `tools/ci/second-engine-check.sh` demands it | ## 6. Tests @@ -153,8 +154,9 @@ measure, TUNE record, upload, aggregation, prior shape in a test manifest). The 9070 XT run are owed: the 5090 the moment Power control is switched on (one prompt, then the tune runs by itself within 2 minutes of steady mining), the 9070 XT when the card is back on the bus. -Run 1 (ember-tune-pc1-1, 22:30 UTC): aborted 46 s in by the installed app quitting (source unnamed by the 0.3.10 app; -not an update, not a job, not a relay task: C35 in the bench log), before any step; nothing set; the "before" snapshots are in the bench log (5090: 450 W of 575, 2,850 MHz core, 3,090 MHz +Run 1 (ember-tune-pc1-1, 22:30 UTC): aborted 46 s in by the installed app quitting, named the next morning: the +second engine's own updater (0.3.9 under min_supported_version = urgent) ran the per-user installer, whose +PrepareToInstall quit the installed app through its api/quit (C35 in the bench log); before any step; nothing set; the "before" snapshots are in the bench log (5090: 450 W of 575, 2,850 MHz core, 3,090 MHz maximum, 14,001 MHz memory; 9070 XT present on bus 98 with OFFSET ranges `gmax_range -500 1000`, `plimit_range -30 10`). The offset finding changed the AMD mapping (054e041): an offset clock range closes the clock knob and the power ladder runs on a percent scale bounded by `plimit_range`. The re-run follows the 0.3.11 rollout. diff --git a/relay/playbooks/ember-tune-pc1.ps1 b/relay/playbooks/ember-tune-pc1.ps1 index 0ce8566c..e9ed9164 100644 --- a/relay/playbooks/ember-tune-pc1.ps1 +++ b/relay/playbooks/ember-tune-pc1.ps1 @@ -95,6 +95,7 @@ Snapshot 'before' $env:IGNEUM_APP_DATA = $root $env:IGNEUM_APP_LOGS = $sLogs $env:IGNEUM_APP_STATUS_SECS = '10' +$env:IGNEUM_APP_NO_OTA = '1' # C35: a second engine never runs the updater (the installer would quit the installed app) # C35 (5 October 2026): the engine's output goes to a FILE, never a pipe. A pipe's write end is inherited by every # process the engine starts (its miners and workers), so after an abort the installed app's jobs runner waits for an # EOF that never comes and hangs in its own quit; and the engine's whole tree is killed at the end (nothing orphaned). diff --git a/relay/playbooks/sweep-5090.ps1 b/relay/playbooks/sweep-5090.ps1 index bb0a3a77..8b68f7d0 100644 --- a/relay/playbooks/sweep-5090.ps1 +++ b/relay/playbooks/sweep-5090.ps1 @@ -60,6 +60,7 @@ if (Test-Path $smi) { $env:IGNEUM_APP_DATA = $root $env:IGNEUM_APP_LOGS = $sLogs $env:IGNEUM_APP_STATUS_SECS = '10' +$env:IGNEUM_APP_NO_OTA = '1' # C35: a second engine never runs the updater (the installer would quit the installed app) # C35 (5 October 2026): the engine's output goes to a FILE, never a pipe. A pipe's write end is inherited by every # process the engine starts (its miners and workers), so after an abort the installed app's jobs runner waits for an # EOF that never comes and hangs in its own quit; and the engine's whole tree is killed at the end (nothing orphaned). diff --git a/tools/ci/second-engine-check.sh b/tools/ci/second-engine-check.sh index 564d390f..e5a42e31 100755 --- a/tools/ci/second-engine-check.sh +++ b/tools/ci/second-engine-check.sh @@ -7,7 +7,8 @@ # output goes to a FILE (Start-Process -RedirectStandardOutput ), never a pipe into the script; (2) the engine's # whole process tree is ended at the end and on the budget (taskkill /T /F), so nothing is orphaned; the installed # app's miners come back only after that (the job runner restarts them when the script ends). This check fails CI when -# a playbook starts an engine without both. +# a playbook starts an engine without both, or without IGNEUM_APP_NO_OTA = '1' (the third line, same night: the second +# engine's updater found itself under min_supported_version and ran the installer, which quit the installed app). set -euo pipefail cd "$(dirname "$0")/../.." fail=0 @@ -19,6 +20,9 @@ while IFS= read -r f; do if ! grep -qE 'taskkill /T /F' "$f"; then echo "second-engine: $f starts an engine without ending its process tree (taskkill /T /F) at the end"; fail=1 fi + if ! grep -qE "IGNEUM_APP_NO_OTA *= *'1'" "$f"; then + echo "second-engine: $f starts an engine without IGNEUM_APP_NO_OTA = '1' (its updater would run the installer, which quits the installed app: PC 1, 5 October 2026, 22:31 UTC)"; fail=1 + fi done < <(git ls-files 'relay/playbooks/**' 'tools/windows/**' 'packaging/**' | grep -E '\.ps1$') [ "$fail" = 0 ] && echo "second-engine: every playbook that starts an engine logs to a file and ends its tree" exit $fail From f5dfdbf443b5b86abd45f7c693be15715327a70a Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Tue, 6 Oct 2026 07:03:26 +0000 Subject: [PATCH 22/63] bench log: the 22:31 UTC installer run was a second install of 0.3.10 over 0.3.10 (PC 1 took 0.3.10 at 21:40:41Z through update-now), not how PC 1 got 0.3.10 Co-Authored-By: Claude Fable 5.1 --- docs/bench-log.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/bench-log.md b/docs/bench-log.md index ff45fe34..7853fa54 100644 --- a/docs/bench-log.md +++ b/docs/bench-log.md @@ -1624,7 +1624,7 @@ Branch `ember-tune` (54ff1bc), docs/plans/ember-tune.md. Every card tuned for MH **Tier consequences** (docs/plans/ember-tune.md section 7): a 9-step full tune costs about 12 minutes once and 3 minutes a week per card, under 1% of the hour, the worker never stops; a rig tunes one card at a time and every card of a known model after the first takes the 3-minute confirm; a pool user gives up the same 1% of shares at most; Apple silicon and AMD on Linux measure only and the row says so. -**The PC 1 run, 22:30 UTC (job ember-tune-pc1-1, engine aeea3228..., PC 1 on 0.3.10):** the job published at 22:29:40Z, the installed app stopped its miners and started the second engine at 22:30:21Z, and at 22:31:06Z the installed app quit (its log: `quit: stopping the miners, then the node`, then `job ember-tune-pc1-1: aborted (the app is quitting)`), 46 s in, before any step. Nothing was set. Corrected the same night (C35), then named the next morning from the second engine's own log (collect ember-c35-collect-1, 06:59Z): the second engine, reporting 0.3.9 (the branch's Cargo version) under the manifest's `min_supported_version`, took the 0.3.10 update as urgent (the "urgent" rule beats the copied `auto_update = false`), downloaded it at 22:31:02Z and started `ota-apply.ps1` with the per-user installer at 22:31:05Z; the installer's PrepareToInstall sent `POST /api/quit` to the installed app, which logged `quit:` at 22:31:06Z. So the source was my own second engine's updater, through the installer, one second before. The first reading (the 0.3.11 rollout) was wrong in the cause and right in the class: an installer. What else is established: the engine's quit then HUNG for 24 minutes in the jobs runner's abort, waiting for EOF on the script's stdout pipe whose write end the second engine and its miners had inherited, and those miners (2 igneum-miner, 2 CUDA workers, 1 OpenCL worker) mined on, orphaned, until the relay lane killed them at about 23:00Z; the second engine also raised one administrator prompt at about 22:30:25Z (`apply_power_limits` at start counted `--sweep` as Power control), 41 s before the quit; PC 2's unexplained quit at 20:01:09Z came 20 s after a cancelled prompt of the same class, so the prompt is the common factor and the morning's test (one prompt raised beside the mining app on PC 2, the stamped quit line read). Fixed on the branch: b671c8b (quit sources, Power control alone decides, no cap at start under `--sweep`), 8ab9068 (no pipe into a second engine, its tree ended, the CI check), and the third close: a second engine never runs the updater (`IGNEUM_APP_NO_OTA=1`, implied by `--sweep`; the playbooks set it; the CI check demands it). What the run did record, the "before" snapshots with the miners stopped: +**The PC 1 run, 22:30 UTC (job ember-tune-pc1-1, engine aeea3228..., PC 1 on 0.3.10):** the job published at 22:29:40Z, the installed app stopped its miners and started the second engine at 22:30:21Z, and at 22:31:06Z the installed app quit (its log: `quit: stopping the miners, then the node`, then `job ember-tune-pc1-1: aborted (the app is quitting)`), 46 s in, before any step. Nothing was set. Corrected the same night (C35), then named the next morning from the second engine's own log (collect ember-c35-collect-1, 06:59Z): the second engine, reporting 0.3.9 (the branch's Cargo version) under the manifest's `min_supported_version`, took the 0.3.10 update as urgent (the "urgent" rule beats the copied `auto_update = false`), downloaded it at 22:31:02Z and started `ota-apply.ps1` with the per-user installer at 22:31:05Z; the installer's PrepareToInstall sent `POST /api/quit` to the installed app, which logged `quit:` at 22:31:06Z. So the source was my own second engine's updater, through the installer, one second before: a second install of 0.3.10 over the 0.3.10 PC 1 had taken through the shipper's update-now at 21:40:41Z (release-0.3.10.md section 8), whose only effect was the quit and the hang. The first reading (the 0.3.11 rollout) was wrong in the cause and right in the class: an installer. What else is established: the engine's quit then HUNG for 24 minutes in the jobs runner's abort, waiting for EOF on the script's stdout pipe whose write end the second engine and its miners had inherited, and those miners (2 igneum-miner, 2 CUDA workers, 1 OpenCL worker) mined on, orphaned, until the relay lane killed them at about 23:00Z; the second engine also raised one administrator prompt at about 22:30:25Z (`apply_power_limits` at start counted `--sweep` as Power control), 41 s before the quit; PC 2's unexplained quit at 20:01:09Z came 20 s after a cancelled prompt of the same class, so the prompt is the common factor and the morning's test (one prompt raised beside the mining app on PC 2, the stamped quit line read). Fixed on the branch: b671c8b (quit sources, Power control alone decides, no cap at start under `--sweep`), 8ab9068 (no pipe into a second engine, its tree ended, the CI check), and the third close: a second engine never runs the updater (`IGNEUM_APP_NO_OTA=1`, implied by `--sweep`; the playbooks set it; the CI check demands it). What the run did record, the "before" snapshots with the miners stopped: | Card | Read back at 22:30:20Z | Meaning | |---|---|---| From 7dade79bcbc3772243ed079d5f3ddf44043b96df Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Tue, 6 Oct 2026 01:30:25 +0000 Subject: [PATCH 23/63] Aggregation cost on the 5090 (5 October, night): the chained aggregation is 2.1 s alone and 9.7 s beside the miner, the batch-log2 curve (2^16 buys 1.6x for a fifth of the hash rate), batch and tree folds estimated, two streams and SP1 knobs closed; the host times the stdin build and names the knobs, --save-shards; the PC 2 job scripts and the readers The statement and the pinned guests are unchanged; every fixture proof verifies as before. The defaults stay (batch-log2 22, SP1 defaults): the one knob that moves a mining card's prover costs a fifth of the hash rate; the plan carries the trade for the project lead and the batch fold for the next pin. Measured: docs/bench-log.md "aggregation cost on the RTX 5090"; the plan line: docs/plans/proving-v1.md "Aggregation cost (5 October, night)". Also: make-package's gate skips the exporter's .node-plan.json side files and takes the run lock for its execute step; the state-reply class (/api/state answering {} once paid_wei passes u64::MAX) found on the way and fixed on the app branch at 42f36b3. Co-Authored-By: Claude Fable 5.1 (cherry picked from commit ea38ece9eaa5949dd657cbfea5308c4948177ff8) --- docs/bench-log.md | 87 ++++ docs/plans/proving-v1.md | 19 + proving/igneum-prove/host/src/main.rs | 30 +- proving/igneum-prove/host/src/proof_system.rs | 14 + proving/windows-wsl2/make-package.sh | 6 +- tools/proving-v1/agg-cost-table.mjs | 50 ++ tools/proving-v1/miner-rate.mjs | 35 ++ tools/proving-v1/pc2-agg-cost-restore.ps1 | 57 +++ tools/proving-v1/pc2-agg-cost.ps1 | 433 ++++++++++++++++++ 9 files changed, 723 insertions(+), 8 deletions(-) create mode 100644 tools/proving-v1/agg-cost-table.mjs create mode 100644 tools/proving-v1/miner-rate.mjs create mode 100644 tools/proving-v1/pc2-agg-cost-restore.ps1 create mode 100644 tools/proving-v1/pc2-agg-cost.ps1 diff --git a/docs/bench-log.md b/docs/bench-log.md index a0ebfa9e..81974853 100644 --- a/docs/bench-log.md +++ b/docs/bench-log.md @@ -1709,3 +1709,90 @@ Reported by the aggregation-cost agent: PC 2's `/api/state` answered `{}` (2 byt Cause: `ProvingState.paid_wei: u128` and serde_json `to_value` (1.0.151, `value/ser.rs` `serialize_u128`: u64 range or an error); the error became `json!({})`. Fix: the field serialises as a decimal string; `state_json` logs the error once. Test `a_paid_total_over_u64_max_still_serialises_the_whole_state` (`cargo test --offline -q paid_wei`, 1 passed). +| The fast-time 3-node harness (`tools/proving-v1/net.mjs`, 29950+, suffix 956, every node in trust mode, three vmine voters, v0 at DAA 60, v1 at DAA 120, 4 blocks a segment, unproven after 60 DAA, a tenth to the aggregator; fork b177718e built on this Mac) | run 2, 19:13:01Z to 19:16:19Z, under the run lock: PASSED, 21 checks in 197.3 s (`tools/proving-v1/report-2026-10-05.json`). v1 start = chain block 119 on all three nodes; the native statement identical on all three. Known-finished: segment 119..122's fresh-chain record submitted to n1 at t=131.1 s, relayed, verified (trust) and PAID on n0 1.0 s later at chain block 129, 253,611,648,000,000,000 wei = a tenth of the four credits, the same on every node, the payout address holding it. Chain rule: segment 123..126's fresh-chain record refused ("does not chain to segment 119..122 ... proven (record paid at chain block 129)"), the continuing one (chain_len 8) accepted and paid. Known-failed: segment 127..130 left without a record: a fresh-chain record for 131..134 refused while 127..130 was pending ("pending until DAA 191"); at DAA 192 the status read unproven, a late record for 127..130 refused ("unproven: carried after the deadline"), the fresh-chain record for 131..134 accepted and paid with chain_len 4; `segmentsInWindow` proven 3, unproven 1. The shard side: a v1 shard's `shardWei` = 90% of its block's credit. Run 1 (19:10Z) failed in its own tooling (the signer's argument order), fixed | + +## 5 October 2026 (night), aggregation cost on the RTX 5090: what a per-block aggregation spends and what each lever gives (proving engineer, agg-cost) + +the project lead, 5 October 2026: "fix everything else in the numbers tonight". The number under test: the chained segment aggregation cost 9.6 to 9.7 s a block on PC 2's 5090 while the card mined (`chain-pc2-pv1c`, the entry above), 2.2 s on 4 October with the card to itself. Target: under 3 s a block, the miner's slowdown of the prover under 1.5x, the proof statement unchanged. Branch `agg-cost` (worktree `igneum-wt-agg-cost`, from `proving-v1` 219517f). Host changes (statement untouched, `elf/` untouched): the aggregation's stdin build timed apart from the prove call, the deferred-proof count and the SP1 knobs in the RESULT lines, `--mode chain --save-shards` (every shard's compressed proof written next to the results, so `--mode aggregate` re-runs the same proofs under other settings). Jobs: `agg-cost-pc2-1` (21:01:20Z to 21:25:11Z, `tools/proving-v1/pc2-agg-cost.ps1`, the package `igneum-prove-wsl2-aggcost.zip` fetched by `fetch-prove-aggcost` 20:55:39Z, built in WSL2 against the live target dir in 5 s, installed to `/opt/igneum-aggcost`, the live `/opt/igneum` untouched, `--mode id` the pinned pair) and `agg-cost-pc2-2` (21:34:00Z, the same script). The live prover was switched OFF for the runs (its sp1-gpu-server would otherwise be shared through `/tmp/sp1-cuda-0.sock` and carry its own environment; `gpu_server_before running=0`) and ON again at the end. Fixtures: four consecutive live blocks cut from PC 2's own node (86165..86168 at tip 86195, one empty shard each, every one MATCHES natively), the same four for every phase of job 1. App 0.3.9 on PC 2 throughout. + +Known-finished case of the host changes before the GPU (this Mac, CPU, run lock, 20:41Z to 20:44Z): `--mode chain` over `fixtures/chain/block-81046.json` with `--save-shards` (shard 38.5 s, aggregate 43.4 s, the proof file written), then `--mode aggregate` over that saved shard proof with `SP1_WORKER_VERIFY_INTERMEDIATES=false` (46.6 s, the same statement `0x3dedb8ea...`), `--mode verify-segment` VERIFIED in 0.027 s; known-failed: a wrong statement NOT VERIFIED in 0.027 s. Unit tests: `cargo test --release -p igneum-prove-core -p igneum-prove-host`: core 8 passed, host 9 passed and 1 ignored (build lock, 20:53Z). + +### Lever 1, the profile: where a per-block aggregation goes + +| What | Measured (job `agg-cost-pc2-1`) | +|---|---| +| The host's own share of an aggregation (the stdin build: the AggInput, the proof clones into the request) | 0.000 s on every block, mining or idle (the `stdin` field of every `RESULT chain block` line): everything is inside the one `prove().compressed()` call to the GPU server | +| The GPU server's log at `RUST_LOG=info` (phase A0, the same chain of 1, stderr captured) | 1 line: sp1-gpu-server 6.8.1 prints no spans and no timings, so the step costs below are read from the deferred-proof count, not from a profiler | +| Aggregation with 1 deferred proof (the first block, no previous proof) against 2 (every chained block), the card mining | 7.9 s against 9.6, 9.6, 9.8 s: the second deferred proof costs 1.7 to 1.9 s under the miner | +| The same, the miners paused (phase C, the same fixtures, 21:05:51Z) | 1.7 s against 2.1, 2.1, 2.2 s: the second deferred proof costs 0.4 to 0.5 s alone | +| The shard proof of an empty shard | 7.4 to 7.8 s mining, 1.9 to 2.2 s alone | +| A whole block (one empty shard plus its aggregation) | 17.1 to 17.3 s mining (end to end 67.4 s for 4 blocks), 4.1 s alone (16.4 s for 4) | +| GPU utilisation over the phase (1-s `nvidia-smi` samples) | 93.9% mining (80 samples, the miner's), 15.8% alone (32 samples): the prover alone keeps the card busy a sixth of the time. Its work is short GPU bursts between CPU phases (the executor, the witness and recursion-program generation run on the CPU inside the server), and the miner's kernels fill the gaps | +| GPU memory peak | 16,195 MiB mining (the miner's 3.4 GB resident), 14,483 MiB alone | +| The slowdown by the miner, same fixtures, same host, 2 min apart | shards 3.6x, the first aggregation 4.6x, a chained aggregation 4.5x, a block 4.2x | +| Setup per host process (client plus two key setups) | 13.0 to 15.7 s, mining or not | + +Reading. An aggregation is three or four recursion steps on the card (the aggregator guest's one core shard, its lift, one deferred program per verified proof, the compose), each a burst of under half a second when the card is free. The chained aggregation's extra deferred proof is the only part that grows with the chain rule, 0.4 to 0.5 s alone. Everything else the 9.7 s holds is the miner: with the card at 94% from the lottery kernels, every prover burst waits for a time slice, and a 2.1-s aggregation becomes 9.7 s. The 4 October 2.2 s (two shards, no previous proof, the card to itself) and tonight's 1.7 s (one shard) and 2.1 s (one shard plus the previous proof) agree within the deferred count. + +### Lever 2, batch and tree folds (estimate from the measured step costs; the statement is pinned, no guest was changed tonight) + +A fold of K blocks' shard proofs plus the previous segment proof in ONE aggregator call would cost one core shard, one lift, K + 1 deferred programs and the compose tree in place of K chained aggregations. From the measured rows (alone: a 1-deferred aggregation 1.7 s, each further deferred proof 0.45 s; mining: 7.9 s and 1.8 s): + +| Fold | Deferred proofs per call | Per block, card alone (estimate) | Per block, card mining (estimate) | Rule | +|---|---|---|---|---| +| chained, as pinned (measured) | 2 | 2.1 s | 9.7 s | one call per block | +| batch of 4 | 5 | (1.7 + 4 x 0.45) / 4 = 0.9 s | (7.9 + 4 x 1.8) / 4 = 3.8 s | one call per 4 blocks | +| batch of 8 | 9 | (1.7 + 8 x 0.45) / 8 = 0.7 s | (7.9 + 8 x 1.8) / 8 = 2.8 s | one call per 8 blocks | +| tree of 4 (2 + 2, then the pair) | 3 per call, 3 calls | 3 x (1.7 + 2 x 0.45) / 4 = 1.9 s | 3 x (7.9 + 2 x 1.8) / 4 = 8.6 s | no gain over the chain: every call pays the fixed part | + +Reading. A batch fold halves to quarters the per-block aggregation but changes the aggregator's statement (`AggInput` carries one block's shards and the guest asserts one block hash), so it is a new pinned guest and a new program id: a provers-off drain and a rollout (proving/README.md, pinned guests). It does not reach 3 s on a mining card by itself (2.8 s at K = 8 is on the line), and the shard proof beside it stays 7.4 s a block on a mining card. The lever that moves both is the card's other job, lever 4. A tree fold gains nothing here because the fixed part of a call (the core shard and the lift) dominates the per-proof part 4 to 1. + +### Levers 3 and 4, two streams and the miner's kernels (job `agg-cost-pc2-2` and the re-run) + +Job `agg-cost-pc2-2` (21:34:00Z to 21:49:22Z) ran with the 5090 idle throughout: job 1's `/api/resume` had left the worker off (below), so the rows that needed the miner (the batch-log2 curve, the two streams beside the miner, the time-slice policy, the chosen combination) are void and wait for a re-run; the idle rows are measured. + +| What | Measured (job `agg-cost-pc2-2`, card idle) | +|---|---| +| Aggregate-only over job 1's four saved shard proofs (`--mode aggregate --proofs b1;b2;b3;b4 --parent ...`, one process, the same statement `0x3a995f24...` as the chain run), default knobs (phase B0, then C1) | 1.7, 2.0, 2.0, 2.0 s (1, 2, 2, 2 deferred proofs), 8.1 s for four; C1: 1.8, 2.1, 2.1, 2.1 s, 8.3 s | +| The same with `SP1_WORKER_VERIFY_INTERMEDIATES=false` (phase B; the server inherits the host's environment, the knob printed in the `sp1 knobs` line) | 1.7, 2.0, 2.0, 2.0 s, 7.8 s for four: no gain (0.3 s over four, inside the run-to-run spread of 0.2 s). The knobs that change the recursion shape (`SP1_WORKER_MAX_COMPOSE_ARITY`, `MAX_REDUCE_ARITY`) were not tried: a different shape is a different recursion key set and the pinned verifier would refuse the proof | +| A 4-deferred aggregation (block-344-shards4, four prototype shards of 6.75 M pgas, phase C2) | shards 42.8 s (10.7 s each, the 4 October 10.2 to 10.7 s), aggregation 2.4 s with 4 deferred proofs; GPU peak 28,402 MiB (the prototype shard's 28.3 GB), utilisation 27.7% over the phase. With 1.7 s at one deferred proof and 2.0 to 2.1 s at two: 0.25 s per further deferred proof alone, so a batch of 8 would cost about 3.5 s a call, 0.45 s a block (estimate, the pinned statement forbids it) | +| Two host processes at once on the one card (phase G0: chains of 2 on disjoint blocks, started 2 s apart) | both connected to ONE sp1-gpu-server (the first process's child; the socket is per device, `/tmp/sp1-cuda-0.sock`): process 1 shard 2.2 and 3.5 s, aggregation 3.0 and 4.0 s (12.9 s for 2 blocks against 8.2 s alone); process 2 shard 3.3 s, aggregation 3.6 s, then its second block died with `CudaClientError: Failed to read the response: early eof` when process 1 finished and its server exited. GPU 24,911 MiB, utilisation 12.4% and 13.1%. Two streams through SP1 6.8.1's server are serialised on one socket and the second dies with the first: no throughput gain (3 blocks in 33 s against 4 in 16.4 s) and a failure mode; lever 3 is closed on this SP1 version | +| Job 3 (`agg-cost-pc2-3`, 22:41:15Z, app 0.3.10, the same script with the socket rule and a card switch): phase A, the app's 5090 miner at 117.0 MH/s mean (n 3, STATUS lines 22:44:45Z to 22:46:11Z), four fresh live blocks 90896..90899 | shards 8.0, 7.8, 7.6, 7.8 s; aggregations 8.0 s (1 deferred), 10.0, 10.0, 10.0 s (2 deferred); 69.5 s for four, 17.8 s a block; GPU 93.8%, peak 16,245 MiB: the job-1 baseline reproduced 100 min later on other blocks | +| Job 3's own-miner phases | void: the state reads came back empty (the class below), the card switch did nothing, phase D launched my miner beside the app's (the app's dropped to 62.2 MH/s, mine read 60.6 MH/s), then PC 2's app restarted at 23:03:30Z and the job died with it; no curve point | +| The GPU time-slice policy (`nvidia-smi compute-policy --set-timeslice`, the restore job `agg-cost-restore-1`, 23:16:53Z) | "Not Supported" on PC 2 (RTX 5090, driver 13.3, the Windows nvidia-smi, not elevated): the lever is closed on this driver; an elevated try is not worth a slot, the error is the driver's, not a permission's | +| The own-miner phases of job 2 | void: no 5090 miner was running to copy the command line from (the worker off since 21:25Z) | + +The curve, job `agg-cost-pc2-6` (01:12:09Z to 01:24:14Z, app 0.3.11, PC 2 to itself; every phase closed before the next job landed on PC 2 at 01:24:21Z). The app's 5090 miner switched off through `/api/cards` (the keys from `settings.json`; the worker was still alive after 120 s, `/api/pause` as the fallback stopped it in 5 s), then the job's OWN miner on the 5090 with the app's command line (`igneum-miner mine ... --worker igneum-worker-cuda.exe --identities 8 --worker-args "--device 0 --pack packs\devnet --race off [--batch-log2 B]"`, the base variant, its STATUS line every 10 s), the same four live blocks 96556..96559 (one empty shard each) proven by `--mode chain` under it, the miner's rate from its own `now=` field (the first two lines skipped). `--batch-log2 B` sets the worker's nonces per kernel launch (2^B; 22 is the worker's default, 4,194,304 nonces, about 35 ms a launch at 120 MH/s; `proto-cuda/nvrtc/worker.cpp`). + +| batch-log2 | Shard proof (4, s) | Aggregation (1 deferred, then 2) (s) | A block (s) | GPU util. (%) | GPU peak (MiB) | Own miner (MH/s wall, n) | Against the card alone (4.1 s a block) | +|---|---|---|---|---|---|---|---| +| 22 (the default), phase D | 8.1, 7.8, 7.9, 7.8 | 8.4; 10.3, 10.0, 10.4 | 18.1 | 95.5 | 16,580 | 103.9 (9) | 4.4x | +| 20, E20 | 8.1, 7.8, 7.8, 7.8 | 8.4; 10.3, 10.1, 10.1 | 18.0 | 94.9 | 16,461 | 103.7 (8) | 4.4x | +| 18, E18 | 7.0, 6.7, 6.7, 6.7 | 7.2; 8.9, 8.8, 8.8 | 15.6 | 91.5 | 16,487 | 99.3 (7), minus 4.4% | 3.8x | +| 16, E16 | 5.1, 4.9, 4.9, 4.9 | 5.0; 6.1, 6.2, 6.2 | 11.1 | 85.3 | 16,519 | 83.8 (6), minus 19% | 2.7x | +| 16 again, phase H (the job's own choice: the shortest chain) | 5.0, 4.9, 4.8, 4.9 | 4.9; 6.1, 6.2, 6.2 | 11.1 | 85.7 | 16,487 | 84.0 (6) | 2.7x | + +Reading. Between 2^22 and 2^20 nothing moves: the card's time-slice scheduler alternates the two contexts whatever the kernel length above a few milliseconds. From 2^18 down the miner's launches get short enough (about 2 ms at 2^18, 0.5 ms at 2^16) that the prover's bursts find the card sooner, and the miner pays in launch overhead and idle gaps: at 2^16 the prover runs 1.6x faster (18.1 to 11.1 s a block, the chained aggregation 10.2 to 6.2 s) for a fifth of the hash rate, and it is still 2.7x slower than on a card to itself. The trade is about 1 MH/s per 0.37 s of block time at the 2^16 point, and the 3-s aggregation and the 1.5x slowdown are not reachable on a mining card by the kernel length; a 2^14 point (approximate, extrapolated) would be about 8 s a block at about 65 MH/s. The phase E0 (a 4-deferred aggregation under the miner) failed in 0.1 s: its proof paths pointed at `/` where job 1 had left its shard proofs, but job 2's block-344 proofs sit in job 2's own folder (`$JOB` was exported from job 2 on); the 4-deferred cost under the miner stays an estimate (lever 2 above). The app's own 5090 miner ran at 117 MH/s (job 3, 22:44Z) and 110 to 129 MH/s (its STATUS lines at 01:10Z) with the prover beside it, against my miner's 104 MH/s at the default batch: my miner runs the base variant with `--race off` (no tuning file on PC 2), so the curve's rates are relative to each other, not to the app's. + +### Lever 5, the host side under WSL2 (what the chain-mode numbers leave out) + +| What | Measured | +|---|---| +| The export (`igneum_exportSegments` 0..tip, 75 to 77 MB over curl.exe to a file on `C:`) | 1.1 to 1.5 s | +| The cut (`igneum-prove-export` replaying from genesis, then `--mode native`), four blocks | 18 s for four including the native checks (21:01:28Z to 21:01:46Z), about 4 s a block; the export's file sits on `/mnt/c` | +| The key setup per host process | 13.0 to 15.7 s on PC 2 (8.0 to 8.5 s on the Mac CPU): `--mode chain` and `--mode aggregate` pay it once per process, the app's loop pays it per shard | +| The proof file write through the WSL2 bridge | the 4 October entry ("shard proving on the RTX 5090"): 24 min of unbuffered `save` across `/mnt/c`, fixed by the 4 MB buffer; tonight `--save-shards` wrote the four 1.27 MB proofs inside the chain phase with no visible gap (the A phase's 80.4 s wall against 67.4 s of proving plus 13.0 s of setup) | +| Native Linux | not measured: no native Linux machine with an NVIDIA card exists in the project tonight, and the 4 October numbers were also WSL2 (Ubuntu 24.04 under PC 2's Windows). The WSL2 cost inside a `prove()` call is not separable from here; the host-side pieces above are what a native box would also skip or keep | + +### What went wrong, measured + +| What | Fixed | +|---|---| +| Job 1's per-phase command ran with `$JOB` empty (the bash variables of `vars.sh` were set, not exported, and the command runs in a child bash): `--out /results-A.json`, the saved shard proofs in `/` on the WSL root, so the aggregate-only phases B0, B, C1 and the prototype-shard phase C2 failed in 0.0 s ("No such file") | `export` in `vars.sh`; job 2 reads the proofs from `/` | +| Job 1's own-miner phases launched the iGPU miner (the first `igneum-miner mine` process matched; the 5090's is the second) and `if (StartMiner ...)` was always true (PowerShell: a function's emitted RESULT strings are part of its output), so D and E ran with the 5090 idle and the AMD iGPU at 3.4 MH/s: three more idle replicates of the chain (2.0 to 2.2 s shards, 1.8 and 2.2 s aggregations), no curve | the miner matched on `igneum-worker-cuda`, the outcome in a script-scope flag, `--race off` for the own miner (no tuning file on PC 2; a race costs up to 120 s a start) | +| Job 1's `/api/resume` at 21:25:11Z answered ok and the 5090 miner stayed off (card state `off`, hash 0.0, 1,760 MiB on the card) until the 0.3.10 restart; job 2 waited its full 600 s for a hash rate and ran its mining phases void | the restore job `tools/proving-v1/pc2-agg-cost-restore.ps1` also posts `/api/start`; the Counter ASIC coordinator opened a task chip for the resume defect | +| Jobs 3 and 4 (`agg-cost-pc2-3` 22:41Z on app 0.3.10, `agg-cost-pc2-4` 00:18Z on 0.3.11): every `/api/state` read came back as the two bytes `{}` (job 4's raw-body print: `raw_len=2`; the same reads gave the full state on 0.3.9 at 21:01Z and the AMD agent saw the empty reply at 22:22Z), so the card switch found no card, the app's 5090 miner kept mining, and job 3 ran a second miner beside it (two miners at about 60 MH/s each) while job 4's double-mining guard voided its own-miner phases. The class is the app's, not the reader's: `state_json()` (engine.rs:180) does `serde_json::to_value(st).unwrap_or(json!({}))`, and the value that fails is `ProvingState.paid_wei: u128` (serde_json 1.0.151 refuses a u128 over u64::MAX, 18.45 IGN; the proving-v1 agent's diagnosis): a paid shard averages 1.23 IGN, so the reply empties about 15 paid shards after every app start and comes back at the next restart, which matches the times (full at 21:01Z with paid_wei 0, empty from 22:22Z after the prover had paid from 22:02Z). Fixed on the app branch proving-v1 at 6714a45 (paid_wei as a decimal string, the error logged, an `{"error":...}` reply on any future failure) | job 5 reads the card keys from the app's `settings.json` (`cards`: key to enabled and identities), restores the 5090's 8 identities first (the restore job of 23:16:53Z had set 2: its parser read the next card's value), refuses before any pause when it cannot name the card, waits on the CUDA worker process count for the card to stop, and checks the worker is back at the end | +| Job 5 (`agg-cost-pc2-5`, 01:10:44Z) failed at PowerShell's parse in 1 s: `$RestoreIdentities:` inside a double-quoted string (a drive-qualified variable); no card or miner touched | `${RestoreIdentities}:`; the other `$name:` shapes are inside single-quoted bash here-strings | +| Job 6's identities step found `settings.json` already at 8 identities under the active key `nvidia:0:NVIDIA GeForce RTX 5090` (a stale key `nvidia:NVIDIA GeForce RTX 5090` carries 2), so no change was sent; job 6's `/api/cards` with the 5090 disabled answered ok but the worker ran on for 120 s, `/api/pause` stopped it in 5 s, and at the end `/api/resume` brought it back in 5 s on 0.3.11 | the card switch keeps the pause as its fallback; the resume path works on 0.3.11 | +| PC 2 ran three jobs at once from 01:24Z (`run-prover-on-pc2-20261006` at 01:24:21Z, the ledger suites build at 01:26:15Z, while agg-cost-pc2-6's closing report was still being uploaded): the app does not serialise jobs, "one job per machine at a time" holds only by the coordinator's word; job 6 had closed at 01:24:14Z, so its rows are clean | nothing of mine to fix; a rule for the job runner | +| The make-package gate ran the exporter's side files (`block-N.json.node-plan.json`) as fixtures and failed; its execute step took the exclusive `measure` lock for a cycle count and queued 25 min behind a packbench run | the glob skips `.node-plan.json`; the execute step runs under the `run` lock (a count, not a time) | diff --git a/docs/plans/proving-v1.md b/docs/plans/proving-v1.md index bedff564..50b64d14 100644 --- a/docs/plans/proving-v1.md +++ b/docs/plans/proving-v1.md @@ -154,3 +154,22 @@ The aggregation-cost agent's jobs read the two bytes `{}` from `/api/state` on P What it means: the dashboard on a proving machine shows nothing within about 12 minutes of its prover's first payout; every PC playbook that reads a card from `/api/state` fails the same way (the agent's job 5 reads settings.json instead). Mining, proving and payouts are untouched; it is the status page only. Fix on the app branch: `paid_wei` serialises as a decimal string (the dashboard already reads it with `Number()`), `state_json` logs `[error] state_json: ...` once instead of answering `{}`, and the reply on any future serialisation error carries `error` and `version` rather than nothing; unit test `a_paid_total_over_u64_max_still_serialises_the_whole_state`. Not in 0.3.11 (that tree closed at 22c2363, master 630da6b, published); 6714a45 heads 0.3.12, the morning's first cut, app only, before PC 1's relaunch (coordinator, counter-asic-2-rollout.md 8a); until then the workaround is settings.json for the card keys. + +## Aggregation cost (5 October, night) + +the project lead, 5 October 2026: "fix everything else in the numbers tonight". Branch `agg-cost`; every number in `docs/bench-log.md`, "aggregation cost on the RTX 5090", with its job id. The proof statement and the pinned guests are unchanged: every existing fixture proof still verifies (`verify-segment` 0.027 s on the Mac, 0.036 to 0.041 s on PC 2). + +| What | Before (5 October evening, `chain-pc2-pv1c`) | After (5 October night) | +|---|---|---| +| Chained aggregation, the card mining | 9.6 to 9.7 s a block | 9.6 to 9.8 s a block, the same (job `agg-cost-pc2-1`, phase A); the miner's presence is the whole cost: 2.1 to 2.2 s a block with the card to itself, 1.7 s unchained | +| Shard proof (empty shard), the card mining | 7.3 to 7.7 s | 7.4 to 7.8 s; 1.9 to 2.2 s with the card to itself | +| The miner's slowdown of the prover | 3 to 4x (against 4 October) | measured on the same fixtures 2 min apart: shards 3.6x, chained aggregation 4.5x, a block 4.2x | +| Where the time goes | not profiled | the host's share 0.000 s (the prove call is everything); the GPU server prints no timings; the second deferred proof (the chain rule) costs 0.4 to 0.5 s alone and 1.7 to 1.9 s under the miner; the prover alone keeps the card busy 15.8% of the time, the miner 93.9% | +| SP1 knobs (`SP1_WORKER_VERIFY_INTERMEDIATES=false`) | not tried | no gain: 7.8 s against 8.1 s over four aggregations, inside the spread; the shape knobs would change the recursion keys the pinned verifier accepts | +| Batch fold (K blocks in one aggregator call) | not estimated | estimate from the measured step costs: 0.9 s a block alone and 3.8 s mining at K = 4, 0.7 and 2.8 s at K = 8 (0.25 s per further deferred proof alone, 1.8 s mining); a tree fold gains nothing. A new pinned guest and program id either way, so not tonight | +| Two prover processes on one card | not tried | closed on SP1 6.8.1: both share one GPU server socket, run slower together (6.9 s a block against 4.1) and the second dies with the first (`early eof`) | +| The miner's kernel length (`--batch-log2` of the CUDA worker, 2^B nonces a launch; job `agg-cost-pc2-6`, the 5090 alone with the job's own miner) | not tried | 2^22 (the default) and 2^20: 18.1 and 18.0 s a block, 10.0 to 10.4 s a chained aggregation, 104 MH/s; 2^18: 15.6 s, 8.8 s, 99 MH/s (minus 4%); 2^16: 11.1 s, 6.1 to 6.2 s, 84 MH/s (minus 19%), reproduced | +| The GPU time-slice policy (`nvidia-smi compute-policy --set-timeslice`) | not tried | "Not Supported" on PC 2 (driver 13.3, Windows): closed | +| The chosen combination | the defaults | the defaults stay: batch-log2 22 and SP1's default knobs. The one knob that moves the prover (2^16) costs a fifth of the hash rate all the time for a prover that is busy a few seconds a minute on the devnet; it is the project lead's trade, not a default (below) | + +Reading. The per-block aggregation is 2.1 s and a block 4.1 s on a 5090 that only proves, 9.7 and 17.5 s on one that also mines; no knob, fold or stream on tonight's SP1 changes the first pair, and only the miner's kernel length changes the second, at 1 MH/s per 0.37 s of block time. So "under 3 s a block" and "under 1.5x" are met on a card that is not mining and are not reachable on one that is. What that means per tier: a 5090 that mines and proves delivers a proven empty block every 17.5 s (6 cards for 1 block/s), the same card proving only every 4.1 s (2 cards, plus the shard work of full blocks: the fleet table above), and a batch fold of the aggregator (a new pinned guest) would bring the proving-only card to about 2.7 s a block and the mining one to about 12 s. What is being done: the app and host defaults are left as measured; the plan's open decision for the project lead is whether a card that holds a shard assignment should drop to 2^16 for the proof's minute (1.6x faster proof, 19% of its hash rate for that minute) or whether proving-only cards carry the aggregation (the clean 2.1 s), and the batch fold goes on the next pin's list. The state class found on the way (`/api/state` answering `{}` once `paid_wei` passes u64::MAX, fixed on the app branch at 6714a45) is in the bench log with the rest. diff --git a/proving/igneum-prove/host/src/main.rs b/proving/igneum-prove/host/src/main.rs index d7ed375b..005182c2 100644 --- a/proving/igneum-prove/host/src/main.rs +++ b/proving/igneum-prove/host/src/main.rs @@ -93,9 +93,12 @@ fn run() -> Result<()> { // proof (the chain rule of design 5.3), the measurement of docs/plans/proving-v1.md step 2 let list = arg("--chain").or_else(|| args.get(1).filter(|a| !a.starts_with("--")).cloned()).context("--chain (consecutive fixtures)")?; let fixtures: Vec = list.split(',').map(|s| s.trim().to_string()).filter(|s| !s.is_empty()).collect(); - return run_chain(&pinned, &fixtures, prover, out_path.as_deref()); + // --save-shards writes every shard's compressed proof next to the results (block-N-shard-i-compressed.bin), + // so `--mode aggregate` can re-run the aggregation of the same proofs under other settings + let save_shards = args.iter().any(|a| a == "--save-shards"); + return run_chain(&pinned, &fixtures, prover, out_path.as_deref(), save_shards); } - let path = args.get(1).filter(|a| !a.starts_with("--")).context("usage: igneum-prove-host [--mode native|execute|shard|compressed|block|all] [--shard N] [--budget ] [--prover 0x..] [--out results.json]; --mode chain --chain [--prover 0x..] [--out results.json]; --mode aggregate --proofs --parent 0x.. [--prev prev.bin] [--out results.json]; --mode verify --proof --statement 0x..; --mode verify-segment --proof --statement 0x..; --mode id")?; + let path = args.get(1).filter(|a| !a.starts_with("--")).context("usage: igneum-prove-host [--mode native|execute|shard|compressed|block|all] [--shard N] [--budget ] [--prover 0x..] [--out results.json]; --mode chain --chain [--prover 0x..] [--out results.json] [--save-shards]; --mode aggregate --proofs --parent 0x.. [--prev prev.bin] [--out results.json]; --mode verify --proof --statement 0x..; --mode verify-segment --proof --statement 0x..; --mode id")?; let shard_index: usize = arg("--shard").map(|s| s.parse()).transpose()?.unwrap_or(0); // `--budget `: re-plan the fixture's block at a TEST budget (the S_p curve of 5 October 2026); the fixture's // own per-shard plan is then not compared (the chain and the sums still are), and `--out` records the cut @@ -577,6 +580,8 @@ fn setup_sp1(pinned: &pinned::Pinned, results: &mut serde_json::Map) -> std::path::PathBuf { /// `--mode chain`: every fixture in order, consecutive on the chain (number and parent hash), each block's shards /// proven compressed and aggregated with the previous block's aggregated proof (`AggInput.prev`, the chain rule), /// every proof verified. One RESULT line per shard, per block (with the running totals) and for the chain. -fn run_chain(pinned: &pinned::Pinned, fixtures: &[String], prover: Address, out_path: Option<&str>) -> Result<()> { +fn run_chain(pinned: &pinned::Pinned, fixtures: &[String], prover: Address, out_path: Option<&str>, save_shards: bool) -> Result<()> { if fixtures.is_empty() { bail!("--chain needs at least one fixture"); } @@ -660,6 +665,7 @@ fn run_chain(pinned: &pinned::Pinned, fixtures: &[String], prover: Address, out_ let mut blocks_json = Vec::with_capacity(built.len()); let (mut shard_total, mut agg_total, mut shards_total) = (0.0f64, 0.0f64, 0usize); let out_dir = out_dir_of(out_path); + let mut shard_files: Vec = Vec::new(); for (number, _hash, shards, _) in &built { let block_t = Instant::now(); let mut proofs: Vec = Vec::with_capacity(shards.len()); @@ -676,12 +682,18 @@ fn run_chain(pinned: &pinned::Pinned, fixtures: &[String], prover: Address, out_ } shard_secs.push(dt); shard_total += dt; + if save_shards { + let file = out_dir.join(format!("block-{number}-shard-{i}-compressed.bin")); + save_proof(&p.proof, file.clone()); + shard_files.push(file.display().to_string()); + } proofs.push(p); } shards_total += proofs.len(); stage(&format!("chain block {number} aggregate {} shards{}", proofs.len(), if prev.is_some() { " with the previous block proof" } else { "" })); let seg = sp1.aggregate(prev.as_ref(), &proofs)?; let adt = sp1.last_timing("aggregate").unwrap_or_default().as_secs_f64(); + let sdt = sp1.last_timing("aggregate-stdin").unwrap_or_default().as_secs_f64(); agg_total += adt; let claim = SegmentClaim::from_block(&seg.output); let ok = sp1.verify_segment(&seg, &claim); @@ -690,9 +702,10 @@ fn run_chain(pinned: &pinned::Pinned, fixtures: &[String], prover: Address, out_ let block_s = block_t.elapsed().as_secs_f64(); let cumulative = chain_t.elapsed().as_secs_f64(); println!( - "RESULT chain block {number}: {} shards ({:.1} s of shard proofs), aggregate prove {adt:.1} s, proof {bytes} bytes, verify {vdt:.3} s, {}; chain_len {}, agg_vk {}; this block {block_s:.1} s, cumulative {cumulative:.1} s over {} block(s) at {}", + "RESULT chain block {number}: {} shards ({:.1} s of shard proofs), aggregate prove {adt:.1} s (stdin {sdt:.3} s, {} deferred proofs), proof {bytes} bytes, verify {vdt:.3} s, {}; chain_len {}, agg_vk {}; this block {block_s:.1} s, cumulative {cumulative:.1} s over {} block(s) at {}", seg.output.shard_count, shard_secs.iter().sum::(), + proofs.len() + usize::from(prev.is_some()), if ok { "VERIFIED (shard program id, aggregator id and claim checked)" } else { "VERIFY FAILED" }, seg.output.chain_len, seg.output.agg_vk, @@ -707,7 +720,7 @@ fn run_chain(pinned: &pinned::Pinned, fixtures: &[String], prover: Address, out_ bail!("block {number}: chain_len {} is not {expected_len}", seg.output.chain_len); } blocks_json.push(serde_json::json!({ - "number": number, "shards": seg.output.shard_count, "shard_prove_seconds": shard_secs, "aggregate_prove_seconds": adt, + "number": number, "shards": seg.output.shard_count, "shard_prove_seconds": shard_secs, "aggregate_prove_seconds": adt, "aggregate_stdin_seconds": sdt, "aggregate_verify_seconds": vdt, "proof_bytes": bytes, "chain_len": seg.output.chain_len, "block_seconds": block_s, "cumulative_seconds": cumulative, "post_root": seg.output.post_root.to_string(), "statement": alloy_primitives::keccak256(seg.output.to_bytes()).to_string(), })); @@ -732,6 +745,7 @@ fn run_chain(pinned: &pinned::Pinned, fixtures: &[String], prover: Address, out_ results.insert("shard_prove_seconds_total".into(), shard_total.into()); results.insert("aggregate_prove_seconds_total".into(), agg_total.into()); results.insert("chain_seconds".into(), total.into()); + results.insert("shard_proof_files".into(), shard_files.into_iter().map(serde_json::Value::from).collect::>().into()); drop(sp1); finish(results, out_path.map(|s| s.to_string())) } @@ -794,16 +808,18 @@ fn run_aggregate(pinned: &pinned::Pinned, proofs: &str, parent: &str, prev_path: for shards in &blocks { let number = shards[0].output.number; stage(&format!("aggregate block {number}, {} shards", shards.len())); + let deferred = shards.len() + usize::from(prev.is_some()); let seg = sp1.aggregate(prev.as_ref(), shards)?; let adt = sp1.last_timing("aggregate").unwrap_or_default().as_secs_f64(); + let sdt = sp1.last_timing("aggregate-stdin").unwrap_or_default().as_secs_f64(); let claim = SegmentClaim::from_block(&seg.output); let ok = sp1.verify_segment(&seg, &claim); let vdt = sp1.last_timing("verify-block").unwrap_or_default().as_secs_f64(); - println!("RESULT aggregate block {number}: {} shards, prove {adt:.1} s, proof {} bytes, verify {vdt:.3} s, {}; chain_len {}, post {} at {}", seg.output.shard_count, bincode::serialize(&seg.proof)?.len(), if ok { "VERIFIED" } else { "VERIFY FAILED" }, seg.output.chain_len, seg.output.post_root, now()); + println!("RESULT aggregate block {number}: {} shards, prove {adt:.1} s (stdin {sdt:.3} s, {deferred} deferred proofs), proof {} bytes, verify {vdt:.3} s, {}; chain_len {}, post {} at {}", seg.output.shard_count, bincode::serialize(&seg.proof)?.len(), if ok { "VERIFIED" } else { "VERIFY FAILED" }, seg.output.chain_len, seg.output.post_root, now()); if !ok { bail!("the aggregated proof of block {number} did not verify"); } - per_block.push(serde_json::json!({ "number": number, "shards": seg.output.shard_count, "aggregate_prove_seconds": adt, "aggregate_verify_seconds": vdt, "chain_len": seg.output.chain_len })); + per_block.push(serde_json::json!({ "number": number, "shards": seg.output.shard_count, "aggregate_prove_seconds": adt, "aggregate_stdin_seconds": sdt, "deferred_proofs": deferred, "aggregate_verify_seconds": vdt, "chain_len": seg.output.chain_len })); prev = Some(seg); } let seg = prev.unwrap(); diff --git a/proving/igneum-prove/host/src/proof_system.rs b/proving/igneum-prove/host/src/proof_system.rs index 840fae0f..bfc60e2e 100644 --- a/proving/igneum-prove/host/src/proof_system.rs +++ b/proving/igneum-prove/host/src/proof_system.rs @@ -259,6 +259,16 @@ impl Sp1ProofSystem { self.timings.lock().unwrap().push((what.to_string(), dt)); } + /// The SP1 prover knobs set in this process's environment (the GPU server inherits them; the names from + /// sp1-core-executor 6.8.1 `opts.rs` and sp1-prover 6.8.1 `worker/config.rs`), for the RESULT lines, so a + /// measurement names the settings it ran under. "none" when the defaults apply. + pub fn env_knobs() -> String { + let fixed = ["SHARD_SIZE", "ELEMENT_THRESHOLD", "HEIGHT_THRESHOLD", "FULL_SIZE_SHARDS", "MINIMAL_TRACE_CHUNK_THRESHOLD", "TRACE_CHUNK_SLOTS", "MEMORY_LIMIT", "WITHOUT_VK_VERIFICATION", "RUST_LOG"]; + let mut out: Vec = std::env::vars().filter(|(k, _)| k.starts_with("SP1_WORKER_") || fixed.contains(&k.as_str())).map(|(k, v)| format!("{k}={v}")).collect(); + out.sort(); + if out.is_empty() { "none".into() } else { out.join(" ") } + } + pub fn last_timing(&self, what: &str) -> Option { self.timings.lock().unwrap().iter().rev().find(|(k, _)| k == what).map(|(_, d)| *d) } @@ -286,6 +296,9 @@ impl ProofSystem for Sp1ProofSystem { /// The aggregator guest over the shard proofs (and the previous segment's proof when given), by recursion. fn aggregate(&self, prev: Option<&Sp1SegmentProof>, shards: &[Sp1ShardProof]) -> Result { let first = shards.first().ok_or_else(|| anyhow!("no shards"))?; + // 5 October 2026 (aggregation cost): the stdin build (the proof clones into the request) is timed apart + // from the prove call, so the host's own share of an aggregation is visible next to the GPU's. + let t_stdin = Instant::now(); let mut stdin = SP1Stdin::new(); let input = AggInput { shard_vk: self.shard_vk_hash(), @@ -302,6 +315,7 @@ impl ProofSystem for Sp1ProofSystem { let SP1Proof::Compressed(proof) = p.proof.proof.clone() else { return Err(anyhow!("the previous block proof is not a compressed proof")) }; stdin.write_proof(*proof, self.agg_vk.vk.clone()); } + self.record("aggregate-stdin", t_stdin.elapsed()); let t = Instant::now(); let proof = self.client.prove(&self.agg_pk, stdin).compressed().run()?; self.record("aggregate", t.elapsed()); diff --git a/proving/windows-wsl2/make-package.sh b/proving/windows-wsl2/make-package.sh index 4cb16deb..7298dde3 100755 --- a/proving/windows-wsl2/make-package.sh +++ b/proving/windows-wsl2/make-package.sh @@ -34,9 +34,13 @@ if [ "${SKIP_GATE:-0}" != "1" ]; then fi H="$ROOT/proving/igneum-prove/target/release/igneum-prove-host" for f in "$ROOT"/proving/fixtures/block-*.json; do + # the exporter's side files (block-N.json.node-plan.json, 5 October 2026) are not fixtures + case "$f" in *.node-plan.json) continue ;; esac if ! "$H" "$f" --mode native >>"$GATE_LOG" 2>&1; then echo "GATE FAILED: native run of $(basename "$f"); see $GATE_LOG"; exit 1; fi done - if ! "$ROOT/tools/lock/with-lock.sh" measure "$H" "$ROOT/proving/fixtures/block-338-shard1.json" --mode execute --shard 0 >>"$GATE_LOG" 2>&1; then + # the execute step reports a cycle count, not a time: the `run` lock (tools/lock/with-lock.sh: counts, not ms), so the + # gate does not queue behind every build and measurement on the Mac (5 October 2026: 25 min behind a packbench run) + if ! "$ROOT/tools/lock/with-lock.sh" run "$H" "$ROOT/proving/fixtures/block-338-shard1.json" --mode execute --shard 0 >>"$GATE_LOG" 2>&1; then echo "GATE FAILED: the guest did not execute the shard fixture (the exact failure the PC hit on 4 October); see $GATE_LOG"; exit 1 fi "$H" --mode id | tee -a "$GATE_LOG" # the pinned program ids this package carries (the PC's build embeds the same elf/ files) diff --git a/tools/proving-v1/agg-cost-table.mjs b/tools/proving-v1/agg-cost-table.mjs new file mode 100644 index 00000000..e1eefca3 --- /dev/null +++ b/tools/proving-v1/agg-cost-table.mjs @@ -0,0 +1,50 @@ +#!/usr/bin/env node +// The aggregation-cost curve from one or more pc2-agg-cost.ps1 jobs (5 October 2026 night): per phase, the shard and +// aggregation seconds per block (the chain and aggregate-only RESULT lines), the GPU utilisation and memory peak of +// the phase, the own miner's rate (MH/s wall) and its batch-log2. Reads the job uploads from the log intake like +// tools/jobs.mjs (DATABASE_URL in ~/.config/igneum/env). Prints a markdown table. +// node tools/proving-v1/agg-cost-table.mjs agg-cost-pc2-1 agg-cost-pc2-2 ... +import { readFileSync } from 'node:fs'; +import { homedir } from 'node:os'; +const ids = process.argv.slice(2); +if (!ids.length) { console.error('usage: agg-cost-table.mjs [...]'); process.exit(2); } +const m = /^DATABASE_URL=(.*)$/m.exec(readFileSync(`${homedir()}/.config/igneum/env`, 'utf8')); +const url = m[1].trim().replace(/^['"]|['"]$/g, ''); +const host = new URL(url).hostname.replace('-pooler', ''); +const sql = async (query, params = []) => { + const r = await fetch(`https://${host}/sql`, { method: 'POST', headers: { 'Neon-Connection-String': url, 'Content-Type': 'application/json' }, body: JSON.stringify({ query, params }) }); + const j = await r.json(); + if (!r.ok) throw new Error(j.message || JSON.stringify(j)); + return j.rows; +}; +const f1 = x => (Math.round(x * 10) / 10).toFixed(1); +const rows = []; +for (const id of ids) { + const ups = await sql('SELECT lines FROM miner_logs WHERE run_id LIKE $1 ORDER BY received_at DESC LIMIT 1', [`job-${id}-%`]); + if (!ups.length) { console.error(`${id}: no upload`); continue; } + const lines = ups[0].lines.split('\n'); + const phases = new Map(); + const ph = l => { if (!phases.has(l)) phases.set(l, { job: id, label: l, shard: [], agg: [], deferred: [], util: '', mem: '', rate: '', batch: '', note: '' }); return phases.get(l); }; + for (const raw of lines) { + const l = raw.replace(/^\d+(\.\d+)? job \S+: /, ''); + let x; + if ((x = /^([A-Z0-9]+): RESULT chain block \d+ shard \d+: compressed prove ([\d.]+) s/.exec(l))) ph(x[1]).shard.push(Number(x[2])); + else if ((x = /^([A-Z0-9]+): RESULT (?:chain|aggregate) block \d+: .*?(?:aggregate )?prove ([\d.]+) s \(stdin [\d.]+ s, (\d+) deferred/.exec(l))) { ph(x[1]).agg.push(Number(x[2])); ph(x[1]).deferred.push(Number(x[3])); } + else if ((x = /^RESULT phase_gpu ([A-Z0-9]+) samples=\d+ memory_used_max_mib=(\d+) util_mean_pct=([\d.]+)/.exec(l))) { ph(x[1]).mem = x[2]; ph(x[1]).util = x[3]; } + else if ((x = /^RESULT ([A-Z0-9]+) miner rate (n=\d+ mean=[\d.]+)/.exec(l))) ph(x[1]).rate = x[2].replace('n=', 'n ').replace(' mean=', ', mean '); + else if ((x = /^RESULT ([A-Z0-9]+) own miner started .* batch_log2=(\d+)/.exec(l))) ph(x[1]).batch = x[2]; + else if ((x = /^RESULT ([A-Z0-9]+) own miner (not hashing|EXITED|: no app miner)/.exec(l))) ph(x[1]).note = 'own miner ' + x[2]; + else if ((x = /^RESULT phase ([A-Z0-9]+) end .* exit (\d+) wall ([\d.]+) s/.exec(l))) { const p = ph(x[1]); p.exit = x[2]; p.wall = x[3]; } + else if ((x = /^RESULT phase ([A-Z0-9]+) skipped/.exec(l))) ph(x[1]).note = 'skipped'; + else if ((x = /^RESULT (H (?:choice|knobs)): (.*)$/.exec(l))) console.error(`${id}: ${x[1]}: ${x[2]}`); + } + for (const p of phases.values()) rows.push(p); +} +const mean = a => a.length ? a.reduce((s, v) => s + v, 0) / a.length : NaN; +console.log('| Job | Phase | batch-log2 | Shard s (each) | Aggregation s (each, deferred proofs) | Block s (shard + aggregation, mean) | GPU util % | GPU peak MiB | Miner MH/s | Note |'); +console.log('|---|---|---|---|---|---|---|---|---|---|'); +for (const p of rows) { + const chained = p.agg.filter((_, i) => p.deferred[i] >= 2); + const block = p.shard.length && p.agg.length ? f1(mean(p.shard) + mean(chained.length ? chained : p.agg)) : ''; + console.log(`| ${p.job} | ${p.label} | ${p.batch || (p.label.startsWith('E') && /^E\d+$/.test(p.label) ? p.label.slice(1) : '')} | ${p.shard.map(f1).join(', ')} | ${p.agg.map((a, i) => `${f1(a)} (${p.deferred[i]})`).join(', ')} | ${block} | ${p.util} | ${p.mem} | ${p.rate} | ${[p.note, p.exit && p.exit !== '0' ? `exit ${p.exit}` : ''].filter(Boolean).join('; ')} |`); +} diff --git a/tools/proving-v1/miner-rate.mjs b/tools/proving-v1/miner-rate.mjs new file mode 100644 index 00000000..051c919c --- /dev/null +++ b/tools/proving-v1/miner-rate.mjs @@ -0,0 +1,35 @@ +#!/usr/bin/env node +// The hash rate of one miner over a UTC window, from its STATUS lines in the log intake (the app uploads the miner's +// log every minute; each STATUS line starts with a unix timestamp and carries `now= MH/s wall`, the last +// interval's rate). For the aggregation-cost phases that keep the app's own miner running (docs/bench-log.md, +// 5 October 2026 night), where the job cannot read the app's state from PowerShell 5.1. +// node tools/proving-v1/miner-rate.mjs