From d8076513220c2c9184ffe252cf159767ff852dd4 Mon Sep 17 00:00:00 2001 From: igneum-josh <337424239+igneum-josh@users.noreply.github.com> Date: Thu, 8 Oct 2026 07:30:27 +0100 Subject: [PATCH 1/2] CA4 hot-table job: the first kit flattened the eight hot packs into one folder (the job exited 2 in 0 s, pack hot32k4 missing); kit b carries packs// for all nine, the fetch and run ids take a b suffix Co-Authored-By: Claude Fable 5.1 --- tools/ca3-v4-amend/pc1-ca4-hot-ldcs.ps1 | 4 ++-- tools/ca3-v4-amend/pc1-publish-20261007.sh | 4 ++-- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/tools/ca3-v4-amend/pc1-ca4-hot-ldcs.ps1 b/tools/ca3-v4-amend/pc1-ca4-hot-ldcs.ps1 index e8a889536..ada0ea435 100644 --- a/tools/ca3-v4-amend/pc1-ca4-hot-ldcs.ps1 +++ b/tools/ca3-v4-amend/pc1-ca4-hot-ldcs.ps1 @@ -17,8 +17,8 @@ function Summary([string] $status, [hashtable] $extra) { $o = [ordered]@{ job = $jobDir = $env:IGNEUM_JOB_DIR; if (-not $jobDir) { $jobDir = Join-Path $env:TEMP 'igneum-ca4-mb' } New-Item -ItemType Directory -Force -Path $jobDir | Out-Null $jobs = Split-Path $jobDir -$packs = Join-Path (Join-Path $jobs 'fetch-ca4-hot-packs-20261008') 'packs' -if (-not (Test-Path $packs)) { "RESULT error packs missing at $packs (the fetch job fetch-ca4-hot-packs-20261008 runs first)"; Summary 'failed' @{ error = 'packs missing' }; exit 2 } +$packs = Join-Path (Join-Path $jobs 'fetch-ca4-hot-packs-20261008b') 'packs' +if (-not (Test-Path $packs)) { "RESULT error packs missing at $packs (the fetch job fetch-ca4-hot-packs-20261008b runs first)"; Summary 'failed' @{ error = 'packs missing' }; exit 2 } $inst = @("$env:LOCALAPPDATA\Programs\Igneum Miner", "$env:ProgramFiles\Igneum Miner") | Where-Object { Test-Path (Join-Path $_ 'igneum-worker-cuda.exe') } | Select-Object -First 1 if (-not $inst) { 'RESULT error no installed igneum-worker-cuda.exe'; Summary 'failed' @{ error = 'no worker' }; exit 2 } $exe = Join-Path $inst 'igneum-worker-cuda.exe' diff --git a/tools/ca3-v4-amend/pc1-publish-20261007.sh b/tools/ca3-v4-amend/pc1-publish-20261007.sh index 8683b7e2c..018ea5404 100755 --- a/tools/ca3-v4-amend/pc1-publish-20261007.sh +++ b/tools/ca3-v4-amend/pc1-publish-20261007.sh @@ -38,8 +38,8 @@ case "$STEP" in v5-kit) $P add --kind fetch --target ae432dc7,1ccfe586 --id fetch-ca3-v5-kit-20261007 --file "$KITS/v5kit/packs-ca3-v5-20261007T221001Z.zip" --dir jobs --extract --title "Class v5 kit (the v5 lane, 7f58af97): packs, workers, SHA256SUMS" --expires-hours 36 $DEPLOY ;; v5-amd) $P add --kind run --target ae432dc7 --id run-ca3-pc1-v5-amd-bench-20261007 --script tools/ca3-v4-amend/pc1-v5-amd-bench-20261007.ps1 --timeout-minutes 12 --title "Class v5 kit fingerprint on the RX 9070 XT (beside the miners; the v5 lane's script)" --expires-hours 36 $DEPLOY ;; v5-intel) $P add --kind run --target 1ccfe586 --id run-ca3-pc2-v5-intel-bench-20261007 --script tools/ca3-v4-amend/pc1-v5-intel-bench-20261007.ps1 --timeout-minutes 12 --title "Class v5 kit fingerprint on PC 2's Arc B580 eGPU (the v5 lane's script, class-v5 a4b08245; beside the miners; only on the shipper's PC 2 clear)" --expires-hours 36 $DEPLOY ;; - hot-kit) $P add --kind fetch --target ae432dc7 --id fetch-ca4-hot-packs-20261008 --file "$KITS/igneum-ca4-hot-packs-20261008.zip" --dir jobs --extract --title "CA4 research: the eight 5 October hot-table packs plus the mx8 control" --expires-hours 36 $DEPLOY ;; - hot-ldcs) $P add --kind run --target ae432dc7 --id run-ca4-pc1-hot-ldcs-5090-20261008 --cards-off "nvidia:NVIDIA GeForce RTX 5090" --script tools/ca3-v4-amend/pc1-ca4-hot-ldcs.ps1 --timeout-minutes 75 --title "CA4 research: the hot table with the streaming hint (base against ldcs) on the RTX 5090, unlocked and at the knee (the card alone)" --expires-hours 36 $DEPLOY ;; + hot-kit) $P add --kind fetch --target ae432dc7 --id fetch-ca4-hot-packs-20261008b --file "$KITS/igneum-ca4-hot-packs-20261008b.zip" --dir jobs --extract --title "CA4 research: the eight 5 October hot-table packs plus the mx8 control" --expires-hours 36 $DEPLOY ;; + hot-ldcs) $P add --kind run --target ae432dc7 --id run-ca4-pc1-hot-ldcs-5090-20261008b --cards-off "nvidia:NVIDIA GeForce RTX 5090" --script tools/ca3-v4-amend/pc1-ca4-hot-ldcs.ps1 --timeout-minutes 75 --title "CA4 research: the hot table with the streaming hint (base against ldcs) on the RTX 5090, unlocked and at the knee (the card alone)" --expires-hours 36 $DEPLOY ;; amd-g1) $P add --kind run --target ae432dc7 --id run-ca3-pc1-v4-sub3-amd-g1-20261007 --cards-off amd:gfx1201 --script tools/ca3-v4-amend/pc1-v4-sub3-amd-g1.ps1 --timeout-minutes 45 --title "CA3 PC 1: G1 on the sub-version 3 kit and the AMD ladder on the 9070 XT (the card alone)" --expires-hours 36 $DEPLOY ;; family) $P add --kind run --target ae432dc7 --id run-ca3-pc1-amd-family-20261007-e --cards-off amd:gfx1201 --script tools/ca3-v4-amend/pc1-amd-family-20261007.ps1 --timeout-minutes 20 --title "CA3 PC 1: item 6 family step costs on the 9070 XT (run e, the card alone)" --expires-hours 36 $DEPLOY ;; tune-5080) $P add --kind run --target ae432dc7 --id run-ca3-pc1-ember-5080-20261007 --script tools/ca3-v4-amend/pc1-ember-card.ps1 --timeout-minutes 35 --title "CA3 PC 1: Ember Tune on the RTX 5080 (the installed app tunes the one card)" --expires-hours 36 $DEPLOY ;; From b38b4af68a9fb4ac56dcac9c60feae71626f3bec Mon Sep 17 00:00:00 2001 From: igneum-josh <337424239+igneum-josh@users.noreply.github.com> Date: Thu, 8 Oct 2026 08:05:19 +0100 Subject: [PATCH 2/2] bench-log: the hot-table packs on the RTX 5090, ldcs against base, unlocked and at the 1,300 MHz lock (36 rows, all PASS; ldcs dead, the hot packs lose 2 percent to the lock where mx8 loses 7.5) Co-Authored-By: Claude Fable 5.1 --- docs/bench-log.md | 23 +++++++++++++++++++++++ 1 file changed, 23 insertions(+) diff --git a/docs/bench-log.md b/docs/bench-log.md index 26b328d83..b14a40039 100644 --- a/docs/bench-log.md +++ b/docs/bench-log.md @@ -2758,3 +2758,26 @@ RTX 5080 (the same shape; 8 October 2026, 00:45 to 01:54 UTC): Reading (5080): the rate holds within 0.3 percent of unlocked down to 1,000 MHz on both classes and falls 5.2 percent at 900 MHz on class v4, so the knee is between 1,000 and 900 MHz, a third of the 2,963 MHz boost and lower than the 5090's (the 5080's 84 SMs have more compute headroom per unit of its memory bandwidth, so the memory wait hides the shadow down to a lower clock). Best MH per watt within the 1 percent rate tolerance: class v4 at 1,100 MHz (71.20 MH/s at 146.6 W, 0.486 MH/W; 106.5 W recovered for 0.29 percent rate), class v3 at 1,000 MHz (71.11 at 103.7 W, 0.686; 66.0 W for 0.25 percent). The class v4 premium is 83.4 W unlocked and 41 W at the best points (146.6 W against 105.6 W at 1,100). The draw floors from 1,500 MHz down (about 147 W on v4, 104 W on v3) with the power governor as the only throttle reason say the clock lever is spent by 1,500 MHz on this card. Consequence per tier: a 5080 owner on class v4 locked near 1,100 MHz pays 147 W instead of 253 for 0.3 percent less rate (MH per watt up 72 percent). What the lever is and is not: nvidia-smi exposes no voltage offset (that is NVAPI's); the clock lock walks the driver's V/F curve, which is where the watts come from; the Mac has no lever (no clock cap on Apple silicon); AMD has the ADLX tune line through igneum-gpu-telemetry or nothing. The knob goes into Ember Tune for 0.3.24 (src/ember.rs: the clock ladder continues below 45 percent of the maximum in 100 MHz steps to a 20 percent floor, the search stops at the first row more than the tolerance under the cap point's rate or on a faulted row, the best MH per watt within tolerance is the point, the fingerprint checked on every step, the result stored per card as the lock_* fields). The 5080 stock rows against the rented 5080 of 7 October (71.16 MH/s at 145 W on driver 580): the rate agrees to 0.4 percent, the watts do not (253 W here); PC 1's three power fields agree, so the difference sits with the rented card's sampler or its cap, the fleet lane's re-measure owed. + +## 8 October 2026, the hot-table packs on the RTX 5090: ld.global.cs against the plain load, unlocked and at the 1,300 MHz lock (branch ca3-v4-amend, the hash lane, job run-ca4-pc1-hot-ldcs-5090-20261008b) + +PC 1, RTX 5090 (170 SMs, driver 13.4, NVRTC 12.8), igneum-worker-cuda 1.0 (4 October 2026). The eight 5 October hot-table packs (proto-cuda/packs-ca2-hot) and the mx8-genesis control (proto-cuda/packs-ca2-mixer), each run twice per state: variant base (plain loads) and variant ldcs (ld.global.cs on the hot-table loads). 16,777,216 nonces per run, nvidia-smi 1 Hz sampler (power.draw; instant and average agreed within 0.5 W on every row), 19 to 30 samples per row. Every one of the 36 rows PASS on its pinned fingerprint. Unlocked 06:34 to 06:45 UTC at 2,865 MHz; lock1300 06:45 to 06:56 UTC at 1,290 MHz through the helper; clocks reset (rgc) at the end. The 5090 was switched off in the app by the runner for the job and back on after. + +| pack | unlocked base MH/s | unlocked base W | unlocked ldcs MH/s | lock1300 base MH/s | lock1300 base W | lock1300 base MH/W | lock1300 ldcs MH/s | lock1300 ldcs MH/W | +|---|---|---|---|---|---|---|---|---| +| mx8-genesis (control) | 137.7 | 312.0 | 137.7 | 127.3 | 211.4 | 0.602 | 127.3 | 0.603 | +| hot32k4 | 147.4 | 318.9 | 147.4 | 144.4 | 215.3 | 0.671 | 144.4 | 0.672 | +| hot32k4a | 118.9 | 320.1 | 118.9 | 116.6 | 214.9 | 0.542 | 116.6 | 0.542 | +| hot64k2 | 137.7 | 318.1 | 137.7 | 135.0 | 214.3 | 0.630 | 135.0 | 0.630 | +| hot64k4 | 141.0 | 318.8 | 141.0 | 138.2 | 214.4 | 0.645 | 138.2 | 0.644 | +| hot64k4a | 115.7 | 319.6 | 115.8 | 113.4 | 214.8 | 0.528 | 113.4 | 0.528 | +| hot64k8 | 164.2 | 326.4 | 164.2 | 160.9 | 219.2 | 0.734 | 160.8 | 0.733 | +| hot96k4 | 139.0 | 318.3 | 139.0 | 136.2 | 214.5 | 0.635 | 136.3 | 0.635 | +| hot96k4a | 114.6 | 318.8 | 114.5 | 112.3 | 214.8 | 0.523 | 112.3 | 0.523 | + +What the rows say. + +1. The ldcs variant changes nothing: every pack reads the same rate and the same watts as its base run within 0.1 MH/s and 1 W, both states. The streaming hint on the hot-table loads is dead as a lever; the table's cache behaviour is already what the hardware gives. No further ldcs rows are owed. +2. The hot packs hold their rate under the lock far better than mx8: the 1,300 MHz lock costs mx8 7.5 percent of rate (137.7 to 127.3) and costs the hot packs 2 percent (hot32k4 147.4 to 144.4, hot64k8 164.2 to 160.9). The hot packs are bound by the table's latency, not by the core; the lock takes a third of the watts off every pack (319 to 215 W) and the hot packs pay almost no rate for it. +3. Per watt at the lock the hot family runs 0.52 to 0.73 MH/W against the control's 0.60: hot64k8 is 22 percent cheaper per hash than mx8 on this card, hot32k4 11 percent cheaper, the "a" packs 10 to 13 percent dearer. Whether a cheaper hash on the GPU is a gain or a loss for resistance is the research lane's call: it is a gain only if the saving comes from the memory path an ASIC would have to buy too. +4. The 5090's locked class v4 reading from the efficiency pass (1,300 MHz: 134.6 MH/s at 223 W, 0.60 MH/W) sits level with the mx8 control here (0.602), so the two passes agree on the control and the hot rows are comparable to the v4 grid.