diff --git a/app/igneum-app/tiers/class-v5-tiers.json b/app/igneum-app/tiers/class-v5-tiers.json index 5827e9b55..56124c4f0 100644 --- a/app/igneum-app/tiers/class-v5-tiers.json +++ b/app/igneum-app/tiers/class-v5-tiers.json @@ -399,9 +399,9 @@ "memory_gb": 24, "label": "estimated", "stock": { - "mhs": 63.0, - "w": 280.0, - "uj": 4.44, + "mhs": 70.63, + "w": 353.5, + "uj": 5.0, "class": "v4" }, "tiers": [ @@ -411,13 +411,13 @@ "power_pct": 60, "mem_mhz": 0, "limit_w": 450, - "mhs": 63.0, - "w": 210.0, - "mhw": 0.3, + "mhs": 70.6, + "w": 257.0, + "mhw": 0.2747, "source": "measured", "label": "estimated", - "uj": 3.4, - "note": "the Ada shape (the 4070's measured tune: v3 x0.72, the premium x0.7): band 3.0 to 3.8; the class v5 kit read 62.76 MH/s on a rented 4090 with the watts unread" + "uj": 3.64, + "note": "the Ada shape (the 4070's measured tune: v3 x0.72, the premium x0.7) on the measured stock rows (class v3 253.4 W, class v4 353.5 W, floor lane 1, 8 October): band 3.3 to 4.0; the occupancy knob is worth 1 to 2 percent on this card and nothing more" }, { "id": "balanced", @@ -425,12 +425,12 @@ "power_pct": 80, "mem_mhz": 0, "limit_w": 450, - "mhs": 63.0, - "w": 235.0, - "mhw": 0.2681, + "mhs": 70.63, + "w": 290.0, + "mhw": 0.2436, "source": "measured", "label": "estimated", - "uj": 3.73, + "uj": 4.11, "note": "the Ada prior (a cap holds the rate until the SM clock falls under about 2,400 MHz)" }, { @@ -439,16 +439,16 @@ "power_pct": 100, "mem_mhz": 0, "limit_w": 450, - "mhs": 63.0, - "w": 280.0, - "mhw": 0.225, + "mhs": 70.63, + "w": 353.5, + "mhw": 0.1998, "source": "stock", - "label": "estimated", - "uj": 4.44, - "note": "the 0.3.12 rented row 52.25 MH/s at 183.1 W and the rig's 57.4 at 205 W were an earlier class; the class v4 stock row is owed to the denominator sweep" + "label": "measured", + "uj": 5.0, + "note": "floor lane 1's rented 4090 at stock: class v3 70.63 MH/s at 253.4 W (3.59), class v4 70.63 at 353.5 W (5.0), idle 34.7 W; the denominator sweep's RunPod 4090 read class v3 70.09 at 222.6 W (3.18): the host spread again" } ], - "src": "docs/analysis/prover-tiers-real-cards.md (6 October); ~/igneum-fleet/cv5-results.txt (the class v5 kit on a 4090)" + "src": "docs/analysis/class-v6/floor/sm-sparse.md (floor lane 1, 8 October); ~/igneum-fleet/cardbench/rows.jsonl (the denominator sweep)" }, { "card": "NVIDIA GeForce RTX 4080", @@ -1398,7 +1398,7 @@ "source": "stock", "label": "measured", "uj": 2.77, - "note": "rented pods 7 and 8 October (class v3 248.9 MH/s at 411.7 W; class v4 250.6 at 695.1 W)" + "note": "rented pods 7 and 8 October (class v3 248.9 MH/s at 411.7 W; class v4 250.6 at 695.1 W); floor lane 1 read class v3 254.6 at 450.4 W (1.77) and class v4 251.9 at 699 W (2.78, the 700 W cap binding, the clock throttled to 1,750 MHz), class v5 250.2 at 698.8 W: v5 = v4 on this card" } ], "src": "~/igneum-fleet/cardbench/rows.jsonl, 8 October" diff --git a/docs/analysis/class-v6/floor/denominator.md b/docs/analysis/class-v6/floor/denominator.md index d4e982f37..265d074c2 100644 --- a/docs/analysis/class-v6/floor/denominator.md +++ b/docs/analysis/class-v6/floor/denominator.md @@ -24,7 +24,10 @@ its source) or estimated (with its method). Nothing here moves a consensus objec the knee, k = 1) stands; the honest sentence per joule is now "under 2x against a 16 GB Blackwell card at its knee for a core as good as a GPU lane, 1.3x against an Apple laptop". 3. **What moves the floor is the operating point and nothing else in software.** The SM-sparse kernel is dead on - measured rows (the draw follows the work, not the SM count; research file 20.3b); the cache-policy hint is dead; the + measured rows (the draw follows the work, not the SM count; research file 20.3b; floor lane 1's rows of 13:25 on a + rented 4090 and H100 say the same at stock: the best occupancy shape saves 1 to 2 percent of the watts and the draw + over idle is flat across the SM count while the rate holds, so the residual is the clock domain, reachable by the + lock and the undervolt only); the cache-policy hint is dead; the memory clock is never touched (a step that drags it down cannot win); the power cap is flat on Blackwell at the hash (0.41 MH/W under every limit on the 5090) and is the only lever on Ampere; the core lock walks the driver's V/F curve and is the lever on Blackwell and Ada; AMD has the ADLX offsets (measured today: 24 percent of the 9070 XT's @@ -96,7 +99,7 @@ Max; the wall more); the NVIDIA and AMD rows are whole-card. | RTX 5070 (12 GB) | 12 GB | 1.75 | estimated | 2.2x / 1.6x | 2.7x / 1.8x | 3.8x / 2.2x | stock class v4 58.9 MH/s at 176.1 W (2.99) measured (the paired class v3 run 113 W); the Blackwell shape gives about 103 W: 1.75 v4; band 1.7 to 2.1 | | RTX 5060 Ti (16 GB) | 16 GB | 2.33 | estimated | 2.9x / 2.1x | 3.6x / 2.4x | 5.0x / 3.0x | stock class v4 30.9 MH/s at 114.8 W (3.72) measured on PC 2 (the Thunderbolt enclosure); the Blackwell shape gives about 72 W: 2.33; band 2.2 to 2.6; the card-in job measures the ladder | | RTX 5060 (8 GB) | 8 GB | 2.15 | estimated | 2.7x / 1.9x | 3.3x / 2.2x | 4.6x / 2.7x | stock class v3 31.27 MH/s at 75.4 W (2.41) measured; class v4 stock about 110 W (3.5) and the knee about 67 W: 2.15; band 2.0 to 2.5 | -| RTX 4090 (24 GB) | 24 GB | 3.40 | estimated | 4.3x / 3.1x | 5.3x / 3.5x | 7.3x / 4.3x | the denominator sweep's class v3 row 70.09 MH/s (RunPod, 8 October 13:07, the watts in section 6); the class v5 kit 62.76 MH/s on another 4090; the Ada shape (the 4070's tune: v3 x0.72, the premium x0.7) gives about 3.3 v4, 3.4 v5; band 3.0 to 3.8 | +| RTX 4090 (24 GB) | 24 GB | 3.64 | estimated (stock measured, lock owed) | 4.6x / 3.3x | 5.6x / 3.8x | 7.8x / 4.6x | stock measured by floor lane 1: class v3 70.63 MH/s at 253.4 W (3.59), class v4 70.63 at 353.5 W (5.0), idle 34.7 W; the denominator sweep's RunPod 4090 read class v3 70.09 at 222.6 W (3.18, the host spread); the Ada shape (the 4070's tune: v3 x0.72, the premium x0.7) gives about 257 W at the knee: 3.64 v5; band 3.3 to 4.0; the occupancy knob 1 to 2 percent (lane 1) | | RTX 4080 (16 GB) | 16 GB | 3.33 | estimated | 4.2x / 3.0x | 5.2x / 3.4x | 7.2x / 4.2x | stock class v3 40.71 MH/s at 128.6 W (3.16) and class v4 40.74 at 185.6 W (4.56) measured; the Ada shape gives about 133 W: 3.26 v4, 3.33 v5; the 4080 Super reads the same (42.6 at 134.9 / 195.7 W) | | RTX 4070 (12 GB) | 12 GB | 3.51 to 3.58 | measured at the tune point | 4.5x / 3.2x | 5.6x / 3.7x | 7.7x / 4.5x | 1,860 MHz lock + the 50 percent cap: class v4 31.08 MH/s at 109.0 W (3.51), v5 3.58; against 3.58 stock on class v3 and 2.57 tuned; the ladder below 1,860 under class v4 is owed (band 3.2 to 3.6) | | RTX 4070 Ti (12 GB) | 12 GB | 3.60 | estimated | 4.6x / 3.2x | 5.6x / 3.7x | 7.8x / 4.6x | stock class v3 31.24 MH/s at 95.4 W (3.05) and class v4 31.26 at 155.3 W (4.97) measured; the Ada shape gives about 111 W: 3.6; the 4070 Super at a 110 W host cap held 31.26 MH/s at 108.4 W under class v4 (3.47): a cap row measured | @@ -111,7 +114,7 @@ Max; the wall more); the NVIDIA and AMD rows are whole-card. | RX 7800 XT (16 GB) | 16 GB | 9.50 | estimated | 12.0x / 8.5x | 14.7x / 9.8x | 20.5x / 12.1x | 16 channels: about 18 MH/s at about 230 W stock (12.8); the knob 9 to 10 | | RX 7600 XT (16 GB) | 16 GB | 13.0 | estimated | 16.5x / 11.7x | 20.2x / 13.4x | 28.0x / 16.5x | 8 channels: about 9 MH/s at about 160 W stock (18); the knob 12 to 14; the card-in queue holds one for a PC measurement | | RX 9060 XT (16 GB) | 16 GB | 10.0 | estimated | 12.7x / 9.0x | 15.5x / 10.3x | 21.6x / 12.7x | 8 channels of GDDR6 on RDNA 4: about 9.5 MH/s at about 130 W stock (13.7); the knob 9.5 to 10.5; the card-in queue holds one | -| H100 SXM (80 GB) | DC | 2.00 | estimated | 2.5x / 1.8x | 3.1x / 2.1x | 4.3x / 2.5x | stock class v3 248.9 MH/s at 411.7 W (1.65) and class v4 250.6 at 695.1 W (2.77) measured; the premium 283 W (11.2 pJ per op); a lock on an owned host (Hopper at 1,980 MHz boost, HBM3-bound): about 500 W (2.0); band 1.9 to 2.4; rented hosts refuse `-lgc` | +| H100 SXM (80 GB) | DC | 2.00 | estimated (stock measured, lock owed) | 2.5x / 1.8x | 3.1x / 2.1x | 4.3x / 2.5x | stock class v3 248.9 MH/s at 411.7 W (1.65) and class v4 250.6 at 695.1 W (2.77) measured (floor lane 1 the same: v3 254.6 at 450.4 W, v4 251.9 at 699 W with the 700 W cap binding and the clock throttled to 1,750 MHz; class v5 = class v4 on this card, 250.2 at 698.8 W; its -lmc was accepted and the clock stayed 2,619 MHz); the premium 283 W (11.2 pJ per op); a lock on an owned host (Hopper at 1,980 MHz boost, HBM3-bound): about 500 W (2.0); band 1.9 to 2.4; rented hosts refuse `-lgc` | | L40S (48 GB) | DC | 3.50 | estimated | 4.4x / 3.1x | 5.4x / 3.6x | 7.5x / 4.4x | stock class v3 56.4 MH/s at 220.7 W (3.91) and class v4 56.4 at 277.6 W (4.92) measured; the Ada shape gives about 198 W (3.5) | | A100 SXM (80 GB) | DC | 3.10 | estimated | 3.9x / 2.8x | 4.8x / 3.2x | 6.7x / 3.9x | stock class v3 138.3 MH/s at 268.6 W (1.94) and class v4 138.0 at 489.3 W (3.54) measured (the premium 221 W = 15.8 pJ per op, the dearest ALU in the record); the card already runs at 1,410 MHz, a cap recovers about 60 W: 3.1 | | Intel Arc B580 (12 GB) | 12 GB | 10.4 | estimated | 13.2x / 9.4x | 16.2x / 10.8x | 22.5x / 13.3x | 10.6 to 11 MH/s measured on PC 2 (the enclosure), the watts unread; about 110 W by the board's class (OWED); no clock lever in the app for Intel | @@ -147,7 +150,7 @@ What the table says: | Architecture | The lever | What it does, measured | The tier points (Efficiency / Balanced / Max) | Known-failed | |---|---|---|---|---| | Blackwell (5090, 5080; 5070 Ti, 5070, 5060 Ti, 5060 by shape) | the core lock; the power cap is flat on this hash | the rate holds within 1.5 percent to 1,300 MHz on the 5090 and within 0.3 percent to 1,000 on the 5080; the draw falls a third on class v3 and the shadow's premium halves | 1,200 / 1,300 / unlocked on the 5090; 1,100 / 1,100 / unlocked on the 5080; 1,100 / 1,300 / unlocked as the prior on the others | the 5090 at 1,100 (-10.5 percent on class v4), the 5080 at 900 (-5.2 percent); the Power Helper's skip-rule fault (0.3.20) needs the padding lines | -| Ada (4090, 4080, 4070 Ti, 4070, 4060 Ti, 4060, L40S) | the core lock with a cap | the 4070's tune: 1,860 MHz and the 50 percent cap hold 30.95 MH/s at 79.5 W on class v3 (28.7 at 102.7 stock); a cap holds the rate until the SM clock falls under about 2,400 MHz (the search prior) | 1,860 at 50 to 60 percent / 2,400 at 80 percent / unlocked | the 4070 Ti Super at a 210 W host limit with the SM at 2,400 lost 10 percent on class v4 (36.95 against 41.24 MH/s): a cap that binds under the shadow costs rate on Ada too | +| Ada (4090, 4080, 4070 Ti, 4070, 4060 Ti, 4060, L40S) | the core lock with a cap | the 4070's tune: 1,860 MHz and the 50 percent cap hold 30.95 MH/s at 79.5 W on class v3 (28.7 at 102.7 stock); a cap holds the rate until the SM clock falls under about 2,400 MHz (the search prior); the 4090's class v4 premium at stock is 100 W (lane 1), under the 5090's 145 for 40 percent fewer counted ops per second | 1,860 at 50 to 60 percent / 2,400 at 80 percent / unlocked | the 4070 Ti Super at a 210 W host limit with the SM at 2,400 lost 10 percent on class v4 (36.95 against 41.24 MH/s): a cap that binds under the shadow costs rate on Ada too | | Ampere (3090, 3080, 3070, 3060, A4000 to A6000, A100) | the power cap only | a cap holds the rate until the SM falls under about 1,800 MHz; stock clocks sit at 1,900 to 2,000, so the lever is 10 to 15 percent of the draw; the 3060 Ti at a 130 W host cap held 33.06 MH/s at 128.7 W on class v4 | 80 percent with the lock at 1,800 as a ceiling / 90 percent / unlocked | the 3080 Ti at 63 percent with the SM at 749 MHz: 24 to 28 percent of rate lost | | Hopper (H100) | the core lock on an owned host | unmeasured: rented hosts refuse `-lgc`; the premium 283 W at stock (11.2 pJ per op) says half of it is the clock | 1,400 at 80 percent / 1,600 at 90 percent / unlocked, all estimated | none measured | | RDNA 4 (9070 XT; 9060 XT by shape) and RDNA 3 (7900 XTX, 7800 XT, 7600 XT by shape) | the ADLX core and power offsets (0.3.25); no memory knob; no elevation | the 9070 XT's grid: the rate flat at 18.9 MH/s across 24 rows, -500 MHz and -30 percent = 149.3 W against 202 | -500 MHz, -30 percent / -300 MHz, -15 percent / stock | Ember run 6 aborted on "card reports 0 W" (fixed bd7fcf4); the ADLX sampler window bug (fixed) | @@ -255,7 +258,7 @@ The re-measure rule after a class flip, as the table carries it (`remeasure_rule ## 6. The denominator sweep (rented, 8 October 13:05 onward, the stock rows the record lacked) -Two drivers of the fleet lane's model sweep (`box-cardbench-v2.sh` for class v3, the v5 kit's fingerprint and the +Three drivers of the fleet lane's model sweep (the third, from 13:21, on the coordinator's full-spend order: 16 card classes in parallel on `box-cardbench-v2m.sh`, the class v5 kit's bench run for 200 batches of 2^24 under the sampler so the class v5 watts are a measured mean, and a memory-clock try (`-lmc` to the card's maximum, a second class v5 run when the host accepts it)) (`box-cardbench-v2.sh` for class v3, the v5 kit's fingerprint and the memprobe ceiling; `box-cardbench-v4watts.sh` for the class v4 shape with the paired class v3 run), one one-shot pod per card class on RunPod or Vast under the USD 1.00 per hour consumer cap, destroyed at the end of each row; the rows appended to `~/igneum-fleet/cardbench/rows.jsonl` with `sweep: 2026-10-08-denominator`; spend inside the lane's USD 100. @@ -264,7 +267,8 @@ table of section 2 as they land; this section lists them. | Card | Class v3 (MH/s at W, microjoules) | Class v4 shape (MH/s at W, microjoules) | v4 over v3 watts | Host, driver, cost | Reading | |---|---|---|---|---|---| -| RTX 4090 24 GB | 70.09 MH/s (RunPod, driver 595.91, 13:07; the watts in the row) | pending | | USD 0.74 per hour, about 0.05 | the class v5 kit on another 4090 read 62.76 MH/s; the rate differs by host | +| RTX 4090 24 GB | 70.09 MH/s at 222.6 W (3.18; RunPod, driver 595.91, 13:07); class v5 kit 70.25 MATCH | pending | | USD 0.74 per hour, 0.03 | floor lane 1's rented 4090 read 70.63 at 253.4 W on the same class: the host spread of 12 percent on the watts at the same rate | +| RTX 5060 8 GB | 24.77 MH/s at 80.7 W (3.26; Vast, 13:11); class v5 kit 31.31 MATCH (a rate the v3 run on this host did not reach) | 16.23 MH/s at 87.8 W (5.41; a second Vast host, 13:10, v4 over v3 +20 percent on that host) | +20 percent | USD 0.02 | the rate varies by host 16 to 31 MH/s on this card class: the 7 October pod read 31.27 on class v3; the per-hash energy at the matched rate stands at 2.4 (v3) | | RTX 3090 24 GB | pending (two Vast hosts rented 13:05) | pending | | USD 0.14 per hour | | | RTX 3080 10 GB | pending | pending | | | the first Vast offer vanished at the rent (HTTP 410) | | RTX 5080 16 GB | the RunPod host failed `cuInit` (a bad host, destroyed); Vast next | pending | | | |