Merge class-v6-floor-denominator-docs 611c3975 into master (gate: green on 611c3975, recorded by tools/ci/pre-push.sh; landed on the box mirror under the exception declared by main: main's ruling, 7 Oct 2026 19:5x UK: the GitHub account is suspended, lanes land on the box mirror's master, the box gate stamp is the verdict; GitHub gets the fast-forward when it answers)

This commit is contained in:
igneum-labs 2026-10-08 13:49:29 +00:00
commit 2d69686ea3
4 changed files with 2322 additions and 0 deletions

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,100 @@
// The class v5 tiers table (app/igneum-app/tiers/class-v5-tiers.json): the loader and the validator the test and the
// app's packaging read it through. The rows are in the shape src/ember.rs tier_from_json reads back from a card's state
// (id, clock_mhz, power_pct, mem_mhz, limit_w, mhs, w), so a card whose own search has not run can be set to a tier
// from this table by /api/tune/tier, and the search replaces the row when it runs. node --test class-v5-tiers.test.mjs
import { readFileSync } from 'node:fs';
import { fileURLToPath } from 'node:url';
import { dirname, join } from 'node:path';
export const TIER_IDS = ['efficiency', 'balanced', 'max'];
export const LABELS = ['measured', 'estimated'];
export const SOURCES = ['measured', 'stock'];
export const VENDORS = ['nvidia', 'amd', 'apple', 'intel'];
// the fields tier_from_json reads, every one a number
export const ROW_FIELDS = ['clock_mhz', 'power_pct', 'mem_mhz', 'limit_w', 'mhs', 'w'];
// the brief's card classes (floor lane 4, 8 October 2026): every one must have an entry
export const REQUIRED_CARDS = ['5090', '5080', '5070 Ti', '5070', '5060 Ti', '5060', '4090', '4080', '4070', '3090', '3080', '3070', '3060',
'9070 XT', '7900 XTX', '7800 XT', '7600 XT', '9060 XT', 'H100', 'L40S', 'A100', 'B580', 'A750', 'M5 Max', 'M4 Max', 'M4 Pro', 'M3 Max'];
export function loadTable(path) {
const p = path || join(dirname(fileURLToPath(import.meta.url)), 'class-v5-tiers.json');
return JSON.parse(readFileSync(p, 'utf8'));
}
/** Every fault in the table as a list of strings; an empty list is a valid table. */
export function validate(t) {
const faults = [];
const f = (s) => faults.push(s);
if (t.class !== 'v5') f(`class is ${t.class}, not v5`);
if (!Array.isArray(t.tier_ids) || t.tier_ids.join() !== TIER_IDS.join()) f('tier_ids must be efficiency, balanced, max');
for (const k of ['stale_when', 'on_flip', 'measured_flip', 'period']) if (!t.remeasure_rule || typeof t.remeasure_rule[k] !== 'string' || !t.remeasure_rule[k]) f(`remeasure_rule.${k} missing`);
if (!t.v5_over_v4 || typeof t.v5_over_v4.watts_pct !== 'number') f('v5_over_v4.watts_pct missing');
if (!Array.isArray(t.cards) || t.cards.length === 0) { f('cards missing'); return faults; }
const seen = new Set();
for (const c of t.cards) {
const name = c.card || '(unnamed)';
if (seen.has(name)) f(`${name}: listed twice`);
seen.add(name);
if (!VENDORS.includes(c.vendor)) f(`${name}: vendor ${c.vendor}`);
if (!LABELS.includes(c.label)) f(`${name}: label ${c.label}`);
if (!Array.isArray(c.match) || c.match.length === 0) f(`${name}: no match list`);
if (typeof c.src !== 'string' || !c.src) f(`${name}: no src`);
if (!c.stock || typeof c.stock.uj !== 'number' || c.stock.uj <= 0) f(`${name}: stock.uj missing`);
if (!Array.isArray(c.tiers) || c.tiers.length === 0) { f(`${name}: no tiers`); continue; }
const ids = c.tiers.map((r) => r.id);
for (const id of ids) if (!TIER_IDS.includes(id)) f(`${name}: tier id ${id}`);
if (new Set(ids).size !== ids.length) f(`${name}: a tier id repeats`);
const lever = c.vendor === 'nvidia' || c.vendor === 'amd';
if (lever && ids.join() !== TIER_IDS.join()) f(`${name}: a card with a lever carries all three tiers in order`);
if (!lever && ids.join() !== 'max') f(`${name}: a card with no lever carries the max tier only`);
for (const r of c.tiers) {
for (const k of ROW_FIELDS) if (typeof r[k] !== 'number' || !Number.isFinite(r[k])) f(`${name}/${r.id}: ${k} is not a number`);
if (typeof r.power_pct === 'number' && (r.power_pct < 50 || r.power_pct > 100)) f(`${name}/${r.id}: power_pct ${r.power_pct} outside 50 to 100`);
if (typeof r.clock_mhz === 'number' && (r.clock_mhz < 0 || r.clock_mhz > 4000)) f(`${name}/${r.id}: clock_mhz ${r.clock_mhz}`);
if (!LABELS.includes(r.label)) f(`${name}/${r.id}: label ${r.label}`);
if (!SOURCES.includes(r.source)) f(`${name}/${r.id}: source ${r.source}`);
if (typeof r.uj !== 'number' || r.uj <= 0) f(`${name}/${r.id}: uj missing`);
if (typeof r.note !== 'string' || !r.note) f(`${name}/${r.id}: no note`);
if (r.mhs > 0 && r.w > 0 && r.uj > 0 && Math.abs(r.w / r.mhs - r.uj) / r.uj > 0.025) f(`${name}/${r.id}: uj ${r.uj} is not w over mhs (${(r.w / r.mhs).toFixed(2)})`);
if (r.mhs > 0 && r.w > 0 && typeof r.mhw === 'number' && Math.abs(r.mhs / r.w - r.mhw) / r.mhw > 0.025) f(`${name}/${r.id}: mhw ${r.mhw} is not mhs over w`);
if (r.id === 'max' && r.source !== 'stock') f(`${name}/max: the max tier is the stock row`);
if (r.id === 'max' && (r.clock_mhz !== 0 || r.power_pct !== 100)) f(`${name}/max: stock is unlocked at 100 percent`);
if (c.vendor === 'amd' && (typeof r.core_offset_mhz !== 'number' || typeof r.power_offset_pct !== 'number')) f(`${name}/${r.id}: an AMD row carries core_offset_mhz and power_offset_pct`);
if (c.vendor === 'amd' && r.core_offset_mhz > 0) f(`${name}/${r.id}: an AMD core offset never raises the clock`);
}
const by = Object.fromEntries(c.tiers.map((r) => [r.id, r]));
if (by.efficiency && by.max && by.efficiency.uj > by.max.uj) f(`${name}: efficiency costs more per hash than stock`);
if (by.efficiency && by.balanced && by.balanced.uj + 1e-9 < by.efficiency.uj) f(`${name}: balanced is cheaper per hash than efficiency`);
if (by.balanced && by.max && by.balanced.mhs > 0 && by.max.mhs > 0 && by.balanced.mhs < by.max.mhs * 0.98) f(`${name}: balanced gives up over 2 percent of the top rate`);
if (c.label === 'measured' && !c.tiers.some((r) => r.label === 'measured' && r.source === 'measured') && lever) f(`${name}: a measured card has no measured tuned row`);
}
for (const want of REQUIRED_CARDS) if (!t.cards.some((c) => c.card.includes(want))) f(`no entry for ${want}`);
return faults;
}
/** The entry a card name matches (the first whose match list hits; the longer match wins, so "5070 Ti" beats "5070"). */
export function entryFor(t, cardName) {
const n = cardName.toUpperCase();
let best = null, bestLen = 0;
for (const c of t.cards) for (const m of c.match) if (n.includes(m.toUpperCase()) && m.length > bestLen) { best = c; bestLen = m.length; }
return best;
}
/** The tier row a card starts from, in the shape tier_from_json reads; null when the table has none. */
export function tierRow(t, cardName, id) {
const c = entryFor(t, cardName);
if (!c) return null;
return c.tiers.find((r) => r.id === id) || null;
}
/** The re-measure verdict the app applies on a class flip (src/ember.rs tiers_stale): stale when both classes are known and differ. */
export function stale(tiersClass, programClass) {
return Boolean(tiersClass) && Boolean(programClass) && tiersClass !== programClass;
}
if (process.argv[1] && fileURLToPath(import.meta.url) === process.argv[1]) {
const t = loadTable();
const faults = validate(t);
if (faults.length) { console.error(faults.join('\n')); process.exit(1); }
console.log(`class ${t.class}: ${t.cards.length} card classes, ${t.cards.filter((c) => c.label === 'measured').length} measured`);
}

View file

@ -0,0 +1,109 @@
// node --test app/igneum-app/tiers/class-v5-tiers.test.mjs
// The class v5 tiers table: the shipped file validates; the measured rows carry the record's numbers; the match rule
// picks the longer name; a class flip reads stale; and the known-failed cases (a bad power rung, a missing field, a
// tuned row dearer than stock, a card class left out) are refused, so a stale or broken table cannot pass.
import { test } from 'node:test';
import assert from 'node:assert/strict';
import { loadTable, validate, entryFor, tierRow, stale, REQUIRED_CARDS, ROW_FIELDS } from './class-v5-tiers.mjs';
const clone = (t) => JSON.parse(JSON.stringify(t));
test('the shipped table validates with no faults', () => {
const t = loadTable();
assert.deepEqual(validate(t), []);
assert.equal(t.class, 'v5');
assert.ok(t.cards.length >= REQUIRED_CARDS.length);
});
test('the measured rows are the record\'s (the 5090 at its 1,300 MHz knee, the 5080 at 1,100, the 4070 at its tune, the 9070 XT grid, the M5 Max meter)', () => {
const t = loadTable();
const r5090 = tierRow(t, 'NVIDIA GeForce RTX 5090', 'balanced');
assert.equal(r5090.clock_mhz, 1300);
assert.equal(r5090.label, 'measured');
assert.ok(Math.abs(r5090.uj - 2.38) < 0.01, `the 5090 balanced row reads 2.38 microjoules under class v5: ${r5090.uj}`);
const e5090 = tierRow(t, 'NVIDIA GeForce RTX 5090', 'efficiency');
assert.equal(e5090.clock_mhz, 1200);
const r5080 = tierRow(t, 'NVIDIA GeForce RTX 5080', 'efficiency');
assert.equal(r5080.clock_mhz, 1100);
assert.ok(Math.abs(r5080.uj - 2.06) < 0.01);
const r4070 = tierRow(t, 'NVIDIA GeForce RTX 4070', 'efficiency');
assert.equal(r4070.clock_mhz, 1860);
assert.equal(r4070.power_pct, 50);
const amd = tierRow(t, 'AMD Radeon RX 9070 XT', 'efficiency');
assert.equal(amd.core_offset_mhz, -500);
assert.equal(amd.power_offset_pct, -30);
assert.ok(Math.abs(amd.w - 149.3) < 0.01);
const apple = entryFor(t, 'Apple M5 Max');
assert.equal(apple.tiers.length, 1);
assert.equal(apple.tiers[0].id, 'max');
assert.ok(Math.abs(apple.tiers[0].uj - 1.40) < 0.01);
});
test('every entry carries the three tiers where the card has a lever, and only max where it has none', () => {
const t = loadTable();
for (const c of t.cards) {
const ids = c.tiers.map((r) => r.id).join();
if (c.vendor === 'nvidia' || c.vendor === 'amd') assert.equal(ids, 'efficiency,balanced,max', c.card);
else assert.equal(ids, 'max', c.card);
for (const r of c.tiers) for (const k of ROW_FIELDS) assert.equal(typeof r[k], 'number', `${c.card}/${r.id}.${k}`);
}
});
test('the match rule takes the longer name: a 5070 Ti is not a 5070, a 4060 Ti is not a 4060, a 9060 XT is not a 9070 XT', () => {
const t = loadTable();
assert.equal(entryFor(t, 'NVIDIA GeForce RTX 5070 Ti').card, 'NVIDIA GeForce RTX 5070 Ti');
assert.equal(entryFor(t, 'NVIDIA GeForce RTX 5070').card, 'NVIDIA GeForce RTX 5070');
assert.equal(entryFor(t, 'NVIDIA GeForce RTX 4060 Ti').card, 'NVIDIA GeForce RTX 4060 Ti');
assert.equal(entryFor(t, 'NVIDIA GeForce RTX 4060').card, 'NVIDIA GeForce RTX 4060');
assert.equal(entryFor(t, 'AMD Radeon RX 9060 XT').card, 'AMD Radeon RX 9060 XT');
assert.equal(entryFor(t, 'amd:gfx1201').card, 'AMD Radeon RX 9070 XT');
assert.equal(entryFor(t, 'NVIDIA GeForce GTX 1080 Ti'), null);
assert.equal(tierRow(t, 'NVIDIA GeForce GTX 1080 Ti', 'max'), null);
});
test('a class flip reads stale exactly as src/ember.rs tiers_stale does', () => {
assert.equal(stale('v4', 'v5'), true);
assert.equal(stale('v5', 'v5'), false);
assert.equal(stale('', 'v5'), false);
assert.equal(stale('v4', ''), false);
const t = loadTable();
assert.ok(t.remeasure_rule.measured_flip.includes('0.0 percent of rate'));
assert.equal(t.v5_over_v4.watts_pct, 2.0);
});
test('known-failed: a power rung under 50 is refused', () => {
const t = clone(loadTable());
t.cards[0].tiers[0].power_pct = 40;
assert.ok(validate(t).some((s) => s.includes('power_pct 40')));
});
test('known-failed: a row missing a field tier_from_json reads is refused', () => {
const t = clone(loadTable());
delete t.cards[1].tiers[1].limit_w;
assert.ok(validate(t).some((s) => s.includes('limit_w is not a number')));
});
test('known-failed: a tuned row dearer per hash than stock is refused, and a max tier that is not stock', () => {
const t = clone(loadTable());
const c = t.cards.find((x) => x.card.includes('4080'));
c.tiers[0].uj = c.tiers[2].uj + 1; c.tiers[0].w = c.tiers[0].uj * c.tiers[0].mhs; c.tiers[0].mhw = c.tiers[0].mhs / c.tiers[0].w;
assert.ok(validate(t).some((s) => s.includes('efficiency costs more per hash than stock')));
const u = clone(loadTable());
u.cards[0].tiers[2].clock_mhz = 1500;
assert.ok(validate(u).some((s) => s.includes('stock is unlocked at 100 percent')));
});
test('known-failed: a card class of the brief left out is refused, and a stale uj (not w over mhs) is refused', () => {
const t = clone(loadTable());
t.cards = t.cards.filter((c) => !c.card.includes('3060'));
assert.ok(validate(t).some((s) => s === 'no entry for 3060'));
const u = clone(loadTable());
u.cards[0].tiers[1].uj = 9.99;
assert.ok(validate(u).some((s) => s.includes('is not w over mhs')));
});
test('known-failed: a table under another class is refused', () => {
const t = clone(loadTable());
t.class = 'v4';
assert.ok(validate(t).some((s) => s.includes('not v5')));
});

View file

@ -0,0 +1,386 @@
# Class v6 floor lane 4: the honest denominator (8 October 2026)
Branch `class-v6-floor-denominator` from `counter-asic-4` (91117093). The question: every chip edge in the record is a
ratio over the honest tier's microjoules per hash, and the honest tier's own operating point is half the resistance,
the half the project controls through Ember's tuning. This file finds how low the honest joule goes per card class
today with software alone, under class v5, and reads the chip edge against each row; models the LPDDR6 and
unified-memory tier five years out; and lands the Ember tiers statement as a table the app reads
(`app/igneum-app/tiers/class-v5-tiers.json`, its validator and test beside it). Every figure is labelled measured (with
its source) or estimated (with its method). Nothing here moves a consensus object or a served text. Times are UK (BST).
## 0. One page
1. **The honest floor under class v5, measured today:** the RTX 5080 at an 1,100 MHz core lock reads 2.06 microjoules
per hash on class v4's shape (71.20 MH/s at 146.6 W) and about 2.10 under class v5; the RTX 5090 at its 1,300 MHz
knee 2.37 (134.76 MH/s at 318.8 W, the v5lock job's +2.0 percent over class v4 applied); the RTX 4070 at its tune
point 3.51 to 3.58; the RX 9070 XT at the ADLX grid's best point 7.9 to 8.1; the Apple M5 Max 1.40 to 1.43 at the
GPU-plus-DRAM meter with no lever. The class v5 STOCK column is now measured on 14 card classes (the rented sweep,
13:4x to 14:0x UK, every host refusing the lock): the 5080 3.48, the 4070 5.82, the 4080 4.92, the 5060 3.76, the
5060 Ti 4.02, the 3090 5.03, the 3060 6.40, the L40S 5.29, the H100 2.58, the A100 2.99, the 4090 5.0 (lane 1), so
the lock is worth 34 to 41 percent of a Blackwell or Ada card's class v5 draw (the 5080 3.48 to 2.10, the 4070 5.82
to 3.58) for under 2 percent of rate, and the stock column is a different card from the floor column. The 5080
beats the 5090 per joule at the knee, and the 16 GB and 12 GB Blackwell cards (5070 Ti, 5070) read lower still by
the Blackwell shape (1.7 to 2.1, estimated): **the honest NVIDIA floor is the mass-market Blackwell card at its
knee, not the 5090.**
2. **The chip edge against the honest floor, class v5, the chip's shadow core priced in absolute terms** (3.2 pJ per
forced op, the record's k = 0.5 at the 5090's knee; and 6.4 pJ, k = 1): the GDDR7 stored-dataset board reads 2.7x /
1.9x against the 5080's floor, 3.0x / 2.1x against the 5090's, 1.8x / 1.3x against the M5 Max; one HBM3 stack 3.3x /
2.2x, 3.6x / 2.4x, 2.2x / 1.5x; the N2 SRAM die 4.5x / 2.7x, 5.0x / 3.0x, 3.1x / 1.8x. The served 2.1x (the 5090 at
the knee, k = 1) stands; the honest sentence per joule is now "under 2x against a 16 GB Blackwell card at its knee
for a core as good as a GPU lane, 1.3x against an Apple laptop".
3. **What moves the floor is the operating point and nothing else in software.** The SM-sparse kernel is dead on
measured rows (the draw follows the work, not the SM count; research file 20.3b; floor lane 1's rows of 13:25 on a
rented 4090 and H100 say the same at stock: the best occupancy shape saves 1 to 2 percent of the watts and the draw
over idle is flat across the SM count while the rate holds, so the residual is the clock domain, reachable by the
lock and the undervolt only); the cache-policy hint is dead; the
memory clock is never touched (a step that drags it down cannot win); the power cap is flat on Blackwell at the hash
(0.41 MH/W under every limit on the 5090) and is the only lever on Ampere; the core lock walks the driver's V/F
curve and is the lever on Blackwell and Ada; AMD has the ADLX offsets (measured today: 24 percent of the 9070 XT's
draw for no rate); Apple and Intel have no lever in the app.
4. **Five years out the honest tier is the unified-memory SoC on LPDDR6**, modelled at about 1.2 microjoules per hash
under class v5 at the meter for an Apple Max-class part (0.27 of DRAM reads, 0.35 of GPU issue, 0.56 of shadow), 1.6
to 1.7 at the package; the AMD, Qualcomm, MediaTek and NVIDIA LPDDR SoCs model 3 to 10 microjoules because their
GPUs hide the latency worse, not because of the memory. Against that tier the GDDR7 chip reads 1.5x / 1.06x, one
HBM3 stack 1.8x / 1.2x, the N2 SRAM die 2.5x / 1.5x, the custom HBM4E base die 2.4x / 1.4x. The class should size the
shadow per tier with the Apple tier as the binding ceiling (it already does by the 2.0 rule: N at or under 130,000
ops) and state resistance against the best honest joule; it should NOT score the acceptance floor per tier (one
network-wide number, the same for every honest card). Pricing the Apple-bound shadow for the 5090 tier: N = 130,000
instead of 100,000 costs the 5090 +0.19 microjoules at the knee (+8 percent of its electricity) and moves its GDDR7
edge at k = 1 from 2.1x to 1.95x.
5. **The tiers for Ember** (`app/igneum-app/tiers/class-v5-tiers.json`): per card class the Efficiency, Balanced and Max
rows in the app's own tier shape, measured on the 5090, 5080, 4070, 9070 XT and M5 Max, estimated from the measured
stock rows and the architecture's shape elsewhere (Blackwell: lock 1,100 to 1,300 MHz, power 100 percent; Ada: lock
1,860 to 2,400 and 50 to 80 percent; Ampere: 80 to 90 percent and no lock below 1,800 MHz; AMD: -500 MHz and -30
percent; Apple and Intel: stock). The re-measure rule: a set measured under one class never applies under another
(the app's `tiers_stale`); the stored point is the provisional start of the re-measure, the Max tier is stock and never
stale; the measured v4 to v5 flip on the 5090 moved the knee by nothing (0.0 percent of rate, +2.0 percent of watts).
## 1. The method and the denominators
| Quantity | Value | Label | Source |
|---|---|---|---|
| Class v5 | `mx8-era<seed>+sh256x27+state`: class v4's shape (64-instruction base, a 256-instruction shadow block x27, 102,100 counted ops per hash) plus the state leaves; the worker uploads the leaves for the dataset build and frees them | measured (the kit's RESULT line on a rented 4090 and 23 other cards: every v5 fingerprint MATCH) | `~/igneum-fleet/cv5-results.txt`; `~/igneum-fleet/cardbench/rows.jsonl` 8 October |
| Class v5 over class v4 at the same point | 0.0 percent of rate, +2.0 percent of watts (the 5090 at the 1,300 MHz lock: 125.93 MH/s at 299.8 W against 125.92 at 294.0) | measured | the hash lane's v5lock job, 8 October; `scratch v4/ca4-v6-cost-rows.md` |
| The 5090 at class v4, the core-clock grid | unlocked 136.84 MH/s at 475.5 W; 1,300 MHz 134.76 at 312.5 (the knee, rate -1.5 percent); 1,200 MHz 133.80 at 305.1 (the best MH per watt, -2.2 percent); 1,100 MHz 122.43 at 287.3 (-10.5 percent); the class v4 premium 145.3 W unlocked, 81.8 W at the best points | measured, PC 1, 7 to 8 October | `docs/bench-log.md`, the class v4 efficiency passes |
| The 5080 at class v4, the core-clock grid | unlocked 71.41 at 253.1 W; 1,100 MHz 71.20 at 146.6 (the best MH per watt, -0.3 percent); 1,000 MHz 71.19 at 149.0; 900 MHz 67.66 at 137.8 (-5.2 percent); the draw floors from 1,500 MHz down; the premium 83.4 W unlocked and 41 W at the best points | measured, PC 1, 8 October 00:45 to 01:54 UTC | the same |
| The 4070 at its Ember tune (1,860 MHz lock, the 50 percent cap) | class v3 30.95 MH/s at 79.5 W (2.57); class v4 31.08 at 109.0 (3.51): +30 W for no rate; stock class v3 28.7 at 102.7 (3.58) | measured, PC 1, 6 October | `docs/plans/counter-asic-3-status.md` item 8; Ember run 6 |
| The 9070 XT on the ADLX grid | 24 rows, the rate flat at 18.9 MH/s, the best point -500 MHz core and -30 percent power = 149.3 W (7.9), 24 percent under the 202 W stock point (10.7) | measured, PC 1, 8 October 12:13 | the hash lane's AMD grid job |
| The M5 Max | class v3 27.08 MH/s at 21.0 W (0.78), class v4 26.67 at 37.3 W (1.40) at the GPU and DRAM channels of the IOReport meter; the package about 17 W more; marginal 6.9 pJ per counted op | measured, 6 October | `docs/analysis/latency-shadow-2026-10-06.md` section 3 |
| The rented stock rows (class v3 and the class v4 shape, 24 card classes) | `rows.jsonl`, 7 and 8 October: three timed runs, an nvidia-smi sampler at 1 Hz, the three power fields; no host allowed `-lgc` ("The current user does not have permission to change clocks") so no rented row carries a knee | measured (stock only) | `~/igneum-fleet/cardbench/rows.jsonl`; `bench-table.md` |
| The rented against the PC spread | the 5080 read 143.4 W at 71.16 MH/s on a rented Linux host and 169.7 W at 71.28 on PC 1 (Windows) on class v3: 18 percent apart at the same rate; the 5090 the same shape (258 W rented at 0.3.12 against 312 W on the PC) | measured; neither corrected | `site/miner-bench.json` notes |
| The chips at zero shadow | GDDR7 board 0.466 microjoules per hash; one HBM3 stack 0.321; the N2 SRAM die 0.14 | modelled | chip-model-v3 5.4 and 5.12; `class-v6/hardware-future.md` section 5 |
| The chip's shadow under class v5 | 101,170 forced ops per hash (102,100 minus the 930 base) at 3.2 pJ (k = 0.5 against the 5090's measured 6.4 pJ per op at the knee: the record's centre band) = 0.324 microjoules; at 6.4 pJ (k = 1) = 0.648 | modelled on the measured GPU rows | research file 15.1a, 20.4 |
| The chips under class v5 | GDDR7 0.79 / 1.11; HBM3 0.65 / 0.97; N2 SRAM 0.46 / 0.79 (k = 0.5 / k = 1) | modelled | this file |
Reading the identity. The chip's energy is `E_mem + c x N` with `c` the chip core's picojoules per forced op and `N`
the op count, a number that does not depend on which card is the reference; so the chip's microjoules per hash are
fixed by the class (0.79 to 1.11 on GDDR7 under class v5) and every row's edge is just that card's microjoules over
them. The record's `k` columns are this `c` divided by the reference card's own picojoules per op: k = 1 at the 5090's
knee is 6.4 pJ, k = 1 at the unlocked 5090 would be 10.8 pJ, k = 1 on the M5 Max 6.9 pJ. Stating the chip in absolute
picojoules keeps the comparison honest across cards.
What "every software knob" is, per vendor. NVIDIA through nvidia-smi: `-pl` (the power cap, percent of the default
limit, floor 50 percent, min 400 W on a 5090), `-lgc 0,<MHz>` (a core-clock cap that walks the driver's V/F curve;
administrator rights, so only through the Power Helper on Windows and never on a rented Linux host), `-lmc` (the
memory clock, never touched: a step whose memory clock falls under 95 percent of baseline is marked and cannot win);
no voltage offset (that is NVAPI's). AMD through the ADLX line in igneum-gpu-telemetry: `--set-gmax` (a core offset)
and `--set-plimit` (a power offset), no memory knob on RDNA 4, no elevation. Apple: measure only. Intel: measure only.
The kernel-side knobs the record closed: the SM-sparse variant (dead: the energy per hash never falls below base on
either class, 20.3b), `ld.global.cs` (dead: equals base within 1 W, 20.3c), the block shape (64 to 256 instructions,
fixed by the class), the occupancy (one warp per block, 24 blocks per SM, fixed by the worker's race).
## 2. The table: every card class under class v5, the lowest microjoules per hash with the knobs, and the chip edge
"Floor" is the lowest class v5 energy per hash the card reaches with the knobs available to it today; "stock" is the
card unlocked at 100 percent under class v5 (the sweep's measured rows, 13:4x to 14:0x UK, say "stock, lock owed" in
the table: no rented host allowed the lock). A measured row cites its job; an estimated row names its method and its
band. The chip columns divide the row's floor by the chip's class v5 energy at three core figures: 1.1 pJ per forced op
(the k lane's synthesised sequencer-core floor at N3, the pessimistic column), 3.2 pJ (the record's k = 0.5) and 6.4
pJ (k = 1, a core as good as a GPU lane); `E_mem` 0.466, 0.321 and 0.14 microjoules. The Apple rows are at the GPU-plus-DRAM meter (the package adds about 17 W on the M5
Max; the wall more); the NVIDIA and AMD rows are whole-card.
| Card | Tier | Class v5 floor, microjoules per hash | Label | Class v5 stock, microjoules per hash (label) | GDDR7 board at 1.1 / 3.2 / 6.4 pJ per forced op | HBM3 one stack | N2 SRAM die | The point, the band and the source |
|---|---|---|---|---|---|---|---|---|
| NVIDIA GeForce RTX 5090 | 32 GB | 2.33 | measured | 3.48 (measured) | 4.0x / 3.0x / 2.1x | 5.4x / 3.6x / 2.4x | 9.3x / 5.0x / 3.0x | 1200 MHz at 100 percent. class v4 best MH per watt (133.80 MH/s at 305.1 W, 2.2 percent of rate given); under class v5 +2.0 percent watts: 311.2 W, 2.33 Stock: the power cap is flat on this hash (0.41 MH/W under every limit), so no power rung is used on any tier |
| NVIDIA GeForce RTX 5080 | 16 GB | 2.06 | measured | 3.48 (measured) | 3.6x / 2.6x / 1.9x | 4.8x / 3.2x / 2.1x | 8.2x / 4.4x / 2.6x | 1100 MHz at 100 percent. class v4 best MH per watt; the draw floors from 1,500 MHz down; the knee is between 1,000 and 900 MHz (900 costs 5.2 percent); class v5 about 2.10 Stock: class v5 measured on a Vast 5080 (13:49 UK): 71.35 MH/s at 248.0 W (3.48, 39 samples, SM 2,769, memory 14,801); class v3 71.16 at 159.4 W on the same host; PC 1 read class v4 71.41 at 253.1 W: the three agree within 3 percent on this card; stock, lock owed on the rented host |
| NVIDIA GeForce RTX 5070 Ti | 16 GB | 1.7 | estimated | 2.84 (measured) | 2.9x / 2.2x / 1.5x | 3.9x / 2.6x / 1.8x | 6.8x / 3.7x / 2.2x | 1100 MHz at 100 percent. the Blackwell shape applied to the measured stock rows (the v3 draw x0.63, the premium x0.5): band 1.6 to 2.0; the rented host refused -lgc, so the knee is the search's to find Stock: rented pod 8 October 10:09Z: class v3 78.69 MH/s at 140.8 W, class v4 78.78 at 224.0 W; the class v5 sweep rented this card twice (14:08 and 14:13 UK) and neither host ever accepted the fleet key |
| NVIDIA GeForce RTX 5070 | 12 GB | 1.75 | estimated | 2.99 (measured) | 3.0x / 2.2x / 1.6x | 4.0x / 2.7x / 1.8x | 7.0x / 3.8x / 2.2x | 1100 MHz at 100 percent. the Blackwell shape on the measured stock rows (paired class v3 113 W): band 1.7 to 2.1 Stock: rented v4watts pod 8 October; the 7 October pod read class v3 52.0 MH/s at 102.8 W (the rate differs by host); the class v5 sweep rented this card twice (14:08 and 14:13 UK) and neither host ever accepted the fleet key |
| NVIDIA GeForce RTX 5060 Ti | 16 GB | 2.36 | estimated | 4.02 (measured) | 4.1x / 3.0x / 2.1x | 5.5x / 3.7x / 2.4x | 9.4x / 5.1x / 3.0x | 1100 MHz at 100 percent. the Blackwell shape on the PC 2 stock row: band 2.2 to 2.6; re-based on the measured class v5 stock row of 13:5x UK (the architecture's shape on the measured v3 draw and the measured premium) Stock: class v5 measured on a Vast 5060 Ti (13:5x UK): 30.84 MH/s at 123.9 W (4.02, 96 samples, SM 2,835); class v3 30.75 at 82.6 W on the same host; PC 2's enclosure row read class v4 30.9 at 114.8 W; stock, lock owed |
| NVIDIA GeForce RTX 5060 | 8 GB | 2.21 | estimated | 3.76 (measured) | 3.8x / 2.8x / 2.0x | 5.1x / 3.4x / 2.3x | 8.8x / 4.8x / 2.8x | 1100 MHz at 100 percent. class v3 measured 31.27 MH/s at 75.4 W; re-based on the measured class v5 stock row of 13:5x UK (the architecture's shape on the measured v3 draw and the measured premium) Stock: class v5 measured (13:51 UK): 30.80 MH/s at 115.7 W (3.76, 99 samples, SM 2,715, memory 13,801); class v3 30.76 at 78.8 W on the same host (the shadow +37 W = 12 pJ per op); stock, lock owed |
| NVIDIA GeForce RTX 4090 | 24 GB | 3.58 | estimated | 5.0 (measured) | 6.2x / 4.5x / 3.2x | 8.3x / 5.6x / 3.7x | 14.2x / 7.7x / 4.5x | 1860 MHz at 60 percent. the Ada shape (the 4070's measured tune: v3 x0.72, the premium x0.7) on the measured stock rows (class v3 253.4 W, class v4 353.5 W, floor lane 1, 8 October): band 3.3 to 4.0; re-based on the measured class v5 stock row of 13:5x UK (the architecture's shape on the measured v3 draw and the measured premium) Stock: floor lane 1's rented 4090 at a 450 W limit: class v3 70.63 MH/s at 253.4 W (3.59), class v4 70.63 at 353.5 W (5.0); the denominator sweep's Vast 4090 carried a 250 W host limit: class v5 62.68 at 249.9 W (3.99, the SM at 2,579 MHz) and class v3 62.40 at 200.5 W, so a 250 W cap held 89 percent of the rate at 71 percent of the draw (an Ada cap row measured); stock, lock owed |
| NVIDIA GeForce RTX 4080 | 16 GB | 3.51 | estimated | 4.92 (measured) | 6.1x / 4.4x / 3.2x | 8.1x / 5.4x / 3.6x | 14.0x / 7.6x / 4.5x | 1860 MHz at 60 percent. the Ada shape on the measured stock rows (class v3 128.6 W, class v4 185.6 W): band 3.1 to 3.7; re-based on the measured class v5 stock row of 13:5x UK (the architecture's shape on the measured v3 draw and the measured premium) Stock: class v5 measured (13:50 UK): 40.80 MH/s at 200.7 W (4.92, 72 samples, SM 2,760); class v3 40.69 at 141.5 W on the same host; the v4watts pod read class v4 40.74 at 185.6 W; stock, lock owed |
| NVIDIA GeForce RTX 4070 | 12 GB | 3.58 | measured | 5.82 (measured) | 6.2x / 4.5x / 3.2x | 8.3x / 5.6x / 3.7x | 14.2x / 7.7x / 4.5x | 1860 MHz at 50 percent. Ember run 6's tune point (1,863 MHz at 50 percent) under class v4 (sh256x27): 31.08 MH/s at 109.0 W, +30 W over class v3 at the same point for no rate; class v5 about 3.58; the ladder below 1,860 under class v4 is owed (band 3.2 to 3.6) Stock: class v5 measured on a Vast 4070 (13:54 UK): 28.43 MH/s at 165.5 W (5.82, 108 samples, SM 2,820, memory 9,801, a 210 W host limit not binding); class v3 28.40 at 116.3 W on the same host (the shadow +49 W = 17 pJ per op at the stock clock); Ember run 6's PC 1 stock on class v3 read 28.72 at 106.0 W; the tune point (1,860 MHz, 50 percent) under class v4 measured 31.08 at 109.0 W (3.51), so the lock takes 34 percent of this card's class v5 draw for no rate |
| NVIDIA GeForce RTX 4070 Ti | 12 GB | 3.6 | estimated | 4.97 (measured) | 6.2x / 4.6x / 3.2x | 8.3x / 5.6x / 3.7x | 14.3x / 7.8x / 4.6x | 1860 MHz at 60 percent. the Ada shape on the measured stock rows (class v3 95.4 W, class v4 155.3 W); the 4070 Super at a 110 W host cap held 31.26 MH/s at 108.4 W under class v4 (3.47): a cap row measured on a sibling Stock: rented pods 8 October |
| NVIDIA GeForce RTX 4060 Ti | 16 GB | 3.81 | estimated | 5.32 (measured) | 6.6x / 4.8x / 3.4x | 8.8x / 5.9x / 3.9x | 15.2x / 8.2x / 4.8x | 1860 MHz at 60 percent. the Ada shape (the 4070's measured tune: v3 x0.72, the premium x0.7) on the measured class v5 stock row; band 3.5 to 4.2 Stock: class v5 measured on a Vast 4060 Ti (14:27 UK): 18.75 MH/s at 99.8 W (5.32) under a 140 W host limit; class v3 18.74 at 79.1 W on the same host (the shadow +21 W = 11 pJ per op; this host read 7 percent under the 7 October pod's 20.10 MH/s); the v4watts pod read class v4 20.10 at 102.3 W (5.09); stock, lock owed |
| NVIDIA GeForce RTX 4060 | 8 GB | 3.77 | estimated | 4.98 (estimated) | 6.5x / 4.8x / 3.4x | 8.7x / 5.8x / 3.9x | 15.0x / 8.1x / 4.8x | 1860 MHz at 60 percent. 19.09 MH/s measured on three rented hosts, the watts unread on every one (no power sensor): the watts are the 4060 Ti's shape; band 3.7 to 4.2 Stock: watts OWED: every rented 4060 host exposed no power sensor |
| NVIDIA GeForce RTX 3090 | 24 GB | 4.51 | estimated | 5.03 (measured) | 7.8x / 5.7x / 4.1x | 10.4x / 7.0x / 4.7x | 17.9x / 9.7x / 5.7x | 1800 MHz at 80 percent. the Ampere cap lever (10 to 15 percent of the draw; a cap that takes the SM under 1,800 MHz costs rate) on the measured class v5 stock row; the 3090 Ti read class v3 61.95 MH/s at 249.5 W (4.03) Stock: the denominator sweep, RunPod, 8 October 13:25 UK: class v3 60.01 MH/s at 294.6 W (4.91), class v5 61.49 at 309.4 W (5.03, 46 samples, SM 1,673 MHz, memory 9,501); the class v5 premium on this card 5 percent at stock; stock, lock owed (the host refused -lgc and -lmc) |
| NVIDIA GeForce RTX 3080 | 10 GB | 4.2 | estimated | 4.54 (estimated) | 7.3x / 5.3x / 3.8x | 9.7x / 6.5x / 4.3x | 16.7x / 9.1x / 5.3x | 1800 MHz at 80 percent. the Ampere cap lever (10 percent, no lower: a 170 W cap took 34 percent of the class v5 rate on this card, a 180 W cap 11 percent) Stock: the sweep's two 3080 hosts both carried a limit under the class v5 draw: at 170 W class v3 50.71 MH/s at 169.2 W (3.34) held the rate and class v5 fell to 33.37 MH/s at 169.9 W (5.09) with the SM at 689 MHz; at 180 W class v3 49.83 at 177.8 W and class v5 44.31; the uncapped class v5 stock by the 3080 Ti's +30 percent: about 230 W (4.54); the Ampere cap under the shadow costs rate one for one: measured twice |
| NVIDIA GeForce RTX 3070 | 8 GB | 4.8 | estimated | 5.29 (measured) | 8.3x / 6.1x / 4.3x | 11.1x / 7.4x / 5.0x | 19.1x / 10.4x / 6.1x | 1800 MHz at 80 percent. the Ampere cap lever on the measured stock rows (class v3 142.3 W, class v4 196.4 W); the 3070 Ti 38.98 MH/s at 177.1 / 267.6 W Stock: rented pods 8 October; the class v5 sweep's two 3070 hosts closed the connection mid-bench |
| NVIDIA GeForce RTX 3060 | 12 GB | 5.77 | estimated | 6.4 (measured) | 10.0x / 7.3x / 5.2x | 13.3x / 8.9x / 6.0x | 23.0x / 12.4x / 7.3x | 1800 MHz at 80 percent. the Ampere cap lever (10 percent) on the measured class v4 stock row; the 3060 Ti at a 130 W host cap held 33.06 MH/s at 128.7 W under class v4 (3.89): a cap row measured on a sibling Stock: class v5 measured on a Vast 3060 (13:53 UK): 26.53 MH/s at 169.8 W (6.40) at the host's 170 W limit (the limit binding: the class v4 pod read 26.89 at 166.1 W, 6.18); class v3 26.53 at 120.4 W on the same host; stock, lock owed |
| AMD Radeon RX 9070 XT | 16 GB | 7.9 | measured | 10.7 (measured) | 13.7x / 10.0x / 7.1x | 18.3x / 12.3x / 8.2x | 31.4x / 17.0x / 10.0x | ADLX -500 MHz, -30 percent. the ADLX grid (24 rows, PC 1): the rate flat at 18.9 MH/s across the grid, the best point -500 MHz core and -30 percent power, 24 percent under stock; class v5 about 8.1 Stock: the app's own power reading at the stock point, 8 October 07:11 UK |
| AMD Radeon RX 7900 XTX | 24 GB | 8.0 | estimated | 10.7 (estimated) | 13.9x / 10.1x / 7.2x | 18.5x / 12.4x / 8.3x | 31.8x / 17.3x / 10.2x | ADLX -500 MHz, -30 percent. not rentable, not owned: the 9070 XT's dependent-read rate per channel (150 M per second per 32-bit channel) on 24 channels; the knob by the 9070 XT grid Stock: modelled |
| AMD Radeon RX 7800 XT | 16 GB | 9.5 | estimated | 12.8 (estimated) | 16.5x / 12.0x / 8.5x | 22.0x / 14.7x / 9.8x | 37.8x / 20.5x / 12.1x | ADLX -500 MHz, -30 percent. 16 channels of GDDR6 at the 9070 XT's per-channel rate; the knob by the 9070 XT grid Stock: modelled |
| AMD Radeon RX 7600 XT | 16 GB | 13.0 | estimated | 17.8 (estimated) | 22.5x / 16.5x / 11.7x | 30.1x / 20.2x / 13.4x | 51.7x / 28.0x / 16.5x | ADLX -500 MHz, -30 percent. 8 channels; the card-in queue holds one for a PC measurement Stock: modelled |
| AMD Radeon RX 9060 XT | 16 GB | 10.0 | estimated | 13.7 (estimated) | 17.3x / 12.7x / 9.0x | 23.1x / 15.5x / 10.3x | 39.8x / 21.6x / 12.7x | ADLX -500 MHz, -30 percent. 8 channels of GDDR6 on RDNA 4; the card-in queue holds one Stock: modelled |
| NVIDIA H100 80GB HBM3 | DC | 2.0 | estimated | 2.58 (measured) | 3.5x / 2.5x / 1.8x | 4.6x / 3.1x / 2.1x | 8.0x / 4.3x / 2.5x | 1400 MHz at 80 percent. the premium 240 to 280 W at stock says half of it is the clock; a lock on an owned host (Hopper at 1,980 MHz boost, HBM3-bound): about 500 W (2.0); band 1.9 to 2.4; rented hosts refuse -lgc Stock: the denominator sweep, RunPod, 13:25 UK: class v3 252.96 MH/s at 382.4 W (1.51), class v5 241.34 at 621.6 W (2.58, 7 samples, the SM throttled to 1,590 MHz under the 700 W cap); the 7 and 8 October pods read class v4 250.6 at 695.1 W (2.77); -lmc answered 'use --lock-memory-clocks-deferred' and the memory clock stayed 2,619; stock, lock owed |
| NVIDIA L40S | DC | 3.78 | estimated | 5.29 (measured) | 6.5x / 4.8x / 3.4x | 8.7x / 5.9x / 3.9x | 15.0x / 8.2x / 4.8x | 1860 MHz at 60 percent. the Ada shape on the measured stock rows (class v3 220.7 W, class v4 277.6 W); re-based on the measured class v5 stock row of 13:5x UK (the architecture's shape on the measured v3 draw and the measured premium) Stock: class v5 measured (13:49 UK): 56.49 MH/s at 298.7 W (5.29, 46 samples, SM 2,520); class v3 56.35 at 221.5 W on the same host; the v4watts pod read class v4 56.4 at 277.6 W; stock, lock owed |
| NVIDIA A100-SXM4-80GB | DC | 2.89 | estimated | 2.99 (measured) | 5.0x / 3.7x / 2.6x | 6.7x / 4.5x / 3.0x | 11.5x / 6.2x / 3.7x | 1200 MHz at 80 percent. the card already runs at 1,410 MHz and the 400 W host limit binds under class v5 (SM clock gives); a cap under it costs rate on Ampere, so the floor is within 5 percent of the measured stock row Stock: the denominator sweep, RunPod, 13:24 UK: class v3 138.06 MH/s at 294.9 W (2.14), class v5 133.11 at 398.1 W (2.99, 19 samples, at the host's 400 W limit, memory 1,593 MHz); the v4watts pod on a 500 W host read class v4 138.0 at 489.3 W (3.54); -lmc 'not supported' on this card; stock, lock owed |
| Intel Arc B580 | 12 GB | 10.4 | estimated | 10.4 (estimated) | 18.0x / 13.2x / 9.3x | 24.1x / 16.1x / 10.7x | 41.4x / 22.4x / 13.2x | stock, no lever. 10.6 to 11 MH/s measured on PC 2 (the enclosure), the watts unread; about 110 W by the board's class (OWED); no clock lever in the app for Intel |
| Intel Arc A750 | 8 GB | 18.8 | estimated | 18.8 (estimated) | 32.6x / 23.8x / 16.9x | 43.5x / 29.2x / 19.4x | 74.8x / 40.5x / 23.9x | stock, no lever. unmeasured; modelled from the B580 and the Alchemist board |
| Apple M5 Max | Apple | 1.4 | measured | 1.4 (measured) | 2.4x / 1.8x / 1.3x | 3.2x / 2.2x / 1.4x | 5.6x / 3.0x / 1.8x | stock, no lever. the GPU and DRAM channels of the IOReport meter (not the wall): class v3 27.08 MH/s at 21.0 W (0.78), class v4 26.67 at 37.3 W (1.40), class v5 about 1.43; the package about 17 W more; no lever (no clock cap on Apple silicon) |
| Apple M4 Max | Apple | 1.55 | estimated | 1.55 (estimated) | 2.7x / 2.0x / 1.4x | 3.6x / 2.4x / 1.6x | 6.2x / 3.3x / 2.0x | stock, no lever. the same 512-bit LPDDR5X (8,533 against 9,600 MT/s) and a 40-core GPU on N3E: about 26 MH/s; class v3 about 0.85 at the meter, class v5 about 1.55 (the M5 Max's rows scaled); a devnet row from an unnamed Apple laptop read 24.3 MH/s on class v4 |
| Apple M4 Pro | Apple | 1.65 | estimated | 1.65 (estimated) | 2.9x / 2.1x / 1.5x | 3.8x / 2.6x / 1.7x | 6.6x / 3.6x / 2.1x | stock, no lever. 256-bit LPDDR5X and a 20-core GPU: about 13.5 MH/s; the shadow's per-hash premium does not shrink with the part, so class v5 reads about 1.65 at the meter |
| Apple M3 Max | Apple | 1.75 | estimated | 1.75 (estimated) | 3.0x / 2.2x / 1.6x | 4.0x / 2.7x / 1.8x | 7.0x / 3.8x / 2.2x | stock, no lever. 512-bit LPDDR5-6400 and a 40-core GPU on N3B: about 24 MH/s; class v5 about 1.75 at the meter |
To fold any other chip core figure (the research lane's convention, 13:5x): the edge on a row is the card's class v5
microjoules over `E_mem + 101,170 x c`, with `E_mem` 0.466 (GDDR7 board), 0.321 (one HBM3 stack), 0.14 (the N2 SRAM
die), 0.18 (the custom HBM4E base die) and `c` the core's picojoules per forced op; at the k lane's synthesised
sequencer-core floor of 1.1 pJ per op at N3 (its per-unit figure before fetch, decode and the register file, due 15:00)
the chip's class v5 energy reads GDDR7 0.577, HBM3 0.432, SRAM 0.251, so the 5080's floor of 2.10 gives 3.6x, 4.9x and
8.4x and the M5 Max's 1.43 gives 2.5x, 3.3x and 5.7x; that figure is a floor on the core and the row at it is the
pessimistic column.
What the table says:
1. The honest floor per tier, measured: 32 GB 2.33 to 2.37 (the 5090 at its knee); 16 GB 2.06 to 2.10 (the 5080 at
1,100 MHz); 12 GB 3.51 to 3.58 (the 4070 at its tune); 16 GB AMD 7.9 to 8.1 (the 9070 XT on the grid); Apple 1.40
to 1.43 (the M5 Max, no lever). The 16 GB NVIDIA tier is cheaper per hash than the 32 GB tier at the knee, and the
estimated 5070 Ti and 5070 rows (1.7 to 2.1) say the 12 and 16 GB Blackwell cards are the honest NVIDIA floor: the
5090's 170 SMs carry the shadow at the same picojoules per op as the 5080's 84 but its memory system draws more
per read (16 GDDR7 devices against 8).
2. The chip edge against the honest floor at k = 1 (a core as good as a GPU lane): GDDR7 1.9x to 2.1x on Blackwell at
the knee, 1.3x on the M5 Max; one HBM3 stack 2.2x to 2.4x and 1.5x; the N2 SRAM die 2.7x to 3.0x and 1.8x. At the
record's centre (k = 0.5): 2.7x to 3.0x, 1.8x; 3.3x to 3.6x, 2.2x; 4.5x to 5.0x, 3.1x. The served 2.1x stands as the
5090 figure; the public sentence can carry the lower honest row.
3. The lock recovers half the shadow's premium on Blackwell (the 5090: 145 W to 82 W; the 5080: 83 W to 41 W) and a
third of the class v3 draw (330 to 223 W; 170 to 104 W); on Ada the 4070's tune took 30 W of 106 on class v3 and the
premium reads 30 W at the tune point; on Ampere the cap is the only lever and it costs rate past 1,800 MHz; on AMD
the ADLX offsets take 24 percent of the draw for no rate; on Apple nothing moves.
4. Every Ada and Ampere consumer row is estimated at its knee because no rented host allows `-lgc` and no founder
machine holds one but the 4070 (whose full class v4 ladder is owed to a PC 1 job). The rented stock rows are
measured and stand; the estimated knees carry bands of about plus or minus 10 percent and the rented-against-PC
watts spread of 15 to 20 percent on top.
## 3. What moves the floor, per architecture (the knob set the measured rows support)
| Architecture | The lever | What it does, measured | The tier points (Efficiency / Balanced / Max) | Known-failed |
|---|---|---|---|---|
| Blackwell (5090, 5080; 5070 Ti, 5070, 5060 Ti, 5060 by shape) | the core lock; the power cap is flat on this hash | the rate holds within 1.5 percent to 1,300 MHz on the 5090 and within 0.3 percent to 1,000 on the 5080; the draw falls a third on class v3 and the shadow's premium halves | 1,200 / 1,300 / unlocked on the 5090; 1,100 / 1,100 / unlocked on the 5080; 1,100 / 1,300 / unlocked as the prior on the others | the 5090 at 1,100 (-10.5 percent on class v4), the 5080 at 900 (-5.2 percent); the Power Helper's skip-rule fault (0.3.20) needs the padding lines |
| Ada (4090, 4080, 4070 Ti, 4070, 4060 Ti, 4060, L40S) | the core lock with a cap | the 4070's tune: 1,860 MHz and the 50 percent cap hold 30.95 MH/s at 79.5 W on class v3 (28.7 at 102.7 stock); a cap holds the rate until the SM clock falls under about 2,400 MHz (the search prior); the 4090's class v4 premium at stock is 100 W (lane 1), under the 5090's 145 for 40 percent fewer counted ops per second | 1,860 at 50 to 60 percent / 2,400 at 80 percent / unlocked | the 4070 Ti Super at a 210 W host limit with the SM at 2,400 lost 10 percent on class v4 (36.95 against 41.24 MH/s): a cap that binds under the shadow costs rate on Ada too |
| Ampere (3090, 3080, 3070, 3060, A4000 to A6000, A100) | the power cap only | a cap holds the rate until the SM falls under about 1,800 MHz; stock clocks sit at 1,900 to 2,000, so the lever is 10 to 15 percent of the draw; the 3060 Ti at a 130 W host cap held 33.06 MH/s at 128.7 W on class v4 | 80 percent with the lock at 1,800 as a ceiling / 90 percent / unlocked | the 3080 Ti at 63 percent with the SM at 749 MHz: 24 to 28 percent of rate lost |
| Hopper (H100) | the core lock on an owned host | unmeasured: rented hosts refuse `-lgc`; the premium 283 W at stock (11.2 pJ per op) says half of it is the clock | 1,400 at 80 percent / 1,600 at 90 percent / unlocked, all estimated | none measured |
| RDNA 4 (9070 XT; 9060 XT by shape) and RDNA 3 (7900 XTX, 7800 XT, 7600 XT by shape) | the ADLX core and power offsets (0.3.25); no memory knob; no elevation | the 9070 XT's grid: the rate flat at 18.9 MH/s across 24 rows, -500 MHz and -30 percent = 149.3 W against 202 | -500 MHz, -30 percent / -300 MHz, -15 percent / stock | Ember run 6 aborted on "card reports 0 W" (fixed bd7fcf4); the ADLX sampler window bug (fixed) |
| Apple (M5 Max measured; M4 Max, M4 Pro, M3 Max by shape) | none | no clock cap on Apple silicon; the GPU draws 11 W at the hash and 27 W with the shadow, the DRAM 10 W throughout | stock only (the max tier, source stock) | none |
| Intel (B580 measured for rate; A750 by shape) | none in the app | measure only; the watts unread on the B580 | stock only | a kit without the Intel rotate-fold rewrite fails its self-test on Arc (a build fault, not an Arc result) |
The one rule under the table: a step that drags the memory clock under 95 percent of baseline is marked and cannot
win, because on this hash the memory clock is the rate and the core clock is only the shadow's cost; the Ember 2
climb may raise the memory clock (5 percent of the range per probe) where the vendor exposes it, and no measured row
yet shows a gain from it (the 5090's memory sits at 13,801 of 14,001 MHz).
## 4. The LPDDR6 and unified-memory tier, five years out
The method. The M5 Max row is the only measured unified-memory point and it splits three ways at the meter: the DRAM
rail 10.2 W at 27.08 MH/s x 128 reads = 2.94 nJ per dependent read (LPDDR5X-9600, 512-bit, 32 channels of 16 bits);
the GPU 11.0 W at the hash = 0.41 microjoules per hash of issue and latency-hiding cost; the shadow +16 W = 0.62
microjoules (6.9 pJ per counted op). Each device below is modelled by scaling those three terms: reads per second by
the part's sub-channel count at the M5 Max's 108 M dependent reads per second per 16-bit channel (the rate is
activate-bound per bank, not pin-rate-bound, so 14.4 Gbps moves it little); DRAM energy per read by the generation
(LPDDR6 at JESD209-6's lower VDD and the 32-byte atom, which halves the over-fetch of a 4-byte read against LPDDR5X's
64-byte burst: 2.0 to 2.3 nJ per read at the meter against 2.94, approximate); the GPU term by the vendor's measured
latency-hiding (Apple 0.41 microjoules per hash; AMD's RDNA reads 2.4 G dependent reads per second at 199 W on the
9070 XT, so its GPU term is about 7 microjoules per hash; NVIDIA's Blackwell at a 128-bit board (the 5060) about 1.5
at the knee); the shadow term by the vendor's ALU picojoules per op (Apple 6.9 measured, NVIDIA 6.4 at the knee, AMD
about 3 W for the whole shadow on the 9070 XT, so near zero at the card's rate). Every row is modelled; none is a
measurement, and the first LPDDR6 part to be sold will replace its row.
| Device class (5 years out) | Memory | Modelled MH/s | DRAM / GPU / shadow, microjoules per hash | Class v5 at the meter, microjoules per hash | GDDR7 board edge k = 0.5 / 1 | HBM3 stack | N2 SRAM die | Custom HBM4E base die (0.18 + shadow) |
|---|---|---|---|---|---|---|---|---|
| Apple Max-class on LPDDR6 (an M7 or M8 Max, 2028 to 2030: 576-bit, 24 channels of 24 bits = 48 sub-channels, a 40 to 48-core GPU at N2) | LPDDR6-14400 | about 41 | 0.27 / 0.35 / 0.56 | **1.18** (the package about 1.6 to 1.7) | 1.5x / 1.06x | 1.8x / 1.2x | 2.5x / 1.5x | 2.4x / 1.4x |
| Apple Pro-class on LPDDR6 (384-bit, 32 sub-channels) | LPDDR6 | about 27 | 0.27 / 0.37 / 0.58 | 1.22 | 1.5x / 1.1x | 1.9x / 1.3x | 2.6x / 1.6x | 2.4x / 1.5x |
| Apple M5 Max today, for the comparison | LPDDR5X-9600 | 27 measured | 0.38 / 0.41 / 0.62 | 1.40 measured (v5 1.43) | 1.8x / 1.3x | 2.2x / 1.5x | 3.1x / 1.8x | 2.8x / 1.7x |
| AMD Strix Halo class and its LPDDR6 successor (256-bit LPDDR5X-8000 today, 40 CU RDNA 3.5; approximate) | LPDDR5X then LPDDR6 | about 12 (the GPU's latency hiding binds, not the memory: the 9070 XT's 64 CU read 2.4 G per second) | 0.35 / about 6 / about 0.2 | about 6.5 (band 5 to 9) | 8x / 6x | 10x / 7x | 14x / 8x | 13x / 8x |
| NVIDIA GB10 class (DGX Spark, N1X: 256-bit LPDDR5X-9400, a 48-SM Blackwell GPU; approximate) | LPDDR5X then LPDDR6 | about 13.5 | 0.33 / about 2.5 at a lock / 0.65 | about 3.5 (band 3 to 4.5) | 4.4x / 3.1x | 5.4x / 3.6x | 7.5x / 4.4x | 7x / 4.2x |
| Qualcomm X Elite class (128-bit LPDDR5X-8448, 8 channels; Adreno X1; no worker exists; approximate) | LPDDR5X then LPDDR6 | about 7 | 0.35 / about 2.5 / about 0.8 | about 3.7 (band 3 to 5) | 4.7x / 3.3x | 5.7x / 3.8x | 8x / 4.7x | 7.4x / 4.5x |
| MediaTek Dimensity class (64-bit LPDDR5X, 4 channels; a phone; approximate) | LPDDR5X then LPDDR6 | about 3.4 | 0.35 / about 2 / about 0.9 | about 3.3 | 4.2x / 3x | 5.1x / 3.4x | 7.1x / 4.2x | 6.6x / 4x |
| LPDDR6 controller chip, 32 channels (lane B's row: the chip built on the honest tier's own memory) | LPDDR6 | about 100 per chip | 0.25 to 0.30 at zero shadow; + 0.32 / 0.65 | 0.57 to 0.62 / 0.90 to 0.95 | against the Apple LPDDR6 row: 2.0x / 1.3x | | | |
What the rows say:
1. The honest tier five years out is Apple's unified-memory part on LPDDR6 at about 1.2 microjoules per hash under class
v5 at the meter, 15 percent under today's M5 Max; the gain is the DRAM term (0.38 to 0.27, the 32-byte atom and the
lower rail) and the GPU term at N2, not the pin rate. The other unified-memory vendors model 3 to 10 microjoules
because their GPUs hide a dependent read's latency worse (AMD's RDNA 8x worse than Apple's per read, measured on
the 9070 XT), and the memory does not rescue a GPU term that large. So "the unified-memory SoC tier" as the reference
joule means the Apple tier specifically, until a measurement on a Strix or GB10 part says otherwise.
2. Against the Apple LPDDR6 row every chip's edge is the lowest in the record: the GDDR7 board 1.5x at k = 0.5 and
1.06x at k = 1 (at parity with a laptop at a core as good as a GPU lane); the N2 SRAM die 2.5x / 1.5x; the custom
HBM4E base die 2.4x / 1.4x; the LPDDR6 controller chip 2.0x / 1.3x. The chips lane B priced at 13x to 17x against
the stock 5090 are 1.5x to 2.5x against the best honest joule with the shadow on.
3. The reference joule should be the best honest joule, stated as a rule, not a fifth layer: the layers change what a
chip can be built as; the denominator changes what the edge is called. The public text carries the edge over the
Apple tier (1.3x at k = 1 today on GDDR7, 1.8x on the SRAM die) and over the 16 GB Blackwell card at its knee (1.9x
and 2.7x), with the stock 5090 figure retired from the headline.
4. Per-tier scoring. (a) The shadow sizing (layer 1's program-length band): YES, and it already is: the 2.0 rule caps N
at the card that loses 5 percent first, which is the Apple tier (130,000 ops on the M5 Max against 210,000 on the
5090 at 431 W and about 150,000 at its knee). Sizing N to the Apple ceiling rather than the 5090's costs the 5090
tier: at N = 130,000 the 5090 at its knee pays 6.4 pJ x 131,170 = 0.84 microjoules of shadow instead of 0.65 (+0.19,
+8 percent of its electricity at 2.37 to 2.56), its GDDR7 edge at k = 1 moves from 2.13x to 1.95x and at k = 0.5
from 3.0x to 2.9x; the M5 Max pays +0.21 (1.43 to 1.64, +15 percent) for 1.3x to 1.25x at k = 1. The shadow's
extra length buys about 0.15x of edge on every tier for 8 to 15 percent of every honest miner's electricity: the
priced answer is to keep N at 100,000 and let the per-era draw move inside the band, not to size it up to the Apple
ceiling. (b) The acceptance floor (layer 4, the (c''') ratio and the F8 uniformity test): NO per-tier scoring. The
floor is a property of the program's address distribution against a chip's on-die SRAM (the 1.067x hot-set ceiling),
the same number whichever honest card runs it; scoring it per tier would change the figure for nobody and would
invite a per-card class menu, which section 6 of the research file shows can never beat the single best class.
5. The dataset stays inside 16 GB unified memory (8 GiB at the top of the schedule) so the honest tier keeps mining:
lane B's point stands and this file adds a number: a 16 GB Mac is out at the candidate 10 GiB step and pays the
measured -12 to -22 percent rate cost of a larger working set before any limit; the Apple LPDDR6 row above is a
36 GB or larger part.
## 5. The tiers statement for Ember (keyed by class v5)
The table the app reads: `app/igneum-app/tiers/class-v5-tiers.json`, 30 card classes, 5 measured (the 5090, 5080,
4070, 9070 XT, M5 Max), the rest estimated from the measured stock rows and the architecture's shape in section 3.
Each entry carries the three tiers as rows in the shape `src/ember.rs tier_from_json` reads back from a card's state
(`id`, `clock_mhz`, `power_pct`, `mem_mhz`, `limit_w`, `mhs`, `w`; `mhw` and `source` as `Tier::json` writes them) plus
`label` (measured or estimated), `uj`, `note` and `src`; the AMD rows carry `core_offset_mhz` and `power_offset_pct`
for the ADLX knob; a card with no lever carries the `max` tier only, source `stock`. The validator
(`class-v5-tiers.mjs`) refuses a class other than v5, a tier outside the three, a power rung outside 50 to 100, a
missing field, a `uj` that is not watts over rate, a tuned row dearer than stock, a max tier that is not stock, an AMD
row without its offsets, and a missing card class of the brief; the test (`class-v5-tiers.test.mjs`, 10 of 10 on
igneum-build-3 at 13:0x) runs the shipped file, the measured rows, the match rule (a 5070 Ti is not a 5070), the stale
rule and the known-failed cases.
| Tier | The rule (the ember-tiers-25 definition) | Blackwell | Ada | Ampere | AMD | Apple, Intel |
|---|---|---|---|---|---|---|
| Efficiency | the best MH per watt row of the card's search outright | the lock one rung past the knee at 100 percent power: 1,200 on the 5090 (measured), 1,100 on the 5080 (measured), 1,100 as the prior elsewhere | the lock at 1,860 with the 50 to 60 percent cap (the 4070 measured) | the 80 percent cap with the lock at 1,800 as a ceiling (no row under it) | -500 MHz core, -30 percent power (the 9070 XT measured) | stock (no lever) |
| Balanced (the default from install) | the best MH per watt row within 1 percent of the top rate | the knee: 1,300 on the 5090 (measured), 1,100 on the 5080 (the same row), 1,300 as the prior elsewhere | 2,400 at 80 percent (the Ada prior) | the 90 percent cap, no lock | -300 MHz, -15 percent (inside the flat band) | stock |
| Max | the stock row (unlocked, 100 percent) | stock | stock | stock | stock | stock |
The re-measure rule after a class flip, as the table carries it (`remeasure_rule`):
1. A set measured under one program class never applies as current under another (`tiers_stale`: both classes known
and different). The stored Efficiency and Balanced points are the provisional START of the re-measure, not its
result: the search begins at the stored point and walks one rung either way on the clock and the power ladder,
the fingerprint checked on every step; the Max tier is stock and is never stale.
2. The measured flip: class v4 to class v5 on the 5090 at the 1,300 MHz lock cost 0.0 percent of rate and +2.0
percent of watts; the knee did not move. The expected flip cost for any class whose shadow stays inside the
latency-bound band (the op count and the mix inside layer 1's draw band) is 0 to 1 rung of knee; a class whose shadow
leaves the band (doubling the op count: the 5090 at 431 W binds at 210,000 ops) re-measures from stock, not from
the stored point.
3. The set also re-measures weekly (`PERIOD_S`) and on a driver major change (the prior key carries driver major and
class), and a confirm check that beats its prior by over 1 percent on MH per watt asks for the full plan.
## 6. The denominator sweep (rented, 8 October 13:05 to 14:1x UK, the stock rows the record lacked)
Three drivers of the fleet lane's model sweep and then a direct ssh runner: `box-cardbench-v2.sh` (class v3, the v5
kit's fingerprint, the memprobe ceiling), `box-cardbench-v4watts.sh` (the class v4 shape with the paired class v3 run)
and, from 13:21 on the coordinator's full-spend order, `box-cardbench-v2m.sh` on 16 card classes in parallel (the
class v5 kit's bench run for 200 batches of 2^24 under the 1 Hz sampler, so the class v5 watts are a measured mean
over busy samples from 8 s in; then `-lmc` to the card's maximum and a second class v5 run when the host accepted it).
One one-shot pod per card class on RunPod or Vast under the consumer cap, destroyed at the end of each row; the rows
in `~/igneum-fleet/cardbench/rows.jsonl` under `sweep: 2026-10-08-denominator` and `-v5`; the lane's spend about USD 5
of its USD 100 (the pods ran 3 to 10 minutes each). No rented host allowed `-lgc`; the two that accepted `-lmc` (the
H100 and A100) left the memory clock where it was, so every row here is stock and the knees of section 2 stay
estimated where no founder machine holds the card.
Two faults found on the way, for the fleet lane: (1) the sweep driver's ssh wait read every Vast pod of the 13:21 wave
as "never answered ssh in 10 minutes" while a direct `ssh` to the same host and port answered at once (13 pods, all
live; the rows were then taken by a direct runner, `~/igneum-fleet/denom-direct.sh`, which keys on the provider's
own ssh host and port and nothing in the registry); (2) the registry lock `~/igneum-fleet/boxes.json.lock` was held
from 13:32 by three `dn3-q05.py` processes of another lane, so every `Registry.patch` from this lane blocked (a
re-rent sat 12 minutes on a live pod and then destroyed it); the lane's scripts now write the registry best-effort
under a 15 s alarm. The rented stock rows against the PC rows on the one card measured both ways (the 5080): class v5
71.35 MH/s at 248.0 W rented against class v4 71.41 at 253.1 W on PC 1, within 3 percent, so the 18 percent class v3
spread of the morning was that host and not the method.
| Card | Host | Class v3 (MH/s at W = microjoules) | Class v4 shape or class v5 (MH/s at W = microjoules; samples) | The knobs on the host | Cost |
|---|---|---|---|---|---|
| RTX 4090 24 GB | runpod m1tstl1h1b7r0o (595.91.07), limit 450.0 W | 70.094 at 222.6 W = 3.18 | class v5 70.253 MH/s (MATCH), watts not sampled on this row | -lgc refused | USD 0.032 |
| RTX 5060 8 GB | vast 54840005 (580.126.09), limit 145.0 W | | class v4 16.234 at 87.8 W = 5.41 (v4 over v3 20.4 percent on this host) | | USD 0.009 |
| RTX 5060 8 GB | vast 54840004 (580.126.09), limit 145.0 W | 24.774 at 80.7 W = 3.26 | class v5 31.312 MH/s (MATCH), watts not sampled on this row | -lgc refused | USD 0.011 |
| RTX 5080 16 GB | vast 54840017 (570.133.07), limit 300.0 W | | class v4 70.951 at 239.8 W = 3.38 (v4 over v3 41.1 percent on this host) | | USD 0.052 |
| RTX 3080 10 GB | vast 54840023 (580.159.03), limit 130.0 W | | class v4 24.007 at 129.2 W = 5.38 (v4 over v3 3.0 percent on this host) | | USD 0.025 |
| RTX 3060 12 GB | vast 54840846 (580.126.20), limit 170.0 W | | class v4 26.886 at 166.1 W = 6.18 (v4 over v3 44.6 percent on this host) | | USD 0.007 |
| RTX 4070 12 GB | vast 54840856 (580.173.02), limit 200.0 W | 30.829 at 107.1 W = 3.47 | class v5 30.878 MH/s (MATCH), watts not sampled on this row | -lgc refused | USD 0.023 |
| A100 80 GB | runpod miuoe6gkcb1rhz (580.126.16), limit 400.0 W | 138.062 at 294.9 W = 2.14 | class v5 133.548 MH/s (MATCH), watts not sampled on this row | -lgc refused | USD 0.075 |
| H100 80 GB | runpod zosqj1s802dk3x (580.126.09), limit 700.0 W | 252.959 at 382.4 W = 1.51 | class v5 241.342 MH/s (MATCH), watts not sampled on this row | -lgc refused | USD 0.216 |
| RTX 3090 24 GB | runpod sbrhhunqui25d0 (580.65.06), limit 310.0 W | 60.008 at 294.6 W = 4.91 | class v5 61.486 MH/s (MATCH), watts not sampled on this row | -lgc refused | USD 0.014 |
| RTX 3080 10 GB | vast 54841582 (580.159.03), limit 180.0 W | 49.832 at 177.8 W = 3.57 | class v5 44.308 MH/s (MATCH), watts not sampled on this row | -lgc refused | USD 0.024 |
| RTX 5080 16 GB | vast 54845199 (570.133.07), limit 300.0 W | 71.16 at 159.4 W = 2.24 | class v5 71.349 at 248.0 W = 3.476; 39 samples, SM 2769, memory 14801 | -lgc refused; -lmc refused | USD 0.014 |
| L40S 48 GB | vast 54845608 (595.71.05), limit 350.0 W | 56.354 at 221.5 W = 3.93 | class v5 56.493 at 298.7 W = 5.287; 46 samples, SM 2520, memory 9001 | -lgc refused; -lmc refused | USD 0.04 |
| RTX 4080 16 GB | vast 54845190 (595.84), limit 320.0 W | 40.694 at 141.5 W = 3.48 | class v5 40.8 at 200.7 W = 4.919; 72 samples, SM 2760, memory 10801 | -lgc refused; -lmc refused | USD 0.017 |
| RTX 3080 10 GB | vast 54845161 (580.178.04), limit 170.0 W | 50.708 at 169.2 W = 3.34 | class v5 33.37 at 169.9 W = 5.091; 90 samples, SM 689, memory 9251 | -lgc refused; -lmc refused | USD 0.008 |
| RTX 4090 24 GB | vast 54845886 (595.84), limit 250.0 W | 62.401 at 200.5 W = 3.21 | class v5 62.684 at 249.9 W = 3.987; 46 samples, SM 2579, memory 10251 | -lgc refused; -lmc refused | USD 0.02 |
| RTX 5060 8 GB | vast 54845177 (595.84), limit 140.0 W | 30.76 at 78.8 W = 2.56 | class v5 30.804 at 115.7 W = 3.756; 99 samples, SM 2715, memory 13801 | -lgc refused; -lmc refused | USD 0.011 |
| RTX 5060 Ti 16 GB | vast 54845162 (580.126.09), limit 180.0 W | 30.747 at 82.6 W = 2.69 | class v5 30.843 at 123.9 W = 4.017; 96 samples, SM 2835, memory 13801 | -lgc refused; -lmc refused | USD 0.013 |
| RTX 3060 12 GB | vast 54845154 (580.126.09), limit 170.0 W | 26.534 at 120.4 W = 4.54 | class v5 26.53 at 169.8 W = 6.4; 117 samples, SM 1806, memory 7301 | -lgc refused; -lmc refused | USD 0.008 |
| RTX 4070 12 GB | vast 54845860 (580.126.09), limit 210.0 W | 28.401 at 116.3 W = 4.09 | class v5 28.433 at 165.5 W = 5.821; 108 samples, SM 2820, memory 9801 | -lgc refused; -lmc refused | USD 0.011 |
| RTX 4060 Ti 16 GB | vast 54849184 (580.159.03), limit 100.0 W | 18.735 at 79.1 W = 4.22 | class v5 18.75 at 99.8 W = 5.323; 163 samples, SM 2668, memory 8751 | -lgc refused; -lmc refused | USD 0.016 |
Not reached by the class v5 sweep (stock, their class v4 rows of the morning stand within 2 percent): the 5070 Ti and
the 5070 (two Vast hosts each, 14:08 and 14:13 UK, never accepted the fleet key in 15 minutes); the 3070 (two hosts
closed the connection mid-bench). The sweep closed at 14:35 UK: 20 rows landed, no pod left on either provider
(audited 14:40), the lane's spend about USD 5 of its USD 100.
## 7. Consequences per tier
| Tier | What this file means | What is being done |
|---|---|---|
| Home miner, one 8 GB card (5060, 4060, 3070) | the Blackwell 8 GB card at its knee is 2.15 microjoules (estimated), inside 10 percent of the 5090's measured floor; the Ada and Ampere 8 GB cards 3.8 to 4.8. The lever is the Efficiency tier: on a 5060 about 40 W of 110 back for no rate (estimated); on a 3070 about 20 W of 196 | the tiers table ships the prior; the card's own search measures it; the 5060 and 4060 watts rows are the sweep's |
| One 12 GB card (5070, 4070, 3060) | the 5070 at its knee about 1.75 (the lowest NVIDIA row in the record, estimated); the 4070 measured 3.51 at its tune; the 3060 4.9 | the 4070's class v4 ladder below 1,860 MHz is owed to a PC 1 job (the hash lane's scripts, not this lane's hands) |
| One 16 GB card (5080, 5070 Ti, 5060 Ti, 4080, 4060 Ti, 9070 XT) | the 5080 measured 2.06 to 2.10 at 1,100 MHz, 42 percent of its stock draw back for 0.3 percent of rate; the 5070 Ti about 1.70 (estimated); the 9070 XT 8.1 measured on the grid, 24 percent of its draw back for nothing | the 5060 Ti ladder on PC 2 (the card-in job); the 9070 XT grid lands in the AMD tier rows |
| One 24 or 32 GB card (5090, 4090, 3090) | the 5090 at its knee 2.33 to 2.37, 33 percent of its electricity back for 1.5 percent of rate; the 4090 about 3.4, the 3090 about 6.4 (estimated: Ampere's cap is a small lever) | the Balanced tier is the default from install; the 4090 and 3090 stock rows are the sweep's |
| Apple (M-series) | the honest best per joule the project owns: 1.40 to 1.43 measured on the M5 Max, nothing to set; the M4 and M3 parts within 25 percent (estimated); five years out about 1.2 on LPDDR6 | the public edge is stated against this tier; the dataset stays inside 16 GB unified memory |
| A rig | per card as above; a 5090 rig at the Balanced tier draws 319 W per card instead of 476 on class v5; a 5080 rig 147 instead of 253 | the tiers table |
| A pool user | nothing changes in shares or payout from any row here | |
| A chip | the honest floor it must beat is 2.1 microjoules on a mass-market 16 GB Blackwell card at its knee (1.9x at k = 1 on GDDR7, 2.7x on the N2 SRAM die) and 1.43 on an Apple laptop (1.3x and 1.8x); five years out 1.2 on the Apple LPDDR6 part (1.06x and 1.5x) | the research lane's synthesis carries the number |
| The public claim | the honest sentence per joule: "a chip that stores the dataset keeps the card's whole-card energy over its memory energy: under 2x against a 16 GB Blackwell card at its knee and 1.3x against an Apple laptop for a core as good as a GPU lane; 2.7x and 1.8x against a 2 nm SRAM store" | to the Counter lane for the chip texts; nothing served moves from this file |
## 8. Unverified and owed
- Every Ada and Ampere knee is estimated: no rented host allows `-lgc`, and the founder's machines hold one Ada card
(the 4070, its class v4 ladder below 1,860 MHz owed to a PC 1 job through the hash lane's scripts) and no Ampere.
The estimated knees carry about plus or minus 10 percent, and the rented stock watts read 15 to 20 percent under a
PC's on the one card measured both ways (the 5080).
- The 5070 Ti and 5070 floors (1.7 to 2.1) are the strongest claims in this file and both are estimated from stock
rows; a 5070 Ti ladder on PC 1 or PC 2 (the card-in job's queue) would settle whether the honest NVIDIA floor is a 16
GB card.
- The AMD rows other than the 9070 XT are modelled on one card; the 9060 XT and 7600 XT in the card-in queue will
replace two of them.
- The Apple M4 Max, M4 Pro and M3 Max rows are modelled on the M5 Max's split; a Metal bench on any of them with the
IOReport meter would replace its row in an hour.
- Section 4's LPDDR6 rows are modelled on JESD209-6's public summary (the bank count and tFAW are behind the
paywall, lane B's note) and on the M5 Max's three-way split; the Strix, GB10 and Qualcomm rows rest on the vendor's
measured latency hiding on other parts and are approximate.
- The 4060's watts (three rented hosts, no power sensor), the B580's watts, the H100's lock, the A750 entirely.
- The class v5 stock rows of the 5070 Ti, the 5070 and the 3070 (section 6: the hosts never took the key or closed
the connection); their class v4 stock rows stand within 2 percent.
- Two fleet-tooling faults for the fleet lane (section 6): the sweep driver's ssh wait misreads live Vast pods, and a
registry flock held for an hour by another lane's processes blocks every `Registry.patch`; both cost this lane 40
minutes and five pods.
## 9. The reading for the 15:45 BST close (five sentences)
The honest joule under class v5 is 2.06 to 2.10 microjoules per hash on a 16 GB Blackwell card at its 1,100 MHz lock
(measured), 2.33 to 2.37 on the 5090 at its knee (measured), about 1.7 to 2.1 on the 12 and 16 GB Blackwell cards
(estimated from their measured stock rows), 1.40 to 1.43 on an Apple M5 Max with nothing to set (measured), and the
only software lever that moves it is the operating point: the core lock on Blackwell and Ada (worth 34 to 41 percent
of the class v5 stock draw measured today on the 5080 and the 4070), the cap on Ampere (10 to 15 percent, and a cap
under the shadow costs rate one for one: the 3080 at 170 W lost a third of its class v5 rate), the ADLX offsets on
AMD (24 percent), nothing on Apple and Intel; the occupancy knob is 1 to 2 percent and the memory clock is untouched.
Against those floors the stored-dataset chips read 1.9x to 2.1x on GDDR7, 2.2x to 2.4x on one HBM3 stack and 2.7x to
3.0x on a 2 nm SRAM die at a core as good as a GPU lane (6.4 pJ per forced op), 2.6x to 3.0x, 3.2x to 3.6x and 4.4x
to 5.0x at the record's 3.2 pJ, and 3.6x to 4.0x, 4.8x to 5.4x and 8.2x to 9.3x at the k lane's 1.1 pJ floor; against
the Apple laptop 1.3x, 1.5x and 1.8x at 6.4 pJ; the served 2.1x stands as the 5090 figure and the honest headline is
under 2x against a mass-market card at a core as good as a GPU lane. Five years out the honest tier is Apple's
unified-memory part on LPDDR6 at about 1.2 microjoules (modelled), against which the GDDR7 chip is at parity at 6.4 pJ
and the SRAM die 1.5x; the other LPDDR SoC vendors model 3 to 10 microjoules because their GPUs hide the latency
worse, so the reference joule is the Apple tier by name. The class should state resistance against the best honest
joule and keep sizing the shadow by the Apple tier's 5 percent point (N at 100,000, the band drawn per era), and
should not score the acceptance floor per tier; sizing N up to the Apple ceiling would cost every honest miner 8 to
15 percent of its electricity for about 0.15x of edge. The tiers table ships the measured points on the 5090, 5080,
4070, 9070 XT and M5 Max, the measured class v5 stock row on 14 classes and an architecture prior for the knee
everywhere else, with the rule that a class flip re-measures from the stored point (the measured v4 to v5 flip moved
the 5090's knee by nothing), and the rows still owed are the Ada and Ampere knees, which no rented host can give.