Merge ca3-coord 55863f81 into master (gate: green on 55863f81, recorded by tools/ci/pre-push.sh; landed on the build mirror)

This commit is contained in:
igneum-labs 2026-10-07 20:16:18 +00:00
commit d74bd5ca21
7 changed files with 129 additions and 59 deletions

View file

@ -202,6 +202,8 @@ the `f = 1` chip: it is the 55 W without the 271.
### 5.4 The curve
Per row: reads per hash = 128 f; items recomputed = 128 (1 - f); ops per hash = 128 (1 - f) x 9,360 + 512.
Correction, 7 October 2026 (the in-house adversarial pass, lane adv-cache-3, report 091edc34): the partial-store rows above and adv-cache's Q1b table price a chip that holds every k-th line of the 64-line chain and recomputes a read at offset o in o evaluations ((k - 1) / 2 on average). The exact pebbling optimum for the chain (dynamic programming, checked against exhaustive search at 10 to 16 lines) sits under that curve: blocks per read 16.0 against 31.5 at f = 1/64 (the one held line belongs at line 32, not line 0), 10.5 against 15.5 at 2/64, 6.09 against 7.5 at 4/64, 3.17 against 3.5 at 8/64, 1.45 against 1.5 at 16/64, equal from f = 1/2. So a chip holding 1/64 of the cache pays 9.3x the item's ops, not 17.4x; at f = 1/2 and above nothing moves, and the SRAM column and the full-store verdict stand (no point on the curve beats the full store under the op budget or under energy).
Memory-bound rate = the ceiling / (128 f). Compute-bound rate = 50 T op/s / ops per hash (the section 1 budget).
The rate is the smaller; "binding" names it. Power = rate x (128 f x E_read + 128 (1 - f) x 6.3 nJ) + static (memory,
controller, and 20 W for the recompute die's clocks and leakage when `f < 1`). Energy per hash = power / rate. "Gain,

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

View file

@ -425,39 +425,56 @@ for (const [file, active] of PAGES) {
const rows = bj.rows;
const fmt = (n) => Number(n).toLocaleString('en-GB', { maximumFractionDigits: 1 });
const fmt3 = (n) => Number(n).toLocaleString('en-GB', { maximumFractionDigits: 3 });
// Nine compact columns sort; the long fields (the class v4 cost, the Hive values with their label, the miner and driver,
// the source and the note) sit in a detail row under each card so a row stays one line wide at 1600 px (the 7 October
// capture showed eleven columns clipped at five, every row inflated by off-screen wrapped text). The detail row moves
// with its data row on a sort.
const heads = [
['Card', 'card', 'text'], ['Generator', 'generator', 'text'], ['Best MH/s', 'mh_s', 'num'], ['Watts', 'watts', 'num'], ['MH per watt', 'mh_per_w', 'num'],
['Class v4 cost', 'v4_cost', 'text'], ['Tuned', 'tuned', 'text'], ['Hive flight sheet (core, mem, PL)', 'hive', 'text'], ['Miner', 'miner', 'text'], ['Date', 'date', 'text'], ['Source', 'source', 'text'], ['Who measured it', 'by', 'text'],
['Class v4 cost', 'v4_cost_short', 'text'], ['Tuned', 'tuned_short', 'text'], ['Hive core / mem / PL', 'hive_short', 'text'], ['Date', 'date', 'text'], ['Who measured it', 'by', 'text'],
];
const shortV4 = (r) => { const v = r.v4_cost || 'not measured'; const m = v.match(/^([^;(]+?)(?:\s*[;(]|$)/); return m ? m[1].trim() : v; };
const shortTuned = (r) => { const v = r.tuned || 'stock, mining'; return v.split(' (')[0]; };
const shortHive = (r) => { const h = r.hive; return (h && h.core_mhz != null) ? fmt(h.core_mhz) + ' / ' + fmt(h.mem_mhz) + ' / ' + fmt(h.pl_w) + ' W' : 'stock'; };
const cells = (r) => [
[r.card, r.card], [r.generator, r.generator], [fmt(r.mh_s), r.mh_s], [r.watts == null ? 'not read' : fmt(r.watts), r.watts ?? -1], [r.mh_per_w == null ? 'not measured' : fmt3(r.mh_per_w), r.mh_per_w ?? -1],
[r.v4_cost || 'not measured', r.v4_cost || ''], [r.tuned || 'stock, mining', r.tuned || ''], [hiveCell(r), r.hive && r.hive.core_mhz ? 'a measured ' + r.hive.core_mhz : 'z stock'], [r.miner + (r.driver_os ? ' (' + r.driver_os + ')' : ''), r.miner], [r.date, r.date], [r.source, r.source], [r.by + (r.note ? '. ' + r.note : ''), r.by],
[shortV4(r), shortV4(r)], [shortTuned(r), shortTuned(r)], [shortHive(r), r.hive && r.hive.core_mhz != null ? 'a ' + r.hive.core_mhz : 'z stock'], [r.date, r.date], [r.by, r.by],
];
const hiveCell = (r) => {
const h = r.hive; if (!h || h.core_mhz == null) return 'stock' + (h && h.label ? ' (' + h.label.replace(/^stock \(|\)$/g, '') + ')' : '');
return 'core lock ' + fmt(h.core_mhz) + ' MHz, mem ' + fmt(h.mem_mhz) + ' MHz, PL ' + fmt(h.pl_w) + ' W (' + h.label + ')';
const detail = (r) => {
const parts = [];
parts.push('<b>Class v4 cost:</b> ' + esc(r.v4_cost || 'not measured'));
parts.push('<b>Tuned:</b> ' + esc(r.tuned || 'stock, mining'));
if (r.hive && r.hive.core_mhz != null) parts.push('<b>Hive flight sheet:</b> core lock ' + esc(fmt(r.hive.core_mhz)) + ' MHz, mem ' + esc(fmt(r.hive.mem_mhz)) + ' MHz, PL ' + esc(fmt(r.hive.pl_w)) + ' W (' + esc(r.hive.label) + ')');
else if (r.hive && r.hive.label) parts.push('<b>Hive flight sheet:</b> ' + esc(r.hive.label));
parts.push('<b>Miner:</b> ' + esc(r.miner + (r.driver_os ? ' (' + r.driver_os + ')' : '')));
parts.push('<b>Source:</b> ' + esc(r.source));
if (r.note) parts.push('<b>Note:</b> ' + esc(r.note));
return parts.join(' · ');
};
const render = (list, id) => '<div class="tbl"><table class="sortable" id="' + id + '"><thead><tr>' +
heads.map(([h, k, t], i) => `<th data-key="${k}" data-type="${t}" aria-sort="${k === 'mh_per_w' ? 'descending' : 'none'}"><button type="button" class="sort">${h}</button></th>`).join('') +
'</tr></thead><tbody>' + list.map(r => '<tr>' + cells(r).map(([c, v]) => `<td data-v="${esc(String(v))}">${esc(String(c))}</td>`).join('') + '</tr>').join('') + '</tbody></table></div>';
const table = render(cur, 'bench-current');
const earlierTable = render(earlier, 'bench-earlier');
const render = (list, id) => '<div class="tbl"><table class="sortable bench" id="' + id + '"><thead><tr>' +
heads.map(([h, k, t]) => `<th data-key="${k}" data-type="${t}" aria-sort="${k === 'mh_per_w' ? 'descending' : 'none'}"><button type="button" class="sort">${h}</button></th>`).join('') +
'</tr></thead><tbody>' + list.map(r => '<tr class="row">' + cells(r).map(([c, v]) => `<td data-v="${esc(String(v))}">${esc(String(c))}</td>`).join('') + '</tr>' +
`<tr class="detail"><td colspan="${heads.length}">${detail(r)}</td></tr>`).join('') + '</tbody></table></div>';
const sortScript = `<script>
(function(){
// header sort on the bench tables: a click sorts by that column (numbers by value, text by locale), a second click flips it
// header sort on the bench tables: a click sorts by that column (numbers by value, text by locale), a second click flips it;
// each data row carries its detail row with it
document.querySelectorAll('table.sortable').forEach(function(t){
var ths=t.querySelectorAll('th'),tb=t.querySelector('tbody');
ths.forEach(function(th,i){th.querySelector('button').addEventListener('click',function(){
var cur=th.getAttribute('aria-sort'),dir=cur==='descending'?'ascending':'descending',num=th.getAttribute('data-type')==='num';
ths.forEach(function(o){o.setAttribute('aria-sort','none');});th.setAttribute('aria-sort',dir);
var rows=Array.prototype.slice.call(tb.querySelectorAll('tr'));
rows.sort(function(a,b){var x=a.children[i].getAttribute('data-v'),y=b.children[i].getAttribute('data-v');var c=num?(parseFloat(x)-parseFloat(y)):x.localeCompare(y,'en');return dir==='descending'?-c:c;});
rows.forEach(function(r){tb.appendChild(r);});
var rows=Array.prototype.slice.call(tb.querySelectorAll('tr.row'));
var pairs=rows.map(function(r){return [r, r.nextElementSibling && r.nextElementSibling.classList.contains('detail') ? r.nextElementSibling : null];});
pairs.sort(function(a,b){var x=a[0].children[i].getAttribute('data-v'),y=b[0].children[i].getAttribute('data-v');var c=num?(parseFloat(x)-parseFloat(y)):x.localeCompare(y,'en');return dir==='descending'?-c:c;});
pairs.forEach(function(p){tb.appendChild(p[0]); if(p[1]) tb.appendChild(p[1]);});
});});
});
})();
</script>`;
const sortStyle = '<style>th .sort{all:unset;cursor:pointer;font:inherit;color:inherit}th .sort::after{content:" \\2195";opacity:.45}th[aria-sort="descending"] .sort::after{content:" \\2193";opacity:1}th[aria-sort="ascending"] .sort::after{content:" \\2191";opacity:1}details.earlier{margin:var(--s-4) 0}details.earlier summary{cursor:pointer;color:var(--bone)}</style>';
const sortStyle = '<style>th .sort{all:unset;cursor:pointer;font:inherit;color:inherit;white-space:nowrap}th .sort::after{content:" \\2195";opacity:.45}th[aria-sort="descending"] .sort::after{content:" \\2193";opacity:1}th[aria-sort="ascending"] .sort::after{content:" \\2191";opacity:1}table.bench{min-width:1040px}table.bench tr.row td{white-space:nowrap;border-bottom:0}table.bench tr.detail td{font-size:13px;color:var(--ash);padding-top:0;overflow-wrap:anywhere;white-space:normal}table.bench tr.detail td b{color:var(--ink-2);font-weight:600}details.earlier{margin:var(--s-4) 0}details.earlier summary{cursor:pointer;color:var(--bone)}</style>';
const table = render(cur, 'bench-current');
const earlierTable = render(earlier, 'bench-earlier');
// Ember Tune's fleet priors (site/miner-priors.json, tools/tuning.mjs --priors --site): one row per card model,
// driver major and program class; a row under the sample floor shows its count and no point
const pj = JSON.parse(readFileSync(join(here, 'miner-priors.json'), 'utf8'));
@ -480,7 +497,7 @@ for (const [file, active] of PAGES) {
'<p><strong>Why the rate fell from the first bench to today.</strong> The genesis program did 104 dependent random 4-byte loads per hash over a 1 GiB dataset; the hourly program and class v3 do 128, with the mixer between them; class v4 adds about 100,000 integer operations per hash that ride in the memory wait. So the hash is bound by random-read bandwidth by design, and a card\'s MH/s is a relative number: the difficulty follows it, and the same card earns the same share of blocks at 136 MH/s on class v3 as it did at 228 MH/s on the genesis program. What a miner compares is hash per watt, and what the chain cares about is the chip edge, which the shadow work is there to cut.</p>',
table,
`<p>Rows on the current class: ${cur.length}. Each row names the engineering log entry or the job it came from.</p>`,
'<p><strong>The Hive flight sheet column.</strong> Where a card has a measured tune point, the column gives the core clock lock, the memory clock and the power limit to copy into a HiveOS flight sheet, labelled measured with the date; stock means no tune point has been measured yet. The Hive package mines at these settings through Hive\'s own overclock controls; the desktop app\'s Ember Tune lands on them by itself.</p>',
'<p><strong>The Hive flight sheet column.</strong> Where a card has a measured tune point, the column gives the core clock lock, the memory clock and the power limit to copy into a HiveOS flight sheet (core / mem / PL); the line under each row carries the label with the date, the class v4 cost in full, the miner and driver, the source and the note. Stock means no tune point has been measured yet. The Hive package mines at these settings through Hive\'s own overclock controls; the desktop app\'s Ember Tune lands on them by itself.</p>',
'<details class="earlier"><summary>Earlier classes (the genesis program, the hourly program, class v3 before the shadow): ' + earlier.length + ' rows, not comparable with the table above</summary>',
'<p>These rows are the bench numbers of 3 and 4 October 2026: the genesis program (104 loads per hash), the hourly program and the first class v3 miner. A higher MH/s here is a different hash, not a faster card.</p>',
earlierTable,

View file

@ -254,7 +254,7 @@ td.mono{font-family:var(--f-mono);font-size:12.5px;min-width:180px}td.iv{color:v
<tr data-status="tested by the team"><td class="n">14</td><td class="claim">Ethereum bytecode runs unchanged, with the documented differences of spec 7.1<div class="where">Homepage Build card; litepaper Building</div></td><td><span class="st st-2">tested by the team</span></td><td class="mono">as row 13; fixes <code>F-exec-A</code>, <code>F-exec-B</code> (spec 7.5)</td><td><code>tools/evm-smoke/smoke.mjs</code>: deploy via viem, <code>increment</code>, <code>hashLoop</code>, <code>eth_estimateGas</code>, <code>eth_getLogs</code>; <code>tools/exec-attacks</code> scenarios 1 and 3; bench-log "execution layer attack fixes"</td><td>Deployment, calls, reverts, logs and gas estimates behave as viem expects; chain id 4463; the prototype pgas table gives 0.0095 to 0.028 pgas per gas, below the design's band before calibration, 3 October 2026. 4 October 2026: a transaction that would cross the block's proving budget is refused by the mempool and, if forced in, aborted and charged with its nonce advanced (25 of 25 checks; 30 of 30 malformed cases). Apple M5 Max. The <code>Prover</code> precompile, proof records and the shard planner are not in the node</td><td class="iv">none yet</td></tr>
<tr data-status="implemented"><td class="n">15</td><td class="claim">Every block is proven, with the proof landing within about a minute at launch<div class="where">Homepage stats ("~60 s to a proof"); litepaper Proving; roadmap phase 3 gate</div></td><td><span class="st st-1">implemented</span></td><td class="mono">repo <code>d7e1f89</code> (GPU proof), <code>e01a3cc</code>, <code>292e800</code>, <code>eedd136</code> (<code>proving/igneum-prove</code>: shard cutter, MPT witnesses, shard and aggregator guests); SP1 6.8.1; spec 7.2, 7.6</td><td><code>proving/windows-wsl2</code> (SETUP-PROVER, PROVE-BLOCK) on the RTX 5090; <code>igneum-prove-host --mode block</code> on <code>proving/fixtures/</code>; bench-log "proving v0 on the RTX 5090" and "proving: devnet v4 shards"</td><td>First GPU proof of an Igneum block, 4 October 2026, RTX 5090 (WSL2, SP1 cuda, mining paused): fixture <code>block-78-increment</code> (2 transactions), core proof 1.4 s (7.3 MB, verify 0.221 s), compressed proof 2.7 s (1.27 MB, verify 0.038 s), post-state and receipts roots identical to the node's; 15.7x and 20.6x faster than a loaded M5 Max CPU. The same day on that CPU (load 38 to 47): a three-shard block proved shard by shard and aggregated by recursion, 19 min (1,139 s) end to end, 245 to 337 s per compressed shard proof, every proof verified. What is not there: no proof is produced, carried or checked on the chain (the devnet prover is a stub that signs claims), the proving pool pays nobody (row 21), the block proven is far below one shard, and the 60-second figure remains a design target; the pass mark is the standard in <code>docs/benchmarks/proving-e2e.md</code>. Second RTX 5090 run, 4 October 2026 evening (job run-20261004-173115): a full shard at the provisional S_p (6.75 M pgas, 60.8 M cycles) executed in 1.63 s, core proof 8.3 s (18.1 MB), compressed proof 10.9 s (1.27 MB, verify 0.040 s); a two-shard block (13.5 M pgas) proved shard by shard (11.7 s and 10.0 s) and aggregated in 2.2 s, 24 s of GPU stages end to end, every proof verified, six tampered witnesses rejected. The two host defects (an abort after the upload, an idle wait that turned out to be an unbuffered 18 MB proof save through the WSL2 file bridge, 24 minutes) are fixed (ledger P20) 5 October 2026, live devnet with real transactions (bench-log "real transactions, the first non-empty shard proven and paid"): block 72704 shard 0, 29 transfers, 5,800 pgas, proven on the RTX 5090 Windows rig in 34 s, verified on the Apple M5 Max in 0.297 s and paid 1.7623 IGN, 53 s after the chain block executed; of about 1,400 blocks in the 20-minute window 36 were proven (the one prover takes the newest shard assigned to it), so "every block" is not yet true; a second content shard (72803, all copies skipped) failed the native-execution veto on the exporter's block structure, fixed with fixtures the same day, the node side pending the 0.3.9 rollout 5 October 2026, evening (bench-log "proving v1"): the aggregated segment record, the chain rule and the unproven rule are implemented behind <code>proving_v1_activation_daa</code> (branch proving-v1, not on the devnet before 0.3.11); on the RTX 5090 a chain of 8 consecutive live blocks proved and aggregated by recursion in 135.6 s with the miner on the card (17 s a block, one proof of 1,272,909 bytes attesting all 8, verified in 0.04 s); the 3-node fast-time harness paid a segment record 1.0 s after submission and refused a late one after its deadline (21 checks); the devnet itself, with one prover, carried proofs for 2.4% of blocks over 30 minutes at a block-to-record latency p50 44 s, p99 52 s. The "within about a minute" holds per proven block; "every block" needs 18 mining 5090s or 6 proving-only cards at empty blocks on the measured rates, and the mandatory rule stays off until the share is one</td><td class="iv">none yet</td></tr>
<tr data-status="designed"><td class="n">16</td><td class="claim">A 12 GB card proves one shard in about 20 s (WITHDRAWN 5 October 2026: a 24 GB card proves a full shard at the adopted size in 4.3 s; 32 GB mines and proves)<div class="where">Litepaper Proving ("The proving budget"); roadmap gate 2</div></td><td><span class="st st-0">designed</span></td><td class="mono">spec 5.1 (Target), 7.6 (<code>S_p</code> provisional, 7,500,000 pgas = <code>B_p</code> / 4)</td><td><code>PROVE-SHARD.bat</code> on the RTX 5090 (pending); the end-to-end standard in <code>docs/benchmarks/proving-e2e.md</code>; bench-log "proving: devnet v4 shards"</td><td>Measured on a 32 GB card, not yet on a 12 GB card. A shard at the provisional <code>S_p</code> is 60.8 M SP1 cycles on the prototype pgas table (9 cycles per pgas, 44 per EVM gas; the modexp entry about 100x its SP1 cost); on an RTX 5090 (4 October 2026 evening, job run-20261004-173115) it executed in 1.63 s and its compressed proof took 10.9 s, verified in 0.040 s, so the 32 GB card is inside the 20 s target with margin. Whether a 12 GB card proves it at all, and in what time, is the next measurement (an RTX 3060 and an RTX 5060 Ti 16 GB are on order). A per-shard time can be met by shrinking the shard, so the project does not use it as a pass mark 5 October 2026, evening (bench-log "proving v1", the S_p curve): measured on the RTX 5090 with SP1 6.8.1's GPU prover, the card to itself, 1-s nvidia-smi samples: an empty shard 13,874 MiB and 2.2 s; a full shard at the ADOPTED v1 budget (30,000 pgas, 4.7 M cycles) 20,434 MiB and 4.3 s; the full prototype shard (6.75 M pgas, 60 M cycles) 28,307 MiB and 10.8 s; beside the miner 15,670 and 30,039 MiB. No environment knob of SP1 moves the 13.9 GB floor and the GPU server has no options of its own, so on this build a 12 GB card proves nothing, a 16 GB card only empty shards, a 24 GB card the adopted full shard alone and beside the miner (22,210 MiB and 13.2 s, measured on the 32 GB card: the 5090's allocation pattern, not yet a run on a 24 GB card) and a 32 GB card the prototype shard beside the miner with 2.5 GB spare. The litepaper line now says so; the 12 GB gate returns when a prover build with a smaller floor is measured on a 12 GB card</td><td class="iv">none yet</td></tr>
<tr data-status="designed"><td class="n">17</td><td class="claim">The chip resistance claim: at launch the strongest chip in the public model reaches 2.1x (k = 1) to 3.9x (k about 0.33) per joule against an RTX 5090 under class v4, live from genesis on the testnet and the mainnet; the ladder's second rung brings it to about 2.8x; class v5 makes the dataset the chain's state so a stateless or stale chip is wrong on every item; the hot-set cache is bounded at 1.067x at the ceiling and the weak-day FPGA at 12 percent on 12 days a century, both routed to the next class; datacentre silicon does not change the question; a stored-dataset chip pays for itself only at about USD 100 M of market cap in two years; without class v4 the same chip would reach 5x to 9x (the class v3 baseline, the devnet's starting state, never the launch state)<div class="where">the home page's chip line, the litepaper's chip section (/litepaper#chip-model), the miner page's line</div></td><td><span class="st st-0">tested by the team (every card, the verifier, the two attack-pass bounds, the H100), the chip itself modelled, class v5 and the ladder designed, the X9 core claimed and never measured</span></td><td class="mono"><code>docs/analysis/chip-model-v3.md</code> 5 and 6; <code>docs/analysis/latency-shadow-2026-10-06.md</code>; <code>docs/plans/counter-asic-3-status.md</code>; <code>docs/analysis/attack-pass/f8-uniform.md</code>, <code>f4-weakday.md</code>, <code>docs/analysis/ca3-v4-uniform.md</code>; <code>docs/design/class-v5-stored-state.md</code>; the H100 and market-cap rows of 7 October; <code>docs/plans/cryptanalysis/in-house-pass.md</code> (the internal adversarial pass)</td><td>the chip model's arithmetic in its file; the card rows by the benchmark package; the attack-pass harnesses <code>tools/attack/f8-uniform</code> and the F4 census; the verifier by <code>igneum-pow bench</code></td><td>136 MH/s at 350 W (5090, bench) and 290 W (app); 27 MH/s at 21 W (M5 Max); 249 MH/s (H100 SXM) at 98 percent of its read ceiling, 1.78x hash, 1.15x MH/W, a third per rented dollar; 2.33 ms per warp; 2.1x, 3.9x, 2.8x at launch; 1.067x at the ceiling; 12 percent on 12 days a century; 10.85 ms at rung 3; USD 100 M; 5.1x to 9.2x the class v3 baseline; 6 and 7 October 2026, the M5 Max, the RTX 5090 Windows rig's RTX 5090, the three-card Windows rig's RX 9070 XT and RTX 4070, a rented H100 SXM, igneum-build-1</td><td class="iv">none yet; the next test is the internal adversarial pass (three lanes new to the hash code, outsider inputs only, reports published whole), and the one outside check is staged and waits on its escrow and the publish word</td></tr>
<tr data-status="designed"><td class="n">17</td><td class="claim">The chip resistance claim: at launch the strongest chip in the public model reaches 2.1x (k = 1) to 3.9x (k about 0.33) per joule against an RTX 5090 under class v4, live from genesis on the testnet and the mainnet; the ladder's second rung brings it to about 2.8x; class v5 makes the dataset the chain's state so a stateless or stale chip is wrong on every item; the hot-set cache is bounded at 1.067x at the ceiling and the weak-day FPGA at 12 percent on 12 days a century, both routed to the next class; datacentre silicon does not change the question; a stored-dataset chip pays for itself only at about USD 100 M of market cap in two years; without class v4 the same chip would reach 5x to 9x (the class v3 baseline, the devnet's starting state, never the launch state)<div class="where">the home page's chip line, the litepaper's chip section (/litepaper#chip-model), the miner page's line</div></td><td><span class="st st-0">tested by the team (every card, the verifier, the two attack-pass bounds, the H100), the chip itself modelled, class v5 and the ladder designed, the X9 core claimed and never measured</span></td><td class="mono"><code>docs/analysis/chip-model-v3.md</code> 5 and 6; <code>docs/analysis/latency-shadow-2026-10-06.md</code>; <code>docs/plans/counter-asic-3-status.md</code>; <code>docs/analysis/attack-pass/f8-uniform.md</code>, <code>f4-weakday.md</code>, <code>docs/analysis/ca3-v4-uniform.md</code>; <code>docs/design/class-v5-stored-state.md</code>; the H100 and market-cap rows of 7 October; <code>docs/plans/cryptanalysis/in-house-pass.md</code> (the internal adversarial pass)</td><td>the chip model's arithmetic in its file; the card rows by the benchmark package; the attack-pass harnesses <code>tools/attack/f8-uniform</code> and the F4 census; the verifier by <code>igneum-pow bench</code></td><td>136 MH/s at 350 W (5090, bench) and 290 W (app); 27 MH/s at 21 W (M5 Max); 249 MH/s (H100 SXM) at 98 percent of its read ceiling, 1.78x hash, 1.15x MH/W, a third per rented dollar; 2.33 ms per warp; 2.1x, 3.9x, 2.8x at launch; 1.067x at the ceiling; 12 percent on 12 days a century; 10.85 ms at rung 3; USD 100 M; 5.1x to 9.2x the class v3 baseline; 6 and 7 October 2026, the M5 Max, the RTX 5090 Windows rig's RTX 5090, the three-card Windows rig's RX 9070 XT and RTX 4070, a rented H100 SXM, igneum-build-1 The k about 0.33 bound is the implied core of Bitmain's Antminer X9 (RandomX; 1,000 KH/s, 2,472 W, 2.47 J per KH, USD 5,600; pre-orders 26 December 2025), withdrawn in mid-May 2026 with buyers refunded before any unit shipped, no independent benchmark, commodity Sophgo SG2044 server SoCs with an AES accelerator, no tapeout: a claimed, unmeasured figure carried as the pessimistic bound, not a calibration point (attack pass AP-F5-1, 7 October 2026).</td><td class="iv">none yet; the next test is the internal adversarial pass (three lanes new to the hash code, outsider inputs only, reports published whole), and the one outside check is staged and waits on its escrow and the publish word</td></tr>
<tr data-status="tested by the team"><td class="n">18</td><td class="claim">The chip resistance measurements: the program is latency-bound (random reads), not bandwidth-bound, on every card we own, and sits beyond a card's on-chip cache<div class="where">Litepaper Mining ("waits on memory latency, not on maths or bandwidth"), vs RandomX; the numbers page</div></td><td><span class="st st-2">tested by the team</span></td><td class="mono">readwidth e752fc7 (<code>docs/plans/read-width.md</code>), ca2-era 78c0ee4, ca2-cache 2de19e5 (<code>docs/plans/hot-table.md</code>)</td><td>The dependent-read probes at 32 to 1,024 MiB and the hash rate per class on the three cards; the latency-bound share = rate over the probe ceiling per load</td><td>Latency-bound share at the 1 GiB dataset: RTX 5090 0.96 (v2) and 1.01 (v3), RX 9070 XT 0.87 and 0.95, M5 Max 1.01 and 1.06; wider reads do not close the AMD gap (the 9070 XT does 2.4 G dependent reads per second at every width; the 5090 goes bandwidth-bound at 64 B, share 0.58); a 32 to 96 MiB hot table is not kept resident by any card while the dataset streams (g 0.80 to 0.87 in the added form). 5 October 2026</td><td class="iv">none yet</td></tr>
<tr data-status="tested by the team"><td class="n">19</td><td class="claim">The lottery hash is sound as a hash: uniform output, deterministic, no out-of-bounds read, fuzzed; class v3 bit-exact on the three vendors<div class="where">Litepaper vs RandomX ("Every number above is measured and logged"), the numbers page</div></td><td><span class="st st-2">tested by the team</span></td><td class="mono">ca2-mixer 1ab8b21 (<code>tests/mixer.rs</code>, <code>tests/scratch.rs</code>), ca2-era 78c0ee4, ca2-soundness a465881 (<code>docs/analysis/scratch-soundness.md</code>), <code>igneum-pow/tests/packs.rs</code></td><td>The crate suite (53 + 4 + 19 + 7), the Metal fuzz, edge, stats and determinism runs on the v3 construction, the pack vectors and 2^24 fingerprints on Metal, Apple OpenCL, the RTX 5090 and the RX 9070 XT, the 1,024-hash CPU re-check per card</td><td>Class v3 (mixer x8 + era): 200-program fuzz 200 of 200 on Metal, every tenth on Apple OpenCL; the pinned v3 packs 3/3 + 3/3 and 96 of 96 lanes on Metal and Apple OpenCL; the six era packs' fingerprints equal on the three vendors (the three-card Windows rig (RTX 5090, RTX 4070, RX 9070 XT) job run-ca2-era-pc1-20261005, 5 October 2026); the v2 exports byte-identical on the v3 crate; the final-class PC rows and the G2 re-check: job run-ca2-era-pc1b-20261005 (pending at the time of writing)</td><td class="iv">none yet</td></tr>
<tr data-status="implemented"><td class="n">20</td><td class="claim">No premine, no pre-sale, no allocation: every coin is minted by the schedule and every coin goes to the block producer (80%) and the proving pool (20%)<div class="where">Homepage stats and Economics tiles; litepaper Supply, Economics</div></td><td><span class="st st-1">implemented</span></td><td class="mono">repo <code>6ac80a3</code>; fork "igneum-node devnet v0"; <code>consensus/core/src/igneum.rs</code>, <code>coinbase.rs</code></td><td><code>cargo test -p kaspa-consensus-core igneum</code> (8 pass: subsidy table, ramp, split, cap) and <code>cargo test -p kaspa-consensus coinbase</code> (8 pass); <code>igneum-miner inspect 40</code>; bench-log "igneum-node devnet v0"</td><td>Coinbases on the devnet: 80/20 exact on 39 of 39 single-payee blocks, the 20% to the <code>igneum-proving-pool-v0</code> output; the per-second schedule sums to under the 4,000,000,000 cap by less than 100 coins; 3,168,808,781 units per DAA second in years 0 to 2, halving at 63,115,200 DAA s. 3 October 2026, Apple M5 Max. The devnet genesis carries no allocation; the mainnet genesis does not exist yet, so the claim is about the code and the stated rule, not a launch that has happened</td><td class="iv">none yet</td></tr>

View file

@ -158,8 +158,8 @@
"date": "2026-10-06",
"source": "bench log: 6 October 2026, Counter ASIC 3.0 item 8, the 5090 rows (the control row)",
"by": "measured by the team",
"note": "the control; 290 W in the app on the same card",
"v4_cost": "about +80 W for 0.2 percent of rate at the unlocked 2,850 MHz core, measured 6 October 2026; the efficiency pass (clock and voltage under class v4) runs 7 October",
"note": "the control, unlocked; the 6 October +80 W reading was at the app's tuned cap; the rate is memory-bound from 2,850 to 1,400 MHz (136.8 to 135.0 MH/s), the best MH per watt at the lowest lock on the grid, so the knee is below 1,400 MHz (the second pass runs to the driver's floor)",
"v4_cost": "+145.3 W at the unlocked core (475.5 against 330.2 W) for +0.18 percent of rate; +88.3 W at the 1,400 MHz lock (316.3 against 228.0 W) for +0.23 percent; measured 7 October 2026 (the class v4 efficiency pass on the team's Windows desk machine, 60 s steps, every fingerprint matched)",
"tuned": "stock, bench only (unlocked core)",
"driver_os": "NVIDIA driver, Windows 11",
"hive": {
@ -777,6 +777,27 @@
"pl_w": null,
"label": "stock (no measured tune point; the 5080 and 9070 XT passes and the class v4 efficiency pass land theirs when read)"
}
},
{
"card": "NVIDIA RTX 5090 (32 GB)",
"generator": "v2",
"mh_s": 134.98,
"watts": 316.3,
"mh_per_w": 0.427,
"miner": "igneum-worker-cuda bench (installed worker 0.3.20), class v4 program v4-devnet-epoch0",
"date": "2026-10-07",
"source": "Counter ASIC 3.0 status: the class v4 efficiency pass (the efficiency pass job of 7 October 2026, 18:40 to 19:16 UTC, on the team's Windows desk machine, the core locked through the installed app's Power Helper task, no prompt)",
"by": "measured by the team",
"note": "the v3 control at the same lock 134.68 MH/s at 228.0 W (0.591 MH/W); recovered 159.2 W for 1.36 percent of rate against the unlocked class v4 point; the best MH per watt on the grid, so the knee is below 1,400 MHz",
"v4_cost": "+88.3 W at this lock for +0.23 percent of rate, measured 7 October 2026",
"tuned": "core lock 1,400 MHz (the efficiency pass's grid floor), memory 13,801 MHz, the driver's power limit untouched",
"driver_os": "NVIDIA driver 617.14, Windows 11",
"hive": {
"core_mhz": 1400,
"mem_mhz": 13801,
"pl_w": 575,
"label": "measured 7 October 2026 (the class v4 efficiency pass: the 1,400 MHz lock, the memory clock as read, the limit as the driver's default since the lock alone set the draw; the knee below 1,400 is the second pass's)"
}
}
]
}

File diff suppressed because one or more lines are too long