From 8117c988e33d08deaa75b78c45b718f49adab60f Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Sat, 3 Oct 2026 16:09:21 +0000 Subject: [PATCH] Chain scene: final label sits at the top and moves out of the way of blocks Co-Authored-By: Claude Fable 5.1 --- proto-metal/MEMHARD.md | 368 +++++++++++++++++++++++++++++++++++++++++ site/index.html | 2 +- 2 files changed, 369 insertions(+), 1 deletion(-) create mode 100644 proto-metal/MEMHARD.md diff --git a/proto-metal/MEMHARD.md b/proto-metal/MEMHARD.md new file mode 100644 index 00000000..22b7a191 --- /dev/null +++ b/proto-metal/MEMHARD.md @@ -0,0 +1,368 @@ +# Memory-hard dataset for the Igneum lottery hash + +Date: 3 October 2026. Machine: Apple M5 Max (40 GPU cores, 12 performance + 6 efficiency CPU cores, 64 GB +unified memory), macOS Darwin 25.6.0, Swift 5.8.1 from Command Line Tools, no Xcode, Metal shaders compiled at +runtime. Build: `swiftc -O -o igneum-bench main.swift -framework Metal`. Every number below was produced on this +machine on this date by the command shown above its table. The CPU figures are one core of a single-threaded +process; nothing in the verifier uses threads. + +Why this exists: `TESTS.md` section 7 measured that the prototype's closed-form dataset element (six integer +operations) lets a miner recompute every dataset word in registers and run about 110x faster than the honest kernel +that reads the 1 GiB buffer. The hash was not memory-hard. The design doc fixes the remedy: a 256 MB RandomX-style +cache and 8 dependent reads per dataset item. This file specifies that construction exactly, and measures it. + +The lottery program itself is unchanged. It still executes `dst ^= dataset[src & MASK]` on 4-byte words of a 1 GiB +dataset (2^28 words, MASK 0x0fffffff). Only where the words come from changed. The closed form stays available +behind `--closed-form` so every result can be compared. + +## 1. The construction + +All arithmetic is on unsigned 32-bit words modulo 2^32. `rotl(x, n)` is a left rotation by n in 1..31. `||` is +concatenation of word vectors. Indices are zero based. + +### 1.1 Day key + +`K[0..7]` are the eight words of `seedWords("day/" + day)`: FNV-1a 64 over the UTF-8 bytes of the string, computed +four times with basis `0xcbf29ce484222325 ^ (salt * 0x9E3779B97F4A7C15)` for salt 0..3, each finalised with +`h ^= h >> 33; h *= 0xff51afd7ed558ccd; h ^= h >> 33`, low word then high word. This is the same derivation the +closed form used; the closed form's `d0, d1` are `K[0], K[1]`. For day `2026-10-03`: +`K = 3067619f 3c269176 84a03b03 f8c63294 ff977c5b e60def3e 63630141 b8fbcb58`. + +### 1.2 Block function B (ChaCha12 core with feed-forward) + +Input `x[0..15]`, output `y[0..15]`: + +``` +y = x +repeat 6 times: + QR(y0, y4, y8, y12) QR(y1, y5, y9, y13) QR(y2, y6, y10, y14) QR(y3, y7, y11, y15) columns + QR(y0, y5, y10, y15) QR(y1, y6, y11, y12) QR(y2, y7, y8, y13) QR(y3, y4, y9, y14) diagonals +y[i] += x[i] for i in 0..15 +``` + +with the standard ChaCha quarter round + +``` +QR(a, b, c, d; r1, r2, r3, r4): + a += b; d ^= a; d = rotl(d, r1) + c += d; b ^= c; b = rotl(b, r2) + a += b; d ^= a; d = rotl(d, r3) + c += d; b ^= c; b = rotl(b, r4) +``` + +and rotations `(16, 12, 8, 7)` in B. Twelve rounds, no key schedule beyond the input block. + +### 1.3 Cache (256 MiB) + +The cache is `2^26` words = `2^22` lines of 16 words (64 bytes). Lines are grouped into `2^16` segments of +`2^6 = 64` lines. Segment `s`, line `j` lives at word offset `(s * 64 + j) * 16`. Each segment is a sequential +chain: + +``` +sigma = (0x61707865, 0x3320646e, 0x79622d32, 0x6b206574) the ChaCha constants +tag = (0x49676e65, 0x756d4d48) "Igne", "umMH" +prev = 0^16 +for j in 0..63: + in = prev XOR (sigma[0..3] || K[0..7] || s || j || tag[0..1]) 16 words + line = B(in) + cache[(s * 64 + j) * 16 .. + 15] = line + prev = line +``` + +Line j of a segment therefore costs j + 1 block evaluations to recompute from nothing, 32.5 on average. The 65,536 +segments are independent, which is the GPU's parallelism for the fill (one thread per segment). + +### 1.4 Mixer parameters drawn from the key + +One SplitMix64 stream seeded with `K[0] | (K[1] << 32)` (the SplitMix64 of `main.swift`: state += 0x9E3779B97F4A7C15, +then the two xor-shift-multiply steps), drawn in this order: + +| Parameter | Count | Draw | +|---|---|---| +| `ROT[0..7]` | 8 | `1 + (next() mod 31)`, so 1..31 | +| `MUL[0..15]` | 16 | `low32(next()) OR 1`, always odd | +| `RC[0..15]` | 16 | `low32(next())` | + +For day `2026-10-03`: `ROT = 20 20 19 4 26 3 3 27`; `MUL` and `RC` are written in full in +`../proto-cuda/packs/igneum-genesis-mh/program.h` (`IGNEUM_MIX_MUL_INIT`, `IGNEUM_MIX_RC_INIT`). + +### 1.5 Mixer M_r (fixed shape, seed-parameterised) + +On a 16-word state `s`, with round key `rk = (r + 1) * 0x9E3779B9`: + +``` +for i in 0..15: s[i] = (s[i] XOR (RC[i] + rk)) * MUL[i] +QR(s0, s4, s8, s12; ROT0..3) QR(s1, s5, s9, s13; ROT0..3) QR(s2, s6, s10, s14; ROT0..3) QR(s3, s7, s11, s15; ROT0..3) +QR(s0, s5, s10, s15; ROT4..7) QR(s1, s6, s11, s12; ROT4..7) QR(s2, s7, s8, s13; ROT4..7) QR(s3, s4, s9, s14; ROT4..7) +``` + +Sixteen odd multiplications (bijective per word), then one ChaCha-shaped double round with the four column +rotations `ROT[0..3]` and the four diagonal rotations `ROT[4..7]`. About 130 integer operations. This is the +prototype's stand-in for RandomX's SuperscalarHash: the shape is fixed, the constants come from the seed. + +### 1.6 Dataset item + +Item `t` (0 <= t < 2^24 for the 1 GiB dataset) is 16 words: + +``` +s[i] = K[i] for i in 0..7 +s[8 + i] = t * MUL[i] + RC[i] for i in 0..7 +for r in 0..7: + s = M_r(s) + a = s[0] AND 0x003fffff cache line index, 2^22 lines + s[i] ^= cache[a * 16 + i] for i in 0..15 +s = M_8(s) +item(t) = s +``` + +Eight dependent cache reads: the address of read r is a function of every earlier read. Nine mixer applications. +The final `M_8` makes every output word depend on all 64 bytes of the last line. + +### 1.7 Dataset word + +`dataset[w] = item(w >> 4)[w AND 15]`. The dataset is `2^28` words; item `t` occupies words `16t .. 16t + 15`. A +smaller dataset (`--dataset-log2 D`) is the prefix of items `0 .. 2^(D-4) - 1`, so an item has the same value at +every dataset size, exactly as the closed form did. + +### 1.8 Where each piece runs + +| Piece | GPU (Metal) | CPU verifier (Swift) | CUDA pack | +|---|---|---|---| +| Cache fill | `igneum_cache_fill`, one thread per segment, 65,536 threads | `cpuFillCache`, one core, 65,536 chains in order | `igneum_cache_fill` kernel; `mh_cache_segment` on the host too | +| Dataset build | `igneum_build`, one thread per item, 2^24 threads | not done: the verifier never holds the dataset | `igneum_build` kernel | +| Word on demand | inline kernel only (`mh_word`, the shortcut measurement) | `MemhardCPU.fetch`: the 32 lanes of a load batched and interleaved round by round | `mh_word` on the host for the self-test | +| Hash | `igneum_hash`, unchanged | `cpuWarp`, unchanged apart from the batched fetch | `igneum_hash`, unchanged | + +The emitted text of B, the cache segment, M_r and the item derivation is produced once by `emitMemhardCore` in two +dialects (Metal, CUDA C++) with every constant as a literal, so the GPU kernels, the pack and its host reference are +the same text. The Swift verifier is a separate implementation of the same definitions; the agreement tests below +are what tie the two together. + +### 1.9 The CPU verifier + +The verifier holds `K`, the mixer parameters and the 256 MiB cache. It computes the cache itself (one core, timed +below) and never touches the dataset buffer. When the interpreter reaches a load it gathers the 32 lane indices, +deduplicates the items, derives them with the 32 chains interleaved round by round (`deriveItems`: all mixers for +round r, then all cache-line xors for round r), and hands each lane its word. Interleaving is what lets the eight +dependent misses of one lane overlap with the other 31 lanes' misses; without it the verifier would pay about +8 x 100 ns of DRAM latency per item in series. + +## 2. Measurements + +### 2.1 Cache fill and dataset build + +Commands: `./igneum-bench --hours 3` (default, memory-hard) and `./igneum-bench --closed-form --hours 2`. + +| Step | GPU, first in process | GPU, second | One CPU core | +|---|---|---|---| +| Compile cache-fill + build kernels (Metal, runtime) | 6.4 ms (3.2 to 6.4 across runs) | | | +| Cache fill, 256 MiB, 65,536 chains x 64 blocks | 1.95 ms (0.62 to 2.03 across 5 runs) | 2.06 ms (0.84 to 2.06) | 184.5 to 190.6 ms (Swift, 5 runs); 161.5 ms (C++ host reference, clang -O2, in the emulator run) | +| Dataset build, 1 GiB, 2^24 items, 8 cache reads each | 29.4 ms | 20.6 ms (20.5 to 20.6) | not done by the verifier | +| Closed-form fill, 1 GiB, for comparison | 2.34 ms | 2.54 ms (394 GB/s write) | | + +The build does 2^27 random 64-byte cache reads in 20.6 ms: 6.5 G line reads per second, about 417 GB/s of cache-line +traffic plus 52 GB/s of dataset writes, so the 256 MiB cache is served largely from on-chip cache during the build. +The CPU fill is 2^22 ChaCha12 blocks at about 44 ns each. The GPU cache fill varied between 0.6 and 2.1 ms across +runs with no change in code; the variation was not chased. The GPU cache was compared word for word with the CPU +cache in every run (all 67,108,864 words equal, FNV-1a 64 `48c4f5bf24166b2e` for day 2026-10-03), and 1,024 sampled +dataset words (including 0, 1, 15, 16 and the last index) were read back from the GPU-built dataset and matched the +CPU derivation in every run. + +### 2.2 Stored dataset versus inline derivation (the shortcut ratio) + +Commands: `./igneum-bench --hours 1`, `./igneum-bench --inline-dataset --hours 1`, and the same two with +`--closed-form`; then both again with `--dataset-log2 26`. Seed `igneum-genesis`, 104 loads per hash, 4 timed +batches of 2^22 hashes after a warm-up batch. The inline kernel replaces every `dataset[a & MASK]` by a recomputation +of that word and never reads the dataset buffer: `ds_elem(a & MASK)` for the closed form, `mh_word(cache, a & MASK)` +(8 dependent 64-byte cache reads plus 9 mixers) for the memory-hard construction. + +| Dataset construction | Dataset | Honest kernel (reads dataset), Mhash/s | Inline kernel (recomputes words), Mhash/s | Inline / honest | +|---|---|---|---|---| +| closed form (before) | 1 GiB | 45.2 | 5,014 wall, 6,281 GPU time | 111x faster | +| memory-hard (after) | 1 GiB | 45.2 | 9.49 | 0.21 (4.8x slower) | +| memory-hard (after) | 256 MiB | 94.8 | 9.48 | 0.10 (10x slower) | + +The honest rate is identical in both constructions (45.2 Mhash/s, 18.8 GB/s useful), as it must be: the hash kernel +text is the same and only the buffer contents differ. The inline figure for the closed form repeats the TESTS.md +section 7 measurement (wall-clock figure dominated by command overhead; the GPU-time figure is the real ALU rate). +The memory-hard inline kernel is bound by the 8 dependent 64-byte reads per word: 104 loads x 8 = 832 dependent +cache-line reads per hash, against 104 independent 4-byte reads for the honest kernel. The result is the same at +256 MiB and 1 GiB because the inline kernel reads only the cache. + +### 2.3 CPU verification time per 32-lane warp + +Command: `./igneum-bench --hours 3` for `igneum-genesis`, and `./igneum-bench --seed igneum-second-seed --hours 2` +for the 144-load program (the top of the generator's usual range). One core, release build, 1 GiB dataset. "Items" +is the number of 64-byte dataset items derived from the cache for the warp (loads per hash x 32 lanes, less the +rare duplicate). "Single" is the first, cold run of each warp; "avg of 20" is the steady figure the gate is judged +on. + +| Seed | Loads/hash | Items derived per warp | CPU verify ms/warp, avg of 20 | Single cold run, ms (3 warps) | Closed form, same program, ms/warp | GPU vs CPU | +|---|---|---|---|---|---|---| +| igneum-genesis | 104 | 3,328 | 0.649 | 1.24 to 1.52 | 0.017 | PASS 3/3 warps | +| igneum-genesis/epoch1 | 104 | 3,182 to 3,224 | 0.631 | 1.16 to 1.22 | 0.017 | PASS 3/3 warps | +| igneum-genesis/epoch2 | 112 | 3,584 | 0.701 | 1.31 to 1.49 | | PASS 3/3 warps | +| igneum-second-seed | 104 | 3,328 | 0.801 | 1.43 to 1.50 | 0.016 (README) | PASS 3/3 warps | +| igneum-second-seed/epoch1 | 144 | 4,608 | 1.205 | 1.70 to 2.11 | 0.017 (README) | PASS 3/3 warps | + +Per item that is 0.19 to 0.26 microseconds, roughly 1,170 integer operations plus 8 dependent cache-line reads with +the misses of 32 lanes overlapped. The 10 ms gate is met by every program measured: the worst steady figure is +1.2 ms (144 loads, 4,608 items), the worst cold single run 2.1 ms, so the margin is about 8x on the steady figure +and about 5x on a cold run. The verifier is about 40x to 70x slower than with the closed form, which is the price of +the derivation. Verifying a block needs one warp (32 hashes, of which the block's nonce is one), so these are +per-block figures. + +### 2.4 Levers (measured, not adopted) + +Both levers keep the default generator byte for byte when their flag is absent (`--load-weight 25`, `--wide-frac 0`), +and neither consumes extra random draws, so a lever changes a program only where it acts. The default generator was +NOT changed. Seeds `igneum-genesis`, `/epoch1`, `/epoch2`; 1 GiB; same commands with the lever flags added. + +Lever (a), `--load-weight 17`: load weight 17 percent (about one load per six instructions), the other ten ops +scaled to 83 by largest remainder (`add=13 xor=11 mul=9 mad=9 shfl=9 rotl=8 sub=7 mulhi=7 rotr=6 or=4`). + +Lever (b), `--wide-frac 50`: half of the load instructions (those whose already-drawn selector bit is below 16) +become `wload`: all 32 lanes read consecutive words of one 128-byte block whose base is lane 0's source register, +masked and aligned down to 32 words (`dataset[(simd_broadcast(a, 0) & (MASK & ~31)) + lane]`; CUDA +`__shfl_sync(0xffffffff, a, 0)`). A wide load touches 2 items per warp instead of 32. + +| Variant | Seed | Loads/hash (wide) | Items per warp | CPU verify ms/warp | GPU Mhash/s | GB/s useful | GPU vs CPU | +|---|---|---|---|---|---|---|---| +| default | igneum-genesis | 104 (0) | 3,328 | 0.649 | 45.2 | 18.8 | PASS | +| default | /epoch1 | 104 (0) | 3,328 | 0.631 | 48.4 | 20.1 | PASS | +| default | /epoch2 | 112 (0) | 3,584 | 0.701 | 40.0 | 17.9 | PASS | +| (a) load weight 17 | igneum-genesis | 72 (0) | 2,304 | 0.457 | 73.4 | 21.1 | PASS | +| (a) load weight 17 | /epoch1 | 80 (0) | 2,560 | 0.481 | 72.7 | 23.3 | PASS | +| (a) load weight 17 | /epoch2 | 80 (0) | 2,560 | 0.512 | 55.0 | 17.6 | PASS | +| (b) wide 50 percent | igneum-genesis | 104 (32) | 2,368 | 0.489 | 56.1 | 23.3 | PASS | +| (b) wide 50 percent | /epoch1 | 104 (72) | 1,168 | 0.233 | 135.2 | 56.2 | PASS | +| (b) wide 50 percent | /epoch2 | 112 (80) | 1,184 | 0.251 | 103.4 | 46.3 | PASS | +| (a) + (b) | igneum-genesis | 72 (32) | 1,344 | 0.276 | 108.2 | 31.2 | PASS | +| (a) + (b) | /epoch1 | 80 (48) | 1,120 | 0.223 | 139.1 | 44.5 | PASS | +| (a) + (b) | /epoch2 | 80 (48) | 1,120 | 0.240 | 106.0 | 33.9 | PASS | + +Note that under lever (a) the programs differ from the default ones (the op roll lands differently), so the rows are +not the same program with fewer loads; under lever (b) the instruction list is the default one with some loads +widened. Reading: (a) cuts CPU time in proportion to loads and raises the GPU rate about 1.6x because the kernel +does fewer random reads per hash; useful bandwidth is unchanged, so the kernel is still memory bound, just with +fewer loads. (b) cuts CPU time up to 2.7x but raises the GPU rate up to 2.8x and the useful bandwidth up to 2.8x: +coalesced 128-byte loads are what GPUs do well, and the random-access bound that the RTX 5090 sweep identified as +the defence (`docs/bench-log.md`, 23.7 G random loads/s past the L2) is partly removed. Lever (b) makes the hash +less random-access bound. Neither lever is needed for the gate. + +### 2.5 Agreement tests re-run with the new dataset + +All commands as in `TESTS.md`, now with the memory-hard dataset as the default. Every test run also compares the +GPU cache with the CPU cache word for word first (`cache PASS` in the summary). + +Fuzz, `./igneum-bench --fuzz 200`: + +| Dataset | Programs | Pass | Fail | +|---|---|---|---| +| 2^24 words (64 MiB) | 63 | 63 | 0 | +| 2^26 words (256 MiB) | 64 | 64 | 0 | +| 2^28 words (1 GiB) | 73 | 73 | 0 | +| all | 200 | 200 | 0 | + +200 programs, 800 warps, 25,600 hashes, 0 mismatches, 0 compile failures, 0 static mask failures, 0 generator +contract failures; loads per hash 64 to 208; the three datasets were built from the cache in 49.9 ms. CPU +interpreter total 1,228.5 ms (16 ms with the closed form in TESTS.md), GPU dispatch 160 ms, wall 3.0 s. FUZZ: PASS. + +Edge, determinism, memcheck, `./igneum-bench --edge --determinism --memcheck`: EDGE PASS 14 of 14 counted cases +(128/128 lanes each, preconditions held); DETERMINISM PASS (identical MSL from two generations, fingerprint +`62a4f0eb018df273` on 5 runs and 3 compiles including one forced cold recompile, 8 CPU warps match, two dataset +builds fingerprint `e8a68cab6c55dce8` both times, 4,096 sampled words including 0, 1, MASK-1, MASK match the CPU +derivation); MEMCHECK PASS (13 of 13 `dataset[rN & MASK]` at three sizes, CUDA twin 13 of 13 with no `ds[` write, +4 wrapping batches at 4 MiB completed, 4 warps verified, 416 of 416 load indices exceeded MASK before masking). +The fingerprints differ from the closed-form run in TESTS.md because the dataset contents differ; that is expected. + +Stats, `./igneum-bench --stats` (2^20 nonces per seed, 1 GiB): + +| Seed | Loads/hash | Bit freq min..max | Max bit z | Avalanche 1k mean / std | Avalanche 16k mean / std | Per-output-bit flip prob | Worst chi2 z | Dups | +|---|---|---|---|---|---|---|---|---| +| igneum-genesis | 104 | 0.4990..0.5008 | 2.13 | 32.01 / 3.91 | 31.979 / 4.00 | 0.490..0.508 | 1.59 | 0 | +| igneum-genesis/stats1 | 128 | 0.4992..0.5010 | 1.98 | 32.06 / 4.03 | 31.969 / 3.97 | 0.491..0.509 | 2.00 | 0 | +| igneum-genesis/stats2 | 176 | 0.4990..0.5010 | 2.14 | 32.22 / 3.76 | 32.006 / 4.00 | 0.492..0.506 | 1.58 | 0 | + +STATS: PASS (the same sanity check as before; not a proof of anything). + +Vectors for the pack: 3 warps (base nonces 0, 4096, 1000000) x 3 seeds across the `--hours 3` run plus the export, +all PASS; in total this session compared 21 bench warps, 800 fuzz warps, 56 edge warps, 8 determinism warps and 4 +memcheck warps against the CPU with the new dataset, zero mismatches. + +### 2.6 CUDA pack + +`./igneum-bench --seed igneum-genesis --export-pack ../proto-cuda/packs/igneum-genesis-mh` wrote the new pack +(`kernel.cu`, `memhard.h`, `program.h`, `vectors.h`, `program.json`, `vectors.json`, `program.metal`, +`memhard.metal`) after: CPU cache == GPU cache (all words), Metal GPU cross-check PASS 3/3 warps, 66 GPU dataset +words == CPU derivation. `vectors.h` carries the 96 hash outputs, dataset head and `[MASK]`, 64 sampled dataset +words, the cache head and last line, and FNV-1a 64 of the whole cache. + +`proto-cuda/emu/emu.sh igneum-genesis-mh --batch-log2 13 --batches 1 --block-warps 2` (clang, host threads, no +GPU): cache check PASS (emulated GPU cache == host cache on all 2^26 words, host FNV `48c4f5bf24166b2e` == Mac, +head and last line == Mac), dataset self-test PASS at 1 GiB (head 16, `[MASK]`, 64 random points vs host +derivation, 64 Mac samples), vectors PASS 3/3 standalone and 2/2 in batch at 2 warps per block. `-Wall -Wextra` +clean. The old closed-form pack `igneum-genesis` still passes in the emulator with the updated `host.cu`. A +closed-form export re-run to a scratch directory produced `kernel.cu` and `program.metal` byte-identical to the +checked-in `igneum-genesis` pack (the headers gain `IGNEUM_DATASET_MODE 0` and the sample arrays). + +## 3. Conclusion + +**10 ms gate: MET.** On one Apple M5 Max core the verifier derives every dataset word a warp needs from the 256 MiB +cache and takes 0.63 to 0.80 ms per 32-lane warp at 104 loads per hash and 1.21 ms at 144 loads (4,608 items). The +worst cold single warp was 2.1 ms. Margin about 8x steady, about 5x cold. The verifier never holds the dataset. +A one-time 185 ms cache fill per day key is additional and amortised over every block of that day. + +**Shortcut ratio: 111x faster before, 4.8x slower after.** With the closed form a kernel that skipped the dataset +ran at 5,014 Mhash/s against 45.2 honest. With the memory-hard dataset the same shortcut runs at 9.49 Mhash/s against +45.2 honest (0.21), and at 0.10 of the honest rate when the honest kernel has a 256 MiB dataset. The honest path is +the fastest path on this GPU. + +**Lever recommendation: none.** Both levers were implemented and measured and both are off by default. The gate is +met without them. Lever (a) would be the one to reach for if a slower CPU or a longer program ever threatened the +gate: it lowers CPU time in proportion and leaves the kernel memory bound. Lever (b) is not recommended: it trades +random 4-byte loads for coalesced 128-byte loads, triples the GPU rate on some programs and raises useful bandwidth +2.8x, which erodes the random-access bound that makes the dataset hard to escape with on-chip memory. The default +generator is unchanged (`--load-weight 25 --wide-frac 0` reproduce it byte for byte; the closed-form pack export is +byte-identical to the pack written before this work). + +**What remains unproven.** + +1. Hardware other than Apple. The CUDA pack passed only the clang emulation on this Mac. The RTX 5090 and the AMD rig + must run `igneum-genesis-mh` before any cross-vendor claim about the new dataset is made; the memory-hard inline + ratio in particular has not been measured on a discrete GPU (the 96 MiB L2 of the 5090 does not hold the 256 MiB + cache, which argues the ratio will hold, but that is an argument, not a number). +2. The shortcut ratio is a measurement against one attacker kernel (recompute every word, never read the dataset). + Time-memory trade-offs in between (store part of the dataset, recompute the rest) were not measured; by + construction each recomputed word costs 8 dependent reads, so partial storage can only interpolate between the + two measured points, but the curve was not drawn. A smarter attacker kernel (for example, hoisting the first mixer + rounds that depend only on t, or caching the first cache line of hot items) was not attempted. +3. Cryptographic strength of M_r and of the chained-block cache. ChaCha12 is a standard primitive used in a + non-standard chaining mode; the mixer is an ad hoc ARX-multiply construction with seed-drawn rotations (for this + day they include two rotations by 3 and two by 20 in the same quarter round). Nothing here is a proof of + preimage resistance, uniformity of the item distribution, or absence of weak keys (a `ROT` draw of all equal + values is possible and untested). The stats run shows no obvious bias in the hash output for three programs only. +4. The cost argument for inlining the cache itself (recompute a cache line from the segment chain instead of + reading it) is an estimate: 32.5 blocks x about 700 operations x 8 reads is about 180,000 operations per + dataset word against one 4-byte load. It was not measured because no kernel doing it was written. +5. Distinct cache lines touched per hash were not measured (TESTS.md asked for it). Analytically a hash touches + up to 832 of 4,194,304 lines; the working set of a warp is 26,624 lines, so a cache-resident shortcut is not + available to a warp, but a census over many nonces was not run. +6. The CPU figure is one M5 Max performance core. A slower core, or a verifier written without the 32-lane + interleaving, will be slower; the single-core Swift fill (185 ms) and the C++ fill (162 ms) bound what a + reasonable implementation should expect on this class of core. +7. The GPU cache fill time varied 0.6 to 2.1 ms between runs with identical code; the cause (likely GPU clock + state) was not isolated. It is a once-per-day cost and does not affect any conclusion. + +## 4. Flags and files + +- `main.swift`: `// MARK: - Memory-hard dataset` (CPU side: `chachaBlock`, `cpuFillCache`, `MixParams`, `mixer`, + `deriveItems`, `MemhardCPU`, `DatasetSource`), `emitMemhardCore` (the one text for Metal and CUDA), `memhardMSL`, + `DatasetContext` (GPU fill, build, checks), `LoadSource` (stored / inline), `GeneratorConfig` (levers), `wload`. +- Flags: default is memory-hard; `--closed-form` restores the original dataset; `--inline-dataset` is the shortcut + kernel for whichever construction is active; `--load-weight W` and `--wide-frac P` are the levers (defaults 25 + and 0 reproduce the original generator exactly). +- `../proto-cuda/packs/igneum-genesis-mh/`: the new pack. `../proto-cuda/host.cu` handles both modes + (`IGNEUM_DATASET_MODE`); old packs default to mode 0. +- Raw logs of every run quoted here were kept in the session scratchpad and are not checked in; each table is + reproducible with the command above it. diff --git a/site/index.html b/site/index.html index 1c96932d..44b87a4d 100644 --- a/site/index.html +++ b/site/index.html @@ -501,7 +501,7 @@ footer .wrap{padding-block:48px 32px} blocks.forEach(function(b){b.parents.forEach(function(q){if(blocks.indexOf(q)<0)return;var both=b.state==='proven'&&q.state==='proven';x.strokeStyle=both?'rgba(242,84,27,0.45)':'rgba(154,154,158,0.22)';x.lineWidth=both?1.5:1;x.beginPath();x.moveTo(b.x,b.y);var mx=(b.x+q.x)/2;x.bezierCurveTo(mx,b.y,mx,q.y,q.x,q.y);x.stroke();});}); // lock line: everything left of the newest locked block is final var lk=null;blocks.forEach(function(b){if(b.locked&&(!lk||b.x>lk.x))lk=b;}); - if(lk){x.strokeStyle='rgba(242,84,27,0.55)';x.lineWidth=1.5;x.setLineDash([5,6]);x.beginPath();x.moveTo(lk.x,6);x.lineTo(lk.x,H-6);x.stroke();x.setLineDash([]);x.fillStyle='rgba(154,154,158,0.9)';x.font='500 '+Math.max(10,Math.round(S*0.42))+'px IBM Plex Mono, monospace';x.textAlign='right';x.fillText('final',lk.x-8,H-10);} + if(lk){x.strokeStyle='rgba(242,84,27,0.55)';x.lineWidth=1.5;x.setLineDash([5,6]);x.beginPath();x.moveTo(lk.x,6);x.lineTo(lk.x,H-6);x.stroke();x.setLineDash([]);var topBusy=blocks.some(function(q){return Math.abs(q.x-lk.x)0){var g=x.createRadialGradient(b.x,b.y,0,b.x,b.y,S*1.6);g.addColorStop(0,'rgba(242,84,27,'+(0.45*b.glow)+')');g.addColorStop(1,'rgba(242,84,27,0)');x.fillStyle=g;x.beginPath();x.arc(b.x,b.y,S*1.6,0,Math.PI*2);x.fill();}