From b98090649e4d2a40fd1a8bdd206ad113dc22469c Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 08:16:48 +0000 Subject: [PATCH] attack-pass F3: PASS (0 of 64 and 0 of 1,024 lines under j+1, curve monotone, f=1 unchanged); AP-H1 box-clean hazard recorded The F3 record and its harness crate (tools/attack/f3-cache: the extracted chain, the exhaustive closure search, the pebbling cross-check, the store-set DP and brute force, the two planted broken chains). The optimal-placement observation (3.17 blocks per read at f=1/8, 16.0 at f=1/64) noted against funding.md B2. Co-Authored-By: Claude Fable 5.1 --- docs/analysis/attack-pass-2026-10.md | 31 +- docs/analysis/attack-pass/f3-cache.md | 148 ++++ tools/attack/f3-cache/Cargo.lock | 14 + tools/attack/f3-cache/Cargo.toml | 20 + tools/attack/f3-cache/run-box.sh | 39 + tools/attack/f3-cache/src/main.rs | 1003 +++++++++++++++++++++++++ 6 files changed, 1252 insertions(+), 3 deletions(-) create mode 100644 docs/analysis/attack-pass/f3-cache.md create mode 100644 tools/attack/f3-cache/Cargo.lock create mode 100644 tools/attack/f3-cache/Cargo.toml create mode 100644 tools/attack/f3-cache/run-box.sh create mode 100644 tools/attack/f3-cache/src/main.rs diff --git a/docs/analysis/attack-pass-2026-10.md b/docs/analysis/attack-pass-2026-10.md index 102c6fc1..4053416c 100644 --- a/docs/analysis/attack-pass-2026-10.md +++ b/docs/analysis/attack-pass-2026-10.md @@ -22,7 +22,7 @@ against the log before quoting it to the project lead. |---|---|---|---|---| | F1 | Shadow block compressibility and shortcut search | best compressed block within 5% of N on every program; no program over 10% compressible | pending evidence map + box run | RUNNING | | F2 | Mixer round margin (SAT/MILP, 1 to 4 keyed applications) | no distinguisher or shortcut beyond 2 of the 8 applications | pending evidence map (ca2-mixer) + box run | RUNNING | -| F3 | Chained cache j+1 bound and storage-vs-recompute curve | no derivation under j+1 blocks; curve monotone; f=1 point unchanged | pending evidence map (ca2-cache) + box run | RUNNING | +| F3 | Chained cache j+1 bound and storage-vs-recompute curve | no derivation under j+1 blocks; curve monotone; f=1 point unchanged | 0 of 64 and 0 of 1,024 lines under j+1 (exhaustive closure search, cross-checked by exhaustive pebbling at 10 lines, 10,240 pairs, 0 mismatches); both planted broken chains fire; curve monotone at both op counts; f=1 point 9,360 ops per item unchanged. Record `docs/analysis/attack-pass/f3-cache.md` | PASS | | F4 | Weak-day census over 2^24 day keys | fraction of days with gain over 1.1x under 2^-20 | pending box census | RUNNING | | F5 | Chip-model sweep + AWS F2 FPGA hour | evidence row 17 holds across the sweep; FPGA row under 27 M reads/s/W | sweep: 2.1x at k=1 GDDR7 reproduces, 3.2x at k=0.5, 4.1x at k=0.3 (matches ledger M32); FPGA row 2.3 to 2.9 G/s, 10 to 20 M reads/s/W (literature). FINDING: the k=0.33 figure is framed as the X9's measured core (M32) and a "measured class" (ladder branch ยง5a); the X9 was withdrawn before launch and never benchmarked. F2 hour SKIPPED: no AWS account | FIXED-AND-PASSED (sweep PASS; AP-F5-1 fixed and re-gated 7 Oct 2026: chip section re-run 2.1x at k=1 unchanged, identity grep 0 hits, site lane concurred); F2 hour SKIPPED | | F6 | Verifier worst case over 10^5 programs + O-1.14 laptop run | worst program under 10 ms cold on the half-core proxy and the laptop | v4 average 4.90 to 5.06 ms one-core cold, 8.23 ms half-core proxy (under 10 ms). Worst-case search over 10^5 and laptop run owed | RUNNING | @@ -57,8 +57,21 @@ between dependent reads. Gate: no distinguisher or shortcut beyond 2 of the 8 ap Method: exhaustive search on a 2^10-line model segment for a line derivable without an earlier line; the curve from f = 1/64 to 1 in ops per item. Known-failed shape: a line (s, j) computable in fewer than j+1 block evaluations without an earlier line (the MTP address-steering break shape). Gate: no derivation under j+1 blocks; curve monotone; -f=1 point unchanged. Result: RUNNING (prior `ca2-cache` evidence to be re-gated). What a failure moves: the chain -construction (a second feed-forward or a cross-segment tie). +f=1 point unchanged. Result, 7 October 2026, 09:10 to 09:12 UK on the box (`docs/analysis/attack-pass/f3-cache.md`; logs +`/srv/builds/igneum-wt-attack/attack-f3/r1-*.log`): PASS on all three clauses. The chain extracted from +`Cache::fill_segment` (verified equal to the code on 16 of 16 key and segment pairs; `block == chacha_block` on +100,000 random inputs) has line j fed by line j - 1 only; the exhaustive closure search finds 0 of 64 and 0 of +1,024 lines under j + 1 (every line costs exactly j + 1), cross-checked by an exhaustive pebbling search at 10 +lines (10,240 configuration and target pairs, 0 mismatches). The two planted chains fire: `skip2` (63 of 64 under +j + 1) and `nofeed` (every line in 1 block). The curve over stored cache lines is monotone non-increasing from +f = 1/64 to 1 at 608 (counted) and 700 (MEMHARD.md) ops per block; the f = 1 point is 9,360 ops per item, +41.7 MH/s at the 50 T op/s budget, unchanged. `ca2-cache` was the hot-table experiment, not a chain analysis, so +there was nothing to re-gate. Observation (coordinator and the F3 record, not a finding): `funding.md` B2 rank 2 +prices the trade-off at the naive placement; the optimal placement of every 8th line costs 3.17 blocks per read, +not 3.5, and 16.0 at f = 1/64, not 31.5 (brute force over 4,426,165,368 sets at n = 8); the chip stays worse than +the full mirror at every f under 1, so the verdict stands, and a `chacha_block` shortcut in chaining mode stays +the paid question (Lot A and B). What a failure would have moved: the chain construction (a second feed-forward or +a cross-segment tie). ### F4. Weak-day census over 2^24 day keys (hash lane, on the box) @@ -187,6 +200,18 @@ rule's monotonicity and its memoisation per seed block. Known-failed shape: a ch either direction; a step down needs the same. Result: RUNNING. What a failure moves: the step rule's text in spec 01 before the ladder is frozen. +## Operating hazards found by the pass + +AP-H1 (box scratch cleaned by builds; found by F3, 7 October 2026, 10:0x UK). `infra/build-server/remote-run.sh` +line 71 runs `git clean -qfd -e target -e 'target-*' ...` on `/srv/builds/` before every remote build, so +an untracked box scratch directory of one row (a venv, a log dir, a crate's `tools/attack/*/target`) is deleted by +the next build from any row. F3 protected its own directory through the box mirror's `.git/info/exclude`; the lane +then added `attack-*/`, `target-attack-*/`, `tools/attack/` and `.build-remote.log` to that file at 10:1x UK, after +which `git clean -fdn` on the mirror lists nothing (the clean has no `-x`, so the exclude file applies). The class +check is owed to the build-server lane: the clean line should spare a lane's declared scratch prefix (`-e 'attack-*'` +style, or read a per-worktree exclude list), and a CI check should fail a remote-run.sh whose clean line lacks it. +OPEN until that check lands (CLAUDE.md: a rule row closes only with its check). + ## Ledger rows AP-F5-1 (algorithm and hash lane, ours; the ladder lane is closed). The k about 0.33 chip-efficiency figure is diff --git a/docs/analysis/attack-pass/f3-cache.md b/docs/analysis/attack-pass/f3-cache.md new file mode 100644 index 00000000..a818af17 --- /dev/null +++ b/docs/analysis/attack-pass/f3-cache.md @@ -0,0 +1,148 @@ +# F3: the chained cache's j + 1 bound and the storage-against-recompute curve + +Attack-pass row F3 of `docs/plans/cryptanalysis.md` section 4.2 (the record is `docs/analysis/attack-pass-2026-10.md`). Run 7 October 2026, 09:10 to 09:12 UK (08:10 to 08:12 UTC in the logs), on igneum-build-1. Verdict: PASS on all three gate clauses. No line (s, j) is derivable in fewer than j + 1 block evaluations without an earlier line, by an exhaustive search over the block dependency graph extracted from the code at 64 and 1,024 lines, cross-checked by an exhaustive pebbling search over every configuration at 10 lines. The storage-against-recompute curve over cache lines is monotone from f = 1/64 to 1. The f = 1 point is unchanged. + +## Target + +| Item | Value | +|---|---| +| Commit | 924288d1 (the brief); the worktree HEAD moved to 11b375a0 during the run; `igneum-pow/src/memhard.rs` is byte-identical at both (blob ad42470b, `git diff --stat 924288d1 HEAD -- igneum-pow/src/memhard.rs` empty) | +| Construction, from the code | `Cache::fill_segment`: `in_j = prev XOR (sigma || K || seg || j || tag)`, `line_j = chacha_block(in_j)` where `chacha_block(x) = ChaCha12core(x) + x`, `prev_0 = 0`, `prev_j = line_{j-1}`; 64 lines per segment, 2^16 segments, 2^26 words (256 MiB) | +| Reads | `derive_items_mask`: 8 dependent reads per item at line index `s[0] AND mask`, so the segment and j of a read are uniform over the 2^22 lines (F8 checks the uniformity) | +| Known-failed shape | a line (s, j) computable in fewer than j + 1 block evaluations without an earlier line of segment s (the address-steering shape of the MTP break, Dinur and Nadler 2017, needs a data-dependent chain; this chain's inputs are fixed by the key, so the shape to search is a structural shortcut on the dependency graph) | +| Gate | no derivation under j + 1 blocks; the curve monotone; the f = 1 point unchanged | +| Prior evidence | none to re-gate: the `ca2-cache` branch named in the status board is the hot-table experiment (`docs/plans/hot-table.md`), not a chain analysis | + +## Method + +The model is the code, not the prose. `tools/attack/f3-cache/src/main.rs` runs one chain function, written in the shape of `memhard.rs` (the quarter round, the 6 double rounds, the feed-forward, the prev XOR, the constant block), generically over two word types: + +| Word type | What it computes | Use | +|---|---|---| +| `u32` | the real arithmetic | `verify`: bit-exact against `Cache::fill_segment` on 16 (key, segment) pairs and against `chacha_block` on 100,000 random inputs | +| taint set | which block outputs a value depends on (add, xor, rotate = union) | `search`: the direct-parent graph of every block, with each computed line relabelled to the single node {j} so parents are direct, not transitive; plus the 16 x 16 (output word, input word) dependency matrix of one block | + +The exhaustive search: for every target line j, the minimum number of block evaluations with nothing stored is the size of the backward closure of j on the extracted graph (every non-stored block in the closure must be evaluated at least once; once each in dependency order suffices). `pebble` checks that formula against an exhaustive 0-1 BFS over every pebble configuration (place on a node whose parents are pebbled at cost 1, remove at cost 0) for all 2^10 stored sets x 10 targets on each of the three graphs: 10,240 pairs per graph, 0 mismatches. Two deliberately broken chains are the known-fail cases: `skip2` (line j fed from line j - 2) and `nofeed` (no previous line fed in). The curve: for f = 1/64 to 1 (fraction of cache LINES held), the blocks per read on the naive pattern of `funding.md` B2 rank 2 (every L/n-th line from line 0) and on the optimal pattern (exact DP over chunk lengths; brute force over every C(64, n) set for n up to 8, 4,426,165,368 sets at n = 8); ops per item = 9,360 mixer ops (chip-model-v3.md 5.2) + 8 reads x blocks per read x ops per block (608 counted from the code: 48 quarter rounds x 12, 16 feed-forward adds, 16 input XORs; also at MEMHARD.md's approximate 700). Three more checks on the real function: single-bit avalanche and a differential-independence test on `chacha_block`, a census of every line of the real 2^22-line cache for the day key 2026-10-03, and one-core timings of a block, a mixer application, an item and a line recompute. + +## Harness + +| Item | Value | +|---|---| +| Crate | `/Users/joshm/Projects/igneum-wt-attack/tools/attack/f3-cache/` (`Cargo.toml` with `igneum-pow = { path = "../../../igneum-pow" }` and an empty `[workspace]`; `src/main.rs`; `run-box.sh`) | +| Build | `cd tools/attack/f3-cache && IGNEUM_AGENT=attack-f3 IGNEUM_TOOLCHAIN_MISMATCH=ok bash /Users/joshm/Projects/igneum/tools/build-remote.sh --artefacts "target/release/attack-f3" --out /attack-f3 -- build --release` (rc 0, 38 s wall, 0 warnings; the Mac's PATH rustc is 1.69 but `~/.cargo/bin/rustc` is 1.99.0, which the script read as "on both sides"; log `/attack-f3/build-1.log`) | +| Binary | box `/srv/builds/igneum-wt-attack/tools/attack/f3-cache/target/release/attack-f3`, sha256 975115a385ca3195...71cc33, 563,104 bytes | +| Run | on the box: `nohup bash run-box.sh r1 > run-r1.log 2>&1 &` from `/srv/builds/igneum-wt-attack/attack-f3/`; every phase as `flock -s /srv/builds/_locks/measure -c "nice -n 10 taskset -c 12-15,60-63 attack-f3 "`, one chunk per phase, the whole run 17 s (08:10:58 to 08:11:15 UTC; box load 54 at start) | +| Phase lines | `verify`; `search --lines 64|1024 --variant real|skip2|nofeed`; `pebble --lines 10`; `store --lines 64 --brute-max 8`; `store --lines 1024 --brute-max 2`; `curve --lines 64`; `curve --lines 64 --ops-block 700`; `curve --lines 1024`; `avalanche --samples 1048576`; `census --day 2026-10-03`; `bench --n 20000000` | +| Logs | box `/srv/builds/igneum-wt-attack/attack-f3/run-r1.log` and `r1-.log`; Mac copies `/private/tmp/claude-501/-Users-joshm/cd75457f-4858-4f86-9634-7481ee056b7b/scratchpad/attack-f3/` | + +Box hygiene: the box checkout of every `build-remote.sh` run on this worktree executes `git clean -fd` at `/srv/builds/igneum-wt-attack` (remote-run.sh `checkout_tree`), which deletes any untracked scratch directory there. `attack-f3/` and `attack-f3-venv/` are listed in that mirror's `.git/info/exclude` so they survive; nothing in the tree was touched. The F1 lane's `attack-f1-venv/` is untracked and unprotected and will be removed by the next build from any agent on this worktree. + +## The two firings and the pass + +| Chain | Direct parents (taint trace) | Lines under j + 1 at 64 lines | Cheapest derivations | Exhaustive pebbling at 10 lines, cost per target | Verdict | Log | +|---|---|---|---|---|---|---| +| real (the code) | j - 1 for all 63 lines after line 0 | 0 of 64 | none; every line costs exactly j + 1 (mean 32.5) | 1, 2, 3, 4, 5, 6, 7, 8, 9, 10 | PASS | `r1-search-real-64.log`, `r1-pebble-10.log` | +| real, 1,024-line model | j - 1 for all 1,023 lines after line 0 | 0 of 1,024 | none (mean 512.5) | same graph rule | PASS | `r1-search-real-1024.log` | +| skip2 (known fail A) | j - 2 for 62 lines, none for 2 | 63 of 64 | j = 1 in 1, j = 63 in 32 (mean 16.5) | 1, 1, 2, 2, 3, 3, 4, 4, 5, 5 | FIRE | `r1-search-skip2-64.log`, `r1-search-skip2-1024.log` | +| nofeed (known fail B) | none for all 64 | 63 of 64 | every line in 1 block (mean 1.0) | 1 x 10 | FIRE | `r1-search-nofeed-64.log`, `r1-search-nofeed-1024.log` | + +`verify` (`r1-verify.log`): the model chain equals `Cache::fill_segment` on keys {day 2026-10-03, 3 random} x segments {0, 1, 12345, 65535} (16 of 16), `block == chacha_block` on 100,000 of 100,000 random inputs, the 1,024-line model's first 64 lines equal the 64-line chain, and both broken variants differ from the real chain from line 1 (line 0 equal, as the rule predicts). The block's word dependency matrix is full on every variant (256 of 256 pairs), so the firings come from the chain rule alone. + +## Derivation cost per line on the model segment (real chain, nothing stored) + +| j | blocks to derive line j | j + 1 | Log | +|---|---|---|---| +| 0 | 1 | 1 | `r1-search-real-1024.log` | +| 1 | 2 | 2 | | +| 3 | 4 | 4 | | +| 7 | 8 | 8 | | +| 15 | 16 | 16 | | +| 31 | 32 | 32 | | +| 63 | 64 | 64 | (the last line of a real segment; `r1-search-real-64.log` lists all 64) | +| 127 | 128 | 128 | | +| 255 | 256 | 256 | | +| 511 | 512 | 512 | | +| 1,023 | 1,024 | 1,024 | | + +All 1,024 lines were searched (0 under j + 1, mean 512.5 = (L + 1) / 2); the 64-line table in `r1-search-real-64.log` has every j from 0 to 63 at exactly j + 1. + +## Store patterns on the real 64-line segment + +Blocks per read averaged over j uniform in 0..63. "Naive" is `funding.md` B2 rank 2's pattern (every k-th line from line 0). "Optimal" is the exact minimum over store sets of that size (DP; brute force over every set for n up to 8, agreeing with the DP on every row it ran). The gap formula equals the closure cost on the extracted graph on 2,000 of 2,000 random stored sets (`r1-store-64.log`). + +| f | Stored lines n | SRAM held | Naive blocks per read | Optimal positions | Optimal blocks per read | Brute force over C(64, n) sets | +|---|---|---|---|---|---|---| +| 1/64 | 1 | 4 MiB | 31.5 | [32] | 16.0 | 16.0 (64 sets) | +| 1/32 | 2 | 8 MiB | 15.5 | [21, 43] | 10.5 | 10.5 (2,016 sets) | +| 1/16 | 4 | 16 MiB | 7.5 | [12, 25, 38, 51] | 6.094 | 6.094 (635,376 sets) | +| 1/8 | 8 | 32 MiB | 3.5 | [7, 15, 22, 29, 36, 43, 50, 57] | 3.172 | 3.172 (4,426,165,368 sets, 11.2 s) | +| 1/4 | 16 | 64 MiB | 1.5 | [3, 7, 11, ..., 55, 58, 61] | 1.453 | not run (DP exact) | +| 1/2 | 32 | 128 MiB | 0.5 | odd lines | 0.5 | not run | +| 1 | 64 | 256 MiB | 0 | all | 0 | not run | + +The 1,024-line model (`r1-store-1024.log`) gives 29.68 / 15.05 / 7.40 / 3.48 / 1.50 / 0.5 / 0 at the same f on the optimal pattern: the naive and optimal patterns converge as the chain lengthens, because the wasted stored line 0 and the end effects are a smaller share. + +## The curve: ops per item against the fraction of cache lines held (real 64-line segment) + +Ops per item = 9,360 (the 72 mixer applications, hoisted, plus the fold: chip-model-v3.md 5.2) + 8 reads x blocks per read x ops per block. `r1-curve-64.log` (608 ops per block, counted) and `r1-curve-64-memhard.log` (700, MEMHARD.md item 4). Ops per hash = 128 x ops per item + 512. MH/s at the chip model's 50 T op/s budget (approximate, chip-model-v3.md section 1). + +| f (lines held) | SRAM | Blocks per read, naive / optimal | Ops per item, naive, 608 | Ops per item, optimal, 608 | Ops per item, optimal, 700 | Ops per hash, optimal, 608 | MH/s at 50 T op/s, optimal, 608 | +|---|---|---|---|---|---|---|---| +| 1/64 | 4 MiB | 31.5 / 16.0 | 162,576 | 87,184 | 98,960 | 11,160,064 | 4.5 | +| 1/32 | 8 MiB | 15.5 / 10.5 | 84,752 | 60,432 | 68,160 | 7,735,808 | 6.5 | +| 1/16 | 16 MiB | 7.5 / 6.094 | 45,840 | 39,000 | 43,485 | 4,992,512 | 10.0 | +| 1/8 | 32 MiB | 3.5 / 3.172 | 26,384 | 24,788 | 27,122 | 3,173,376 | 15.8 | +| 1/4 | 64 MiB | 1.5 / 1.453 | 16,656 | 16,428 | 17,498 | 2,103,296 | 23.8 | +| 1/2 | 128 MiB | 0.5 / 0.5 | 11,792 | 11,792 | 12,160 | 1,509,888 | 33.1 | +| 1 | 256 MiB | 0 / 0 | 9,360 | 9,360 | 9,360 | 1,198,592 | 41.7 | + +Monotone: ops per item is non-increasing in f on both patterns at both op counts and on the 1,024-line model (`CURVE ... monotone non-increasing` in all three curve logs). The f = 1 point: 9,360 ops per item, 1,198,592 ops per hash, 41.7 MH/s at 50 T op/s, which is the chip-model-v3.md section 5.4 row "none, f = 0" of the published ITEM curve (the on-die-cache recompute chip of sections 1 to 3). The published item curve stores dataset ITEMS and is a different curve: its f = 1 point (GDDR7, 166.4 MH/s, 0.466 microjoules per hash) contains no cache read and no mixer op, so nothing in this row touches it. `funding.md` B2 rank 2's arithmetic reproduces on the naive pattern at 700 ops per block: 3.5 blocks per read, 2,450 ops per line, 19,600 per item on top of the mixer, 13.5 MH/s (50 T / (128 x 28,960 + 512)). + +## Measured times, one box core (`r1-bench.log`, `r1-census.log`; nice 10, cores 12-15,60-63, box load 54) + +| What | Measured | Note | +|---|---|---| +| One ChaCha12 block, dependent chain of 20,000,000 | 66.64 ns | | +| One mixer application (class v4 parameters), dependent chain of 20,000,000 | 17.29 ns | block / application = 3.85 (counted ops 608 / 128 = 4.75) | +| One item against the 256 MiB cache, batches of 32 | 1,326 ns | 72 applications = 1,245 ns; the 8 dependent reads and the fold add 81 ns because the batch overlaps them | +| One line recomputed from nothing, 312,500 random (seg, j) | 2,734 ns | 32.5 blocks per line on average, 84.1 ns per block inside the chain | +| The 256 MiB cache fill, one thread | 0.36 to 0.4 s | 86 ns per block with the writes | + +In measured time, holding every 8th line at the optimal placement makes an item cost 72 + 8 x 3.172 x 3.85 = 170 mixer-application equivalents against 72, a 2.36x penalty per item (2.65x in counted ops). Holding one line in 64 costs 72 + 8 x 16 x 3.85 = 565, a 7.8x penalty. + +## Checks on the real function (`r1-avalanche.log`, `r1-census.log`) + +| Check | Result | +|---|---| +| Single-bit avalanche of `chacha_block`, 1,048,576 flips | mean 256.00 of 512 output bits change (ideal 256), min 200, max 312 | +| (output word, input word) pairs where an output word did not change | worst count 0 of 1,048,576 | +| Chain step: one bit of line j - 1 flipped | line j changes 255.96 bits, line j + 1 changes 255.94 (131,072 flips) | +| Differential independence: B(x ^ d) ^ B(x) == B(y ^ d) ^ B(y) over 262,144 (x, y, single-bit d) | 0 cases | +| Census of the real cache, day key 2026-10-03 | 4,194,304 of 4,194,304 lines distinct, 0 all-zero lines: no two chains merge and no block input repeats | + +## Gate + +| Clause | Result | Where | +|---|---|---| +| No derivation under j + 1 blocks | 0 of 64 and 0 of 1,024 lines under j + 1 on the extracted graph; the formula exact on 10,240 of 10,240 exhaustive pebbling cases; both known-fail chains fire | `r1-search-real-64.log`, `r1-search-real-1024.log`, `r1-pebble-10.log` | +| The curve monotone | non-increasing on both patterns, both op counts, both segment lengths | the three `r1-curve-*.log` | +| The f = 1 point unchanged | 9,360 ops per item = chip-model-v3.md 5.4 "none, f = 0" row; the item curve's GDDR7 f = 1 row (166.4 MH/s, 0.466 microjoules) untouched | `r1-curve-64.log` | + +Verdict: PASS. + +## Observations that are not findings + +| Observation | Number | What it means | What I propose | +|---|---|---|---| +| `funding.md` B2 rank 2 prices the honest trade-off at the naive placement | 3.5 blocks per read at f = 1/8 against 3.17 optimal (9.4 percent less); 31.5 against 16.0 at f = 1/64 (2.0x less, because storing line 0 is worthless: it costs 1 block anyway) | the chip at f = 1/8 reads 15.8 MH/s (608 ops per block, optimal placement) or 14.4 (700, optimal) against `funding.md`'s 13.5 (700, naive); still 0.38x of the full SRAM mirror's 41.7 and 0.12x of the 5090's 136.1 (chip-model-v3.md section 2); the curve stays monotone, so the published verdict (the partial chip is not the threat, the full mirror beats it) stands | one sentence in `funding.md` B2 rank 2: "holding every 8th line at the best placement costs 3.2 blocks per read (3.5 for every 8th line from line 0)". Not edited here: outside this row's two files; for main to serialise | +| The chain's hardness per line is sequential time, not memory | one pebble (64 bytes) over j + 1 steps: the cumulative memory of deriving a line is about 64 x (j + 1) byte-steps | the chain protects the cache by op count, which is exactly what the curve prices in ops; parallel attackers pipeline items and pay E(f) x 608 ops per read in throughput, E(f) block latencies in latency; a chip that holds nothing (f = 0) pays 32.5 x 608 = 19,760 ops per read, 158,080 per item, 167,440 with the mixer (17.9x the mixer alone), 2.3 MH/s at 50 T op/s | nothing to move; the public model should keep quoting ops, never bytes, for this piece | +| What this row does not cover | a cryptanalytic shortcut inside `chacha_block` in this chaining mode (the differential and avalanche tests are sanity checks, not a bound) | the paid engagement's rank 2 question (`funding.md` B2) stays worth the money; plan 4.2 says the internal pass cannot prove the chain's trade-off curve | none | + +## Consequences per user tier + +| Tier | What this row changes | +|---|---| +| Home miner, one 8 / 12 / 16 / 24 or 32 GB card, any vendor, any OS | nothing: the honest miner holds the dataset, the verifier holds the 256 MiB cache; no memory, hash rate, or power figure moves | +| Rig, pool user | nothing | +| Chip builder | the partial-cache chip is priced 9 percent better at f = 1/8 and 2x better at f = 1/64 than `funding.md` says, and is still worse than the full SRAM mirror at every f below 1; the public per-joule sentence (evidence row 17, 2.1x at k = 1) rests on the item curve's f = 1 point, which this row leaves untouched | +| The paid review | the firm receives this record and the harness; rank 2's open question is the block function in chaining mode, not the graph | diff --git a/tools/attack/f3-cache/Cargo.lock b/tools/attack/f3-cache/Cargo.lock new file mode 100644 index 00000000..ecf634df --- /dev/null +++ b/tools/attack/f3-cache/Cargo.lock @@ -0,0 +1,14 @@ +# This file is automatically @generated by Cargo. +# It is not intended for manual editing. +version = 4 + +[[package]] +name = "attack-f3" +version = "0.1.0" +dependencies = [ + "igneum-pow", +] + +[[package]] +name = "igneum-pow" +version = "0.2.0" diff --git a/tools/attack/f3-cache/Cargo.toml b/tools/attack/f3-cache/Cargo.toml new file mode 100644 index 00000000..bd5f80ca --- /dev/null +++ b/tools/attack/f3-cache/Cargo.toml @@ -0,0 +1,20 @@ +[package] +name = "attack-f3" +version = "0.1.0" +edition = "2021" +description = "Attack-pass row F3: the chained cache's j + 1 bound (exhaustive derivation search on a taint-extracted block DAG) and the storage-against-recompute curve over cache lines" +publish = false + +[[bin]] +name = "attack-f3" +path = "src/main.rs" + +[dependencies] +igneum-pow = { path = "../../../igneum-pow" } + +[workspace] + +[profile.release] +opt-level = 3 +lto = true +codegen-units = 1 diff --git a/tools/attack/f3-cache/run-box.sh b/tools/attack/f3-cache/run-box.sh new file mode 100644 index 00000000..eb278dad --- /dev/null +++ b/tools/attack/f3-cache/run-box.sh @@ -0,0 +1,39 @@ +#!/usr/bin/env bash +# Attack-pass row F3: run every phase of attack-f3 on igneum-build-1, on this agent's cores, under the shared +# measure lock in chunks (each phase is one chunk, all well under 30 minutes), one log per phase under +# /srv/builds/igneum-wt-attack/attack-f3/. Usage on the box: nohup bash run-box.sh > .../run-.log 2>&1 & +set -u +RUN="${1:-run}" +BIN=/srv/builds/igneum-wt-attack/tools/attack/f3-cache/target/release/attack-f3 +OUT=/srv/builds/igneum-wt-attack/attack-f3 +LOCK=/srv/builds/_locks/measure +CORES=12-15,60-63 +mkdir -p "$OUT" +[ -x "$BIN" ] || { echo "no binary at $BIN"; exit 2; } +phase() { + local name="$1"; shift + local log="$OUT/$RUN-$name.log" + echo "$(date -u +%FT%TZ) phase $name start -> $log" + flock -s "$LOCK" -c "nice -n 10 taskset -c $CORES $BIN $* > '$log' 2>&1" + local rc=$? + echo "$(date -u +%FT%TZ) phase $name exit $rc" + grep -E '^(VERIFY|SEARCH|PEBBLE|CURVE|AVALANCHE|CENSUS|BENCH) |FIRE|FAIL|panicked|error' "$log" || true +} +echo "$(date -u +%FT%TZ) run $RUN start; binary sha256 $(sha256sum "$BIN" | cut -c1-16); host $(hostname); load $(cut -d' ' -f1-3 /proc/loadavg)" +phase verify verify +phase search-real-64 search --lines 64 --variant real +phase search-real-1024 search --lines 1024 --variant real +phase search-skip2-64 search --lines 64 --variant skip2 +phase search-skip2-1024 search --lines 1024 --variant skip2 +phase search-nofeed-64 search --lines 64 --variant nofeed +phase search-nofeed-1024 search --lines 1024 --variant nofeed +phase pebble-10 pebble --lines 10 +phase store-64 store --lines 64 --brute-max 8 +phase store-1024 store --lines 1024 --brute-max 2 +phase curve-64 curve --lines 64 +phase curve-64-memhard curve --lines 64 --ops-block 700 +phase curve-1024 curve --lines 1024 +phase avalanche avalanche --samples 1048576 +phase census census --day 2026-10-03 +phase bench bench --n 20000000 +echo "$(date -u +%FT%TZ) run $RUN DONE" diff --git a/tools/attack/f3-cache/src/main.rs b/tools/attack/f3-cache/src/main.rs new file mode 100644 index 00000000..092a5102 --- /dev/null +++ b/tools/attack/f3-cache/src/main.rs @@ -0,0 +1,1003 @@ +//! Attack-pass row F3 (`docs/plans/cryptanalysis.md` section 4.2): the chained cache's j + 1 bound and the +//! storage-against-recompute curve over cache LINES. +//! +//! Target: `igneum-pow/src/memhard.rs` at 924288d1, `Cache::fill_segment` and `chacha_block`: +//! `in_j = prev XOR (sigma || K || seg || j || tag)`, `line_j = ChaCha12core(in_j) + in_j`, `prev_0 = 0`, 64 lines +//! per segment, 2^16 independent segments. The model here is the code, run generically over `u32` (bit-exact +//! against igneum-pow, `verify`) and over taint sets (the block dependency graph the exhaustive search runs on, +//! `search`). Two deliberately broken variants are the known-fail firings: `skip2` (line j fed from line j - 2, +//! so a line costs about j / 2 + 1 blocks) and `nofeed` (no previous line fed in, so every line costs 1 block). +//! +//! Sub-commands: verify, search, pebble, store, curve, avalanche, census, bench. Every number is printed with +//! its inputs so a log line is the citation. + +use igneum_pow::memhard::{ + chacha_block, derive_items, mixer, round_key_mult, Cache, MixParams, Shape, CACHE_LINES_PER_SEGMENT, + CACHE_LOG2_WORDS, CACHE_TAG, CHACHA_SIGMA, ITEM_ROUNDS, +}; +use igneum_pow::seed::{day_key, SplitMix64}; +use std::collections::{HashMap, VecDeque}; +use std::time::Instant; + +const MAX_LINES: usize = 1024; +const TW: usize = MAX_LINES / 64; + +/// Operations per chained line, counted from `memhard.rs`: 48 quarter rounds of 12 (4 add, 4 xor, 4 rotate) +/// = 576, the 16 feed-forward adds, the 16 input XORs with the previous line. The constant assembly (sigma, key, +/// seg, j, tag) is wiring on a chip and is not counted. +const OPS_PER_BLOCK: u64 = 48 * 12 + 16 + 16; +/// MEMHARD.md section 3 item 4's approximate figure, the one funding.md B2 rank 2 prices (2,450 = 3.5 x 700). +const OPS_PER_BLOCK_MEMHARD: u64 = 700; +/// Mixer ops per item under class v4 (`mixer_mult = 8`): 72 applications x 128 hoisted + 144 (chip-model-v3.md 5.2). +const OPS_PER_ITEM_MIXER: u64 = 9_360; +/// Program ops per hash beside the items (chip-model-v3.md 5.4: "ops per hash = 128 (1 - f) x 9,360 + 512"). +const OPS_PER_HASH_PROGRAM: u64 = 512; +const ITEMS_PER_HASH: u64 = 128; +const READS_PER_ITEM: u64 = ITEM_ROUNDS as u64; + +// ------------------------------------------------------------------------------------------------------------ +// Words: u32 (the real arithmetic) and taint sets (which block outputs a value depends on) +// ------------------------------------------------------------------------------------------------------------ + +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +struct Taint([u64; TW]); + +impl Taint { + const NONE: Taint = Taint([0; TW]); + fn of(i: usize) -> Taint { + let mut t = [0u64; TW]; + t[i / 64] |= 1u64 << (i % 64); + Taint(t) + } + fn union(self, o: Taint) -> Taint { + let mut t = self.0; + for k in 0..TW { + t[k] |= o.0[k]; + } + Taint(t) + } + fn has(&self, i: usize) -> bool { + (self.0[i / 64] >> (i % 64)) & 1 == 1 + } + fn members(&self, n: usize) -> Vec { + (0..n).filter(|&i| self.has(i)).collect() + } +} + +trait Word: Copy { + fn add(self, o: Self) -> Self; + fn xor(self, o: Self) -> Self; + fn rotl(self, n: u32) -> Self; +} + +impl Word for u32 { + fn add(self, o: u32) -> u32 { + self.wrapping_add(o) + } + fn xor(self, o: u32) -> u32 { + self ^ o + } + fn rotl(self, n: u32) -> u32 { + self.rotate_left(n) + } +} + +impl Word for Taint { + fn add(self, o: Taint) -> Taint { + self.union(o) + } + fn xor(self, o: Taint) -> Taint { + self.union(o) + } + fn rotl(self, _n: u32) -> Taint { + self + } +} + +// ------------------------------------------------------------------------------------------------------------ +// The block function and the chain, the shape of memhard.rs +// ------------------------------------------------------------------------------------------------------------ + +#[allow(clippy::too_many_arguments)] +fn qr(s: &mut [W; 16], a: usize, b: usize, c: usize, d: usize, r1: u32, r2: u32, r3: u32, r4: u32) { + s[a] = s[a].add(s[b]); + s[d] = s[d].xor(s[a]); + s[d] = s[d].rotl(r1); + s[c] = s[c].add(s[d]); + s[b] = s[b].xor(s[c]); + s[b] = s[b].rotl(r2); + s[a] = s[a].add(s[b]); + s[d] = s[d].xor(s[a]); + s[d] = s[d].rotl(r3); + s[c] = s[c].add(s[d]); + s[b] = s[b].xor(s[c]); + s[b] = s[b].rotl(r4); +} + +/// `y = ChaCha12 core(x) + x`, the shape of `memhard::chacha_block`. +fn block(x: &[W; 16]) -> [W; 16] { + let mut y = *x; + for _ in 0..6 { + qr(&mut y, 0, 4, 8, 12, 16, 12, 8, 7); + qr(&mut y, 1, 5, 9, 13, 16, 12, 8, 7); + qr(&mut y, 2, 6, 10, 14, 16, 12, 8, 7); + qr(&mut y, 3, 7, 11, 15, 16, 12, 8, 7); + qr(&mut y, 0, 5, 10, 15, 16, 12, 8, 7); + qr(&mut y, 1, 6, 11, 12, 16, 12, 8, 7); + qr(&mut y, 2, 7, 8, 13, 16, 12, 8, 7); + qr(&mut y, 3, 4, 9, 14, 16, 12, 8, 7); + } + for i in 0..16 { + y[i] = y[i].add(x[i]); + } + y +} + +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +enum Variant { + /// The shipped chain: line j fed from line j - 1. + Real, + /// Known-fail A: line j fed from line j - 2 (lines 0 and 1 from zero). + Skip2, + /// Known-fail B: no previous line fed in; every line is a keystream block. + NoFeed, +} + +impl Variant { + fn parse(s: &str) -> Variant { + match s { + "real" => Variant::Real, + "skip2" => Variant::Skip2, + "nofeed" => Variant::NoFeed, + _ => panic!("variant must be real, skip2 or nofeed"), + } + } + fn name(&self) -> &'static str { + match self { + Variant::Real => "real", + Variant::Skip2 => "skip2", + Variant::NoFeed => "nofeed", + } + } +} + +/// The chain of `Cache::fill_segment` with `lines` lines under `v`. `const_of(j)` is the 16-word constant +/// block of line j; `relabel(j, line)` is what later steps see of line j (identity for u32; for taints the +/// single node {j}, so `on_input` sees DIRECT parents and not the transitive closure); `on_input(j, in_j)` +/// sees the block input before the block runs. +fn chain( + lines: usize, + v: Variant, + zero: W, + const_of: &dyn Fn(usize) -> [W; 16], + relabel: &dyn Fn(usize, [W; 16]) -> [W; 16], + on_input: &mut dyn FnMut(usize, &[W; 16]), +) -> Vec<[W; 16]> { + let mut out: Vec<[W; 16]> = Vec::with_capacity(lines); + let mut kept: Vec<[W; 16]> = Vec::with_capacity(lines); + for j in 0..lines { + let prev = match v { + Variant::Real => { + if j >= 1 { + kept[j - 1] + } else { + [zero; 16] + } + } + Variant::Skip2 => { + if j >= 2 { + kept[j - 2] + } else { + [zero; 16] + } + } + Variant::NoFeed => [zero; 16], + }; + let mut x = const_of(j); + for i in 0..16 { + x[i] = x[i].xor(prev[i]); + } + on_input(j, &x); + let y = block(&x); + kept.push(relabel(j, y)); + out.push(y); + } + out +} + +fn real_const(key: [u32; 8], seg: u32) -> impl Fn(usize) -> [u32; 16] { + move |j| { + let mut x = [0u32; 16]; + x[..4].copy_from_slice(&CHACHA_SIGMA); + x[4..12].copy_from_slice(&key); + x[12] = seg; + x[13] = j as u32; + x[14] = CACHE_TAG[0]; + x[15] = CACHE_TAG[1]; + x + } +} + +fn u32_chain(lines: usize, v: Variant, key: [u32; 8], seg: u32) -> Vec<[u32; 16]> { + let c = real_const(key, seg); + let identity = |_: usize, y: [u32; 16]| y; + let mut nothing = |_: usize, _: &[u32; 16]| {}; + chain(lines, v, 0u32, &c, &identity, &mut nothing) +} + +// ------------------------------------------------------------------------------------------------------------ +// The dependency graph and the derivation search +// ------------------------------------------------------------------------------------------------------------ + +struct Dag { + /// Direct parents of block j: the earlier blocks whose outputs appear in in_j. + parents: Vec>, + /// word_dep[o][i]: output word o of one block depends on input word i (taint). + word_dep: [[bool; 16]; 16], +} + +fn extract_dag(lines: usize, v: Variant) -> Dag { + assert!(lines <= MAX_LINES); + let mut parents: Vec> = vec![Vec::new(); lines]; + let const_of = |_j: usize| [Taint::NONE; 16]; + let relabel = |j: usize, _y: [Taint; 16]| [Taint::of(j); 16]; + let mut on_input = |j: usize, x: &[Taint; 16]| { + let mut u = Taint::NONE; + for w in x.iter() { + u = u.union(*w); + } + parents[j] = u.members(lines); + }; + chain(lines, v, Taint::NONE, &const_of, &relabel, &mut on_input); + // one block at word granularity: input word i carries taint {i} + let mut x = [Taint::NONE; 16]; + for (i, w) in x.iter_mut().enumerate() { + *w = Taint::of(i); + } + let y = block(&x); + let mut word_dep = [[false; 16]; 16]; + for o in 0..16 { + for i in 0..16 { + word_dep[o][i] = y[o].has(i); + } + } + Dag { parents, word_dep } +} + +/// The minimum number of block evaluations that produce block `target` when the blocks in `stored` are held: +/// every non-stored block in the backward closure of the target must be evaluated at least once (its value is +/// needed and not held), and evaluating each exactly once in dependency order suffices. `pebble` checks this +/// formula against an exhaustive search over every pebbling sequence at small sizes. +fn min_blocks(parents: &[Vec], target: usize, stored: &[bool]) -> usize { + if stored[target] { + return 0; + } + let n = parents.len(); + let mut seen = vec![false; n]; + let mut stack = vec![target]; + seen[target] = true; + let mut count = 0usize; + while let Some(v) = stack.pop() { + count += 1; + for &p in &parents[v] { + if !seen[p] && !stored[p] { + seen[p] = true; + stack.push(p); + } + } + } + count +} + +/// Exhaustive 0-1 BFS over every pebble configuration of a DAG of at most 20 nodes: place a pebble on a node +/// whose parents are pebbled (cost 1), remove any pebble (cost 0); the minimum cost to pebble `target` from the +/// configuration `start`. +fn pebble_min(parents: &[Vec], target: usize, start: u32) -> u32 { + let n = parents.len(); + assert!(n <= 20); + let states = 1usize << n; + let mut dist = vec![u32::MAX; states]; + let mut dq: VecDeque = VecDeque::new(); + dist[start as usize] = 0; + dq.push_back(start); + while let Some(m) = dq.pop_front() { + let d = dist[m as usize]; + if (m >> target) & 1 == 1 { + return d; + } + for v in 0..n { + let bit = 1u32 << v; + if m & bit != 0 { + let nm = m & !bit; + if dist[nm as usize] > d { + dist[nm as usize] = d; + dq.push_front(nm); + } + } else if parents[v].iter().all(|&p| m & (1u32 << p) != 0) { + let nm = m | bit; + if dist[nm as usize] > d + 1 { + dist[nm as usize] = d + 1; + dq.push_back(nm); + } + } + } + } + u32::MAX +} + +// ------------------------------------------------------------------------------------------------------------ +// Store patterns on the chain +// ------------------------------------------------------------------------------------------------------------ + +fn total_cost(parents: &[Vec], stored: &[bool]) -> u64 { + (0..parents.len()).map(|j| min_blocks(parents, j, stored) as u64).sum() +} + +fn stored_from(lines: usize, positions: &[usize]) -> Vec { + let mut s = vec![false; lines]; + for &p in positions { + s[p] = true; + } + s +} + +/// funding.md B2 rank 2's pattern: every k-th line from line 0. +fn naive_positions(lines: usize, n: usize) -> Vec { + let k = lines / n; + (0..n).map(|i| i * k).collect() +} + +/// The cost of a stored set on the real path chain by the gap formula: lines before the first stored line cost +/// j + 1, a line after stored line p costs j - p. Equal to the closure cost on the extracted DAG (checked in +/// `store` on random sets). +fn path_cost(lines: usize, sorted_positions: &[usize]) -> u64 { + let mut total = 0u64; + let mut last: Option = None; + let mut idx = 0usize; + for j in 0..lines { + if idx < sorted_positions.len() && sorted_positions[idx] == j { + last = Some(j); + idx += 1; + } + total += match last { + Some(p) => (j - p) as u64, + None => (j + 1) as u64, + }; + } + total +} + +/// Exhaustive minimum over every n-subset of the L lines, no pruning: C(L, n) sets, each costed incrementally +/// by the gap formula (the chunk before the first stored line p1 costs C(p1 + 1, 2), a chunk of g lines after a +/// stored line costs C(g, 2); `path_cost` is the same sum line by line and the two agree on every set `store` +/// samples). +fn brute_force_best(lines: usize, n: usize) -> (u64, Vec) { + fn c2(g: usize) -> u64 { + (g * g.saturating_sub(1) / 2) as u64 + } + #[allow(clippy::too_many_arguments)] + fn rec(lines: usize, n: usize, start: usize, prev: Option, partial: u64, chosen: &mut Vec, best: &mut (u64, Vec), count: &mut u64) { + let remaining = n - chosen.len(); + if remaining == 1 { + for p in start..lines { + let gap = match prev { + Some(q) => c2(p - q), + None => c2(p + 1), + }; + let c = partial + gap + c2(lines - p); + *count += 1; + if c < best.0 { + chosen.push(p); + *best = (c, chosen.clone()); + chosen.pop(); + } + } + return; + } + for p in start..=(lines - remaining) { + let gap = match prev { + Some(q) => c2(p - q), + None => c2(p + 1), + }; + chosen.push(p); + rec(lines, n, p + 1, Some(p), partial + gap, chosen, best, count); + chosen.pop(); + } + } + let mut best = (u64::MAX, Vec::new()); + let mut count = 0u64; + rec(lines, n, 0, None, 0, &mut Vec::new(), &mut best, &mut count); + eprintln!(" brute force: {count} subsets of size {n} of {lines} lines enumerated"); + best +} + +/// Exact minimum over store sets of size n on the path by dynamic programming over the n + 1 chunks: the +/// chunk before the first stored line costs g0 (g0 + 1) / 2, every later chunk of h lines (starting at a stored +/// line) costs h (h - 1) / 2; chunk lengths sum to L. Returns (cost, positions). +fn dp_best(lines: usize, n: usize) -> (u64, Vec) { + // chunks: c0 = g0 + 1 in 1..=L+1, c_i = h_i in 1..; sum of the n + 1 chunk sizes = L + 1; cost = sum C(c, 2) + let total = lines + 1; + let chunks = n + 1; + let c2 = |c: usize| ((c * (c.saturating_sub(1))) / 2) as u64; + let inf = u64::MAX / 4; + let mut dp = vec![vec![inf; total + 1]; chunks + 1]; + let mut arg = vec![vec![0usize; total + 1]; chunks + 1]; + dp[0][0] = 0; + for i in 1..=chunks { + for r in i..=total { + let mut best = inf; + let mut bh = 0; + for h in 1..=(r - (i - 1)) { + let prev = dp[i - 1][r - h]; + if prev >= inf { + continue; + } + let c = prev + c2(h); + if c < best { + best = c; + bh = h; + } + } + dp[i][r] = best; + arg[i][r] = bh; + } + } + let cost = dp[chunks][total]; + // reconstruct chunk sizes + let mut sizes = Vec::with_capacity(chunks); + let mut r = total; + for i in (1..=chunks).rev() { + let h = arg[i][r]; + sizes.push(h); + r -= h; + } + sizes.reverse(); + let g0 = sizes[0] - 1; + let mut positions = Vec::with_capacity(n); + let mut p = g0; + for h in sizes.iter().skip(1) { + positions.push(p); + p += h; + } + (cost, positions) +} + +// ------------------------------------------------------------------------------------------------------------ +// Commands +// ------------------------------------------------------------------------------------------------------------ + +fn cmd_verify(args: &HashMap) { + let day = args.get("day").map(String::as_str).unwrap_or("2026-10-03"); + let seed: u64 = args.get("seed").map(|s| s.parse().unwrap()).unwrap_or(1); + let mut rng = SplitMix64::new(seed); + let mut keys = vec![day_key(day)]; + for _ in 0..3 { + let mut k = [0u32; 8]; + for w in k.iter_mut() { + *w = rng.next() as u32; + } + keys.push(k); + } + let segs = [0usize, 1, 12_345, 65_535]; + let mut ok = 0usize; + let mut bad = 0usize; + for (ki, key) in keys.iter().enumerate() { + for &seg in &segs { + let mut words = vec![0u32; (seg + 1) * CACHE_LINES_PER_SEGMENT * 16]; + Cache::fill_segment(&mut words, seg, key); + let base = seg * CACHE_LINES_PER_SEGMENT * 16; + let model = u32_chain(CACHE_LINES_PER_SEGMENT, Variant::Real, *key, seg as u32); + let flat: Vec = model.iter().flat_map(|l| l.iter().copied()).collect(); + let same = flat[..] == words[base..base + CACHE_LINES_PER_SEGMENT * 16]; + if same { + ok += 1; + } else { + bad += 1; + } + println!("verify key{ki} seg {seg}: model chain == Cache::fill_segment: {same}"); + } + } + // the block itself on random inputs + let mut same = 0usize; + let n = 100_000usize; + for _ in 0..n { + let mut x = [0u32; 16]; + for w in x.iter_mut() { + *w = rng.next() as u32; + } + if block(&x) == chacha_block(&x) { + same += 1; + } + } + println!("verify block == memhard::chacha_block on {n} random inputs: {same}/{n}"); + // the 1024-line model's first 64 lines are the real segment + let long = u32_chain(1024, Variant::Real, keys[0], 7); + let short = u32_chain(64, Variant::Real, keys[0], 7); + let prefix = long[..64] == short[..]; + println!("verify 1024-line model prefix == 64-line chain: {prefix}"); + // the broken variants differ from the real one (so the firings are on different functions) + let s2 = u32_chain(64, Variant::Skip2, keys[0], 7); + let nf = u32_chain(64, Variant::NoFeed, keys[0], 7); + println!( + "verify variants differ from real: skip2 {} nofeed {} (line 0 equal to real: skip2 {} nofeed {})", + s2 != short, + nf != short, + s2[0] == short[0], + nf[0] == short[0] + ); + let pass = bad == 0 && same == n && prefix && s2 != short && nf != short; + println!("VERIFY {} ({ok} segment comparisons equal, {bad} unequal)", if pass { "PASS" } else { "FAIL" }); + if !pass { + std::process::exit(2); + } +} + +fn cmd_search(args: &HashMap) { + let lines: usize = args.get("lines").map(|s| s.parse().unwrap()).unwrap_or(64); + let v = Variant::parse(args.get("variant").map(String::as_str).unwrap_or("real")); + let t0 = Instant::now(); + let dag = extract_dag(lines, v); + // word-level dependency of one block + let mut missing = 0usize; + for o in 0..16 { + for i in 0..16 { + if !dag.word_dep[o][i] { + missing += 1; + } + } + } + println!("search variant {} lines {}: block word dependency matrix: {} of 256 (output word, input word) pairs dependent, {} missing", v.name(), lines, 256 - missing, missing); + // parents summary + let mut hist: HashMap = HashMap::new(); + for (j, p) in dag.parents.iter().enumerate() { + let shape = p.iter().map(|&q| format!("j-{}", j - q)).collect::>().join(","); + *hist.entry(if shape.is_empty() { "none".to_string() } else { shape }).or_insert(0) += 1; + } + let mut hv: Vec<_> = hist.into_iter().collect(); + hv.sort(); + println!("search parents of block j (direct, from the taint trace): {:?}", hv); + // exhaustive: every target, nothing stored + let none = vec![false; lines]; + let mut under = Vec::new(); + let mut costs = Vec::with_capacity(lines); + for j in 0..lines { + let c = min_blocks(&dag.parents, j, &none); + costs.push(c); + if c < j + 1 { + under.push((j, c)); + } + } + let show: Vec = if lines <= 64 { (0..lines).collect() } else { vec![0, 1, 2, 3, 7, 15, 31, 63, 64, 127, 255, 511, 1023].into_iter().filter(|&j| j < lines).collect() }; + println!("| j | blocks to derive line j, nothing stored | j + 1 | under |"); + println!("|---|---|---|---|"); + for &j in &show { + println!("| {} | {} | {} | {} |", j, costs[j], j + 1, if costs[j] < j + 1 { "YES" } else { "no" }); + } + let total: usize = costs.iter().sum(); + println!( + "search variant {} lines {}: {} of {} lines derivable under j + 1 blocks; mean blocks per line {:.3} (j + 1 mean {:.3}); {:.2} s", + v.name(), + lines, + under.len(), + lines, + total as f64 / lines as f64, + (lines as f64 + 1.0) / 2.0, + t0.elapsed().as_secs_f64() + ); + if !under.is_empty() { + let first: Vec = under.iter().take(8).map(|(j, c)| format!("j={j}:{c}")).collect(); + println!("search FIRE: cheapest derivations under j + 1: {}", first.join(" ")); + } + let gate = under.is_empty() && missing == 0; + println!("SEARCH variant {} lines {}: {}", v.name(), lines, if gate { "PASS (no line under j + 1 blocks)" } else { "FIRE (derivation under j + 1 found)" }); +} + +fn cmd_pebble(args: &HashMap) { + let lines: usize = args.get("lines").map(|s| s.parse().unwrap()).unwrap_or(10); + assert!(lines <= 16, "pebble enumerates 2^lines configurations; keep lines at most 16"); + for v in [Variant::Real, Variant::Skip2, Variant::NoFeed] { + let dag = extract_dag(lines, v); + let t0 = Instant::now(); + let mut checked = 0u64; + let mut mismatch = 0u64; + let mut example = String::new(); + let mut target_cost_none = Vec::new(); + for start in 0u32..(1u32 << lines) { + let stored: Vec = (0..lines).map(|i| (start >> i) & 1 == 1).collect(); + for target in 0..lines { + let a = min_blocks(&dag.parents, target, &stored) as u32; + let b = pebble_min(&dag.parents, target, start); + checked += 1; + if a != b { + mismatch += 1; + if example.is_empty() { + example = format!("start {start:#b} target {target}: closure {a} pebbling {b}"); + } + } + if start == 0 { + target_cost_none.push(b); + } + } + } + println!( + "pebble variant {} lines {}: {} (stored set, target) pairs, exhaustive pebbling == closure formula on {} of them, {} mismatches {}; nothing stored, cost per target: {:?}; {:.1} s", + v.name(), + lines, + checked, + checked - mismatch, + mismatch, + example, + target_cost_none, + t0.elapsed().as_secs_f64() + ); + let under: Vec = (0..lines).filter(|&j| (target_cost_none[j] as usize) < j + 1).collect(); + println!("PEBBLE variant {} lines {}: {}", v.name(), lines, if under.is_empty() { "PASS (no target under j + 1 by exhaustive pebbling)".to_string() } else { format!("FIRE (targets under j + 1 by exhaustive pebbling: {under:?})") }); + } +} + +fn fractions() -> Vec<(usize, &'static str)> { + vec![(64, "1/64"), (32, "1/32"), (16, "1/16"), (8, "1/8"), (4, "1/4"), (2, "1/2"), (1, "1")] +} + +fn cmd_store(args: &HashMap) { + let lines: usize = args.get("lines").map(|s| s.parse().unwrap()).unwrap_or(64); + let v = Variant::parse(args.get("variant").map(String::as_str).unwrap_or("real")); + let brute_max: usize = args.get("brute-max").map(|s| s.parse().unwrap()).unwrap_or(if lines <= 64 { 8 } else { 2 }); + let seed: u64 = args.get("seed").map(|s| s.parse().unwrap()).unwrap_or(1); + let dag = extract_dag(lines, v); + // the gap formula against the closure cost on random sets (real variant only: the formula is the path's) + if v == Variant::Real { + let mut rng = SplitMix64::new(seed); + let mut agree = 0usize; + let trials = 2_000usize; + for _ in 0..trials { + let n = 1 + rng.below(lines as u64 / 2) as usize; + let mut pos: Vec = Vec::new(); + while pos.len() < n { + let p = rng.below(lines as u64) as usize; + if !pos.contains(&p) { + pos.push(p); + } + } + pos.sort_unstable(); + let a = path_cost(lines, &pos); + let b = total_cost(&dag.parents, &stored_from(lines, &pos)); + if a == b { + agree += 1; + } + } + println!("store lines {lines}: gap formula == DAG closure cost on {agree}/{trials} random stored sets"); + } + println!("| f | stored lines n | naive (every L/n-th from 0) blocks per read | optimal positions (DP) | optimal blocks per read | brute force (exhaustive over C(L, n) sets) | DP == brute |"); + println!("|---|---|---|---|---|---|---|"); + for (den, label) in fractions() { + if lines < den { + continue; + } + let n = lines / den; + let naive = naive_positions(lines, n); + let naive_cost = total_cost(&dag.parents, &stored_from(lines, &naive)); + let (dp_cost, dp_pos) = dp_best(lines, n); + let dp_cost_dag = total_cost(&dag.parents, &stored_from(lines, &dp_pos)); + let brute = if n <= brute_max { + let t0 = Instant::now(); + let (c, p) = brute_force_best(lines, n); + Some((c, p, t0.elapsed().as_secs_f64())) + } else { + None + }; + let pos_str = if dp_pos.len() <= 16 { format!("{dp_pos:?}") } else { format!("{:?}..{:?} ({} lines)", &dp_pos[..4], &dp_pos[dp_pos.len() - 2..], dp_pos.len()) }; + let brute_str = match &brute { + Some((c, _, s)) => format!("{:.4} ({:.1} s)", *c as f64 / lines as f64, s), + None => "not run".to_string(), + }; + let eq = match &brute { + Some((c, _, _)) => if *c == dp_cost { "yes" } else { "NO" }, + None => "n/a", + }; + println!( + "| {} | {} | {:.4} | {} | {:.4} (DAG {:.4}) | {} | {} |", + label, + n, + naive_cost as f64 / lines as f64, + pos_str, + dp_cost as f64 / lines as f64, + dp_cost_dag as f64 / lines as f64, + brute_str, + eq + ); + } +} + +fn cmd_curve(args: &HashMap) { + let lines: usize = args.get("lines").map(|s| s.parse().unwrap()).unwrap_or(64); + let ops_block: u64 = args.get("ops-block").map(|s| s.parse().unwrap()).unwrap_or(OPS_PER_BLOCK); + let cache_mib: f64 = args.get("cache-mib").map(|s| s.parse().unwrap()).unwrap_or(256.0); + let dag = extract_dag(lines, Variant::Real); + println!("curve: ops per block {ops_block} (counted {OPS_PER_BLOCK}; MEMHARD.md approximate {OPS_PER_BLOCK_MEMHARD}), mixer ops per item {OPS_PER_ITEM_MIXER}, {READS_PER_ITEM} dependent reads per item, {ITEMS_PER_HASH} items per hash + {OPS_PER_HASH_PROGRAM}"); + println!("| f (cache lines stored) | SRAM MiB | blocks per read, naive | blocks per read, optimal | recompute ops per item, naive | recompute ops per item, optimal | ops per item, naive | ops per item, optimal | ops per hash, optimal | MH/s at 50 T op/s, optimal |"); + println!("|---|---|---|---|---|---|---|---|---|---|"); + let mut prev_opt: Option = None; + let mut prev_naive: Option = None; + let mut monotone = true; + for (den, label) in fractions() { + if lines < den { + continue; + } + let n = lines / den; + let naive_cost = total_cost(&dag.parents, &stored_from(lines, &naive_positions(lines, n))) as f64 / lines as f64; + let (dp_cost, _) = dp_best(lines, n); + let opt_cost = dp_cost as f64 / lines as f64; + let rec_naive = READS_PER_ITEM as f64 * naive_cost * ops_block as f64; + let rec_opt = READS_PER_ITEM as f64 * opt_cost * ops_block as f64; + let item_naive = OPS_PER_ITEM_MIXER as f64 + rec_naive; + let item_opt = OPS_PER_ITEM_MIXER as f64 + rec_opt; + let hash_opt = ITEMS_PER_HASH as f64 * item_opt + OPS_PER_HASH_PROGRAM as f64; + let mhs = 50.0e12 / hash_opt / 1.0e6; + if let Some(p) = prev_opt { + if item_opt > p { + monotone = false; + } + } + if let Some(p) = prev_naive { + if item_naive > p { + monotone = false; + } + } + prev_opt = Some(item_opt); + prev_naive = Some(item_naive); + println!( + "| {} | {:.1} | {:.4} | {:.4} | {:.0} | {:.0} | {:.0} | {:.0} | {:.0} | {:.1} |", + label, + cache_mib / den as f64, + naive_cost, + opt_cost, + rec_naive, + rec_opt, + item_naive, + item_opt, + hash_opt, + mhs + ); + } + println!("CURVE lines {lines}: ops per item {} with f on both patterns; f = 1 point = {} ops per item, {} ops per hash (the chip-model-v3.md 5.4 'none, f = 0' row of the ITEM curve: 1,198,592 ops per hash, 41.7 MH/s at 50 T op/s)", if monotone { "monotone non-increasing" } else { "NOT monotone" }, OPS_PER_ITEM_MIXER, ITEMS_PER_HASH * OPS_PER_ITEM_MIXER + OPS_PER_HASH_PROGRAM); +} + +fn cmd_avalanche(args: &HashMap) { + let samples: u64 = args.get("samples").map(|s| s.parse().unwrap()).unwrap_or(1 << 20); + let seed: u64 = args.get("seed").map(|s| s.parse().unwrap()).unwrap_or(1); + let mut rng = SplitMix64::new(seed); + let t0 = Instant::now(); + // single-bit avalanche of the block + let mut sum = 0u64; + let mut min = u32::MAX; + let mut max = 0u32; + let mut word_miss = [[0u64; 16]; 16]; + let mut per_word = [0u64; 16]; + for _ in 0..samples { + let mut x = [0u32; 16]; + for w in x.iter_mut() { + *w = rng.next() as u32; + } + let bit = rng.below(512) as usize; + let mut x2 = x; + x2[bit / 32] ^= 1u32 << (bit % 32); + let y = chacha_block(&x); + let y2 = chacha_block(&x2); + let mut d = 0u32; + for o in 0..16 { + let dd = (y[o] ^ y2[o]).count_ones(); + d += dd; + if dd == 0 { + word_miss[o][bit / 32] += 1; + } + } + per_word[bit / 32] += 1; + sum += d as u64; + min = min.min(d); + max = max.max(d); + } + let mean = sum as f64 / samples as f64; + let worst_miss = word_miss.iter().flat_map(|r| r.iter()).copied().max().unwrap_or(0); + println!("avalanche block: {samples} single-bit flips, output bits changed mean {:.2} of 512 (ideal 256), min {min}, max {max}; (output word, input word) pairs where the output word did not change: worst count {worst_miss} (ideal about {:.1} = samples / 16 / 2^32)", mean, samples as f64 / 16.0 / 4_294_967_296.0); + // the chained step: flip one bit of line j - 1, measure line j (x_j = line_{j-1} XOR const, the same block) + // and line j + 1 (two blocks on): both should be at 256 + let key = day_key("2026-10-03"); + let c = real_const(key, 5); + let mut sum1 = 0u64; + let mut sum2 = 0u64; + let n2 = samples / 8; + for _ in 0..n2 { + let mut prev = [0u32; 16]; + for w in prev.iter_mut() { + *w = rng.next() as u32; + } + let j = 1 + rng.below(62) as usize; + let bit = rng.below(512) as usize; + let mut prev2 = prev; + prev2[bit / 32] ^= 1u32 << (bit % 32); + let step = |p: &[u32; 16], jj: usize| { + let mut x = c(jj); + for i in 0..16 { + x[i] ^= p[i]; + } + chacha_block(&x) + }; + let a1 = step(&prev, j); + let b1 = step(&prev2, j); + let a2 = step(&a1, j + 1); + let b2 = step(&b1, j + 1); + sum1 += (0..16).map(|i| (a1[i] ^ b1[i]).count_ones() as u64).sum::(); + sum2 += (0..16).map(|i| (a2[i] ^ b2[i]).count_ones() as u64).sum::(); + } + println!("avalanche chain step: {n2} flips of one bit of line j-1: line j changes {:.2} bits of 512, line j+1 changes {:.2} bits", sum1 as f64 / n2 as f64, sum2 as f64 / n2 as f64); + // differential linearity: does B(x ^ d) ^ B(x) depend on x? (a block whose differences were x-independent + // would let a stored line of one segment derive another's) + let mut equal = 0u64; + let n3 = samples / 4; + for _ in 0..n3 { + let mut x = [0u32; 16]; + let mut y = [0u32; 16]; + let mut d = [0u32; 16]; + for i in 0..16 { + x[i] = rng.next() as u32; + y[i] = rng.next() as u32; + } + let bit = rng.below(512) as usize; + d[bit / 32] = 1u32 << (bit % 32); + let mut xd = x; + let mut yd = y; + for i in 0..16 { + xd[i] ^= d[i]; + yd[i] ^= d[i]; + } + let bx = chacha_block(&x); + let bxd = chacha_block(&xd); + let by = chacha_block(&y); + let byd = chacha_block(&yd); + let mut same = true; + for i in 0..16 { + if (bx[i] ^ bxd[i]) != (by[i] ^ byd[i]) { + same = false; + break; + } + } + if same { + equal += 1; + } + } + println!("avalanche differential: {n3} (x, y, single-bit d): B(x^d)^B(x) == B(y^d)^B(y) in {equal} cases (ideal 0); {:.1} s", t0.elapsed().as_secs_f64()); + let pass = worst_miss <= 4 && (mean - 256.0).abs() < 1.0 && equal == 0; + println!("AVALANCHE {}", if pass { "PASS" } else { "FAIL" }); +} + +fn cmd_census(args: &HashMap) { + let day = args.get("day").map(String::as_str).unwrap_or("2026-10-03"); + let t0 = Instant::now(); + let key = day_key(day); + let cache = Cache::fill(key); + let fill_s = t0.elapsed().as_secs_f64(); + let words = cache.words(); + let lines = cache.lines(); + let mut v: Vec<[u32; 16]> = Vec::with_capacity(lines); + for l in 0..lines { + let mut a = [0u32; 16]; + a.copy_from_slice(&words[l * 16..l * 16 + 16]); + v.push(a); + } + let zero = v.iter().filter(|a| a.iter().all(|&w| w == 0)).count(); + v.sort_unstable(); + let mut distinct = 1usize; + let mut dup_pairs = 0usize; + for i in 1..v.len() { + if v[i] != v[i - 1] { + distinct += 1; + } else { + dup_pairs += 1; + } + } + println!( + "census day {day} key {:08x}..: {} lines in {} segments (2^{} words), fill {:.1} s on one thread ({:.0} ns per block); distinct lines {distinct}, duplicate adjacent pairs {dup_pairs}, all-zero lines {zero}; {:.1} s", + key[0], + lines, + cache.segments(), + CACHE_LOG2_WORDS, + fill_s, + fill_s * 1e9 / lines as f64, + t0.elapsed().as_secs_f64() + ); + println!("CENSUS {}", if distinct == lines && zero == 0 { "PASS (every line distinct: no two chains merge, no repeated block input)" } else { "FAIL" }); +} + +fn cmd_bench(args: &HashMap) { + let n: u64 = args.get("n").map(|s| s.parse().unwrap()).unwrap_or(20_000_000); + let day = args.get("day").map(String::as_str).unwrap_or("2026-10-03"); + let key = day_key(day); + let shape = Shape { mixer_mult: 8, cache_log2_words: CACHE_LOG2_WORDS as u32, derive_len: 0 }; + let mp = MixParams::with_shape(key, shape); + // dependent ChaCha12 blocks + let mut x = [0u32; 16]; + x[..8].copy_from_slice(&key); + let t0 = Instant::now(); + for _ in 0..n { + x = chacha_block(&x); + } + let block_ns = t0.elapsed().as_nanos() as f64 / n as f64; + println!("bench block: {n} dependent chacha_block: {:.2} ns each (sink {:08x})", block_ns, x[0]); + // dependent mixer applications under class v4's parameters + let mut s = [0u32; 16]; + s[..8].copy_from_slice(&key); + let t0 = Instant::now(); + for i in 0..n { + let rk = round_key_mult((i % 9) as usize, (i % 8) as usize, 8); + mixer(&mut s, rk, &mp); + } + let mixer_ns = t0.elapsed().as_nanos() as f64 / n as f64; + println!("bench mixer: {n} dependent applications: {:.2} ns each (sink {:08x}); block / mixer application = {:.2} (counted ops {} / 128 = {:.2})", mixer_ns, s[0], block_ns / mixer_ns, OPS_PER_BLOCK, OPS_PER_BLOCK as f64 / 128.0); + // whole items against the real 256 MiB cache (72 mixer applications + 8 dependent DRAM reads) + let t0 = Instant::now(); + let cache = Cache::fill(key); + println!("bench cache fill: {:.2} s on one thread", t0.elapsed().as_secs_f64()); + let items: u64 = (n / 200).max(32); + let mut rng = SplitMix64::new(7); + let mut out = [[0u32; 16]; 32]; + let mut sink = 0u32; + let t0 = Instant::now(); + let mut done = 0u64; + while done < items { + let ts: Vec = (0..32).map(|_| rng.next() as u32).collect(); + derive_items(&ts, &mp, &cache, &mut out); + sink ^= out[0][0]; + done += 32; + } + let item_ns = t0.elapsed().as_nanos() as f64 / done as f64; + println!("bench item: {done} items (batches of 32) against the 256 MiB cache: {:.1} ns each (72 x mixer = {:.1} ns; remainder {:.1} ns = 8 dependent reads and the fold; sink {:08x})", item_ns, 72.0 * mixer_ns, item_ns - 72.0 * mixer_ns, sink); + // a line recomputed from nothing: j + 1 blocks, j uniform + let lines_n = (n / 64).max(1); + let mut rng = SplitMix64::new(9); + let mut sink2 = 0u32; + let mut blocks = 0u64; + let t0 = Instant::now(); + for _ in 0..lines_n { + let seg = rng.below(65_536) as u32; + let j = rng.below(64) as usize; + let c = u32_chain(j + 1, Variant::Real, key, seg); + sink2 ^= c[j][0]; + blocks += (j + 1) as u64; + } + let line_ns = t0.elapsed().as_nanos() as f64 / lines_n as f64; + println!("bench line recompute from nothing: {lines_n} random (seg, j): {:.1} ns per line, {:.1} blocks per line on average, {:.2} ns per block (sink {:08x})", line_ns, blocks as f64 / lines_n as f64, line_ns * lines_n as f64 / blocks as f64, sink2); + println!("BENCH block {:.2} ns, mixer application {:.2} ns, item {:.1} ns, ratio block/mixer {:.2}", block_ns, mixer_ns, item_ns, block_ns / mixer_ns); +} + +fn main() { + let argv: Vec = std::env::args().collect(); + if argv.len() < 2 { + eprintln!("usage: attack-f3 [--lines N] [--variant real|skip2|nofeed] [--samples N] [--seed N] [--day D] [--n N] [--brute-max N] [--ops-block N]"); + std::process::exit(1); + } + let cmd = argv[1].clone(); + let mut args: HashMap = HashMap::new(); + let mut i = 2; + while i < argv.len() { + let k = argv[i].trim_start_matches("--").to_string(); + let v = argv.get(i + 1).cloned().unwrap_or_default(); + args.insert(k, v); + i += 2; + } + println!("attack-f3 {} {:?} (igneum-pow memhard: {} lines per segment, tag {:08x}{:08x})", cmd, args, CACHE_LINES_PER_SEGMENT, CACHE_TAG[0], CACHE_TAG[1]); + match cmd.as_str() { + "verify" => cmd_verify(&args), + "search" => cmd_search(&args), + "pebble" => cmd_pebble(&args), + "store" => cmd_store(&args), + "curve" => cmd_curve(&args), + "avalanche" => cmd_avalanche(&args), + "census" => cmd_census(&args), + "bench" => cmd_bench(&args), + _ => { + eprintln!("unknown command {cmd}"); + std::process::exit(1); + } + } +}