Compare commits
5 commits
master
...
adv-accept
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
17bfe929d5 | ||
|
|
34da77c99a | ||
|
|
93a789f651 | ||
|
|
472648ff37 | ||
|
|
64a5bf6fe6 |
75 changed files with 2075 additions and 0 deletions
|
|
@ -0,0 +1,2 @@
|
||||||
|
[2026-10-07T18:44:04Z] diffuse: prog=devnet id=0xa785001687d8688a pairs=256
|
||||||
|
[2026-10-07T18:44:05Z] one-bit H flip: 1.0000 of the 4,096 unit addresses change (1.0 = every address moved; a header with no path would read ~0)
|
||||||
|
|
@ -0,0 +1,2 @@
|
||||||
|
[2026-10-07T18:53:14Z] diffuse: prog=devnet id=0xa785001687d8688a pairs=4096 plant=none
|
||||||
|
[2026-10-07T18:53:21Z] one-bit H flip: 1.0000 of the 4,096 unit addresses change (1.0 = every address moved; a header with no path would read ~0)
|
||||||
|
|
@ -0,0 +1,2 @@
|
||||||
|
[2026-10-07T18:53:35Z] diffuse: prog=devnet id=0xa785001687d8688a pairs=1024 plant=header-blind
|
||||||
|
[2026-10-07T18:53:37Z] one-bit H flip: 0.0000 of the 4,096 unit addresses change (1.0 = every address moved; a header with no path would read ~0)
|
||||||
|
|
@ -0,0 +1,2 @@
|
||||||
|
[2026-10-07T18:53:24Z] diffuse: prog=devnet3 id=0xfce15bf61030be57 pairs=4096 plant=none
|
||||||
|
[2026-10-07T18:53:31Z] one-bit H flip: 1.0000 of the 4,096 unit addresses change (1.0 = every address moved; a header with no path would read ~0)
|
||||||
|
|
@ -0,0 +1,5 @@
|
||||||
|
[2026-10-07T18:43:43Z] devnet: program_id=0xa785001687d8688a attempt=1 generator=4 day=20730 sites=[1, 4, 6, 10, 12, 20, 27, 30, 35, 40, 41, 43, 45, 52, 53, 54]
|
||||||
|
[2026-10-07T18:43:47Z] devnet3: program_id=0xfce15bf61030be57 attempt=0 generator=4 day=20733 sites=[3, 8, 14, 15, 20, 26, 28, 35, 40, 43, 47, 49, 52, 53, 61, 62]
|
||||||
|
[2026-10-07T18:43:52Z] drawn:0: id=0x5d7cc2b09fc6922a attempt=0
|
||||||
|
[2026-10-07T18:43:56Z] drawn:1: id=0x2c81972d33e22ad2 attempt=1
|
||||||
|
[2026-10-07T18:44:00Z] drawn:2: id=0xd3fead516b1ab00c attempt=0
|
||||||
|
|
@ -0,0 +1,65 @@
|
||||||
|
[2026-10-07T19:13:18Z] dump: prog=drawn:0 id=0x5d7cc2b09fc6922a attempt=0 era M=0xcead9fb7 R=7 pos=[2, 5, 8, 15]
|
||||||
|
0: mul dst=7 src=6 src2=1 imm=0xbb06b7ce imm2=0x8908c573 rot=30 bit= 7 mask= 4 win=0 off=0
|
||||||
|
1: xor dst=5 src=1 src2=5 imm=0x84024852 imm2=0x0fbcc2ec rot= 5 bit=21 mask= 1 win=0 off=0
|
||||||
|
2: sub dst=7 src=5 src2=1 imm=0x8367cf58 imm2=0xf987e76c rot= 6 bit=29 mask= 2 win=0 off=0
|
||||||
|
3: mulhi dst=0 src=6 src2=4 imm=0x0fd5d40e imm2=0xf62060ce rot= 4 bit= 5 mask= 1 win=0 off=0
|
||||||
|
4: load dst=0 src=5 src2=6 imm=0x5f194e25 imm2=0x7e8699cf rot= 6 bit=20 mask= 4 win=2 off=0
|
||||||
|
5: xor dst=3 src=2 src2=6 imm=0xdcb63ba7 imm2=0x64fa2d9f rot=28 bit=23 mask= 2 win=0 off=0
|
||||||
|
6: load dst=5 src=3 src2=2 imm=0x8e9f65b3 imm2=0xbaae1db0 rot=22 bit= 0 mask= 2 win=2 off=1
|
||||||
|
7: sub dst=7 src=5 src2=3 imm=0xce74e0d7 imm2=0x1f8d6a1e rot=23 bit=12 mask= 4 win=0 off=0
|
||||||
|
8: load dst=0 src=5 src2=5 imm=0x226d1310 imm2=0x716e3fb1 rot=30 bit= 3 mask=16 win=0 off=0
|
||||||
|
9: xor dst=0 src=3 src2=2 imm=0x04418872 imm2=0x385cc6f1 rot=21 bit=18 mask= 4 win=0 off=0
|
||||||
|
10: mul dst=1 src=7 src2=3 imm=0x6d475338 imm2=0x92f6e34b rot=12 bit=29 mask= 4 win=0 off=0
|
||||||
|
11: xor dst=2 src=5 src2=2 imm=0x4de7383c imm2=0xfb477cb3 rot=31 bit=26 mask=16 win=0 off=0
|
||||||
|
12: mul dst=2 src=3 src2=7 imm=0x1b360c5a imm2=0x6b445c29 rot=25 bit=11 mask= 1 win=0 off=0
|
||||||
|
13: load dst=1 src=7 src2=7 imm=0x5183ab93 imm2=0x297dbb71 rot=11 bit= 3 mask=16 win=0 off=0
|
||||||
|
14: mul dst=1 src=4 src2=4 imm=0xab513337 imm2=0xf0b3130e rot=23 bit=24 mask=16 win=0 off=0
|
||||||
|
15: sub dst=5 src=0 src2=3 imm=0xa55850f8 imm2=0x9ecbb999 rot=31 bit= 0 mask= 1 win=0 off=0
|
||||||
|
16: add dst=5 src=7 src2=5 imm=0x42097584 imm2=0x12bc3f39 rot=12 bit=12 mask= 1 win=0 off=0
|
||||||
|
17: add dst=5 src=6 src2=2 imm=0x8c56720b imm2=0xdea77358 rot= 2 bit= 7 mask= 1 win=0 off=0
|
||||||
|
18: mul dst=0 src=6 src2=1 imm=0x0c941e90 imm2=0xffe34977 rot=18 bit=18 mask= 8 win=0 off=0
|
||||||
|
19: add dst=4 src=0 src2=7 imm=0x81e609d0 imm2=0xa4fc4d89 rot=22 bit=10 mask= 8 win=0 off=0
|
||||||
|
20: load dst=2 src=5 src2=1 imm=0xbb0a8b67 imm2=0xee61e196 rot=18 bit= 4 mask=16 win=2 off=1
|
||||||
|
21: mulhi dst=1 src=3 src2=6 imm=0xd163c3a0 imm2=0x6d419f57 rot=26 bit= 9 mask=16 win=0 off=0
|
||||||
|
22: rotr dst=3 src=1 src2=5 imm=0x0d6bf6b7 imm2=0x54c7a364 rot= 3 bit=27 mask=16 win=0 off=0
|
||||||
|
23: mulhi dst=2 src=6 src2=5 imm=0x1da083e8 imm2=0x565324ac rot=28 bit= 2 mask= 2 win=0 off=0
|
||||||
|
24: mul dst=0 src=3 src2=1 imm=0x39caa142 imm2=0x02b79f06 rot=27 bit=18 mask=16 win=0 off=0
|
||||||
|
25: mulhi dst=4 src=5 src2=3 imm=0xd77e548e imm2=0xd9fbd2b1 rot=14 bit= 3 mask= 8 win=0 off=0
|
||||||
|
26: load dst=6 src=3 src2=0 imm=0xcb01fc62 imm2=0x9dd8a804 rot=12 bit= 5 mask= 4 win=1 off=0
|
||||||
|
27: load dst=7 src=6 src2=3 imm=0x9b9db423 imm2=0x0b71abd5 rot=24 bit=13 mask= 1 win=1 off=1
|
||||||
|
28: load dst=0 src=7 src2=3 imm=0x074b73dc imm2=0x47f2c7c0 rot=10 bit= 2 mask= 1 win=1 off=1
|
||||||
|
29: mulhi dst=6 src=3 src2=4 imm=0x7f80751d imm2=0x8c89eec4 rot= 4 bit=15 mask= 4 win=0 off=0
|
||||||
|
30: shfl dst=0 src=7 src2=0 imm=0x52d1d8cf imm2=0xd4f28db1 rot= 1 bit=26 mask= 4 win=0 off=0
|
||||||
|
31: or dst=0 src=2 src2=0 imm=0x30ab8037 imm2=0x6b824a5e rot=28 bit=28 mask= 8 win=0 off=0
|
||||||
|
32: rotr dst=7 src=2 src2=1 imm=0xcd174ddf imm2=0xd372cba8 rot=27 bit= 2 mask= 8 win=0 off=0
|
||||||
|
33: rotr dst=1 src=2 src2=4 imm=0xcb0795bd imm2=0x77ea9661 rot=27 bit=31 mask=16 win=0 off=0
|
||||||
|
34: mul dst=2 src=3 src2=1 imm=0x3f4a6c2b imm2=0x17818d49 rot=10 bit=23 mask= 2 win=0 off=0
|
||||||
|
35: add dst=3 src=2 src2=1 imm=0x8ebb7cf4 imm2=0xbe8ec3a5 rot=18 bit= 7 mask= 4 win=0 off=0
|
||||||
|
36: sub dst=4 src=0 src2=1 imm=0x0a3c7516 imm2=0xb397e5d3 rot=22 bit=28 mask=16 win=0 off=0
|
||||||
|
37: xor dst=7 src=4 src2=5 imm=0xbdb0382d imm2=0x29362561 rot= 3 bit=30 mask= 1 win=0 off=0
|
||||||
|
38: load dst=6 src=3 src2=4 imm=0xdf2b1cb8 imm2=0xb32d9d72 rot=14 bit=12 mask= 1 win=2 off=0
|
||||||
|
39: add dst=0 src=1 src2=7 imm=0x9a58d1b9 imm2=0x74e51a62 rot=27 bit=11 mask= 8 win=0 off=0
|
||||||
|
40: load dst=5 src=7 src2=7 imm=0x73f5d2ab imm2=0x014ad2b1 rot=15 bit=15 mask= 2 win=0 off=0
|
||||||
|
41: sub dst=4 src=3 src2=7 imm=0x60936be6 imm2=0x70bf5414 rot=13 bit=26 mask= 1 win=0 off=0
|
||||||
|
42: rotl dst=4 src=6 src2=7 imm=0x6fb8a166 imm2=0x7090770d rot= 4 bit=30 mask= 1 win=0 off=0
|
||||||
|
43: load dst=6 src=5 src2=3 imm=0x4869f478 imm2=0x597a30a0 rot=11 bit= 3 mask= 8 win=2 off=0
|
||||||
|
44: sub dst=2 src=0 src2=5 imm=0x7a641989 imm2=0x3815533a rot= 4 bit=27 mask= 4 win=0 off=0
|
||||||
|
45: xor dst=3 src=5 src2=0 imm=0x8a938ecb imm2=0xc78e7a76 rot=15 bit=12 mask= 8 win=0 off=0
|
||||||
|
46: load dst=1 src=4 src2=5 imm=0xb1649fab imm2=0x9410d37a rot=18 bit=26 mask=16 win=1 off=0
|
||||||
|
47: mul dst=5 src=1 src2=0 imm=0x8c267740 imm2=0x64244e45 rot= 2 bit=27 mask= 4 win=0 off=0
|
||||||
|
48: mulhi dst=5 src=7 src2=4 imm=0x755ab59b imm2=0x30187285 rot= 4 bit= 0 mask= 8 win=0 off=0
|
||||||
|
49: add dst=3 src=5 src2=6 imm=0x615d934a imm2=0x7a0bfa51 rot=13 bit= 0 mask=16 win=0 off=0
|
||||||
|
50: load dst=3 src=1 src2=3 imm=0x175c1725 imm2=0x1b6e7b76 rot=30 bit=26 mask= 1 win=1 off=0
|
||||||
|
51: mad dst=4 src=3 src2=4 imm=0x7a8feb6b imm2=0xd1b6ab2b rot= 6 bit=12 mask= 1 win=0 off=0
|
||||||
|
52: load dst=4 src=3 src2=5 imm=0xf8d26e9b imm2=0x90d4a62d rot= 3 bit= 3 mask= 4 win=1 off=0
|
||||||
|
53: add dst=6 src=4 src2=5 imm=0x8b1231c1 imm2=0x69a4453a rot=24 bit= 6 mask=16 win=0 off=0
|
||||||
|
54: xor dst=7 src=6 src2=7 imm=0x59da6c68 imm2=0xa8786856 rot=24 bit= 3 mask= 2 win=0 off=0
|
||||||
|
55: mad dst=6 src=0 src2=7 imm=0xc0c4dc9a imm2=0x981726de rot=11 bit=16 mask= 1 win=0 off=0
|
||||||
|
56: sub dst=3 src=6 src2=3 imm=0xf80670d2 imm2=0x15cede99 rot=13 bit=13 mask= 8 win=0 off=0
|
||||||
|
57: mulhi dst=3 src=1 src2=1 imm=0x722658c1 imm2=0x4391d7dc rot= 3 bit=14 mask= 8 win=0 off=0
|
||||||
|
58: rotl dst=6 src=5 src2=5 imm=0xa3e8fdef imm2=0x6141f682 rot= 2 bit=29 mask=16 win=0 off=0
|
||||||
|
59: load dst=7 src=4 src2=1 imm=0x30f577b1 imm2=0x5962c42e rot= 7 bit=17 mask= 1 win=1 off=0
|
||||||
|
60: mulhi dst=3 src=5 src2=4 imm=0x89df950e imm2=0xac92aa6e rot=25 bit=14 mask= 4 win=0 off=0
|
||||||
|
61: shfl dst=0 src=6 src2=5 imm=0xc96200c1 imm2=0x083a4180 rot=20 bit=26 mask= 4 win=0 off=0
|
||||||
|
62: load dst=0 src=6 src2=6 imm=0xaf77797d imm2=0x159756e0 rot=20 bit=15 mask= 2 win=2 off=1
|
||||||
|
63: mulhi dst=5 src=4 src2=5 imm=0x633b127c imm2=0x28008a24 rot= 9 bit=27 mask= 1 win=0 off=0
|
||||||
|
|
@ -0,0 +1,2 @@
|
||||||
|
[2026-10-07T19:22:25Z] export: best 2048 of 31250 units by distinct 8 KiB rows: rows8 from 3982 to 4005 (median unit 4019)
|
||||||
|
[2026-10-07T19:22:26Z] export: wrote 2048 units (33554432 bytes) to tables/devnet-best2048-of-1e6.bin, prog=devnet plant=none best=true, mean distinct 8 KiB rows 4001.37
|
||||||
|
|
@ -0,0 +1 @@
|
||||||
|
[2026-10-07T19:22:32Z] export: wrote 2048 units (33554432 bytes) to tables/devnet-const2048.bin, prog=devnet plant=const-site best=false, mean distinct 8 KiB rows 3776.81
|
||||||
|
|
@ -0,0 +1 @@
|
||||||
|
[2026-10-07T19:22:29Z] export: wrote 2048 units (33554432 bytes) to tables/devnet-rand2048.bin, prog=devnet plant=none best=false, mean distinct 8 KiB rows 4018.21
|
||||||
|
|
@ -0,0 +1 @@
|
||||||
|
[2026-10-07T19:22:36Z] export: wrote 2048 units (33554432 bytes) to tables/devnet-tiny2048.bin, prog=devnet plant=tiny-window best=false, mean distinct 8 KiB rows 3988.22
|
||||||
|
|
@ -0,0 +1,13 @@
|
||||||
|
rowbench on NVIDIA RTX A6000, 2026-10-07T19:23:47Z
|
||||||
|
rand2048 rep1 units=2048 rounds=200 ms=253.907 reads/s=6.6076e+09 unit-hashes/s=1.6132e+06 (32 lanes x 128 dependent 4-byte reads per unit)
|
||||||
|
best2048-of-1e6 rep1 units=2048 rounds=200 ms=253.698 reads/s=6.6131e+09 unit-hashes/s=1.6145e+06 (32 lanes x 128 dependent 4-byte reads per unit)
|
||||||
|
tiny2048 rep1 units=2048 rounds=200 ms=252.587 reads/s=6.6422e+09 unit-hashes/s=1.6216e+06 (32 lanes x 128 dependent 4-byte reads per unit)
|
||||||
|
const2048 rep1 units=2048 rounds=200 ms=238.940 reads/s=7.0215e+09 unit-hashes/s=1.7142e+06 (32 lanes x 128 dependent 4-byte reads per unit)
|
||||||
|
rand2048 rep2 units=2048 rounds=200 ms=254.018 reads/s=6.6047e+09 unit-hashes/s=1.6125e+06 (32 lanes x 128 dependent 4-byte reads per unit)
|
||||||
|
best2048-of-1e6 rep2 units=2048 rounds=200 ms=253.669 reads/s=6.6138e+09 unit-hashes/s=1.6147e+06 (32 lanes x 128 dependent 4-byte reads per unit)
|
||||||
|
tiny2048 rep2 units=2048 rounds=200 ms=252.671 reads/s=6.6399e+09 unit-hashes/s=1.6211e+06 (32 lanes x 128 dependent 4-byte reads per unit)
|
||||||
|
const2048 rep2 units=2048 rounds=200 ms=238.982 reads/s=7.0203e+09 unit-hashes/s=1.7139e+06 (32 lanes x 128 dependent 4-byte reads per unit)
|
||||||
|
rand2048 rep3 units=2048 rounds=200 ms=253.937 reads/s=6.6068e+09 unit-hashes/s=1.6130e+06 (32 lanes x 128 dependent 4-byte reads per unit)
|
||||||
|
best2048-of-1e6 rep3 units=2048 rounds=200 ms=253.703 reads/s=6.6129e+09 unit-hashes/s=1.6145e+06 (32 lanes x 128 dependent 4-byte reads per unit)
|
||||||
|
tiny2048 rep3 units=2048 rounds=200 ms=252.662 reads/s=6.6402e+09 unit-hashes/s=1.6211e+06 (32 lanes x 128 dependent 4-byte reads per unit)
|
||||||
|
const2048 rep3 units=2048 rounds=200 ms=239.020 reads/s=7.0192e+09 unit-hashes/s=1.7137e+06 (32 lanes x 128 dependent 4-byte reads per unit)
|
||||||
|
|
@ -0,0 +1 @@
|
||||||
|
lease: 0 of 8 pool cores free (88 leased); waiting
|
||||||
|
|
@ -0,0 +1,4 @@
|
||||||
|
lease: 0 of 8 pool cores free (88 leased); waiting
|
||||||
|
lease: holding 16 pool cores (45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60, waited 425 s, class adv): adv-accept-2 1e8 locality tail and rotclass census
|
||||||
|
Terminated
|
||||||
|
lease: released 16 pool cores after 1878 s, exit 143
|
||||||
10
docs/analysis/cryptanalysis/logs/adv-accept-2/prefix.log
Normal file
10
docs/analysis/cryptanalysis/logs/adv-accept-2/prefix.log
Normal file
|
|
@ -0,0 +1,10 @@
|
||||||
|
[2026-10-07T18:52:31Z] prefix: prog=devnet id=0xa785001687d8688a: load-independent sites in iteration 0 = 4 of 16 (instructions [1, 10, 12, 30]); in later iterations = 0
|
||||||
|
[2026-10-07T18:52:35Z] prefix: prog=devnet3 id=0xfce15bf61030be57: load-independent sites in iteration 0 = 4 of 16 (instructions [3, 8, 14, 15]); in later iterations = 0
|
||||||
|
[2026-10-07T18:52:39Z] prefix: prog=drawn:0 id=0x5d7cc2b09fc6922a: load-independent sites in iteration 0 = 2 of 16 (instructions [4, 6]); in later iterations = 0
|
||||||
|
[2026-10-07T18:52:44Z] prefix: prog=drawn:1 id=0x2c81972d33e22ad2: load-independent sites in iteration 0 = 2 of 16 (instructions [2, 5]); in later iterations = 0
|
||||||
|
[2026-10-07T18:52:48Z] prefix: prog=drawn:2 id=0xd3fead516b1ab00c: load-independent sites in iteration 0 = 3 of 16 (instructions [2, 4, 18]); in later iterations = 0
|
||||||
|
[2026-10-07T18:52:53Z] prefix: prog=drawn:3 id=0xb7350750beb0ded9: load-independent sites in iteration 0 = 2 of 16 (instructions [4, 25]); in later iterations = 0
|
||||||
|
[2026-10-07T18:52:57Z] prefix: prog=drawn:4 id=0x180a596485ad0153: load-independent sites in iteration 0 = 3 of 16 (instructions [4, 7, 12]); in later iterations = 0
|
||||||
|
[2026-10-07T18:53:02Z] prefix: prog=drawn:5 id=0x1a77e8de160b6db2: load-independent sites in iteration 0 = 2 of 16 (instructions [3, 10]); in later iterations = 0
|
||||||
|
[2026-10-07T18:53:06Z] prefix: prog=drawn:6 id=0x50b76694674c0d97: load-independent sites in iteration 0 = 3 of 16 (instructions [2, 8, 14]); in later iterations = 0
|
||||||
|
[2026-10-07T18:53:10Z] prefix: prog=drawn:7 id=0xd694dd0bafb2a724: load-independent sites in iteration 0 = 2 of 16 (instructions [1, 19]); in later iterations = 0
|
||||||
19
docs/analysis/cryptanalysis/logs/adv-accept-2/price.log
Normal file
19
docs/analysis/cryptanalysis/logs/adv-accept-2/price.log
Normal file
|
|
@ -0,0 +1,19 @@
|
||||||
|
Q3 price model (internal adversarial pass, not an independent review)
|
||||||
|
A card is bound by random 4-byte reads: rate R_hash = (reads/s) / loads_per_hash, loads_per_hash = 128.
|
||||||
|
A grind finds a header whose hash saves dL loads (distinct items below 128). On the card the found hash
|
||||||
|
then costs 128 - dL useful reads, but every searched header is itself a full 128-read hash evaluation.
|
||||||
|
If the saving dL >= t appears with probability 1/S (the tail rate), one found hash costs S search hashes,
|
||||||
|
each 128 reads, and yields one useful hash of 128 - dL reads. Net rate vs honest:
|
||||||
|
gain = 128 / ((128 - dL) + S * 128) (the search reads amortise over one found hash only; a found
|
||||||
|
header mines ONE 32-lane group, not a stream, because the address set is fixed by (program, I, g)).
|
||||||
|
|
||||||
|
tail rate 1/S dL saved useful reads net rate vs honest over 1%?
|
||||||
|
1000 8 120 0.000999x no
|
||||||
|
10000 16 112 0.000100x no
|
||||||
|
100000 32 96 0.000010x no
|
||||||
|
10000 2 126 0.000100x no
|
||||||
|
100000 4 124 0.000010x no
|
||||||
|
|
||||||
|
Even a 32-load saving at the 1e-5 tail nets 128 / (96 + 1e5*128) = 1.0e-5x: the search cost dwarfs the
|
||||||
|
saving by five orders. A found header mines one group, so the search never amortises. GPU confirmation
|
||||||
|
is BLOCKED tonight (no card); the bound is analytic from the read counts and holds for any dL < 128.
|
||||||
|
|
@ -0,0 +1,13 @@
|
||||||
|
[2026-10-07T19:06:28Z] repeats: prog=devnet id=0xa785001687d8688a: 12 same-word repeats over 200000 hashes (0.00006 per hash); top pairs (load k = iteration*16 + site):
|
||||||
|
loads 88 and 121: iteration 5 site 8 (instr 35) with iteration 7 site 9 (instr 40): 1 repeats
|
||||||
|
loads 61 and 116: iteration 3 site 13 (instr 52) with iteration 7 site 4 (instr 12): 1 repeats
|
||||||
|
loads 49 and 60: iteration 3 site 1 (instr 4) with iteration 3 site 12 (instr 45): 1 repeats
|
||||||
|
loads 11 and 75: iteration 0 site 11 (instr 43) with iteration 4 site 11 (instr 43): 1 repeats
|
||||||
|
loads 115 and 123: iteration 7 site 3 (instr 10) with iteration 7 site 11 (instr 43): 1 repeats
|
||||||
|
loads 10 and 30: iteration 0 site 10 (instr 41) with iteration 1 site 14 (instr 53): 1 repeats
|
||||||
|
loads 32 and 109: iteration 2 site 0 (instr 1) with iteration 6 site 13 (instr 52): 1 repeats
|
||||||
|
loads 8 and 96: iteration 0 site 8 (instr 35) with iteration 6 site 0 (instr 1): 1 repeats
|
||||||
|
loads 36 and 114: iteration 2 site 4 (instr 12) with iteration 7 site 2 (instr 6): 1 repeats
|
||||||
|
loads 32 and 120: iteration 2 site 0 (instr 1) with iteration 7 site 8 (instr 35): 1 repeats
|
||||||
|
loads 62 and 123: iteration 3 site 14 (instr 53) with iteration 7 site 11 (instr 43): 1 repeats
|
||||||
|
loads 71 and 123: iteration 4 site 7 (instr 30) with iteration 7 site 11 (instr 43): 1 repeats
|
||||||
|
|
@ -0,0 +1,13 @@
|
||||||
|
[2026-10-07T19:06:16Z] repeats: prog=drawn:0 id=0x5d7cc2b09fc6922a: 24830 same-word repeats over 200000 hashes (0.12415 per hash); top pairs (load k = iteration*16 + site):
|
||||||
|
loads 97 and 101: iteration 6 site 1 (instr 6) with iteration 6 site 5 (instr 26): 3137 repeats
|
||||||
|
loads 65 and 69: iteration 4 site 1 (instr 6) with iteration 4 site 5 (instr 26): 3130 repeats
|
||||||
|
loads 1 and 5: iteration 0 site 1 (instr 6) with iteration 0 site 5 (instr 26): 3122 repeats
|
||||||
|
loads 81 and 85: iteration 5 site 1 (instr 6) with iteration 5 site 5 (instr 26): 3107 repeats
|
||||||
|
loads 33 and 37: iteration 2 site 1 (instr 6) with iteration 2 site 5 (instr 26): 3106 repeats
|
||||||
|
loads 49 and 53: iteration 3 site 1 (instr 6) with iteration 3 site 5 (instr 26): 3080 repeats
|
||||||
|
loads 113 and 117: iteration 7 site 1 (instr 6) with iteration 7 site 5 (instr 26): 3072 repeats
|
||||||
|
loads 17 and 21: iteration 1 site 1 (instr 6) with iteration 1 site 5 (instr 26): 3067 repeats
|
||||||
|
loads 49 and 61: iteration 3 site 1 (instr 6) with iteration 3 site 13 (instr 52): 1 repeats
|
||||||
|
loads 19 and 25: iteration 1 site 3 (instr 13) with iteration 1 site 9 (instr 40): 1 repeats
|
||||||
|
loads 90 and 104: iteration 5 site 10 (instr 43) with iteration 6 site 8 (instr 38): 1 repeats
|
||||||
|
loads 44 and 88: iteration 2 site 12 (instr 50) with iteration 5 site 8 (instr 38): 1 repeats
|
||||||
|
|
@ -0,0 +1,74 @@
|
||||||
|
drawn:0 id=0x5d7cc2b09fc6922a attempt=0: loads at order 6 and 26 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:1 id=0x2c81972d33e22ad2 attempt=1: loads at order 24 and 44 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:7 id=0xd694dd0bafb2a724 attempt=0: loads at order 1 and 19 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:26 id=0x604c5f1ec8101b51 attempt=7: loads at order 11 and 20 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:26 id=0x604c5f1ec8101b51 attempt=7: loads at order 44 and 53 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:31 id=0xa882aa8aceff369d attempt=1: loads at order 27 and 31 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:33 id=0x273b5b17001effb7 attempt=2: loads at order 30 and 48 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:36 id=0x81e122557a85c854 attempt=2: loads at order 12 and 23 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:42 id=0xcf995d9d87c28c2b attempt=0: loads at order 19 and 33 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:45 id=0x10456ea4dbb45fb6 attempt=4: loads at order 30 and 48 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:52 id=0x4bf852f46bbe830f attempt=8: loads at order 48 and 57 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:53 id=0xfcd9b9abf4ebac92 attempt=8: loads at order 22 and 34 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:65 id=0xf2f22a1bb619c7a5 attempt=3: loads at order 46 and 50 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:72 id=0xb7a4fa92098b7716 attempt=2: loads at order 18 and 35 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:72 id=0xb7a4fa92098b7716 attempt=2: loads at order 40 and 46 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:76 id=0x1033a7635904b1cc attempt=3: loads at order 1 and 37 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:77 id=0xccad6129b648c62e attempt=1: loads at order 46 and 62 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:78 id=0xe7785475b4798064 attempt=3: loads at order 21 and 36 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:81 id=0x447270e8f468dba9 attempt=2: loads at order 3 and 21 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:88 id=0xa26ae16caccd6b0a attempt=0: loads at order 17 and 33 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:90 id=0x5ab1b1884bdec1f0 attempt=0: loads at order 48 and 57 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:93 id=0xe307a169db2a04f7 attempt=0: loads at order 4 and 6 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:99 id=0x4a46626f38d0f37e attempt=3: loads at order 29 and 49 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:100 id=0x9e43817d3cdbe05b attempt=4: loads at order 37 and 48 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:102 id=0x0b3148f343ea87d7 attempt=0: loads at order 10 and 26 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:103 id=0x7db3033f31c15bea attempt=1: loads at order 39 and 62 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:106 id=0x31781a41bd30f514 attempt=1: loads at order 9 and 35 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:108 id=0xffede3d7f736c7ef attempt=7: loads at order 33 and 54 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:108 id=0xffede3d7f736c7ef attempt=7: loads at order 37 and 44 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:111 id=0xc87ec4eb51489e05 attempt=1: loads at order 53 and 59 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:114 id=0x1785de0cfa228d5d attempt=0: loads at order 38 and 51 read one register with 3 rotr write(s) between: identity probability 32^-3
|
||||||
|
drawn:119 id=0x884bbac4110e21b5 attempt=8: loads at order 1 and 8 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:119 id=0x884bbac4110e21b5 attempt=8: loads at order 45 and 51 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:121 id=0xae374fcc547964f3 attempt=1: loads at order 11 and 37 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:129 id=0x89954aa9c6660a98 attempt=3: loads at order 5 and 10 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:130 id=0x54d60b06db5c2961 attempt=4: loads at order 19 and 26 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:136 id=0xeb43c55ac62cf8f0 attempt=0: loads at order 3 and 6 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:136 id=0xeb43c55ac62cf8f0 attempt=0: loads at order 44 and 48 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:144 id=0x98e30a1493e34e2d attempt=0: loads at order 12 and 20 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:147 id=0xb25c1b2eb1f5176b attempt=0: loads at order 2 and 4 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:147 id=0xb25c1b2eb1f5176b attempt=0: loads at order 25 and 38 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:149 id=0x9172cdbe8ff08447 attempt=7: loads at order 41 and 53 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:153 id=0xd1555ab82965865c attempt=1: loads at order 1 and 3 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:158 id=0xd8bb18068c5f4ef2 attempt=4: loads at order 4 and 28 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:164 id=0xa40d4fafa1ce7c43 attempt=0: loads at order 22 and 31 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:166 id=0x67c1dbdae6558ec8 attempt=1: loads at order 40 and 46 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:168 id=0x24d20b96f553cbfe attempt=2: loads at order 17 and 32 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:176 id=0xad63f1aea202a608 attempt=1: loads at order 12 and 18 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:181 id=0xe97116d4279e613b attempt=5: loads at order 46 and 62 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:182 id=0xc9688b1bdbd1aeed attempt=3: loads at order 35 and 46 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:185 id=0xd57149b7cb250457 attempt=0: loads at order 38 and 42 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:188 id=0x99f7026f8eb3c247 attempt=0: loads at order 12 and 21 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:189 id=0xba7dde81ac8081c1 attempt=0: loads at order 27 and 35 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:195 id=0x441e34a82aa1a496 attempt=2: loads at order 26 and 58 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:202 id=0x455f4d334381b91e attempt=10: loads at order 43 and 48 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:206 id=0xce8408ba9bfe79e0 attempt=0: loads at order 30 and 47 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:222 id=0x38b0f6c27d7d11eb attempt=0: loads at order 4 and 21 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:222 id=0x38b0f6c27d7d11eb attempt=0: loads at order 29 and 36 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:224 id=0x46d38a1fefb93d0a attempt=1: loads at order 8 and 28 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:227 id=0xc1b21e0964a55a5e attempt=9: loads at order 50 and 57 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:228 id=0x20795e5a6037607f attempt=0: loads at order 16 and 28 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:228 id=0x20795e5a6037607f attempt=0: loads at order 34 and 40 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:238 id=0xc2bb5c513709ccec attempt=13: loads at order 9 and 44 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:238 id=0xc2bb5c513709ccec attempt=13: loads at order 30 and 58 read one register with 2 rotr write(s) between: identity probability 32^-2
|
||||||
|
drawn:247 id=0xa0afbc7b48b0c01d attempt=0: loads at order 12 and 40 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:248 id=0xa2605e0580cb4466 attempt=3: loads at order 45 and 53 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:254 id=0xa74c6c1fa228dc21 attempt=4: loads at order 43 and 53 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:258 id=0x803ab16d31b21d4c attempt=1: loads at order 4 and 9 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:262 id=0x99b4d06d49d64195 attempt=7: loads at order 37 and 52 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:268 id=0xed6a85c17173bf31 attempt=3: loads at order 47 and 55 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:274 id=0x3edddaebeafa7558 attempt=3: loads at order 22 and 46 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:285 id=0xde945a9effb8bd42 attempt=1: loads at order 40 and 43 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
drawn:285 id=0xde945a9effb8bd42 attempt=1: loads at order 48 and 57 read one register with 1 rotr write(s) between: identity probability 32^-1
|
||||||
|
[2026-10-07T20:14:10Z] rotclass: 300 drawn accepted class v4 programs: 63 affected (21.00 percent), 73 load-site pairs whose only intervening writes are rotr; by rotr count: {1: 71, 2: 1, 3: 1}
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:49:49Z] rows: prog=devnet id=0xa785001687d8688a attempt=1 sites=16 plant=none live=false hashes=100000000 shard=0/16
|
||||||
|
[2026-10-07T19:52:41Z] done 195313 units (6250016 lane-hashes) in 172.5s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [2, 2, 2, 1, 0, 0, 0, 0, 2, 0, 2, 0, 2, 1, 0, 2]
|
||||||
|
== PER HASH (128 loads) prog=devnet plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.981 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.924 124 126 126 125 | 127.924 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 126 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 126 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.400 4054 4061 4058 4054 | 4076.384 4054 4062 4058 4055 | 4080.055 4060
|
||||||
|
rows8KiB 4018.433 3978 3991 3985 3981 | 4018.404 3974 3991 3985 3977 | 4032.665 3995
|
||||||
|
lines64B 4095.385 4090 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.385 4090 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2973 5
|
||||||
|
pairs row8KiB 1.1811 8
|
||||||
|
pairs line64B 0.0094 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:49:49Z] rows: prog=devnet id=0xa785001687d8688a attempt=1 sites=16 plant=none live=false hashes=100000000 shard=1/16
|
||||||
|
[2026-10-07T19:52:41Z] done 195313 units (6250016 lane-hashes) in 171.7s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [2, 2, 2, 1, 0, 0, 0, 0, 2, 0, 2, 0, 2, 1, 0, 2]
|
||||||
|
== PER HASH (128 loads) prog=devnet plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.981 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.924 124 126 126 125 | 127.924 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.403 4055 4062 4058 4055 | 4076.384 4054 4062 4058 4055 | 4080.055 4060
|
||||||
|
rows8KiB 4018.431 3978 3991 3985 3979 | 4018.404 3974 3991 3985 3977 | 4032.665 3995
|
||||||
|
lines64B 4095.385 4089 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.385 4089 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2971 5
|
||||||
|
pairs row8KiB 1.1833 9
|
||||||
|
pairs line64B 0.0093 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:49:49Z] rows: prog=devnet id=0xa785001687d8688a attempt=1 sites=16 plant=none live=false hashes=100000000 shard=10/16
|
||||||
|
[2026-10-07T19:53:03Z] done 195313 units (6250016 lane-hashes) in 193.4s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [2, 2, 2, 1, 0, 0, 0, 0, 2, 0, 2, 0, 2, 1, 0, 2]
|
||||||
|
== PER HASH (128 loads) prog=devnet plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.981 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.924 124 126 126 125 | 127.924 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 126 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 126 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.396 4053 4061 4057 4054 | 4076.384 4054 4062 4058 4055 | 4080.055 4060
|
||||||
|
rows8KiB 4018.419 3977 3990 3984 3978 | 4018.404 3974 3991 3985 3977 | 4032.665 3995
|
||||||
|
lines64B 4095.386 4089 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.386 4089 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2971 5
|
||||||
|
pairs row8KiB 1.1831 8
|
||||||
|
pairs line64B 0.0096 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:49:50Z] rows: prog=devnet id=0xa785001687d8688a attempt=1 sites=16 plant=none live=false hashes=100000000 shard=11/16
|
||||||
|
[2026-10-07T19:53:41Z] done 195313 units (6250016 lane-hashes) in 230.8s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [2, 2, 2, 1, 0, 0, 0, 0, 2, 0, 2, 0, 2, 1, 0, 2]
|
||||||
|
== PER HASH (128 loads) prog=devnet plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.981 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.924 124 126 126 125 | 127.924 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 126 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 126 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.383 4054 4061 4058 4056 | 4076.384 4054 4062 4058 4055 | 4080.055 4060
|
||||||
|
rows8KiB 4018.404 3975 3990 3984 3975 | 4018.404 3974 3991 3985 3977 | 4032.665 3995
|
||||||
|
lines64B 4095.385 4089 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.385 4089 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2951 5
|
||||||
|
pairs row8KiB 1.1821 8
|
||||||
|
pairs line64B 0.0091 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:49:50Z] rows: prog=devnet id=0xa785001687d8688a attempt=1 sites=16 plant=none live=false hashes=100000000 shard=12/16
|
||||||
|
[2026-10-07T19:52:51Z] done 195313 units (6250016 lane-hashes) in 180.7s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [2, 2, 2, 1, 0, 0, 0, 0, 2, 0, 2, 0, 2, 1, 0, 2]
|
||||||
|
== PER HASH (128 loads) prog=devnet plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.981 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.924 124 126 126 125 | 127.924 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 126 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 126 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.392 4052 4062 4058 4053 | 4076.384 4054 4062 4058 4055 | 4080.055 4060
|
||||||
|
rows8KiB 4018.441 3973 3991 3985 3980 | 4018.404 3974 3991 3985 3977 | 4032.665 3995
|
||||||
|
lines64B 4095.383 4090 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.383 4090 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2944 5
|
||||||
|
pairs row8KiB 1.1798 9
|
||||||
|
pairs line64B 0.0090 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:49:49Z] rows: prog=devnet id=0xa785001687d8688a attempt=1 sites=16 plant=none live=false hashes=100000000 shard=13/16
|
||||||
|
[2026-10-07T19:52:47Z] done 195313 units (6250016 lane-hashes) in 177.8s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [2, 2, 2, 1, 0, 0, 0, 0, 2, 0, 2, 0, 2, 1, 0, 2]
|
||||||
|
== PER HASH (128 loads) prog=devnet plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.981 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.924 124 126 126 125 | 127.924 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 126 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 126 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.396 4052 4062 4058 4052 | 4076.384 4054 4062 4058 4055 | 4080.055 4060
|
||||||
|
rows8KiB 4018.447 3978 3991 3984 3978 | 4018.404 3974 3991 3985 3977 | 4032.665 3995
|
||||||
|
lines64B 4095.384 4089 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.384 4089 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2925 5
|
||||||
|
pairs row8KiB 1.1757 9
|
||||||
|
pairs line64B 0.0090 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:49:49Z] rows: prog=devnet id=0xa785001687d8688a attempt=1 sites=16 plant=none live=false hashes=100000000 shard=14/16
|
||||||
|
[2026-10-07T19:52:49Z] done 195313 units (6250016 lane-hashes) in 179.2s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [2, 2, 2, 1, 0, 0, 0, 0, 2, 0, 2, 0, 2, 1, 0, 2]
|
||||||
|
== PER HASH (128 loads) prog=devnet plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 124 127 126 126 | 127.981 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.924 123 126 126 125 | 127.924 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.402 4055 4062 4058 4055 | 4076.384 4054 4062 4058 4055 | 4080.055 4060
|
||||||
|
rows8KiB 4018.430 3977 3991 3986 3981 | 4018.404 3974 3991 3985 3977 | 4032.665 3995
|
||||||
|
lines64B 4095.387 4090 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.387 4090 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2973 5
|
||||||
|
pairs row8KiB 1.1823 8
|
||||||
|
pairs line64B 0.0096 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:49:49Z] rows: prog=devnet id=0xa785001687d8688a attempt=1 sites=16 plant=none live=false hashes=100000000 shard=15/16
|
||||||
|
[2026-10-07T19:52:46Z] done 195305 units (6249760 lane-hashes) in 176.4s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [2, 2, 2, 1, 0, 0, 0, 0, 2, 0, 2, 0, 2, 1, 0, 2]
|
||||||
|
== PER HASH (128 loads) prog=devnet plant=none live=false n=6249760
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.981 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.924 124 126 126 125 | 127.924 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 126 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 126 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet plant=none live=false n=195305
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.390 4054 4061 4058 4055 | 4076.384 4054 4062 4058 4055 | 4080.055 4060
|
||||||
|
rows8KiB 4018.418 3978 3991 3984 3978 | 4018.403 3974 3991 3985 3977 | 4032.665 3995
|
||||||
|
lines64B 4095.385 4090 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.385 4090 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2944 5
|
||||||
|
pairs row8KiB 1.1786 8
|
||||||
|
pairs line64B 0.0091 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:49:49Z] rows: prog=devnet id=0xa785001687d8688a attempt=1 sites=16 plant=none live=false hashes=100000000 shard=2/16
|
||||||
|
[2026-10-07T19:52:41Z] done 195313 units (6250016 lane-hashes) in 172.5s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [2, 2, 2, 1, 0, 0, 0, 0, 2, 0, 2, 0, 2, 1, 0, 2]
|
||||||
|
== PER HASH (128 loads) prog=devnet plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.981 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.924 124 126 126 125 | 127.924 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.396 4053 4061 4058 4054 | 4076.384 4054 4062 4058 4055 | 4080.055 4060
|
||||||
|
rows8KiB 4018.427 3979 3991 3985 3980 | 4018.404 3974 3991 3985 3977 | 4032.665 3995
|
||||||
|
lines64B 4095.387 4090 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.387 4090 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2959 5
|
||||||
|
pairs row8KiB 1.1792 8
|
||||||
|
pairs line64B 0.0093 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:49:49Z] rows: prog=devnet id=0xa785001687d8688a attempt=1 sites=16 plant=none live=false hashes=100000000 shard=3/16
|
||||||
|
[2026-10-07T19:52:42Z] done 195313 units (6250016 lane-hashes) in 172.9s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [2, 2, 2, 1, 0, 0, 0, 0, 2, 0, 2, 0, 2, 1, 0, 2]
|
||||||
|
== PER HASH (128 loads) prog=devnet plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.981 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.924 124 126 126 125 | 127.924 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.410 4052 4062 4058 4054 | 4076.384 4054 4062 4058 4055 | 4080.055 4060
|
||||||
|
rows8KiB 4018.418 3977 3991 3985 3978 | 4018.404 3974 3991 3985 3977 | 4032.665 3995
|
||||||
|
lines64B 4095.383 4090 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.383 4090 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2942 5
|
||||||
|
pairs row8KiB 1.1777 9
|
||||||
|
pairs line64B 0.0092 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:49:49Z] rows: prog=devnet id=0xa785001687d8688a attempt=1 sites=16 plant=none live=false hashes=100000000 shard=4/16
|
||||||
|
[2026-10-07T19:53:53Z] done 195313 units (6250016 lane-hashes) in 243.6s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [2, 2, 2, 1, 0, 0, 0, 0, 2, 0, 2, 0, 2, 1, 0, 2]
|
||||||
|
== PER HASH (128 loads) prog=devnet plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.981 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.924 123 126 126 125 | 127.924 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 126 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 126 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.409 4055 4061 4058 4055 | 4076.384 4054 4062 4058 4055 | 4080.055 4060
|
||||||
|
rows8KiB 4018.456 3977 3991 3985 3978 | 4018.404 3974 3991 3985 3977 | 4032.665 3995
|
||||||
|
lines64B 4095.388 4089 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.388 4089 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2950 5
|
||||||
|
pairs row8KiB 1.1805 8
|
||||||
|
pairs line64B 0.0095 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:49:49Z] rows: prog=devnet id=0xa785001687d8688a attempt=1 sites=16 plant=none live=false hashes=100000000 shard=5/16
|
||||||
|
[2026-10-07T19:53:21Z] done 195313 units (6250016 lane-hashes) in 211.9s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [2, 2, 2, 1, 0, 0, 0, 0, 2, 0, 2, 0, 2, 1, 0, 2]
|
||||||
|
== PER HASH (128 loads) prog=devnet plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.981 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.924 124 126 126 125 | 127.924 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 126 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 126 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.400 4054 4061 4058 4054 | 4076.384 4054 4062 4058 4055 | 4080.055 4060
|
||||||
|
rows8KiB 4018.452 3977 3990 3984 3980 | 4018.404 3974 3991 3985 3977 | 4032.665 3995
|
||||||
|
lines64B 4095.385 4090 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.385 4090 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2973 4
|
||||||
|
pairs row8KiB 1.1800 8
|
||||||
|
pairs line64B 0.0093 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:49:50Z] rows: prog=devnet id=0xa785001687d8688a attempt=1 sites=16 plant=none live=false hashes=100000000 shard=6/16
|
||||||
|
[2026-10-07T19:52:49Z] done 195313 units (6250016 lane-hashes) in 179.9s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [2, 2, 2, 1, 0, 0, 0, 0, 2, 0, 2, 0, 2, 1, 0, 2]
|
||||||
|
== PER HASH (128 loads) prog=devnet plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.981 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.924 124 126 126 125 | 127.924 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.413 4055 4061 4058 4055 | 4076.384 4054 4062 4058 4055 | 4080.055 4060
|
||||||
|
rows8KiB 4018.444 3978 3991 3985 3979 | 4018.404 3974 3991 3985 3977 | 4032.665 3995
|
||||||
|
lines64B 4095.386 4090 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.386 4090 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2928 5
|
||||||
|
pairs row8KiB 1.1748 9
|
||||||
|
pairs line64B 0.0091 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:49:49Z] rows: prog=devnet id=0xa785001687d8688a attempt=1 sites=16 plant=none live=false hashes=100000000 shard=7/16
|
||||||
|
[2026-10-07T19:52:46Z] done 195313 units (6250016 lane-hashes) in 177.6s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [2, 2, 2, 1, 0, 0, 0, 0, 2, 0, 2, 0, 2, 1, 0, 2]
|
||||||
|
== PER HASH (128 loads) prog=devnet plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.981 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.924 124 126 126 125 | 127.924 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.386 4051 4061 4058 4053 | 4076.384 4054 4062 4058 4055 | 4080.055 4060
|
||||||
|
rows8KiB 4018.434 3978 3991 3985 3980 | 4018.404 3974 3991 3985 3977 | 4032.665 3995
|
||||||
|
lines64B 4095.386 4090 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.386 4090 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2972 6
|
||||||
|
pairs row8KiB 1.1837 9
|
||||||
|
pairs line64B 0.0094 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:49:49Z] rows: prog=devnet id=0xa785001687d8688a attempt=1 sites=16 plant=none live=false hashes=100000000 shard=8/16
|
||||||
|
[2026-10-07T19:52:59Z] done 195313 units (6250016 lane-hashes) in 189.8s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [2, 2, 2, 1, 0, 0, 0, 0, 2, 0, 2, 0, 2, 1, 0, 2]
|
||||||
|
== PER HASH (128 loads) prog=devnet plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.981 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.924 124 126 126 125 | 127.924 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.389 4050 4061 4058 4054 | 4076.384 4054 4062 4058 4055 | 4080.055 4060
|
||||||
|
rows8KiB 4018.440 3967 3991 3984 3975 | 4018.404 3974 3991 3985 3977 | 4032.665 3995
|
||||||
|
lines64B 4095.385 4089 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.385 4089 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2959 6
|
||||||
|
pairs row8KiB 1.1806 9
|
||||||
|
pairs line64B 0.0093 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:49:49Z] rows: prog=devnet id=0xa785001687d8688a attempt=1 sites=16 plant=none live=false hashes=100000000 shard=9/16
|
||||||
|
[2026-10-07T19:52:34Z] done 195313 units (6250016 lane-hashes) in 165.8s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [2, 2, 2, 1, 0, 0, 0, 0, 2, 0, 2, 0, 2, 1, 0, 2]
|
||||||
|
== PER HASH (128 loads) prog=devnet plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.981 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.924 124 126 126 125 | 127.924 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 126 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 126 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.396 4054 4062 4058 4055 | 4076.384 4054 4062 4058 4055 | 4080.055 4060
|
||||||
|
rows8KiB 4018.418 3976 3991 3985 3979 | 4018.404 3974 3991 3985 3977 | 4032.665 3995
|
||||||
|
lines64B 4095.384 4090 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.384 4090 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2975 5
|
||||||
|
pairs row8KiB 1.1850 9
|
||||||
|
pairs line64B 0.0095 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:54:54Z] rows: prog=devnet3 id=0xfce15bf61030be57 attempt=0 sites=16 plant=none live=false hashes=100000000 shard=0/16
|
||||||
|
[2026-10-07T19:58:36Z] done 195313 units (6250016 lane-hashes) in 222.0s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [0, 2, 1, 0, 1, 1, 1, 0, 2, 1, 1, 0, 2, 0, 1, 1]
|
||||||
|
== PER HASH (128 loads) prog=devnet3 plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.982 125 127 126 126 | 127.982 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.926 123 126 126 125 | 127.926 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet3 plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.861 4055 4062 4059 4058 | 4076.886 4057 4062 4059 4057 | 4080.055 4060
|
||||||
|
rows8KiB 4020.286 3976 3993 3987 3979 | 4020.328 3981 3993 3987 3983 | 4032.665 3995
|
||||||
|
lines64B 4095.399 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.399 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2545 5
|
||||||
|
pairs row8KiB 1.0026 8
|
||||||
|
pairs line64B 0.0079 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:55:08Z] rows: prog=devnet3 id=0xfce15bf61030be57 attempt=0 sites=16 plant=none live=false hashes=100000000 shard=1/16
|
||||||
|
[2026-10-07T19:58:24Z] done 195313 units (6250016 lane-hashes) in 196.5s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [0, 2, 1, 0, 1, 1, 1, 0, 2, 1, 1, 0, 2, 0, 1, 1]
|
||||||
|
== PER HASH (128 loads) prog=devnet3 plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.982 125 127 126 126 | 127.982 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.926 124 126 126 125 | 127.926 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 127 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 127 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet3 plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.860 4052 4062 4059 4057 | 4076.886 4057 4062 4059 4057 | 4080.055 4060
|
||||||
|
rows8KiB 4020.333 3971 3993 3987 3983 | 4020.328 3981 3993 3987 3983 | 4032.665 3995
|
||||||
|
lines64B 4095.400 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.400 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2530 5
|
||||||
|
pairs row8KiB 1.0006 7
|
||||||
|
pairs line64B 0.0078 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:55:20Z] rows: prog=devnet3 id=0xfce15bf61030be57 attempt=0 sites=16 plant=none live=false hashes=100000000 shard=10/16
|
||||||
|
[2026-10-07T19:58:31Z] done 195313 units (6250016 lane-hashes) in 191.2s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [0, 2, 1, 0, 1, 1, 1, 0, 2, 1, 1, 0, 2, 0, 1, 1]
|
||||||
|
== PER HASH (128 loads) prog=devnet3 plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.982 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.926 124 126 126 125 | 127.926 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet3 plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.853 4053 4062 4058 4054 | 4076.886 4057 4062 4059 4057 | 4080.055 4060
|
||||||
|
rows8KiB 4020.343 3979 3992 3986 3980 | 4020.328 3981 3993 3987 3983 | 4032.665 3995
|
||||||
|
lines64B 4095.399 4089 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.399 4089 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2516 5
|
||||||
|
pairs row8KiB 0.9963 8
|
||||||
|
pairs line64B 0.0081 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:55:26Z] rows: prog=devnet3 id=0xfce15bf61030be57 attempt=0 sites=16 plant=none live=false hashes=100000000 shard=11/16
|
||||||
|
[2026-10-07T19:58:35Z] done 195313 units (6250016 lane-hashes) in 189.5s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [0, 2, 1, 0, 1, 1, 1, 0, 2, 1, 1, 0, 2, 0, 1, 1]
|
||||||
|
== PER HASH (128 loads) prog=devnet3 plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.982 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.926 124 126 126 125 | 127.926 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet3 plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.869 4055 4062 4059 4055 | 4076.886 4057 4062 4059 4057 | 4080.055 4060
|
||||||
|
rows8KiB 4020.327 3981 3993 3987 3981 | 4020.328 3981 3993 3987 3983 | 4032.665 3995
|
||||||
|
lines64B 4095.399 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.399 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2528 4
|
||||||
|
pairs row8KiB 1.0015 8
|
||||||
|
pairs line64B 0.0082 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:55:31Z] rows: prog=devnet3 id=0xfce15bf61030be57 attempt=0 sites=16 plant=none live=false hashes=100000000 shard=12/16
|
||||||
|
[2026-10-07T19:58:35Z] done 195313 units (6250016 lane-hashes) in 183.9s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [0, 2, 1, 0, 1, 1, 1, 0, 2, 1, 1, 0, 2, 0, 1, 1]
|
||||||
|
== PER HASH (128 loads) prog=devnet3 plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.982 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.926 124 126 126 125 | 127.926 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 127 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 127 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet3 plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.855 4054 4062 4058 4056 | 4076.886 4057 4062 4059 4057 | 4080.055 4060
|
||||||
|
rows8KiB 4020.307 3978 3993 3986 3979 | 4020.328 3981 3993 3987 3983 | 4032.665 3995
|
||||||
|
lines64B 4095.397 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.397 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2529 5
|
||||||
|
pairs row8KiB 1.0009 8
|
||||||
|
pairs line64B 0.0078 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:55:52Z] rows: prog=devnet3 id=0xfce15bf61030be57 attempt=0 sites=16 plant=none live=false hashes=100000000 shard=13/16
|
||||||
|
[2026-10-07T19:59:01Z] done 195313 units (6250016 lane-hashes) in 189.9s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [0, 2, 1, 0, 1, 1, 1, 0, 2, 1, 1, 0, 2, 0, 1, 1]
|
||||||
|
== PER HASH (128 loads) prog=devnet3 plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.982 125 127 126 126 | 127.982 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.926 124 126 126 125 | 127.926 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 127 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 127 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet3 plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.879 4056 4062 4059 4056 | 4076.886 4057 4062 4059 4057 | 4080.055 4060
|
||||||
|
rows8KiB 4020.342 3971 3993 3988 3981 | 4020.328 3981 3993 3987 3983 | 4032.665 3995
|
||||||
|
lines64B 4095.403 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.403 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2530 4
|
||||||
|
pairs row8KiB 0.9997 9
|
||||||
|
pairs line64B 0.0075 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:56:21Z] rows: prog=devnet3 id=0xfce15bf61030be57 attempt=0 sites=16 plant=none live=false hashes=100000000 shard=14/16
|
||||||
|
[2026-10-07T19:59:30Z] done 195313 units (6250016 lane-hashes) in 188.8s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [0, 2, 1, 0, 1, 1, 1, 0, 2, 1, 1, 0, 2, 0, 1, 1]
|
||||||
|
== PER HASH (128 loads) prog=devnet3 plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.982 125 127 126 126 | 127.982 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.926 124 126 126 125 | 127.926 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet3 plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.875 4055 4062 4059 4056 | 4076.886 4057 4062 4059 4057 | 4080.055 4060
|
||||||
|
rows8KiB 4020.345 3980 3993 3987 3982 | 4020.328 3981 3993 3987 3983 | 4032.665 3995
|
||||||
|
lines64B 4095.398 4089 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.398 4089 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2552 5
|
||||||
|
pairs row8KiB 1.0001 9
|
||||||
|
pairs line64B 0.0079 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:56:21Z] rows: prog=devnet3 id=0xfce15bf61030be57 attempt=0 sites=16 plant=none live=false hashes=100000000 shard=15/16
|
||||||
|
[2026-10-07T20:00:09Z] done 195305 units (6249760 lane-hashes) in 228.1s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [0, 2, 1, 0, 1, 1, 1, 0, 2, 1, 1, 0, 2, 0, 1, 1]
|
||||||
|
== PER HASH (128 loads) prog=devnet3 plant=none live=false n=6249760
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.982 125 127 126 126 | 127.982 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.926 124 126 126 125 | 127.926 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet3 plant=none live=false n=195305
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.859 4054 4062 4059 4055 | 4076.886 4057 4062 4059 4057 | 4080.055 4060
|
||||||
|
rows8KiB 4020.331 3979 3993 3988 3980 | 4020.328 3981 3993 3987 3983 | 4032.665 3995
|
||||||
|
lines64B 4095.399 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.399 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2536 4
|
||||||
|
pairs row8KiB 0.9976 8
|
||||||
|
pairs line64B 0.0077 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:55:06Z] rows: prog=devnet3 id=0xfce15bf61030be57 attempt=0 sites=16 plant=none live=false hashes=100000000 shard=2/16
|
||||||
|
[2026-10-07T19:58:17Z] done 195313 units (6250016 lane-hashes) in 191.4s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [0, 2, 1, 0, 1, 1, 1, 0, 2, 1, 1, 0, 2, 0, 1, 1]
|
||||||
|
== PER HASH (128 loads) prog=devnet3 plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.982 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.926 124 126 126 125 | 127.926 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 127 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 127 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet3 plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.869 4053 4062 4059 4056 | 4076.886 4057 4062 4059 4057 | 4080.055 4060
|
||||||
|
rows8KiB 4020.336 3982 3993 3987 3982 | 4020.328 3981 3993 3987 3983 | 4032.665 3995
|
||||||
|
lines64B 4095.401 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.401 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2535 5
|
||||||
|
pairs row8KiB 0.9999 8
|
||||||
|
pairs line64B 0.0081 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:55:09Z] rows: prog=devnet3 id=0xfce15bf61030be57 attempt=0 sites=16 plant=none live=false hashes=100000000 shard=3/16
|
||||||
|
[2026-10-07T19:58:45Z] done 195313 units (6250016 lane-hashes) in 216.2s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [0, 2, 1, 0, 1, 1, 1, 0, 2, 1, 1, 0, 2, 0, 1, 1]
|
||||||
|
== PER HASH (128 loads) prog=devnet3 plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.982 125 127 126 126 | 127.982 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.926 124 126 126 125 | 127.926 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet3 plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.869 4055 4062 4059 4055 | 4076.886 4057 4062 4059 4057 | 4080.055 4060
|
||||||
|
rows8KiB 4020.338 3975 3993 3987 3980 | 4020.328 3981 3993 3987 3983 | 4032.665 3995
|
||||||
|
lines64B 4095.399 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.399 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2528 5
|
||||||
|
pairs row8KiB 0.9974 8
|
||||||
|
pairs line64B 0.0081 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:55:11Z] rows: prog=devnet3 id=0xfce15bf61030be57 attempt=0 sites=16 plant=none live=false hashes=100000000 shard=4/16
|
||||||
|
[2026-10-07T19:58:24Z] done 195313 units (6250016 lane-hashes) in 192.7s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [0, 2, 1, 0, 1, 1, 1, 0, 2, 1, 1, 0, 2, 0, 1, 1]
|
||||||
|
== PER HASH (128 loads) prog=devnet3 plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.982 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.926 124 126 126 125 | 127.926 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet3 plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.845 4053 4062 4058 4055 | 4076.886 4057 4062 4059 4057 | 4080.055 4060
|
||||||
|
rows8KiB 4020.320 3980 3993 3986 3982 | 4020.328 3981 3993 3987 3983 | 4032.665 3995
|
||||||
|
lines64B 4095.399 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.399 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2526 4
|
||||||
|
pairs row8KiB 1.0002 8
|
||||||
|
pairs line64B 0.0079 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:55:11Z] rows: prog=devnet3 id=0xfce15bf61030be57 attempt=0 sites=16 plant=none live=false hashes=100000000 shard=5/16
|
||||||
|
[2026-10-07T19:58:41Z] done 195313 units (6250016 lane-hashes) in 209.9s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [0, 2, 1, 0, 1, 1, 1, 0, 2, 1, 1, 0, 2, 0, 1, 1]
|
||||||
|
== PER HASH (128 loads) prog=devnet3 plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.982 125 127 126 126 | 127.982 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.926 124 126 126 125 | 127.926 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet3 plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.873 4054 4062 4059 4055 | 4076.886 4057 4062 4059 4057 | 4080.055 4060
|
||||||
|
rows8KiB 4020.368 3982 3993 3987 3983 | 4020.328 3981 3993 3987 3983 | 4032.665 3995
|
||||||
|
lines64B 4095.401 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.401 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2516 4
|
||||||
|
pairs row8KiB 0.9969 8
|
||||||
|
pairs line64B 0.0079 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:55:16Z] rows: prog=devnet3 id=0xfce15bf61030be57 attempt=0 sites=16 plant=none live=false hashes=100000000 shard=6/16
|
||||||
|
[2026-10-07T19:58:28Z] done 195313 units (6250016 lane-hashes) in 191.6s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [0, 2, 1, 0, 1, 1, 1, 0, 2, 1, 1, 0, 2, 0, 1, 1]
|
||||||
|
== PER HASH (128 loads) prog=devnet3 plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.982 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.926 123 126 126 125 | 127.926 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet3 plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.863 4051 4062 4059 4056 | 4076.886 4057 4062 4059 4057 | 4080.055 4060
|
||||||
|
rows8KiB 4020.339 3980 3993 3986 3981 | 4020.328 3981 3993 3987 3983 | 4032.665 3995
|
||||||
|
lines64B 4095.401 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.401 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2510 5
|
||||||
|
pairs row8KiB 0.9947 8
|
||||||
|
pairs line64B 0.0081 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:55:17Z] rows: prog=devnet3 id=0xfce15bf61030be57 attempt=0 sites=16 plant=none live=false hashes=100000000 shard=7/16
|
||||||
|
[2026-10-07T19:58:43Z] done 195313 units (6250016 lane-hashes) in 205.9s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [0, 2, 1, 0, 1, 1, 1, 0, 2, 1, 1, 0, 2, 0, 1, 1]
|
||||||
|
== PER HASH (128 loads) prog=devnet3 plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.982 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.926 124 126 126 125 | 127.926 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet3 plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.846 4055 4062 4059 4056 | 4076.886 4057 4062 4059 4057 | 4080.055 4060
|
||||||
|
rows8KiB 4020.312 3980 3993 3987 3980 | 4020.328 3981 3993 3987 3983 | 4032.665 3995
|
||||||
|
lines64B 4095.399 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.399 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2520 5
|
||||||
|
pairs row8KiB 0.9999 9
|
||||||
|
pairs line64B 0.0079 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:55:16Z] rows: prog=devnet3 id=0xfce15bf61030be57 attempt=0 sites=16 plant=none live=false hashes=100000000 shard=8/16
|
||||||
|
[2026-10-07T19:58:25Z] done 195313 units (6250016 lane-hashes) in 188.2s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [0, 2, 1, 0, 1, 1, 1, 0, 2, 1, 1, 0, 2, 0, 1, 1]
|
||||||
|
== PER HASH (128 loads) prog=devnet3 plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.982 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.926 124 126 126 125 | 127.926 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet3 plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.870 4054 4062 4059 4054 | 4076.886 4057 4062 4059 4057 | 4080.055 4060
|
||||||
|
rows8KiB 4020.322 3980 3993 3987 3981 | 4020.328 3981 3993 3987 3983 | 4032.665 3995
|
||||||
|
lines64B 4095.397 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.397 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2537 5
|
||||||
|
pairs row8KiB 1.0006 8
|
||||||
|
pairs line64B 0.0081 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T19:55:17Z] rows: prog=devnet3 id=0xfce15bf61030be57 attempt=0 sites=16 plant=none live=false hashes=100000000 shard=9/16
|
||||||
|
[2026-10-07T19:58:31Z] done 195313 units (6250016 lane-hashes) in 193.7s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [0, 2, 1, 0, 1, 1, 1, 0, 2, 1, 1, 0, 2, 0, 1, 1]
|
||||||
|
== PER HASH (128 loads) prog=devnet3 plant=none live=false n=6250016
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.982 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.926 124 126 126 125 | 127.926 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet3 plant=none live=false n=195313
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.853 4053 4062 4059 4053 | 4076.886 4057 4062 4059 4057 | 4080.055 4060
|
||||||
|
rows8KiB 4020.344 3980 3993 3987 3981 | 4020.328 3981 3993 3987 3983 | 4032.665 3995
|
||||||
|
lines64B 4095.399 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
items64B 4095.399 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.500 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2551 4
|
||||||
|
pairs row8KiB 1.0022 8
|
||||||
|
pairs line64B 0.0083 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T18:51:07Z] rows: prog=devnet id=0xa785001687d8688a attempt=1 sites=16 plant=none live=false hashes=10000000 shard=0/1
|
||||||
|
[2026-10-07T18:57:24Z] done 312500 units (10000000 lane-hashes) in 376.9s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [2, 2, 2, 1, 0, 0, 0, 0, 2, 0, 2, 0, 2, 1, 0, 2]
|
||||||
|
== PER HASH (128 loads) prog=devnet plant=none live=false n=10000000
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.981 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.924 124 126 126 125 | 127.924 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 126
|
||||||
|
items64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 126
|
||||||
|
== PER UNIT (4096 loads) prog=devnet plant=none live=false n=312500
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.401 4054 4062 4058 4055 | 4076.387 4054 4062 4058 4055 | 4080.051 4060
|
||||||
|
rows8KiB 4018.434 3978 3991 3985 3981 | 4018.409 3974 3991 3985 3978 | 4032.667 3994
|
||||||
|
lines64B 4095.384 4089 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.499 4090
|
||||||
|
items64B 4095.384 4089 4092 4091 4090 | 4095.385 4089 4092 4091 4090 | 4095.499 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2973 5
|
||||||
|
pairs row8KiB 1.1827 9
|
||||||
|
pairs line64B 0.0093 2
|
||||||
|
|
@ -0,0 +1,21 @@
|
||||||
|
[2026-10-07T18:51:08Z] rows: prog=devnet id=0xa785001687d8688a attempt=1 sites=16 plant=none live=true hashes=100000 shard=0/1
|
||||||
|
[2026-10-07T18:51:08Z] building the memory-hard day cache (day 20730)...
|
||||||
|
[2026-10-07T18:51:57Z] done 3125 units (100000 lane-hashes) in 48.6s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [2, 2, 2, 1, 0, 0, 0, 0, 2, 0, 2, 0, 2, 1, 0, 2]
|
||||||
|
== PER HASH (128 loads) prog=devnet plant=none live=true n=100000
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 126 127 126 126 | 127.982 126 127 126 126 | 127.984 126
|
||||||
|
rows8KiB 127.926 125 126 126 125 | 127.924 124 126 126 125 | 127.937 125
|
||||||
|
lines64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet plant=none live=true n=3125
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.469 4056 4061 4056 4056 | 4076.435 4058 4062 4058 4058 | 4080.081 4065
|
||||||
|
rows8KiB 4018.703 3983 3989 3983 3983 | 4018.301 3986 3990 3986 3986 | 4032.703 4003
|
||||||
|
lines64B 4095.400 4092 4092 4092 4092 | 4095.391 4091 4092 4091 4091 | 4095.477 4092
|
||||||
|
items64B 4095.400 4092 4092 4092 4092 | 4095.391 4091 4092 4091 4091 | 4095.477 4092
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2909 3
|
||||||
|
pairs row8KiB 1.1731 6
|
||||||
|
pairs line64B 0.0118 1
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T18:51:08Z] rows: prog=devnet id=0xa785001687d8688a attempt=1 sites=16 plant=const-site live=false hashes=1000000 shard=0/1
|
||||||
|
[2026-10-07T18:51:46Z] done 31250 units (1000000 lane-hashes) in 38.9s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [2, 2, 2, 1, 0, 0, 0, 0, 2, 0, 2, 0, 2, 1, 0, 2]
|
||||||
|
== PER HASH (128 loads) prog=devnet plant=const-site live=false n=1000000
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 126 127 126 126 | 127.981 126 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.925 124 126 126 125 | 127.924 124 126 126 125 | 127.938 125
|
||||||
|
lines64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet plant=const-site live=false n=31250
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 3828.528 3809 3813 3811 3809 | 4076.408 4056 4062 4059 4056 | 4080.037 4062
|
||||||
|
rows8KiB 3776.783 3739 3750 3743 3739 | 4018.427 3984 3990 3986 3984 | 4032.599 3997
|
||||||
|
lines64B 3845.497 3836 3839 3837 3836 | 4095.389 4090 4092 4091 4090 | 4095.492 4090
|
||||||
|
items64B 3845.497 3836 3839 3837 3836 | 4095.389 4090 4092 4091 4090 | 4095.492 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 3970.4153 3985
|
||||||
|
pairs row8KiB 3971.1977 3987
|
||||||
|
pairs line64B 3970.1610 3984
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T18:51:07Z] rows: prog=devnet id=0xa785001687d8688a attempt=1 sites=16 plant=tiny-window live=false hashes=1000000 shard=0/1
|
||||||
|
[2026-10-07T18:51:46Z] done 31250 units (1000000 lane-hashes) in 38.6s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [2, 2, 2, 1, 0, 0, 0, 0, 2, 0, 2, 0, 2, 1, 0, 2]
|
||||||
|
== PER HASH (128 loads) prog=devnet plant=tiny-window live=false n=1000000
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.726 123 125 124 124 | 127.981 126 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.653 122 125 124 123 | 127.924 124 126 126 125 | 127.938 125
|
||||||
|
lines64B 127.749 123 125 125 124 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 127.749 123 125 125 124 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet plant=tiny-window live=false n=31250
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4062.717 4039 4044 4041 4039 | 4076.408 4056 4062 4059 4056 | 4080.037 4062
|
||||||
|
rows8KiB 3988.338 3947 3957 3952 3947 | 4018.427 3984 3990 3986 3984 | 4032.599 3997
|
||||||
|
lines64B 4087.210 4069 4077 4074 4069 | 4095.389 4090 4092 4091 4090 | 4095.492 4090
|
||||||
|
items64B 4087.210 4069 4077 4074 4069 | 4095.389 4090 4092 4091 4090 | 4095.492 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.4853 6
|
||||||
|
pairs row8KiB 1.9393 9
|
||||||
|
pairs line64B 0.0152 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T18:51:08Z] rows: prog=devnet3 id=0xfce15bf61030be57 attempt=0 sites=16 plant=none live=false hashes=10000000 shard=0/1
|
||||||
|
[2026-10-07T18:57:36Z] done 312500 units (10000000 lane-hashes) in 388.7s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [0, 2, 1, 0, 1, 1, 1, 0, 2, 1, 1, 0, 2, 0, 1, 1]
|
||||||
|
== PER HASH (128 loads) prog=devnet3 plant=none live=false n=10000000
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.982 125 127 126 126 | 127.981 125 127 126 126 | 127.984 125
|
||||||
|
rows8KiB 127.926 123 126 126 125 | 127.926 124 126 126 125 | 127.938 124
|
||||||
|
lines64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 126
|
||||||
|
items64B 127.999 126 128 127 127 | 127.999 126 128 127 127 | 128.000 126
|
||||||
|
== PER UNIT (4096 loads) prog=devnet3 plant=none live=false n=312500
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.860 4055 4062 4059 4058 | 4076.883 4053 4062 4059 4055 | 4080.051 4060
|
||||||
|
rows8KiB 4020.301 3976 3993 3987 3984 | 4020.334 3978 3993 3987 3983 | 4032.667 3994
|
||||||
|
lines64B 4095.400 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.499 4090
|
||||||
|
items64B 4095.400 4090 4092 4091 4090 | 4095.404 4088 4092 4091 4090 | 4095.499 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2539 5
|
||||||
|
pairs row8KiB 1.0013 8
|
||||||
|
pairs line64B 0.0078 2
|
||||||
|
|
@ -0,0 +1,21 @@
|
||||||
|
[2026-10-07T18:51:07Z] rows: prog=devnet3 id=0xfce15bf61030be57 attempt=0 sites=16 plant=none live=true hashes=100000 shard=0/1
|
||||||
|
[2026-10-07T18:51:07Z] building the memory-hard day cache (day 20733)...
|
||||||
|
[2026-10-07T18:51:57Z] done 3125 units (100000 lane-hashes) in 49.1s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [0, 2, 1, 0, 1, 1, 1, 0, 2, 1, 1, 0, 2, 0, 1, 1]
|
||||||
|
== PER HASH (128 loads) prog=devnet3 plant=none live=true n=100000
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.982 126 127 126 126 | 127.982 126 127 126 126 | 127.984 126
|
||||||
|
rows8KiB 127.927 125 126 126 125 | 127.928 125 126 126 125 | 127.937 125
|
||||||
|
lines64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=devnet3 plant=none live=true n=3125
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.780 4061 4062 4061 4061 | 4076.919 4057 4061 4057 4057 | 4080.081 4065
|
||||||
|
rows8KiB 4020.383 3989 3993 3989 3989 | 4020.403 3989 3990 3989 3989 | 4032.703 4003
|
||||||
|
lines64B 4095.408 4091 4092 4091 4091 | 4095.397 4091 4092 4091 4091 | 4095.477 4092
|
||||||
|
items64B 4095.408 4091 4092 4091 4091 | 4095.397 4091 4092 4091 4091 | 4095.477 4092
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2506 3
|
||||||
|
pairs row8KiB 1.0061 6
|
||||||
|
pairs line64B 0.0077 1
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T18:51:14Z] rows: prog=drawn:0 id=0x5d7cc2b09fc6922a attempt=0 sites=16 plant=none live=false hashes=2000000 shard=0/1
|
||||||
|
[2026-10-07T18:52:48Z] done 62500 units (2000000 lane-hashes) in 93.9s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [2, 2, 0, 0, 2, 1, 1, 1, 2, 0, 2, 1, 1, 1, 1, 2]
|
||||||
|
== PER HASH (128 loads) prog=drawn:0 plant=none live=false n=2000000
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.855 123 126 125 125 | 127.980 125 127 126 126 | 127.985 125
|
||||||
|
rows8KiB 127.794 123 125 125 124 | 127.919 124 126 126 125 | 127.938 125
|
||||||
|
lines64B 127.875 124 126 125 125 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 127.875 124 126 125 125 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=drawn:0 plant=none live=false n=62500
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4071.083 4048 4054 4050 4048 | 4075.007 4054 4059 4057 4054 | 4080.044 4060
|
||||||
|
rows8KiB 4009.283 3968 3980 3973 3968 | 4013.028 3972 3984 3977 3972 | 4032.651 3997
|
||||||
|
lines64B 4091.350 4081 4083 4082 4081 | 4095.349 4090 4092 4091 4090 | 4095.495 4090
|
||||||
|
items64B 4091.350 4081 4083 4082 4081 | 4095.349 4090 4092 4091 4090 | 4095.495 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.3105 4
|
||||||
|
pairs row8KiB 1.2392 8
|
||||||
|
pairs line64B 0.0094 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T18:51:14Z] rows: prog=drawn:1 id=0x2c81972d33e22ad2 attempt=1 sites=16 plant=none live=false hashes=2000000 shard=0/1
|
||||||
|
[2026-10-07T18:52:50Z] done 62500 units (2000000 lane-hashes) in 96.1s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [1, 2, 1, 2, 1, 2, 0, 2, 2, 1, 1, 2, 1, 1, 0, 0]
|
||||||
|
== PER HASH (128 loads) prog=drawn:1 plant=none live=false n=2000000
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.984 125 127 126 126 | 127.984 126 127 126 126 | 127.985 125
|
||||||
|
rows8KiB 127.937 125 126 126 125 | 127.937 123 126 126 125 | 127.938 125
|
||||||
|
lines64B 128.000 127 128 127 127 | 128.000 127 128 127 127 | 128.000 127
|
||||||
|
items64B 128.000 127 128 127 127 | 128.000 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=drawn:1 plant=none live=false n=62500
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4079.514 4058 4066 4062 4058 | 4079.484 4061 4066 4062 4061 | 4080.044 4060
|
||||||
|
rows8KiB 4030.569 3996 4005 4000 3996 | 4030.494 3991 4005 3999 3991 | 4032.651 3997
|
||||||
|
lines64B 4095.482 4091 4092 4091 4091 | 4095.487 4091 4092 4092 4091 | 4095.495 4090
|
||||||
|
items64B 4095.482 4091 4092 4091 4091 | 4095.487 4091 4092 4092 4091 | 4095.495 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.3082 5
|
||||||
|
pairs row8KiB 1.2380 9
|
||||||
|
pairs line64B 0.0091 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T18:51:14Z] rows: prog=drawn:2 id=0xd3fead516b1ab00c attempt=0 sites=16 plant=none live=false hashes=2000000 shard=0/1
|
||||||
|
[2026-10-07T18:52:48Z] done 62500 units (2000000 lane-hashes) in 94.3s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [1, 0, 2, 2, 2, 0, 1, 0, 1, 1, 1, 1, 2, 2, 1, 1]
|
||||||
|
== PER HASH (128 loads) prog=drawn:2 plant=none live=false n=2000000
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.983 126 127 126 126 | 127.983 125 127 126 126 | 127.985 125
|
||||||
|
rows8KiB 127.930 124 126 126 125 | 127.930 124 126 126 125 | 127.938 125
|
||||||
|
lines64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=drawn:2 plant=none live=false n=62500
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4077.901 4051 4063 4061 4051 | 4077.882 4049 4064 4061 4049 | 4080.044 4060
|
||||||
|
rows8KiB 4024.255 3989 3997 3991 3989 | 4024.238 3985 3997 3991 3985 | 4032.651 3997
|
||||||
|
lines64B 4095.428 4090 4092 4091 4090 | 4095.434 4090 4092 4091 4090 | 4095.495 4090
|
||||||
|
items64B 4095.428 4090 4092 4091 4090 | 4095.434 4090 4092 4091 4090 | 4095.495 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2938 5
|
||||||
|
pairs row8KiB 1.1842 8
|
||||||
|
pairs line64B 0.0094 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T18:51:14Z] rows: prog=drawn:3 id=0xb7350750beb0ded9 attempt=1 sites=16 plant=none live=false hashes=2000000 shard=0/1
|
||||||
|
[2026-10-07T18:52:47Z] done 62500 units (2000000 lane-hashes) in 93.5s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [1, 0, 1, 2, 1, 2, 1, 2, 2, 2, 0, 2, 2, 2, 0, 2]
|
||||||
|
== PER HASH (128 loads) prog=drawn:3 plant=none live=false n=2000000
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.981 125 127 126 126 | 127.981 125 127 126 126 | 127.985 125
|
||||||
|
rows8KiB 127.924 124 126 126 125 | 127.925 124 126 126 125 | 127.938 125
|
||||||
|
lines64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=drawn:3 plant=none live=false n=62500
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4076.391 4055 4062 4059 4055 | 4076.382 4054 4061 4058 4054 | 4080.044 4060
|
||||||
|
rows8KiB 4018.374 3980 3991 3986 3980 | 4018.383 3974 3990 3984 3974 | 4032.651 3997
|
||||||
|
lines64B 4095.384 4089 4092 4091 4089 | 4095.389 4090 4092 4091 4090 | 4095.495 4090
|
||||||
|
items64B 4095.384 4089 4092 4091 4089 | 4095.389 4090 4092 4091 4090 | 4095.495 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.3589 4
|
||||||
|
pairs row8KiB 1.4340 9
|
||||||
|
pairs line64B 0.0120 3
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T18:51:14Z] rows: prog=drawn:4 id=0x180a596485ad0153 attempt=2 sites=16 plant=none live=false hashes=2000000 shard=0/1
|
||||||
|
[2026-10-07T18:52:54Z] done 62500 units (2000000 lane-hashes) in 99.4s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [1, 0, 0, 0, 0, 1, 0, 0, 1, 0, 1, 0, 1, 2, 2, 2]
|
||||||
|
== PER HASH (128 loads) prog=drawn:4 plant=none live=false n=2000000
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.983 125 127 126 126 | 127.983 125 127 126 126 | 127.985 125
|
||||||
|
rows8KiB 127.932 124 126 126 125 | 127.932 124 126 126 125 | 127.938 125
|
||||||
|
lines64B 127.999 127 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 127 128 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=drawn:4 plant=none live=false n=62500
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4078.476 4059 4065 4061 4059 | 4078.427 4059 4064 4061 4059 | 4080.044 4060
|
||||||
|
rows8KiB 4026.434 3992 4000 3995 3992 | 4026.405 3982 4000 3995 3982 | 4032.651 3997
|
||||||
|
lines64B 4095.450 4090 4092 4091 4090 | 4095.452 4090 4092 4091 4090 | 4095.495 4090
|
||||||
|
items64B 4095.450 4090 4092 4091 4090 | 4095.452 4090 4092 4091 4090 | 4095.495 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2292 4
|
||||||
|
pairs row8KiB 0.9131 7
|
||||||
|
pairs line64B 0.0074 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T18:51:14Z] rows: prog=drawn:5 id=0x1a77e8de160b6db2 attempt=1 sites=16 plant=none live=false hashes=2000000 shard=0/1
|
||||||
|
[2026-10-07T18:52:43Z] done 62500 units (2000000 lane-hashes) in 89.3s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [2, 2, 0, 1, 1, 0, 2, 0, 1, 1, 1, 0, 2, 2, 0, 0]
|
||||||
|
== PER HASH (128 loads) prog=drawn:5 plant=none live=false n=2000000
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.984 125 127 126 126 | 127.984 126 127 126 126 | 127.985 125
|
||||||
|
rows8KiB 127.937 125 126 126 125 | 127.937 124 126 126 125 | 127.938 125
|
||||||
|
lines64B 128.000 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 128.000 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=drawn:5 plant=none live=false n=62500
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4079.672 4062 4066 4063 4062 | 4079.680 4062 4066 4063 4062 | 4080.044 4060
|
||||||
|
rows8KiB 4031.158 3994 4006 4001 3994 | 4031.257 3998 4006 4001 3998 | 4032.651 3997
|
||||||
|
lines64B 4095.491 4090 4092 4091 4090 | 4095.492 4090 4092 4091 4090 | 4095.495 4090
|
||||||
|
items64B 4095.491 4090 4092 4091 4090 | 4095.492 4090 4092 4091 4090 | 4095.495 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2720 4
|
||||||
|
pairs row8KiB 1.0893 8
|
||||||
|
pairs line64B 0.0084 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T18:51:13Z] rows: prog=drawn:6 id=0x50b76694674c0d97 attempt=6 sites=16 plant=none live=false hashes=2000000 shard=0/1
|
||||||
|
[2026-10-07T18:52:50Z] done 62500 units (2000000 lane-hashes) in 96.9s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [1, 1, 2, 0, 2, 0, 0, 1, 1, 1, 0, 1, 0, 1, 0, 2]
|
||||||
|
== PER HASH (128 loads) prog=drawn:6 plant=none live=false n=2000000
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.984 125 127 126 126 | 127.984 126 127 126 126 | 127.985 125
|
||||||
|
rows8KiB 127.934 124 126 126 125 | 127.934 124 126 126 125 | 127.938 125
|
||||||
|
lines64B 128.000 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
items64B 128.000 127 128 127 127 | 127.999 127 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=drawn:6 plant=none live=false n=62500
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4078.894 4059 4065 4062 4059 | 4078.925 4059 4065 4062 4059 | 4080.044 4060
|
||||||
|
rows8KiB 4028.227 3990 4002 3997 3990 | 4028.316 3989 4002 3997 3989 | 4032.651 3997
|
||||||
|
lines64B 4095.468 4091 4092 4091 4091 | 4095.471 4090 4092 4092 4090 | 4095.495 4090
|
||||||
|
items64B 4095.468 4091 4092 4091 4091 | 4095.471 4090 4092 4092 4090 | 4095.495 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.2440 4
|
||||||
|
pairs row8KiB 0.9740 7
|
||||||
|
pairs line64B 0.0073 2
|
||||||
|
|
@ -0,0 +1,20 @@
|
||||||
|
[2026-10-07T18:51:14Z] rows: prog=drawn:7 id=0xd694dd0bafb2a724 attempt=0 sites=16 plant=none live=false hashes=2000000 shard=0/1
|
||||||
|
[2026-10-07T18:52:32Z] done 62500 units (2000000 lane-hashes) in 78.6s
|
||||||
|
sites win (0 = whole dataset, 1 = half, 2 = quarter): [2, 2, 1, 1, 2, 2, 1, 2, 1, 0, 0, 1, 2, 2, 1, 0]
|
||||||
|
== PER HASH (128 loads) prog=drawn:7 plant=none live=false n=2000000
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 127.983 125 127 126 126 | 127.984 126 127 126 126 | 127.985 125
|
||||||
|
rows8KiB 127.933 124 126 126 125 | 127.934 125 126 126 125 | 127.938 125
|
||||||
|
lines64B 127.999 126 127 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
items64B 127.999 126 127 127 127 | 127.999 126 128 127 127 | 128.000 127
|
||||||
|
== PER UNIT (4096 loads) prog=drawn:7 plant=none live=false n=62500
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min
|
||||||
|
rows2KiB 4078.851 4059 4065 4061 4059 | 4078.875 4058 4064 4062 4058 | 4080.044 4060
|
||||||
|
rows8KiB 4028.108 3995 4002 3997 3995 | 4028.053 3988 4002 3997 3988 | 4032.651 3997
|
||||||
|
lines64B 4095.437 4091 4092 4091 4091 | 4095.464 4091 4092 4091 4091 | 4095.495 4090
|
||||||
|
items64B 4095.437 4091 4092 4091 4091 | 4095.464 4091 4092 4091 4091 | 4095.495 4090
|
||||||
|
== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)
|
||||||
|
metric mean max (expected uniform: 2KiB 0.1211, 8KiB 0.4844, 64B 0.00378 per unit)
|
||||||
|
pairs row2KiB 0.3300 6
|
||||||
|
pairs row8KiB 1.3000 9
|
||||||
|
pairs line64B 0.0115 2
|
||||||
|
|
@ -0,0 +1,14 @@
|
||||||
|
[2026-10-07T18:44:09Z] rows: prog=devnet id=0xa785001687d8688a attempt=1 sites=16 plant=none live=false hashes=3200 shard=0/1
|
||||||
|
[2026-10-07T18:44:09Z] done 100 units (3200 lane-hashes) in 0.1s
|
||||||
|
== PER HASH (128 loads) prog=devnet plant=none live=false
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | baseline mean min q1e-3
|
||||||
|
rows2KiB 127.977 126 127 126 126 | 127.983 127 127
|
||||||
|
rows8KiB 127.916 126 126 126 126 | 127.934 126 126
|
||||||
|
lines64B 128.000 128 128 128 128 | 128.000 128 128
|
||||||
|
items64B 128.000 128 128 128 128 | 128.000 128 128
|
||||||
|
== PER UNIT (4096 loads) prog=devnet plant=none live=false
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | baseline mean min q1e-3
|
||||||
|
rows2KiB 4076.560 4064 4064 4064 4064 | 4080.000 4069 4069
|
||||||
|
rows8KiB 4018.980 3998 3998 3998 3998 | 4032.680 4018 4018
|
||||||
|
lines64B 4095.500 4093 4093 4093 4093 | 4095.450 4093 4093
|
||||||
|
items64B 4095.500 4093 4093 4093 4093 | 4095.450 4093 4093
|
||||||
|
|
@ -0,0 +1,14 @@
|
||||||
|
[2026-10-07T18:44:13Z] rows: prog=devnet id=0xa785001687d8688a attempt=1 sites=16 plant=const-site live=false hashes=3200 shard=0/1
|
||||||
|
[2026-10-07T18:44:13Z] done 100 units (3200 lane-hashes) in 0.1s
|
||||||
|
== PER HASH (128 loads) prog=devnet plant=const-site live=false
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | baseline mean min q1e-3
|
||||||
|
rows2KiB 127.981 126 127 126 126 | 127.983 127 127
|
||||||
|
rows8KiB 127.927 126 126 126 126 | 127.934 126 126
|
||||||
|
lines64B 127.999 127 127 127 127 | 128.000 128 128
|
||||||
|
items64B 127.999 127 127 127 127 | 128.000 128 128
|
||||||
|
== PER UNIT (4096 loads) prog=devnet plant=const-site live=false
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | baseline mean min q1e-3
|
||||||
|
rows2KiB 3828.780 3814 3814 3814 3814 | 4080.000 4069 4069
|
||||||
|
rows8KiB 3777.230 3762 3762 3762 3762 | 4032.680 4018 4018
|
||||||
|
lines64B 3845.530 3839 3839 3839 3839 | 4095.450 4093 4093
|
||||||
|
items64B 3845.530 3839 3839 3839 3839 | 4095.450 4093 4093
|
||||||
|
|
@ -0,0 +1,14 @@
|
||||||
|
[2026-10-07T18:44:17Z] rows: prog=devnet id=0xa785001687d8688a attempt=1 sites=16 plant=tiny-window live=false hashes=3200 shard=0/1
|
||||||
|
[2026-10-07T18:44:17Z] done 100 units (3200 lane-hashes) in 0.2s
|
||||||
|
== PER HASH (128 loads) prog=devnet plant=tiny-window live=false
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | baseline mean min q1e-3
|
||||||
|
rows2KiB 127.724 125 125 125 125 | 127.983 127 127
|
||||||
|
rows8KiB 127.645 124 125 124 124 | 127.934 126 126
|
||||||
|
lines64B 127.742 125 125 125 125 | 128.000 128 128
|
||||||
|
items64B 127.742 125 125 125 125 | 128.000 128 128
|
||||||
|
== PER UNIT (4096 loads) prog=devnet plant=tiny-window live=false
|
||||||
|
metric mean min q1e-3 q1e-4 q1e-5 | baseline mean min q1e-3
|
||||||
|
rows2KiB 4062.790 4051 4051 4051 4051 | 4080.000 4069 4069
|
||||||
|
rows8KiB 3987.810 3968 3968 3968 3968 | 4032.680 4018 4018
|
||||||
|
lines64B 4086.890 4081 4081 4081 4081 | 4095.450 4093 4093
|
||||||
|
items64B 4086.890 4081 4081 4081 4081 | 4095.450 4093 4093
|
||||||
139
docs/analysis/cryptanalysis/report-acceptance-rule-2.md
Normal file
139
docs/analysis/cryptanalysis/report-acceptance-rule-2.md
Normal file
|
|
@ -0,0 +1,139 @@
|
||||||
|
# Report: header grinding for locality (acceptance rule and memory access, lane adv-accept-2)
|
||||||
|
|
||||||
|
internal adversarial pass, not an independent review
|
||||||
|
|
||||||
|
## Header
|
||||||
|
|
||||||
|
| Item | Value |
|
||||||
|
|---|---|
|
||||||
|
| Target commit | 017e70376489251e18564c0abce7e466e606c8b3 (class v4 sub-version 3, object byte 7) |
|
||||||
|
| Base commit of this branch | 04c4d9bc (build/master merged at 19:46 UK). `git diff --quiet 017e7037... HEAD -- igneum-pow` prints nothing: igneum-pow is byte-identical to the frozen object, verified at 7a7caa34 and again at 04c4d9bc |
|
||||||
|
| Crate built | igneum-pow at the frozen commit (this worktree's copy). Harness tools/attack/adv-accept-2 (igneum-pow by path; mirrors verify.rs; the real draw Epoch::chain_program(ProgramClass::V4) and the real address map verify::load_index come from the library) |
|
||||||
|
| Binary sha256 (the sweeps) | 20e0000eb927f094c7d0918cd3b7caea22ab4d2a18be7ca6d123075de22a63f8 (rows sweeps); 0bc825b5cbd4df2d6b866b8d2396d2bd38d67344b330ad054361fb5c70f0ffb1 (prefix, diffuse); 4421f760b37fb0eb9bf906457c8e1fcf97545335224831efba104040d3b73364 (repeats); da8c5e00c4939a7b6a533518f53c31572a0f9fd5d94509317b29258d71779a3d (dump); c0164d66aa556707cf4efc02cba6a8f4e4e4bf10080bfc706c4d87594fb0520f (export) |
|
||||||
|
| Boxes | build-1 (real programs, plants, live confirmation, probes); build-2 (eight drawn programs). Every run single-thread, nice 10, cores 8 to 95. GPU: RunPod RTX A6000 48 GB (driver 570.195.03, CUDA 12.8), rented by the fleet lane 20:06 to about 21:00 UK |
|
||||||
|
| Logs | docs/analysis/cryptanalysis/logs/adv-accept-2/ (copied from /srv/builds/_adv-adv-accept-2/ on each box and /root/fleet/out/ on the pod) |
|
||||||
|
| Box-hours | about 9.0 core-hours in all (the 1e8 pass 8.3: 16 cores x 1,878 s; the first pass 0.7) (first pass: build-1 0.45: two 1e7 sweeps of 377 and 389 s, two plants of 39 s, two live runs of 49 s, probes and exports about 10 min; build-2 0.21: eight 2e6 sweeps of 89 to 99 s). Seven 8 s builds. Pod: about 0.3 h of a 3 h rental |
|
||||||
|
| Clock | UK time from `TZ=Europe/London date` throughout |
|
||||||
|
|
||||||
|
Rule-change ledger (honest box-hours): 19:3x UK no SIGSTOP yield, taskset cores 8 to 95, logs outside the worktree mirror; 19:55 UK one sweep per box under sweep.lock; 20:2x UK kill hand-started sweeps and start nothing until `lease pool` (live 20:22 UK). At 20:23 UK nothing of mine was running on either box (pid check: 0 processes); nothing was killed, lost or re-queued; no box run was started after that.
|
||||||
|
|
||||||
|
## Draw-path validation (passed)
|
||||||
|
|
||||||
|
| Program | Through Epoch::chain_program(V4) | Pack | Match |
|
||||||
|
|---|---|---|---|
|
||||||
|
| Shared devnet epoch 0 | id 0xa785001687d8688a, attempt 1, generator 4, sites 1 4 6 10 12 20 27 30 35 40 41 43 45 52 53 54 | v4-devnet-epoch0/program.json | yes |
|
||||||
|
| Devnet 3 epoch 0 | id 0xfce15bf61030be57, attempt 0, generator 4, sites 3 8 14 15 20 26 28 35 40 43 47 49 52 53 61 62 | v4-devnet3-epoch0 (zip sha256 e025750f... verified on build-1) | yes |
|
||||||
|
|
||||||
|
Log: logs/adv-accept-2/draw-check.log.
|
||||||
|
|
||||||
|
## Status board
|
||||||
|
|
||||||
|
| Q | Method | Known-failed shape (fired?) | Gate | Result | Status |
|
||||||
|
|---|---|---|---|---|---|
|
||||||
|
| Q1 | Header to address: read bind.rs and spec 1.6; one-bit flips of H over 4,096 header pairs, fraction of the 4,096 unit addresses that change; dataflow count of header-predictable load sites | header-blind plant (init = program seed) must read 0: read 0.0000 | real path near 1.0 | 1.0000 of addresses change on both real programs; 2 to 4 of 16 sites in iteration 0 are predictable from the header alone, 0 in iterations 1 to 7 | PASS (bound) |
|
||||||
|
| Q2 | Distinct 2 KiB rows, 8 KiB rows, 64 B lines, items per hash and per unit over 1e7 hashes on each real program (closed form), 1e5 on the live dataset, 2e6 on eight drawn programs; tails 1e-3, 1e-4, 1e-5 against a windowed random baseline | const-site and tiny-window plants must collapse the counts at once: const-site 4095.4 to 3845.5 items per unit, tiny-window 4018 to 3988 rows8KiB | best tail inside the baseline's own spread | every clean program sits on the windowed baseline at mean, min and all three tails, per hash and per unit; the live dataset agrees with the closed form | PASS (bound) |
|
||||||
|
| Q3 | Price: search cost in hashes per found group against loads saved; one card point on the A6000 | a 10 percent saving at the 1e-5 tail nets far under 1 percent: 1.0e-5x | no grind nets over 1 percent | the best 1 in 15 header groups run 0.09 percent faster on the card and cost 15 full hashes each: net about 0.07x; the 1e-5 tail saves 37 of 4,096 rows (0.9 percent) at 1e5 hashes each: net 1e-5x | PASS (bound); card point MEASURED |
|
||||||
|
| Q4 | Does (c)/(c'') bound per-hash or only per-program locality: read accept.rs; tiny-window plant through the rule; the per-program repeat drawn:0 shows | tiny-window plant: the rule's LaneConstantSite and ratio tests are evaluated on the program, not the header, so a header cannot move them (confirmed: no header changes a clean tail) | rule rejects planted clustering; headers do not move a clean tail | the rule bounds the PROGRAM (mean distinct over 120 of 128; per-site ratio 0.98 on 2^20 fixed evaluations); it has no per-header term and needs none: the construction (every load after the first 2 to 4 depends on loaded data) bounds the per-hash side. One accepted drawn program repeats a word in 1.55 percent of hashes per site pair (FINDING, 0.1 percent of loads) | PASS, one small FINDING |
|
||||||
|
| GPU | Dependent-read throughput of 2,048 units x 200 rounds per table, three repetitions | const-site plant must run faster: +6.3 percent (its 31 duplicate lanes per site coalesce) | the best ground groups against random | random 1.6130e6 units/s; best ground 1.6145e6 (+0.09 percent); tiny-window 1.6211e6 (+0.5); const-site 1.7139e6 (+6.3) | MEASURED |
|
||||||
|
|
||||||
|
## Q1. What of the header reaches the load addresses
|
||||||
|
|
||||||
|
From bind.rs and spec 1.6: the miner's header bytes (coinbase, extra nonce, timestamp) and the high 32 bits of the nonce reach the hash only as `I = seed_words_from_bytes("igneum-block/" || H || nonce_hi_le32)`: FNV-1a 64 over 49 bytes under four salted bases, each finalised by `h ^= h >> 33; h *= 0xff51afd7ed558ccd; h ^= h >> 33`. Every header byte enters all eight words of I. Register init is `r[i] = splitmix32((n XOR I[i]) + 0x9e3779b9 (i+1)) XOR I[(i+1) & 7]` per lane. The program (from the epoch seed) and the dataset (from the day) do not depend on the header. Load address (verify::load_index, spec 1.13.1): `y = rotl(x M, R)`, masked into the site's window (the dataset, a half or a quarter) and offset; physical address 4 idx bytes under any era interleave.
|
||||||
|
|
||||||
|
Measured: a one-bit flip of H changes 1.0000 of the 4,096 unit addresses on both real programs (4,096 pairs each; logs diffuse-devnet-4096.log, diffuse-devnet3-4096.log). The header-blind plant (I = program seed, the pack vectors' form) reads 0.0000 (diffuse-devnet-headerblind-1024.log), so the probe distinguishes.
|
||||||
|
|
||||||
|
The predictable prefix (prefix.log): a load site whose source has no dataflow path from an earlier load can be addressed from (I, n) with ALU work alone. Iteration 0 has 4 such sites on each real program (devnet instructions 1, 10, 12, 30; Devnet 3 instructions 3, 8, 14, 15) and 2 to 3 on the eight drawn ones; iterations 1 to 7 have none. So at most 4 x 32 = 128 of a unit's 4,096 loads (3.1 percent) can be chosen by header search without executing memory; every other address needs the loads before it.
|
||||||
|
|
||||||
|
Consequence: a header search can select on at most 3.1 percent of a unit's loads for free; everything else costs the full hash it is trying to save.
|
||||||
|
|
||||||
|
## Q2. The locality distributions
|
||||||
|
|
||||||
|
Per hash (128 loads of one lane) and per unit (4,096 loads), distinct counts, dataset 2^28 words, rows modelled as contiguous 2 KiB (512 words) and 8 KiB (2,048 words), lines 64 B (16 words), items 64 B. "Windowed baseline": uniform y masked into the program's own site windows, the same sample count. Lower is more clustered; q1e-k is the lowest value with at most that fraction below it.
|
||||||
|
|
||||||
|
Real programs, closed form, 1e7 hashes each (312,500 units); logs rows-devnet-1e7.log, rows-devnet3-1e7.log:
|
||||||
|
|
||||||
|
| Program | Scope | Metric | mean | min | q1e-3 | q1e-4 | q1e-5 | baseline mean | baseline min | baseline q1e-5 |
|
||||||
|
|---|---|---|---|---|---|---|---|---|---|---|
|
||||||
|
| devnet | per hash | rows8KiB | 127.924 | 124 | 126 | 126 | 125 | 127.924 | 124 | 125 |
|
||||||
|
| devnet | per hash | items | 127.999 | 126 | 128 | 127 | 127 | 127.999 | 126 | 127 |
|
||||||
|
| devnet | per unit | rows2KiB | 4076.40 | 4054 | 4062 | 4058 | 4055 | 4076.39 | 4054 | 4055 |
|
||||||
|
| devnet | per unit | rows8KiB | 4018.43 | 3978 | 3991 | 3985 | 3981 | 4018.41 | 3974 | 3978 |
|
||||||
|
| devnet | per unit | lines/items | 4095.38 | 4089 | 4092 | 4091 | 4090 | 4095.39 | 4089 | 4090 |
|
||||||
|
| devnet3 | per hash | rows8KiB | 127.926 | 123 | 126 | 126 | 125 | 127.926 | 124 | 125 |
|
||||||
|
| devnet3 | per unit | rows2KiB | 4076.86 | 4055 | 4062 | 4059 | 4058 | 4076.88 | 4053 | 4055 |
|
||||||
|
| devnet3 | per unit | rows8KiB | 4020.30 | 3976 | 3993 | 3987 | 3984 | 4020.33 | 3978 | 3983 |
|
||||||
|
| devnet3 | per unit | lines/items | 4095.40 | 4090 | 4092 | 4091 | 4090 | 4095.40 | 4088 | 4090 |
|
||||||
|
|
||||||
|
The plain uniform baseline (no windows) reads 4080.05 rows2KiB and 4032.67 rows8KiB per unit: the 4 to 14 row deficit of the real programs against it is the era's quarter and half windows (spec 1.13.1), a per-program property, present on every header and in the windowed baseline.
|
||||||
|
|
||||||
|
Live memory-hard dataset, 1e5 hashes each (logs rows-devnet-live-1e5.log, rows-devnet3-live-1e5.log): devnet per unit rows8KiB mean 4018.70, min 3983 (baseline 4018.30, 3986); devnet3 4020.38, min 3989 (baseline 4020.40, 3989); per hash identical to the closed form to three decimals. The loaded values do not change the locality distribution, as argued in the harness header (the address is computed from the source register before its own load).
|
||||||
|
|
||||||
|
Eight drawn programs, 2e6 hashes each on build-2 (logs rows-drawn0..7-2e6.log): seven sit on their windowed baseline at every quantile (per-unit rows8KiB means 4009 to 4031 against baselines 4013 to 4031, mins within 6 of the baseline's). drawn:0 (id 0x5d7cc2b09fc6922a, attempt 0, accepted) reads 127.875 items per hash against 127.999, and 4091.35 per unit against 4095.35, on every header: see Q4.
|
||||||
|
|
||||||
|
Same-instruction coalescing (lane pairs of one load site sharing a row or line, the only hits a row buffer or a coalescer sees together), per unit: devnet mean 0.297 pairs per 2 KiB row, 1.18 per 8 KiB row, 0.0093 per 64 B line (max over 312,500 units: 5, 9, 2); Devnet 3 0.254, 1.00, 0.0078 (max 5, 8, 2). The windowed expectation is of this size (quarter windows raise the uniform 0.12 / 0.48 / 0.0038). The const-site plant reads 3,970 line pairs per unit (31 x 32 / 2 x 8 = 3,968 expected): the metric fires.
|
||||||
|
|
||||||
|
Plants (logs rows-devnet-plant-const-1e6.log, rows-devnet-plant-tiny-1e6.log): const-site, items per unit 3845.50 (every header), tiny-window rows8KiB 3988.34 against 4018.43, both far outside the clean spread at the first unit.
|
||||||
|
|
||||||
|
Result: BOUND. Over 2.6e7 header-chosen hashes on ten accepted programs the most clustered unit found saves 40 of 4,096 8 KiB rows (1.0 percent), 26 of 4,096 2 KiB rows (0.6 percent) and 7 of 4,096 lines or items (0.17 percent) against the mean, and the windowed random baseline reaches the same values at the same sample size. Header choice adds no locality beyond chance.
|
||||||
|
|
||||||
|
## Q3. The price
|
||||||
|
|
||||||
|
A card bound by random 4-byte reads mines at (reads per second) / 128 hashes. A header search evaluates candidate groups; each evaluation is a full hash (128 reads per lane) except for the 2 to 4 header-predictable sites of iteration 0, which can be addressed without memory. A found group is one 32-lane unit: the address set is fixed by (program, I, g), so it mines once and the search does not amortise. Net rate against honest, with dL loads saved in the found unit and S candidates searched per find: `128 / ((128 - dL) + S 128)`.
|
||||||
|
|
||||||
|
| Tail | Rows saved per unit (8 KiB, from Q2) | S (hashes per find) | Net rate vs honest |
|
||||||
|
|---|---|---|---|
|
||||||
|
| best 1 in 15 (the exported top 2,048 of 31,250) | 17 of 4,096 (0.4 percent) | 15 | 0.066x |
|
||||||
|
| 1e-3 | 27 (0.7 percent) | 1,000 | 1.0e-3x |
|
||||||
|
| 1e-5 | 37 (0.9 percent) | 100,000 | 1.0e-5x |
|
||||||
|
| any dL below 128 per hash | | S | below 1 / S |
|
||||||
|
|
||||||
|
The free-prefix strategy (select on the 2 to 4 predictable sites only, never execute a rejected candidate): the gain is bounded by collisions among at most 128 predictable addresses, 128^2 / 2 / 2^17 = 0.06 expected 8 KiB row pairs per unit; even all 128 in one row (probability about 2^-17 x 127) saves 127 of 4,096 loads, 3.1 percent, and the realistic 1e-5 tail of a Poisson(0.06) is 4 pairs, 0.1 percent. No strategy nets 1 percent.
|
||||||
|
|
||||||
|
Card point (logs/adv-accept-2/gpu-rowbench-a6000-2048x200.log; kernel tools/attack/adv-accept-2/gpu/rowbench.cu; RTX A6000, 2,048 units x 200 rounds, three repetitions each, spread under 0.03 percent):
|
||||||
|
|
||||||
|
| Table (devnet program) | Mean distinct 8 KiB rows per unit | Unit-hashes per second | Against random |
|
||||||
|
|---|---|---|---|
|
||||||
|
| random headers | 4018.2 | 1.6130e6 | 1.000 |
|
||||||
|
| best 2,048 of 31,250 headers by fewest rows | 4001.4 | 1.6145e6 | 1.0009 |
|
||||||
|
| tiny-window plant | 3988.2 | 1.6211e6 | 1.0050 |
|
||||||
|
| const-site plant (one lane-constant site) | 3776.8 | 1.7139e6 | 1.0626 |
|
||||||
|
|
||||||
|
The card gains 0.09 percent on the best header groups, which cost 15 full hashes each to find: net 0.07x of honest. The const-site plant's 6.3 percent is the coalescer serving 31 duplicate lanes per planted site from one transaction, the shape rule (c) rejects. Caveats: the kernel reads the same 2,048 tables for 200 rounds (the 268 MB of touched sectors exceed the 6 MB L2, so cross-round hits are a few percent at most and equal for all four tables); the A6000 reads 6.6 G dependent 4-byte words per second here, about 0.38 of the RTX 5090's 17.5 G in chip-model-v3 5.1; the DRAM address mapping of the card is not the contiguous-row model, which is why the measured gain follows lines, not rows.
|
||||||
|
|
||||||
|
Result: BOUND. No header-grinding strategy nets above 1 percent of rate; the measured card point agrees.
|
||||||
|
|
||||||
|
## Q4. The acceptance rule and per-hash locality
|
||||||
|
|
||||||
|
Rule (c) interprets the program for 64 fixed units (base nonces from a seed-keyed stream, init words = the seed words) and rejects a load site that reads one address in all 32 lanes of any of those units, and a program whose distinct-address mean over the 2,048 evaluations is 120 of 128 or under. Rule (c'') rejects a site whose distinct word indices over 2^20 fixed evaluations fall under 0.98 of uniform on its window. Neither test has a header term: they bound the PROGRAM on the acceptance stream, and the miner's init words are different words. So the rule bounds per-program locality only. The per-hash side is bounded by the construction, not the rule: every load after the 2 to 4 predictable ones depends on loaded data (Q1), every address is a bijective image of a register (M odd, R a rotation), and Q2 shows no header moves a clean program off chance. The planted clustering (tiny-window, const-site) is a program property the rule's (c) tests see on their own stream.
|
||||||
|
|
||||||
|
FINDING (small, per program, public to every miner): drawn:0 (id 0x5d7cc2b09fc6922a, attempt 0, accepted by the frozen rule) reads the same word at load sites 1 and 5 (instructions 6 and 26) in 1.55 percent of hashes per iteration, 0.124 repeats per hash (logs repeats-drawn0-2e5.log, dump-drawn0.log). Mechanism: both loads read r3; the only write to r3 between them is `rotr r3 by (r1 AND 31)` at instruction 22, the identity when r1 AND 31 = 0 (1 in 32); site 1's quarter window (win 2, off 1) lies inside site 5's half window (win 1, off 0), so the two indices coincide when bit 26 of y is set (1 in 2): 1/64 per iteration per lane, 0.125 per hash, as measured. Rule (a) counts the rotate as a write; rule (a') counts rotr as freshness-preserving; rule (c) admits it because the mean stays at 127.875 of 128. Gain: a chip or card that serves the second read from the first saves 0.1 percent of loads on this program; the rule's floor of 120 admits up to 6.25 percent per program in principle. The devnet program shows 12 chance repeats in 200,000 hashes (0.00006 per hash). Header grinding cannot move this: it is the same on every header.
|
||||||
|
|
||||||
|
Result: PASS with one FINDING of 0.1 percent (a data-dependent rotate by a possibly zero amount as the only write between two loads from one register).
|
||||||
|
|
||||||
|
## Consequences per user tier
|
||||||
|
|
||||||
|
| Tier | What this means | What is being done |
|
||||||
|
|---|---|---|
|
||||||
|
| Home miner (one card, any size, any vendor, any OS), rig, pool user | Nobody gains from grinding headers for locality: the best group a search can find runs 0.09 percent faster on a card and costs 15 hashes to find. Mining stays one header, every nonce. The drawn:0 repeat is 0.1 percent for every miner alike, no tier favoured | Nothing to change in the miner. The repeat class is handed to the rule's owners as a one-line note (a rotate by a register amount as the sole write between two loads from one register) |
|
||||||
|
| A chip that stores the dataset (chip-model-v3 5) | Header choice gives it nothing either; its per-hash read count stays 128 (Q2) and the items per unit stay at 4,095 of 4,096 | The analytic bound of Q3 and the card point stand on their own |
|
||||||
|
|
||||||
|
## The 1e8 pass (lease pool, build-1): the 1e-6 tail and the prevalence of the finding
|
||||||
|
|
||||||
|
Ran 20:48 to 21:15 UK on 16 leased pool cores of build-1 (`lease pool 32 --min 8`, class adv, owner adv-accept-2; the waiter sat 7 minutes behind the attack-pass F9 leases, then held 16 cores for 1,878 s). 16 shards of 6.25e6 hashes per real program (1e8 hashes, 3.125e6 units each; logs logs/adv-accept-2/rows-1e8/rows-<prog>-1e8-s<k>.log) and the 300-program census (rotclass-300.log). Binary sha256 003e540eee23c8621eb1282c2c9a1b4aca981e81337daa20f63e8469d01b1e7e. Ledger: the box-1 lease was killed by its pid file at 21:16 UK on the coordinator's move order, AFTER all 33 jobs had finished (the kill is the "Terminated, exit 143" in lease-1e8.log); the duplicate resubmission on build-2 was killed by its pid file at 21:17 UK before it took cores; nothing was lost and nothing is queued.
|
||||||
|
|
||||||
|
| Program | Scope | Metric | mean over 1e8 | min over 3.1e6 units (about the 3e-7 quantile) | baseline min, same sample | per-shard q1e-5 range | baseline q1e-5 range |
|
||||||
|
|---|---|---|---|---|---|---|---|
|
||||||
|
| devnet | per unit | rows8KiB | 4018.44 | 3967 | 3974 | 3975 to 3980 | 3977 |
|
||||||
|
| devnet | per unit | items | 4095.38 | 4089 | 4089 | 4090 | 4090 |
|
||||||
|
| devnet | per hash | rows8KiB | 127.924 | 123 | 124 | 125 | 125 |
|
||||||
|
| devnet3 | per unit | rows8KiB | 4020.34 | 3971 | 3981 | 3981 to 3983 | 3983 |
|
||||||
|
| devnet3 | per unit | items | 4095.40 | 4089 | 4088 | 4090 | 4090 |
|
||||||
|
| devnet3 | per hash | rows8KiB | 127.926 | 123 | 124 | 125 | 125 |
|
||||||
|
|
||||||
|
The deepest unit in 1e8 header-chosen hashes saves 51 of 4,096 8 KiB rows (1.2 percent) against the mean, 7 rows more than the windowed baseline's own deepest unit at the same sample size (3967 against 3974; 3971 against 3981 on Devnet 3), a one-sample extreme inside the spread of such minima; at the 1e-5 quantile the two agree to 2 rows. Items per unit and per hash have identical minima to the baseline. Priced through Q3: 51 rows at a 3e-7 tail is 3.3e6 hashes per find for a 1.2 percent saving on one unit, net 3e-7x. The bound stands at 1e8.
|
||||||
|
|
||||||
|
Prevalence of the rotate-identity class (Q4's finding) over 300 drawn accepted class v4 programs through the chain draw path: 63 programs (21.0 percent) hold at least one pair of load sites reading one register whose only intervening writes are `rotr` by a register amount (73 pairs: 71 with one rotr, 1 with two, 1 with three). Each such pair repeats its address with probability 32^-r per iteration times the window-overlap factor (1 for equal windows, 1/2 or 1/4 for nested ones, 0 for disjoint offsets), so one pair costs at most 1 of 128 loads in 1 of 32 iterations: 0.024 percent of loads per pair, 0.1 percent on drawn:0 (two overlapping windows, one rotr). The mean over accepted programs is about 0.006 percent of loads. The two real programs hold no such pair (repeats-devnet-2e5.log: 12 chance repeats in 2e5 hashes). The class is a note for the rule's owners (rule (a) counts a rotate as a write, rule (a') treats rotr as freshness-preserving; both are right about entropy and silent about identity), not a gain anyone mines: a chip or card that serves the second read from the first saves under 0.1 percent on the worst program of 300 and nothing on the real ones.
|
||||||
|
|
||||||
|
## What a longer pass would add
|
||||||
|
|
||||||
|
A 1e9-hash sweep per program moves the 3e-7 tail to 3e-8 on the same baseline; a 3,000-program census refines the 21 percent prevalence and tabulates the window-overlap factor per pair; a card point on an RTX 5090 instead of the A6000 reproduces the 0.09 percent at the production read rate. None of these changes the bound.
|
||||||
51
docs/plans/cryptanalysis/plan-acceptance-rule-2.md
Normal file
51
docs/plans/cryptanalysis/plan-acceptance-rule-2.md
Normal file
|
|
@ -0,0 +1,51 @@
|
||||||
|
# Attack plan: header grinding for locality (acceptance rule and memory access, lane adv-accept-2)
|
||||||
|
|
||||||
|
internal adversarial pass, not an independent review
|
||||||
|
|
||||||
|
Lane adv-accept-2. Branch adv-accept-2 off build/master (7a7caa34). Written 7 October 2026, 19:24 to 19:40 UK (first push 9b86d4e2 at about 19:40 UK, ahead of the 20:25 line; base commit then merged to 04c4d9bc, igneum-pow still identical). Clock: TZ=Europe/London date. Target commit 017e70376489251e18564c0abce7e466e606c8b3 (class v4 sub-version 3, object byte 7). I am an outsider with the public kit; I have never worked on the hash code. Every sentence here that could be quoted in public carries the label above.
|
||||||
|
|
||||||
|
## 0. The outsider rule, applied
|
||||||
|
|
||||||
|
| Check | Result |
|
||||||
|
|---|---|
|
||||||
|
| `git diff --stat 017e7037... HEAD -- igneum-pow` at HEAD 7a7caa34 | prints nothing. igneum-pow/ is byte-identical to the frozen commit on this worktree. (Sibling lanes saw a 6-file divergence at an earlier master 3f0afcd5; master has since moved and the crate at my HEAD matches frozen, so my harness depends on the worktree's own igneum-pow by path.) |
|
||||||
|
| Public kit | proto-cuda/packs-ca3-v4/, eight packs at the frozen commit; v4-devnet-epoch0 id 0xa785001687d8688a, attempt 1, class mx8-erad810f22d+sh256x27. |
|
||||||
|
| Devnet 3 pack | /srv/artefacts/packs/v4-devnet3-epoch0.zip on build-1, sha256 e025750f... verified. id 0xfce15bf61030be57, attempt 0, generator 4, sub-version 3, day bytes le64(20733). Copied read-only. |
|
||||||
|
| Boxes | build-2 (96 threads, load 62 at 20:27 UK) is my run box; build-1 read-only for the pack. No GPU: the GPU row is BLOCKED. |
|
||||||
|
|
||||||
|
### Files opened (complete list)
|
||||||
|
|
||||||
|
verify.rs, accept.rs, bind.rs, seed.rs (in full); generator.rs and memhard.rs (public items, the draw and the dataset fetch); lib.rs, Cargo.toml, main.rs (subcommands). docs/spec/01-lottery-hash.md at 017e7037 (1.4 to 1.4.6, 1.5, 1.6, 1.7, 1.8.5, 1.9, 1.13). docs/analysis/chip-model-v3.md at HEAD (1, 2, 5, 6). The eight packs' program.json and v4-era-0/seeds.txt. tools/attack/f8-uniform and f4-weakday (headers and patterns). tools/build-remote.sh, infra/build-server/{lib.sh, remote-run.sh, capacity/run.sh, capacity/lib.sh}. Sibling plans/reports on build/adv-accept, build/adv-mixer, build/adv-cache. Not opened: anything else under docs/, site/, proto-metal/, git log, other branches.
|
||||||
|
|
||||||
|
## 1. The target, restated from the spec and the code
|
||||||
|
|
||||||
|
### 1.1 What of the header reaches the load addresses
|
||||||
|
|
||||||
|
The miner's header bytes and nonce reach the hash ONLY through the init words (bind.rs, spec 1.6):
|
||||||
|
|
||||||
|
I = seed_words_from_bytes("igneum-block/" || H || nonce_hi_le32)
|
||||||
|
|
||||||
|
H is the 32-byte pre-PoW hash, nonce_hi the high 32 bits of the nonce; the low 32 bits are the lane nonce n. seed_words_from_bytes is FNV-1a 64 under four salted bases plus a murmur finaliser, so every header byte enters all eight words of I. Registers init as r[i] = splitmix32((n XOR I[i]) + 0x9e3779b9*(i+1)) XOR I[(i+1)&7] (verify.rs). The program (from the epoch seed) does NOT depend on the header; only I and n do. The attacker's levers are H (a fresh 256-bit value per header, one header-hash each) and nonce_hi (a free 32-bit re-derivation of I, one FNV pass, no header hash).
|
||||||
|
|
||||||
|
Load address (verify.rs load_index, spec 1.13.1): y = rotl(x*M, R); idx = ((y & (MASK>>k)) | (off<<(D-k))) & MASK, x the source register, D=28 on devnet, k=min(win,2). Physical byte address is 4*idx under any era interleave. A load site whose source has no dataflow path from an earlier load in the same evaluation has an address computable from I and n with ALU only (predictable); a site that reads a loaded value cannot be addressed without the load first. I count the predictable set per program.
|
||||||
|
|
||||||
|
### 1.2 The counts I move
|
||||||
|
|
||||||
|
A unit is 32 lanes x 8 iterations x 16 sites = 4,096 loads of 4 bytes from 2^28 words. Honest denominator (chip-model-v3 s1): 128 distinct items per hash. The question: can a header search find a 32-lane group whose 128 loads per hash (or 4,096 per unit) cluster into fewer DRAM rows (2KiB and 8KiB), cache lines (64B), or items (64B) than a random group, and mine it above the honest rate on a card bound by random 4-byte reads.
|
||||||
|
|
||||||
|
## 2. Questions, in order
|
||||||
|
|
||||||
|
| # | Question | Method | Tool | Known-failed shape (must fire) | Gate | Box-hours |
|
||||||
|
|---|---|---|---|---|---|---|
|
||||||
|
| Q1 | What of the header reaches the address, and through how much mixing | Read + a diffusion probe: flip one bit of H / nonce_hi, measure Hamming weight of the change in each I word, each register, and each of the 4,096 load indices of a unit, over 2^12 header pairs on the two real programs | adv-accept-2 diffuse | a harness-mirror patch that sets idx = f(I) only (no register dependence) must show flips propagate to the address; the real path must show full avalanche (addresses near 50% changed) | every address bit flips with prob 0.5 +/- 3 sigma on the real path | 0.3 |
|
||||||
|
| Q2 | Distribution of distinct DRAM rows (2KiB=512 words, 8KiB=2048 words), lines (64B=16 words) and items per 32-lane unit and per hash over >=10^6 pre-PoW hashes, for the two real programs and several drawn ones; tails at 1e-3, 1e-4, 1e-5 vs a random baseline | Mirror the warp interpreter (as f8-uniform does), feed 10^6 random H (and a nonce_hi sweep), record per-unit and per-hash distinct rows/lines/items; histogram and quantiles; a SplitMix64 random-address control of the same shape | adv-accept-2 rows | --plant const-site (one lane-constant load site) and --plant tiny-window (win=2 on every site) must show the clustering at once (distinct counts collapse); clean programs sit at the random baseline | best-case tail within the random baseline's own extreme-value spread | 6 (sharded over idle cores, one log per shard) |
|
||||||
|
| Q3 | The price: search cost in hashes per found group vs loads saved; does any grind net >1% of rate on a card bound by random 4-byte reads | Analytic from Q2's tail: if the best 1e-k group saves dL loads, a card at R reads/s bound mines the found group at 128/(128-dL) higher, but the search costs ~10^k hashes per found group, each hash is itself 128 reads; net rate = gain / (1 + search_reads/useful_reads). State the model; GPU measurement BLOCKED | adv-accept-2 price (arithmetic beside rows) | a hand check: a 10% loads-saved group found at rate 1e-4 nets <<1% after search cost | no grind nets >1% | 0.1 |
|
||||||
|
| Q4 | Does rule (c) / (c'') bound per-hash locality or only per-program | Read: (c) LaneConstantSite bounds per (iteration, instruction) lane spread on the 64 FIXED accept nonces; (c'') bounds per-site distinct INDICES over 2^20 evaluations. Neither is keyed on the header. Measure whether a header outside the 64 accept nonces can cluster a unit that the rule passed; compare the rule's own distinct-index ratio to Q2's per-unit row counts | reuses Q2 | the planted tiny-window program must be REJECTED by accept::check (the rule catches the per-program clustering) while Q2 shows a clean program's per-unit tail is header-independent | the rule rejects the plant; headers do not move a clean program's tail | 0.5 |
|
||||||
|
|
||||||
|
Known-failed shape overall: a planted program with a lane-constant load site or a tiny window; Q2's harness must find its clustering at once, and Q4 must show accept::check rejects it. A result is a BREAK (method, counted gain, command, seed) or a BOUND (what was searched, how far, the margin). "Nothing found" counts only with its effort.
|
||||||
|
|
||||||
|
## 3. Running
|
||||||
|
|
||||||
|
Build through tools/build-remote.sh --box 2 -- build --release. Runs over 10 min start from run-box.sh in the harness dir with nohup nice -n 10, a pid file beside the log under /srv/builds/igneum-wt-adv-accept-2/adv/, and the yield rule: poll /srv/builds/_locks every 5 s, RULE CHANGE by the build-server lane at about 19:3x UK: no SIGSTOP yield (a build slot is held nearly continuously); every sweep runs at nice 10 on cores 8 to 95 only (taskset), logs and pid files OUTSIDE the worktree mirror in /srv/builds/_adv-adv-accept-2/. Kill by pid file only. The 10^6 row sweep is sharded by header range across idle cores, one log per shard. Queue files go to /srv/builds/_adv/accept/queue/NN-adv-accept-2-<name>.sh, claimed with mkdir on /srv/builds/_adv/accept/claims/<name>. Box-hours: 8 a reading, 16 the ask line. No GPU: the Q3 per-card confirmation is BLOCKED and says so.
|
||||||
|
|
||||||
|
Commit as igneum-labs; push only `git push build adv-accept-2`. No em dashes, short sentences, numbers in tables.
|
||||||
1
tools/attack/adv-accept-2/.gitignore
vendored
Normal file
1
tools/attack/adv-accept-2/.gitignore
vendored
Normal file
|
|
@ -0,0 +1 @@
|
||||||
|
target-remote/
|
||||||
14
tools/attack/adv-accept-2/Cargo.lock
generated
Normal file
14
tools/attack/adv-accept-2/Cargo.lock
generated
Normal file
|
|
@ -0,0 +1,14 @@
|
||||||
|
# This file is automatically @generated by Cargo.
|
||||||
|
# It is not intended for manual editing.
|
||||||
|
version = 4
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "attack-adv-accept-2"
|
||||||
|
version = "0.1.0"
|
||||||
|
dependencies = [
|
||||||
|
"igneum-pow",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "igneum-pow"
|
||||||
|
version = "0.2.0"
|
||||||
21
tools/attack/adv-accept-2/Cargo.toml
Normal file
21
tools/attack/adv-accept-2/Cargo.toml
Normal file
|
|
@ -0,0 +1,21 @@
|
||||||
|
[package]
|
||||||
|
name = "attack-adv-accept-2"
|
||||||
|
version = "0.1.0"
|
||||||
|
edition = "2021"
|
||||||
|
description = "adv-accept-2: header grinding for locality. Distinct DRAM rows (2 KiB, 8 KiB), cache lines (64 B) and items per 32-lane unit and per hash over >=10^6 pre-PoW hashes of the real programs and drawn ones, tails against a random baseline, the price model, and the per-hash vs per-program question (spec 01 1.6, 1.7, 1.13.1; igneum-pow by path)"
|
||||||
|
license = "MIT"
|
||||||
|
publish = false
|
||||||
|
|
||||||
|
[[bin]]
|
||||||
|
name = "adv-accept-2"
|
||||||
|
path = "src/main.rs"
|
||||||
|
|
||||||
|
[dependencies]
|
||||||
|
igneum-pow = { path = "../../../igneum-pow" }
|
||||||
|
|
||||||
|
[workspace]
|
||||||
|
|
||||||
|
[profile.release]
|
||||||
|
opt-level = 3
|
||||||
|
lto = true
|
||||||
|
codegen-units = 1
|
||||||
43
tools/attack/adv-accept-2/gpu/rowbench.cu
Normal file
43
tools/attack/adv-accept-2/gpu/rowbench.cu
Normal file
|
|
@ -0,0 +1,43 @@
|
||||||
|
// adv-accept-2 GPU point: dependent 4-byte reads over exported address tables.
|
||||||
|
// internal adversarial pass, not an independent review.
|
||||||
|
// Each warp runs one unit: lane l performs 128 dependent reads buf[addr[k*32+l] + (v & zero)], v ^= word.
|
||||||
|
// The dependence is real (zero is a runtime 0 the compiler cannot fold). Many units run concurrently, as a miner does.
|
||||||
|
// Usage: rowbench <units.bin> <units> <rounds> (1 GiB buffer of 2^28 u32 filled pseudo-randomly)
|
||||||
|
#include <cstdio>
|
||||||
|
#include <cstdlib>
|
||||||
|
#include <cstdint>
|
||||||
|
#include <vector>
|
||||||
|
#include <cuda_runtime.h>
|
||||||
|
#define CK(x) do { cudaError_t e = (x); if (e != cudaSuccess) { fprintf(stderr, "CUDA %s at %d\n", cudaGetErrorString(e), __LINE__); exit(1); } } while (0)
|
||||||
|
__global__ void fill(uint32_t* buf, uint32_t n) { uint32_t i = blockIdx.x * blockDim.x + threadIdx.x; for (; i < n; i += gridDim.x * blockDim.x) { uint32_t x = i * 0x9E3779B1u; x ^= x >> 15; x *= 0x85EBCA77u; x ^= x >> 13; buf[i] = x; } }
|
||||||
|
__global__ void chase(const uint32_t* __restrict__ buf, const uint32_t* __restrict__ tab, uint32_t units, uint32_t zero, uint32_t* out) {
|
||||||
|
uint32_t w = (blockIdx.x * blockDim.x + threadIdx.x) >> 5; uint32_t lane = threadIdx.x & 31;
|
||||||
|
if (w >= units) return;
|
||||||
|
const uint32_t* a = tab + (size_t)w * 4096;
|
||||||
|
uint32_t v = lane;
|
||||||
|
#pragma unroll 4
|
||||||
|
for (int k = 0; k < 128; k++) { uint32_t idx = (a[k * 32 + lane] + (v & zero)) & 0x0fffffffu; v ^= buf[idx]; }
|
||||||
|
out[w * 32 + lane] = v;
|
||||||
|
}
|
||||||
|
int main(int argc, char** argv) {
|
||||||
|
if (argc < 4) { fprintf(stderr, "rowbench <units.bin> <units> <rounds>\n"); return 2; }
|
||||||
|
uint32_t units = atoi(argv[2]); int rounds = atoi(argv[3]);
|
||||||
|
size_t tabn = (size_t)units * 4096;
|
||||||
|
std::vector<uint32_t> tab(tabn);
|
||||||
|
FILE* f = fopen(argv[1], "rb"); if (!f) { perror("open"); return 1; }
|
||||||
|
if (fread(tab.data(), 4, tabn, f) != tabn) { fprintf(stderr, "short read\n"); return 1; } fclose(f);
|
||||||
|
uint32_t *dbuf, *dtab, *dout; const uint32_t n = 1u << 28;
|
||||||
|
CK(cudaMalloc(&dbuf, (size_t)n * 4)); CK(cudaMalloc(&dtab, tabn * 4)); CK(cudaMalloc(&dout, (size_t)units * 32 * 4));
|
||||||
|
fill<<<4096, 256>>>(dbuf, n); CK(cudaDeviceSynchronize());
|
||||||
|
CK(cudaMemcpy(dtab, tab.data(), tabn * 4, cudaMemcpyHostToDevice));
|
||||||
|
int threads = 256; int blocks = (units * 32 + threads - 1) / threads;
|
||||||
|
chase<<<blocks, threads>>>(dbuf, dtab, units, 0, dout); CK(cudaDeviceSynchronize()); // warm
|
||||||
|
cudaEvent_t e0, e1; cudaEventCreate(&e0); cudaEventCreate(&e1);
|
||||||
|
cudaEventRecord(e0);
|
||||||
|
for (int r = 0; r < rounds; r++) chase<<<blocks, threads>>>(dbuf, dtab, units, 0, dout);
|
||||||
|
cudaEventRecord(e1); CK(cudaEventSynchronize(e1));
|
||||||
|
float ms; cudaEventElapsedTime(&ms, e0, e1);
|
||||||
|
double reads = (double)units * 4096 * rounds;
|
||||||
|
printf("units=%u rounds=%d ms=%.3f reads/s=%.4e unit-hashes/s=%.4e (32 lanes x 128 dependent 4-byte reads per unit)\n", units, rounds, ms, reads / (ms / 1e3), (double)units * rounds / (ms / 1e3));
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
17
tools/attack/adv-accept-2/run/run-box.sh
Executable file
17
tools/attack/adv-accept-2/run/run-box.sh
Executable file
|
|
@ -0,0 +1,17 @@
|
||||||
|
#!/usr/bin/env bash
|
||||||
|
# adv-accept-2 runner. No SIGSTOP yield (build-server lane ruling, 7 Oct 2026 ~20:1x BST: a build slot is held nearly
|
||||||
|
# continuously, so the yield kept sweeps paused almost all the time). Every sweep runs at nice 10 on cores 8 to 95
|
||||||
|
# only (cores 0 to 7 are reserved for release builds, the seed and the observer). Logs and pid files live OUTSIDE the
|
||||||
|
# worktree mirror (/srv/builds/igneum-wt-adv-accept-2 is rsync-deleted by build-remote.sh): they go to
|
||||||
|
# /srv/builds/_adv-adv-accept-2/. The binary stays under the mirror's target dir.
|
||||||
|
# run-box.sh <name> <bin-args...>
|
||||||
|
set -euo pipefail
|
||||||
|
BIN="${ADV_BIN:-/srv/builds/igneum-wt-adv-accept-2/tools/attack/adv-accept-2/target/release/adv-accept-2}"
|
||||||
|
OUT=/srv/builds/_adv-adv-accept-2
|
||||||
|
mkdir -p "$OUT"
|
||||||
|
name="$1"; shift
|
||||||
|
log="$OUT/$name.log"; pidf="$OUT/$name.pid"
|
||||||
|
if [ -f "$pidf" ] && kill -0 "$(cat "$pidf")" 2>/dev/null; then echo "already running: $name (pid $(cat "$pidf"))"; exit 0; fi
|
||||||
|
nohup nice -n 10 taskset -c 8-95 "$BIN" "$@" > "$log" 2>&1 &
|
||||||
|
echo $! > "$pidf"
|
||||||
|
echo "started $name pid $(cat "$pidf") log $log"
|
||||||
595
tools/attack/adv-accept-2/src/main.rs
Normal file
595
tools/attack/adv-accept-2/src/main.rs
Normal file
|
|
@ -0,0 +1,595 @@
|
||||||
|
//! adv-accept-2: header grinding for locality.
|
||||||
|
//!
|
||||||
|
//! internal adversarial pass, not an independent review.
|
||||||
|
//!
|
||||||
|
//! The question: a miner chooses the header bytes behind the pre-PoW hash H and the nonce. H and nonce_hi reach the
|
||||||
|
//! hash only through the init words I = seed_words_from_bytes("igneum-block/" || H || nonce_hi_le32) (bind.rs, spec
|
||||||
|
//! 1.6); the program (from the epoch seed) does not change. Can a cheap search over H find a 32-lane group whose 128
|
||||||
|
//! loads per hash, or 4,096 per unit, cluster into fewer DRAM rows (2 KiB, 8 KiB), cache lines (64 B) or dataset
|
||||||
|
//! items (64 B) than a random group, and mine it above the honest rate on a card bound by random 4-byte reads.
|
||||||
|
//!
|
||||||
|
//! The harness mirrors the warp interpreter of verify.rs instruction for instruction (register-major), recording the
|
||||||
|
//! load index of every load site. It never modifies igneum-pow; the real draw (Epoch::chain_program, ProgramClass::V4)
|
||||||
|
//! and the real address map (verify::load_index) and era layout come from the library. Loaded VALUES use the
|
||||||
|
//! closed-form dataset_elem (the acceptance rule's own stand-in, spec 1.4.6): the locality DISTRIBUTION is a property
|
||||||
|
//! of the address map y = rotl(x * M, R) masked and the register distribution, not of the dataset values, since a
|
||||||
|
//! load's address is computed from its source register BEFORE that load and the dataset value only enters the NEXT
|
||||||
|
//! load's source; both datasets give pseudo-random register values. --live confirms on the memory-hard dataset.
|
||||||
|
//!
|
||||||
|
//! Commands:
|
||||||
|
//! adv-accept-2 draw-check derive the two real programs, print id and the 16 load sites
|
||||||
|
//! adv-accept-2 diffuse [--pairs N] Q1: one-bit flips of H and nonce_hi, avalanche into I, registers, addresses
|
||||||
|
//! adv-accept-2 rows --prog <sel> --hashes N [--shard k/of] [--plant P] [--live] Q2: distinct rows/lines/items
|
||||||
|
//! adv-accept-2 price Q3: the search-cost vs loads-saved model, arithmetic
|
||||||
|
//!
|
||||||
|
//! <sel>: devnet | devnet3 | drawn:<k> (devnet = the shared devnet epoch-0 v4 program; devnet3 the Devnet 3 one;
|
||||||
|
//! drawn:k a label-derived epoch and era seed, the chain draw path)
|
||||||
|
//! plant: none | const-site | tiny-window (the known-fail firings; const-site forces load site 0 lane-constant,
|
||||||
|
//! tiny-window forces win=2 on every load site)
|
||||||
|
|
||||||
|
use igneum_pow::bind::{block_init_words, day_bytes, unhex};
|
||||||
|
use igneum_pow::generator::{Op, Program, ProgramClass, ITERATIONS, LANES};
|
||||||
|
use igneum_pow::seed::seed_words_from_bytes;
|
||||||
|
use igneum_pow::verify::{dataset_elem, load_index, splitmix32, DatasetSource, Epoch};
|
||||||
|
use std::io::Write;
|
||||||
|
use std::time::{Instant, SystemTime, UNIX_EPOCH};
|
||||||
|
|
||||||
|
const DATASET_LOG2: u32 = 28;
|
||||||
|
const MASK: u32 = (1u32 << DATASET_LOG2) - 1;
|
||||||
|
/// The shared devnet epoch-0 seed (devnet genesis hash), epoch seed and era seed (v4-devnet-epoch0 pack, seeds.txt).
|
||||||
|
const DEVNET_EPOCH_HEX: &str = "edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07";
|
||||||
|
const DEVNET_DAY: u64 = 20730;
|
||||||
|
/// The Devnet 3 epoch-0 seed (v4-devnet3-epoch0 pack on build-1; verified sha e025750f..., program id fce15bf6...).
|
||||||
|
const DEVNET3_EPOCH_HEX: &str = "4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925";
|
||||||
|
const DEVNET3_DAY: u64 = 20733;
|
||||||
|
|
||||||
|
#[derive(Clone, Copy, PartialEq, Eq)]
|
||||||
|
enum Plant { None, ConstSite, TinyWindow, HeaderBlind }
|
||||||
|
impl Plant {
|
||||||
|
fn parse(s: &str) -> Plant {
|
||||||
|
match s { "none" => Plant::None, "const-site" => Plant::ConstSite, "tiny-window" => Plant::TinyWindow, "header-blind" => Plant::HeaderBlind, _ => panic!("unknown plant {s}") }
|
||||||
|
}
|
||||||
|
fn name(self) -> &'static str { match self { Plant::None => "none", Plant::ConstSite => "const-site", Plant::TinyWindow => "tiny-window", Plant::HeaderBlind => "header-blind" } }
|
||||||
|
}
|
||||||
|
|
||||||
|
fn utc_now() -> String {
|
||||||
|
let s = SystemTime::now().duration_since(UNIX_EPOCH).unwrap().as_secs();
|
||||||
|
let (d, t) = (s / 86400, s % 86400);
|
||||||
|
let z = d as i64 + 719468; let era = z.div_euclid(146097); let doe = z.rem_euclid(146097);
|
||||||
|
let yoe = (doe - doe / 1460 + doe / 36524 - doe / 146096) / 365; let y = yoe + era * 400;
|
||||||
|
let doy = doe - (365 * yoe + yoe / 4 - yoe / 100); let mp = (5 * doy + 2) / 153;
|
||||||
|
let dd = doy - (153 * mp + 2) / 5 + 1; let mm = if mp < 10 { mp + 3 } else { mp - 9 };
|
||||||
|
let yy = if mm <= 2 { y + 1 } else { y };
|
||||||
|
format!("{yy:04}-{mm:02}-{dd:02}T{:02}:{:02}:{:02}Z", t / 3600, (t / 60) % 60, t % 60)
|
||||||
|
}
|
||||||
|
macro_rules! log { ($($a:tt)*) => {{ println!("[{}] {}", utc_now(), format!($($a)*)); std::io::stdout().flush().ok(); }}; }
|
||||||
|
|
||||||
|
#[inline(always)]
|
||||||
|
fn mulhi32(a: u32, b: u32) -> u32 { ((a as u64 * b as u64) >> 32) as u32 }
|
||||||
|
|
||||||
|
/// The two real programs and the drawn ones, through the chain draw path.
|
||||||
|
fn program_of(sel: &str) -> (Program, u64) {
|
||||||
|
if let Some(k) = sel.strip_prefix("drawn:") {
|
||||||
|
let epoch = seed_words_from_bytes(format!("igneum-adv-accept-2/epoch/{k}").as_bytes());
|
||||||
|
let era = seed_words_from_bytes(format!("igneum-adv-accept-2/era/{k}").as_bytes());
|
||||||
|
let eb: Vec<u8> = epoch.iter().flat_map(|w| w.to_le_bytes()).collect();
|
||||||
|
let erab: Vec<u8> = era.iter().flat_map(|w| w.to_le_bytes()).collect();
|
||||||
|
let p = Epoch::chain_program(&eb, Some(&erab), ProgramClass::V4, "adv-accept-2-drawn");
|
||||||
|
return (p, DEVNET_DAY);
|
||||||
|
}
|
||||||
|
let (hex, day) = match sel {
|
||||||
|
"devnet" => (DEVNET_EPOCH_HEX, DEVNET_DAY),
|
||||||
|
"devnet3" => (DEVNET3_EPOCH_HEX, DEVNET3_DAY),
|
||||||
|
_ => panic!("unknown program selector {sel}"),
|
||||||
|
};
|
||||||
|
let eb = unhex(hex).unwrap();
|
||||||
|
// the real packs carry era seed = epoch seed (seeds.txt); chain_program draws the era from it
|
||||||
|
let p = Epoch::chain_program(&eb, Some(&eb), ProgramClass::V4, sel);
|
||||||
|
(p, day)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Load-site count (16) and the sites' (win, era) read from the program.
|
||||||
|
fn load_sites(p: &Program) -> Vec<usize> {
|
||||||
|
p.instrs.iter().enumerate().filter(|(_, i)| i.op.is_load()).map(|(k, _)| k).collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A faithful interpreter mirror that records the 4,096 load indices of one unit at base lane nonce `g` under init
|
||||||
|
/// words `I`. Loaded values are the closed-form dataset_elem(idx, d0, d1) (the acceptance rule's stand-in) unless a
|
||||||
|
/// live DatasetSource is given. The shadow block is executed as verify.rs does. `plant` forces a known-fail shape.
|
||||||
|
#[allow(clippy::too_many_arguments)]
|
||||||
|
fn unit_addresses(p: &Program, init: &[u32; 8], g: u32, d0: u32, d1: u32, live: Option<&DatasetSource>, plant: Plant, out: &mut Vec<u32>) {
|
||||||
|
out.clear();
|
||||||
|
let era = p.class.era;
|
||||||
|
let layout = p.class.layout();
|
||||||
|
let sites = load_sites(p);
|
||||||
|
let first_site = *sites.first().unwrap_or(&usize::MAX);
|
||||||
|
let init: &[u32; 8] = if plant == Plant::HeaderBlind { &p.seed } else { init };
|
||||||
|
let mut r = [[0u32; LANES]; 8];
|
||||||
|
for lane in 0..LANES {
|
||||||
|
let n = g.wrapping_add(lane as u32);
|
||||||
|
for i in 0..8 {
|
||||||
|
let mut x = n ^ init[i];
|
||||||
|
x = x.wrapping_add(0x9e3779b9u32.wrapping_mul(i as u32 + 1));
|
||||||
|
x = splitmix32(x);
|
||||||
|
r[i][lane] = x ^ init[(i + 1) & 7];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let shadow_reps = p.shadow_reps();
|
||||||
|
let mut idx = [0u32; LANES];
|
||||||
|
for _it in 0..ITERATIONS {
|
||||||
|
let sel = r[0];
|
||||||
|
let run = p.instrs.iter().enumerate()
|
||||||
|
.chain((0..shadow_reps).flat_map(|_| p.shadow.iter().enumerate().map(|(k, i)| (64 + k, i))));
|
||||||
|
for (k, ins) in run {
|
||||||
|
let d = ins.dst as usize; let a = ins.src as usize;
|
||||||
|
match ins.op {
|
||||||
|
Op::Add => { let (im, im2, bit) = (ins.imm, ins.imm2, ins.bit as u32); let s = r[a];
|
||||||
|
for l in 0..LANES { let c = if (sel[l] >> bit) & 1 != 0 { im2 } else { im }; r[d][l] = r[d][l].wrapping_add(s[l]).wrapping_add(c); } }
|
||||||
|
Op::Sub => { let s = r[a]; for l in 0..LANES { r[d][l] = r[d][l].wrapping_sub(s[l]); } }
|
||||||
|
Op::Mul => { let s = r[a]; for l in 0..LANES { r[d][l] = r[d][l].wrapping_mul(s[l]); } }
|
||||||
|
Op::MulHi => { let s = r[a]; for l in 0..LANES { r[d][l] = mulhi32(r[d][l], s[l]); } }
|
||||||
|
Op::Xor => { let s = r[a]; for l in 0..LANES { r[d][l] ^= s[l]; } }
|
||||||
|
Op::Or => { let s = r[a]; for l in 0..LANES { r[d][l] |= s[l]; } }
|
||||||
|
Op::Rotl => { let n = ins.rot; for l in 0..LANES { r[d][l] = r[d][l].rotate_left(n); } }
|
||||||
|
Op::Rotr => { let s = r[a]; for l in 0..LANES { r[d][l] = r[d][l].rotate_right(s[l] & 31); } }
|
||||||
|
Op::Mad => { let s = r[a]; let s2 = r[ins.src2 as usize]; for l in 0..LANES { r[d][l] = s[l].wrapping_mul(s2[l]).wrapping_add(r[d][l]); } }
|
||||||
|
Op::Shfl => { let s = r[a]; let m = ins.mask as usize; for l in 0..LANES { r[d][l] ^= s[l ^ m]; } }
|
||||||
|
Op::Load => {
|
||||||
|
let mut ins_eff = *ins;
|
||||||
|
if plant == Plant::TinyWindow { ins_eff.win = 2; }
|
||||||
|
for l in 0..LANES { idx[l] = load_index(era.as_ref(), &ins_eff, r[a][l], MASK, DATASET_LOG2); }
|
||||||
|
if plant == Plant::ConstSite && k == first_site { let v = idx[0]; for l in 0..LANES { idx[l] = v; } }
|
||||||
|
for l in 0..LANES {
|
||||||
|
let w = match live { Some(ds) => ds.word_at(layout, idx[l]), None => dataset_elem(idx[l], d0, d1) };
|
||||||
|
r[d][l] ^= w;
|
||||||
|
out.push(idx[l]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Op::WLoad | Op::Scratch | Op::Hot => { /* not drawn in class v4 */ }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Same-instruction coalescing: over the 128 load sites of a unit (32 lanes each), the number of lane pairs that share
|
||||||
|
/// a 2 KiB row, an 8 KiB row and a 64 B line. These are the only hits a DRAM row buffer or a coalescer sees together.
|
||||||
|
fn coalesced(addrs: &[u32]) -> (u32, u32, u32) {
|
||||||
|
let (mut r2, mut r8, mut ln) = (0u32, 0u32, 0u32);
|
||||||
|
for site in addrs.chunks_exact(LANES) {
|
||||||
|
for i in 0..LANES { for j in i + 1..LANES {
|
||||||
|
let (a, b) = (site[i], site[j]);
|
||||||
|
if a >> 9 == b >> 9 { r2 += 1; }
|
||||||
|
if a >> 11 == b >> 11 { r8 += 1; }
|
||||||
|
if a >> 4 == b >> 4 { ln += 1; }
|
||||||
|
} }
|
||||||
|
}
|
||||||
|
(r2, r8, ln)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Distinct rows (2 KiB = 512 words, 8 KiB = 2048 words), lines (16 words) and items (16 words) among a set of indices.
|
||||||
|
fn distinct(idxs: &[u32]) -> (u32, u32, u32, u32) {
|
||||||
|
let mut rows2 = Vec::with_capacity(idxs.len());
|
||||||
|
let mut rows8 = Vec::with_capacity(idxs.len());
|
||||||
|
let mut lines = Vec::with_capacity(idxs.len());
|
||||||
|
let mut items = Vec::with_capacity(idxs.len());
|
||||||
|
for &w in idxs { rows2.push(w >> 9); rows8.push(w >> 11); lines.push(w >> 4); items.push(w >> 4); }
|
||||||
|
let d = |v: &mut Vec<u32>| { v.sort_unstable(); v.dedup(); v.len() as u32 };
|
||||||
|
(d(&mut rows2), d(&mut rows8), d(&mut lines), d(&mut items))
|
||||||
|
}
|
||||||
|
|
||||||
|
struct Hist { // per-metric: histogram over small distinct counts, and a sorted reservoir for tails
|
||||||
|
min: u32, max: u32, sum: u64, n: u64, counts: Vec<u64>,
|
||||||
|
}
|
||||||
|
impl Hist {
|
||||||
|
fn new(cap: usize) -> Self { Hist { min: u32::MAX, max: 0, sum: 0, n: 0, counts: vec![0; cap + 1] } }
|
||||||
|
fn add(&mut self, v: u32) { self.min = self.min.min(v); self.max = self.max.max(v); self.sum += v as u64; self.n += 1; let i = (v as usize).min(self.counts.len() - 1); self.counts[i] += 1; }
|
||||||
|
fn mean(&self) -> f64 { self.sum as f64 / self.n.max(1) as f64 }
|
||||||
|
/// lowest value v such that at most q fraction lie strictly below it (the low tail: a smaller distinct count is more clustered)
|
||||||
|
fn low_quantile(&self, q: f64) -> u32 { let target = (q * self.n as f64).floor() as u64; let mut c = 0u64; for (v, &cnt) in self.counts.iter().enumerate() { if c + cnt > target { return v as u32; } c += cnt; } 0 }
|
||||||
|
}
|
||||||
|
|
||||||
|
fn cmd_rows(prog: &str, hashes: u64, shard: (u64, u64), plant: Plant, live: bool) {
|
||||||
|
let (p, day) = program_of(prog);
|
||||||
|
let (d0, d1) = (p.seed[0], p.seed[1]);
|
||||||
|
log!("rows: prog={prog} id={:#018x} attempt={} sites={} plant={} live={} hashes={} shard={}/{}",
|
||||||
|
p.program_id(), p.attempt, load_sites(&p).len(), plant.name(), live, hashes, shard.0, shard.1);
|
||||||
|
let ds = if live {
|
||||||
|
log!("building the memory-hard day cache (day {day})...");
|
||||||
|
Some(Epoch::chain_dataset_day(&day_bytes(day), ProgramClass::V4, 0, DATASET_LOG2))
|
||||||
|
} else { None };
|
||||||
|
let units = hashes.div_ceil(LANES as u64);
|
||||||
|
// shard the header range
|
||||||
|
let per = units.div_ceil(shard.1);
|
||||||
|
let lo = shard.0 * per; let hi = (lo + per).min(units);
|
||||||
|
let t0 = Instant::now();
|
||||||
|
// per-hash (128 loads) and per-unit (4096 loads) histograms, for rows2/rows8/lines/items
|
||||||
|
let mut ph = [Hist::new(128), Hist::new(128), Hist::new(128), Hist::new(128)]; // rows2,rows8,lines,items
|
||||||
|
let mut pu = [Hist::new(4096), Hist::new(4096), Hist::new(4096), Hist::new(4096)];
|
||||||
|
let mut addrs = Vec::with_capacity(4096);
|
||||||
|
let mut lane_idx = Vec::with_capacity(128);
|
||||||
|
let mut co = [Hist::new(512), Hist::new(512), Hist::new(512)]; // same-site pairs sharing a 2 KiB row, 8 KiB row, 64 B line, per unit
|
||||||
|
for u in lo..hi {
|
||||||
|
// a fresh pre-PoW hash H per unit: splitmix of the unit index into 32 bytes (a stand-in for a random header hash)
|
||||||
|
let mut h = [0u8; 32];
|
||||||
|
let mut s = u.wrapping_mul(0x9E3779B97F4A7C15).wrapping_add(0xD1B54A32D192ED03);
|
||||||
|
for c in h.chunks_mut(8) { s = (s ^ (s >> 30)).wrapping_mul(0xBF58476D1CE4E5B9); s = (s ^ (s >> 27)).wrapping_mul(0x94D049BB133111EB); s ^= s >> 31; c.copy_from_slice(&s.to_le_bytes()); }
|
||||||
|
let init = block_init_words(&h, 0);
|
||||||
|
let g = 0u32; // lane nonce base; the unit is lanes 0..31, the header is the lever
|
||||||
|
unit_addresses(&p, &init, g, d0, d1, ds.as_ref(), plant, &mut addrs);
|
||||||
|
let (r2, r8, ln, it) = distinct(&addrs);
|
||||||
|
pu[0].add(r2); pu[1].add(r8); pu[2].add(ln); pu[3].add(it);
|
||||||
|
let (c2, c8, cl) = coalesced(&addrs);
|
||||||
|
co[0].add(c2); co[1].add(c8); co[2].add(cl);
|
||||||
|
// per hash: each lane's own 128 loads are addrs[lane], addrs[lane+32], ... (one lane per push group of 32)
|
||||||
|
for lane in 0..LANES {
|
||||||
|
lane_idx.clear();
|
||||||
|
let mut j = lane; while j < addrs.len() { lane_idx.push(addrs[j]); j += LANES; }
|
||||||
|
let (r2, r8, ln, it) = distinct(&lane_idx);
|
||||||
|
ph[0].add(r2); ph[1].add(r8); ph[2].add(ln); ph[3].add(it);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let secs = t0.elapsed().as_secs_f64();
|
||||||
|
log!("done {} units ({} lane-hashes) in {:.1}s", hi - lo, (hi - lo) * LANES as u64, secs);
|
||||||
|
// random baseline of the same shape
|
||||||
|
let br = baseline(128, (hi - lo) * LANES as u64, 0xBA5E1);
|
||||||
|
let bu = baseline(4096, hi - lo, 0xBA5E2);
|
||||||
|
let wr = baseline_windowed(&p, true, (hi - lo) * LANES as u64, 0xBA5E3);
|
||||||
|
let wu = baseline_windowed(&p, false, hi - lo, 0xBA5E4);
|
||||||
|
let names = ["rows2KiB", "rows8KiB", "lines64B", "items64B"];
|
||||||
|
let sites_win: Vec<u8> = p.instrs.iter().filter(|i| i.op.is_load()).map(|i| i.win).collect();
|
||||||
|
println!("sites win (0 = whole dataset, 1 = half, 2 = quarter): {:?}", sites_win);
|
||||||
|
println!("== PER HASH (128 loads) prog={prog} plant={} live={} n={}", plant.name(), live, ph[0].n);
|
||||||
|
println!("metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min");
|
||||||
|
for m in 0..4 {
|
||||||
|
println!("{:14} {:8.3} {:5} {:5} {:5} {:5} | {:8.3} {:5} {:5} {:5} {:5} | {:8.3} {:5}", names[m], ph[m].mean(), ph[m].min,
|
||||||
|
ph[m].low_quantile(1e-3), ph[m].low_quantile(1e-4), ph[m].low_quantile(1e-5),
|
||||||
|
wr[m].mean(), wr[m].min, wr[m].low_quantile(1e-3), wr[m].low_quantile(1e-4), wr[m].low_quantile(1e-5), br[m].mean(), br[m].min);
|
||||||
|
}
|
||||||
|
println!("== PER UNIT (4096 loads) prog={prog} plant={} live={} n={}", plant.name(), live, pu[0].n);
|
||||||
|
println!("metric mean min q1e-3 q1e-4 q1e-5 | windowed baseline mean min q1e-3 q1e-4 q1e-5 | uniform mean min");
|
||||||
|
for m in 0..4 {
|
||||||
|
println!("{:14} {:8.3} {:5} {:5} {:5} {:5} | {:8.3} {:5} {:5} {:5} {:5} | {:8.3} {:5}", names[m], pu[m].mean(), pu[m].min,
|
||||||
|
pu[m].low_quantile(1e-3), pu[m].low_quantile(1e-4), pu[m].low_quantile(1e-5),
|
||||||
|
wu[m].mean(), wu[m].min, wu[m].low_quantile(1e-3), wu[m].low_quantile(1e-4), wu[m].low_quantile(1e-5), bu[m].mean(), bu[m].min);
|
||||||
|
}
|
||||||
|
println!("== SAME-INSTRUCTION COALESCING per unit (lane pairs of one load site sharing a row or line; 128 sites x 496 pairs)");
|
||||||
|
println!("metric mean max (expected uniform: 2KiB {:.4}, 8KiB {:.4}, 64B {:.5} per unit)", 128.0 * 496.0 / (1u64 << 19) as f64, 128.0 * 496.0 / (1u64 << 17) as f64, 128.0 * 496.0 / (1u64 << 24) as f64);
|
||||||
|
for (m, nm) in ["pairs row2KiB", "pairs row8KiB", "pairs line64B"].iter().enumerate() {
|
||||||
|
println!("{:14} {:8.4} {:6}", nm, co[m].mean(), co[m].max);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A windowed random baseline: `samples` units of 4,096 (or lanes of 128) loads whose address is a uniform y masked
|
||||||
|
/// into the program's own site window (verify::window), so the control carries the era's quarter and half windows.
|
||||||
|
fn baseline_windowed(p: &Program, per_hash: bool, samples: u64, seed: u64) -> [Hist; 4] {
|
||||||
|
let sites: Vec<(u32, u32)> = p.instrs.iter().filter(|i| i.op.is_load()).map(|i| igneum_pow::verify::window(i, MASK, DATASET_LOG2)).collect();
|
||||||
|
let loads = if per_hash { 128 } else { 4096 };
|
||||||
|
let mut h = [Hist::new(loads), Hist::new(loads), Hist::new(loads), Hist::new(loads)];
|
||||||
|
let mut s = seed | 1;
|
||||||
|
let mut idxs = vec![0u32; loads];
|
||||||
|
for _ in 0..samples {
|
||||||
|
for (k, x) in idxs.iter_mut().enumerate() {
|
||||||
|
s = (s ^ (s >> 30)).wrapping_mul(0xBF58476D1CE4E5B9); s = (s ^ (s >> 27)).wrapping_mul(0x94D049BB133111EB); s ^= s >> 31;
|
||||||
|
// per hash: load k is site k mod 16; per unit: the unit records lane-minor, so load k is site (k / 32) mod 16
|
||||||
|
let site = if per_hash { k % 16 } else { (k / LANES) % 16 };
|
||||||
|
let (wm, off) = sites[site];
|
||||||
|
*x = ((s as u32) & wm) | off;
|
||||||
|
}
|
||||||
|
let (r2, r8, ln, it) = distinct(&idxs);
|
||||||
|
h[0].add(r2); h[1].add(r8); h[2].add(ln); h[3].add(it);
|
||||||
|
}
|
||||||
|
h
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A uniform random baseline: `samples` sets of `loads` uniform indices in 2^28, the same four distinct metrics.
|
||||||
|
fn baseline(loads: usize, samples: u64, seed: u64) -> [Hist; 4] {
|
||||||
|
let cap = loads;
|
||||||
|
let mut h = [Hist::new(cap), Hist::new(cap), Hist::new(cap), Hist::new(cap)];
|
||||||
|
let mut s = seed | 1;
|
||||||
|
let mut idxs = vec![0u32; loads];
|
||||||
|
for _ in 0..samples {
|
||||||
|
for x in idxs.iter_mut() { s = (s ^ (s >> 30)).wrapping_mul(0xBF58476D1CE4E5B9); s = (s ^ (s >> 27)).wrapping_mul(0x94D049BB133111EB); s ^= s >> 31; *x = (s as u32) & MASK; }
|
||||||
|
let (r2, r8, ln, it) = distinct(&idxs);
|
||||||
|
h[0].add(r2); h[1].add(r8); h[2].add(ln); h[3].add(it);
|
||||||
|
}
|
||||||
|
h
|
||||||
|
}
|
||||||
|
|
||||||
|
fn cmd_diffuse(prog: &str, pairs: u64, plant: Plant) {
|
||||||
|
let (p, _day) = program_of(prog);
|
||||||
|
let (d0, d1) = (p.seed[0], p.seed[1]);
|
||||||
|
log!("diffuse: prog={prog} id={:#018x} pairs={pairs} plant={}", p.program_id(), plant.name());
|
||||||
|
// flip one random bit of H (and separately of nonce_hi); measure the Hamming weight of the change in the 4,096
|
||||||
|
// addresses of the unit. Full avalanche ~ 50% of address bits flip; a header with no path to the address shows ~0.
|
||||||
|
let mut rng = 0x1234_5678_9abc_def0u64;
|
||||||
|
let mut next = || { rng = (rng ^ (rng >> 30)).wrapping_mul(0xBF58476D1CE4E5B9); rng = (rng ^ (rng >> 27)).wrapping_mul(0x94D049BB133111EB); rng ^ (rng >> 31) };
|
||||||
|
let mut a = Vec::new(); let mut b = Vec::new();
|
||||||
|
let (mut sum_changed, mut n) = (0u64, 0u64);
|
||||||
|
for _ in 0..pairs {
|
||||||
|
let mut h = [0u8; 32]; for c in h.iter_mut() { *c = (next() & 0xff) as u8; }
|
||||||
|
let bit = (next() % 256) as usize;
|
||||||
|
let mut h2 = h; h2[bit / 8] ^= 1 << (bit % 8);
|
||||||
|
let i1 = block_init_words(&h, 0); let i2 = block_init_words(&h2, 0);
|
||||||
|
unit_addresses(&p, &i1, 0, d0, d1, None, plant, &mut a);
|
||||||
|
unit_addresses(&p, &i2, 0, d0, d1, None, plant, &mut b);
|
||||||
|
for (x, y) in a.iter().zip(b.iter()) { sum_changed += (x != y) as u64; n += 1; }
|
||||||
|
}
|
||||||
|
log!("one-bit H flip: {:.4} of the 4,096 unit addresses change (1.0 = every address moved; a header with no path would read ~0)", sum_changed as f64 / n as f64);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn cmd_draw_check() {
|
||||||
|
for sel in ["devnet", "devnet3"] {
|
||||||
|
let (p, day) = program_of(sel);
|
||||||
|
log!("{sel}: program_id={:#018x} attempt={} generator={} day={} sites={:?}",
|
||||||
|
p.program_id(), p.attempt, p.generator, day, load_sites(&p));
|
||||||
|
}
|
||||||
|
for k in 0..3 { let (p, _) = program_of(&format!("drawn:{k}")); log!("drawn:{k}: id={:#018x} attempt={}", p.program_id(), p.attempt); }
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Repeats: which (site_i, site_j) pairs of one lane's 128 loads read the same word index (the per-program structural
|
||||||
|
/// repeat drawn:0 shows), counted over `hashes` hashes, with the iteration pair. A header-independent repeat is a
|
||||||
|
/// program property the rule's (c) distinct-address mean (above 120 of 128) admits.
|
||||||
|
fn cmd_repeats(prog: &str, hashes: u64) {
|
||||||
|
let (p, _) = program_of(prog);
|
||||||
|
let (d0, d1) = (p.seed[0], p.seed[1]);
|
||||||
|
let units = hashes.div_ceil(LANES as u64);
|
||||||
|
let mut pairs: std::collections::HashMap<(usize, usize), u64> = std::collections::HashMap::new();
|
||||||
|
let mut addrs = Vec::new();
|
||||||
|
let mut total = 0u64;
|
||||||
|
for u in 0..units {
|
||||||
|
let mut h = [0u8; 32];
|
||||||
|
let mut s = u.wrapping_mul(0x9E3779B97F4A7C15).wrapping_add(0xD1B54A32D192ED03);
|
||||||
|
for c in h.chunks_mut(8) { s = (s ^ (s >> 30)).wrapping_mul(0xBF58476D1CE4E5B9); s = (s ^ (s >> 27)).wrapping_mul(0x94D049BB133111EB); s ^= s >> 31; c.copy_from_slice(&s.to_le_bytes()); }
|
||||||
|
let init = block_init_words(&h, 0);
|
||||||
|
unit_addresses(&p, &init, 0, d0, d1, None, Plant::None, &mut addrs);
|
||||||
|
for lane in 0..LANES {
|
||||||
|
// load k of this lane is addrs[k * 32 + lane]; k = iteration * 16 + site
|
||||||
|
for a in 0..128 { for b in a + 1..128 {
|
||||||
|
if addrs[a * LANES + lane] == addrs[b * LANES + lane] { *pairs.entry((a, b)).or_insert(0) += 1; total += 1; }
|
||||||
|
} }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let mut v: Vec<_> = pairs.into_iter().collect();
|
||||||
|
v.sort_by(|x, y| y.1.cmp(&x.1));
|
||||||
|
log!("repeats: prog={prog} id={:#018x}: {total} same-word repeats over {} hashes ({:.5} per hash); top pairs (load k = iteration*16 + site):", p.program_id(), units * LANES as u64, total as f64 / (units * LANES as u64) as f64);
|
||||||
|
for ((a, b), c) in v.iter().take(12) {
|
||||||
|
let (ia, sa, ib, sb) = (a / 16, a % 16, b / 16, b % 16);
|
||||||
|
let sites = load_sites(&p);
|
||||||
|
println!(" loads {a:3} and {b:3}: iteration {ia} site {sa} (instr {}) with iteration {ib} site {sb} (instr {}): {c} repeats", sites[sa], sites[sb]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The instruction listing of a program (base program only), for the mechanism of a repeat.
|
||||||
|
fn cmd_dump(prog: &str, upto: usize) {
|
||||||
|
let (p, _) = program_of(prog);
|
||||||
|
log!("dump: prog={prog} id={:#018x} attempt={} era M={:#010x} R={} pos={:?}", p.program_id(), p.attempt,
|
||||||
|
p.class.era.map(|e| e.stride_mul).unwrap_or(0), p.class.era.map(|e| e.stride_rot).unwrap_or(0), p.class.era.map(|e| e.pos).unwrap_or([0; 4]));
|
||||||
|
for (k, i) in p.instrs.iter().enumerate().take(upto) {
|
||||||
|
println!("{k:3}: {:6} dst={} src={} src2={} imm={:#010x} imm2={:#010x} rot={:2} bit={:2} mask={:2} win={} off={}", i.op.name(), i.dst, i.src, i.src2, i.imm, i.imm2, i.rot, i.bit, i.mask, i.win, i.off);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Queue 92 (owner adv-accept-3, unspawned): program-id determinism. (1) FNV-1a 64 program ids over `pairs` (seed,
|
||||||
|
/// attempt) pairs through the library's program_id_class for class v4: sorted, any duplicate reported. (2) Verdict
|
||||||
|
/// stability: accept::check evaluated twice on `seeds` class v4 candidates (attempt 0, the chain's era draw), and the
|
||||||
|
/// chosen attempt of the full draw re-derived, every verdict and attempt compared. (3) A code read for
|
||||||
|
/// platform-dependent behaviour is in the report (the one f64 in distinct_ratio_pass).
|
||||||
|
fn cmd_idcheck(pairs: u64, seeds: u64) {
|
||||||
|
use igneum_pow::generator::{program_id_class, V4_CLASS, GENERATOR_VERSION_V4, attempt_words};
|
||||||
|
let t0 = Instant::now();
|
||||||
|
let mut ids: Vec<u64> = Vec::with_capacity(pairs as usize);
|
||||||
|
let per_seed = 16u32;
|
||||||
|
let n_seeds = pairs / per_seed as u64;
|
||||||
|
for k in 0..n_seeds {
|
||||||
|
let sb: Vec<u8> = seed_words_from_bytes(format!("igneum-adv-accept-2/id/{k}").as_bytes()).iter().flat_map(|w| w.to_le_bytes()).collect();
|
||||||
|
for a in 0..per_seed {
|
||||||
|
let words = attempt_words(&sb, a);
|
||||||
|
ids.push(program_id_class(GENERATOR_VERSION_V4, &words, a, &V4_CLASS));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
ids.sort_unstable();
|
||||||
|
let dups = ids.windows(2).filter(|w| w[0] == w[1]).count();
|
||||||
|
log!("idcheck (1): {} program ids over {} seeds x {} attempts: {} duplicates (birthday expectation {:.2e}) in {:.1}s", ids.len(), n_seeds, per_seed, dups, (ids.len() as f64).powi(2) / 2.0 / 2f64.powi(64), t0.elapsed().as_secs_f64());
|
||||||
|
// known-fail shape for (1): the same ids with the generator word forced to 3 must ALL differ from the generator-4 ids
|
||||||
|
let mut cross = 0u64;
|
||||||
|
for k in 0..1000u64 {
|
||||||
|
let sb: Vec<u8> = seed_words_from_bytes(format!("igneum-adv-accept-2/id/{k}").as_bytes()).iter().flat_map(|w| w.to_le_bytes()).collect();
|
||||||
|
let w = attempt_words(&sb, 0);
|
||||||
|
if program_id_class(GENERATOR_VERSION_V4, &w, 0, &V4_CLASS) == program_id_class(3, &w, 0, &V4_CLASS) { cross += 1; }
|
||||||
|
}
|
||||||
|
log!("idcheck (1) plant: generator 3 vs 4 ids equal on {cross} of 1000 seeds (must be 0; the id binds the generator word)");
|
||||||
|
// (2) verdict stability on real chain-path candidates
|
||||||
|
let t1 = Instant::now();
|
||||||
|
let (mut mismatch, mut accepted, mut attempt_mismatch) = (0u64, 0u64, 0u64);
|
||||||
|
let mut reasons: std::collections::BTreeMap<String, u64> = std::collections::BTreeMap::new();
|
||||||
|
for k in 0..seeds {
|
||||||
|
let epoch = seed_words_from_bytes(format!("igneum-adv-accept-2/vd/epoch/{k}").as_bytes());
|
||||||
|
let era = seed_words_from_bytes(format!("igneum-adv-accept-2/vd/era/{k}").as_bytes());
|
||||||
|
let eb: Vec<u8> = epoch.iter().flat_map(|w| w.to_le_bytes()).collect();
|
||||||
|
let erab: Vec<u8> = era.iter().flat_map(|w| w.to_le_bytes()).collect();
|
||||||
|
let class = igneum_pow::generator::LoadClass::era(V4_CLASS, &erab, &igneum_pow::generator::V3_ALLOWED);
|
||||||
|
let c0 = igneum_pow::generator::candidate_class("vd", &eb, 0, class);
|
||||||
|
let v1 = igneum_pow::accept::check(&c0).map(|_| ()).map_err(|e| format!("{e:?}"));
|
||||||
|
let v2 = igneum_pow::accept::check(&c0).map(|_| ()).map_err(|e| format!("{e:?}"));
|
||||||
|
if v1 != v2 { mismatch += 1; }
|
||||||
|
let key = match &v1 { Ok(()) => { accepted += 1; "accepted".to_string() } Err(e) => e.split(['{', ' ']).next().unwrap_or("?").to_string() };
|
||||||
|
*reasons.entry(key).or_insert(0) += 1;
|
||||||
|
// the full draw's chosen attempt, re-derived twice, must agree
|
||||||
|
let p1 = Epoch::chain_program(&eb, Some(&erab), ProgramClass::V4, "vd");
|
||||||
|
let p2 = Epoch::chain_program(&eb, Some(&erab), ProgramClass::V4, "vd");
|
||||||
|
if p1.attempt != p2.attempt || p1.program_id() != p2.program_id() || p1.instrs != p2.instrs { attempt_mismatch += 1; }
|
||||||
|
}
|
||||||
|
log!("idcheck (2): {seeds} class v4 attempt-0 candidates checked twice: {mismatch} verdict disagreements; {accepted} accepted at attempt 0; full draw re-derived twice: {attempt_mismatch} disagreements; {:.1}s", t1.elapsed().as_secs_f64());
|
||||||
|
for (r, c) in &reasons { println!(" attempt-0 verdict {r}: {c}"); }
|
||||||
|
}
|
||||||
|
|
||||||
|
/// GPU point: export address tables for the card measurement. Writes `units` units of 4,096 u32 word indices
|
||||||
|
/// (lane-minor within a load, load-major: index [k * 32 + lane], k = iteration * 16 + site) as raw little-endian u32
|
||||||
|
/// to `out`, for the program and plant given. With `--best`, the sweep runs over `hashes` headers first and exports the
|
||||||
|
/// `units` units with the FEWEST distinct 8 KiB rows (the ground groups a header search would pick).
|
||||||
|
fn cmd_export(prog: &str, hashes: u64, units: usize, plant: Plant, best: bool, out: &str) {
|
||||||
|
let (p, _) = program_of(prog);
|
||||||
|
let (d0, d1) = (p.seed[0], p.seed[1]);
|
||||||
|
let header = |u: u64| -> [u8; 32] {
|
||||||
|
let mut h = [0u8; 32];
|
||||||
|
let mut s = u.wrapping_mul(0x9E3779B97F4A7C15).wrapping_add(0xD1B54A32D192ED03);
|
||||||
|
for c in h.chunks_mut(8) { s = (s ^ (s >> 30)).wrapping_mul(0xBF58476D1CE4E5B9); s = (s ^ (s >> 27)).wrapping_mul(0x94D049BB133111EB); s ^= s >> 31; c.copy_from_slice(&s.to_le_bytes()); }
|
||||||
|
h
|
||||||
|
};
|
||||||
|
let mut addrs = Vec::new();
|
||||||
|
let chosen: Vec<u64> = if best {
|
||||||
|
let n = hashes.div_ceil(LANES as u64);
|
||||||
|
let mut scored: Vec<(u32, u64)> = Vec::with_capacity(n as usize);
|
||||||
|
for u in 0..n {
|
||||||
|
unit_addresses(&p, &block_init_words(&header(u), 0), 0, d0, d1, None, plant, &mut addrs);
|
||||||
|
let (_, r8, _, _) = distinct(&addrs);
|
||||||
|
scored.push((r8, u));
|
||||||
|
}
|
||||||
|
scored.sort_unstable();
|
||||||
|
log!("export: best {} of {} units by distinct 8 KiB rows: rows8 from {} to {} (median unit {})", units, n, scored[0].0, scored[units - 1].0, scored[scored.len() / 2].0);
|
||||||
|
scored.iter().take(units).map(|x| x.1).collect()
|
||||||
|
} else { (0..units as u64).collect() };
|
||||||
|
let mut bytes = Vec::with_capacity(units * 4096 * 4);
|
||||||
|
let mut r8sum = 0u64;
|
||||||
|
for &u in &chosen {
|
||||||
|
unit_addresses(&p, &block_init_words(&header(u), 0), 0, d0, d1, None, plant, &mut addrs);
|
||||||
|
assert_eq!(addrs.len(), 4096);
|
||||||
|
let (_, r8, _, _) = distinct(&addrs); r8sum += r8 as u64;
|
||||||
|
for &a in &addrs { bytes.extend_from_slice(&a.to_le_bytes()); }
|
||||||
|
}
|
||||||
|
std::fs::write(out, &bytes).unwrap();
|
||||||
|
log!("export: wrote {} units ({} bytes) to {out}, prog={prog} plant={} best={best}, mean distinct 8 KiB rows {:.2}", units, bytes.len(), plant.name(), r8sum as f64 / units as f64);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Prevalence of the rotate-identity repeat class (the drawn:0 finding): over `count` drawn class v4 programs through
|
||||||
|
/// the chain draw path (accepted programs), count ordered pairs of load sites (i before j, cyclic over the 64 base
|
||||||
|
/// instructions plus the shadow block in execution order) that read the SAME source register where every write to that
|
||||||
|
/// register between them is a `rotr` (identity when its amount register AND 31 is 0: probability 1/32 each). Any other
|
||||||
|
/// op between them (add, sub, xor, or, mul, mulhi, mad, shfl, load, rotl by 1..31) changes the value except with
|
||||||
|
/// probability about 2^-32 and ends the chain. Reports pairs, the identity probability per pair, and programs affected.
|
||||||
|
fn cmd_rotclass(count: u64) {
|
||||||
|
let (mut affected, mut pairs_total) = (0u64, 0u64);
|
||||||
|
let mut by_k: std::collections::BTreeMap<usize, u64> = std::collections::BTreeMap::new();
|
||||||
|
for k in 0..count {
|
||||||
|
let (p, _) = program_of(&format!("drawn:{k}"));
|
||||||
|
// one iteration in execution order: base then shadow x reps; a second pass closes the wrap
|
||||||
|
let order: Vec<&igneum_pow::generator::Instr> = p.instrs.iter().chain((0..p.shadow_reps()).flat_map(|_| p.shadow.iter())).collect();
|
||||||
|
let mut found = Vec::new();
|
||||||
|
for (ai, a) in order.iter().enumerate() {
|
||||||
|
if !a.op.is_load() { continue; }
|
||||||
|
let reg = a.src;
|
||||||
|
let mut rotrs = 0usize;
|
||||||
|
let n = order.len();
|
||||||
|
for step in 1..n {
|
||||||
|
let b = order[(ai + step) % n];
|
||||||
|
if b.op.is_load() && b.src == reg {
|
||||||
|
found.push((ai, (ai + step) % n, rotrs));
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
if b.dst == reg {
|
||||||
|
if b.op == Op::Rotr { rotrs += 1; continue; }
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// a pair with zero rotr writes cannot exist (rule (a) forces a write between two loads from one register)
|
||||||
|
let real: Vec<_> = found.iter().filter(|(_, _, r)| *r > 0).collect();
|
||||||
|
if !real.is_empty() { affected += 1; }
|
||||||
|
for (a, b, r) in &real {
|
||||||
|
pairs_total += 1;
|
||||||
|
*by_k.entry(*r).or_insert(0) += 1;
|
||||||
|
if real.len() <= 4 { println!(" drawn:{k} id={:#018x} attempt={}: loads at order {a} and {b} read one register with {r} rotr write(s) between: identity probability 32^-{r}", p.program_id(), p.attempt); }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
log!("rotclass: {count} drawn accepted class v4 programs: {affected} affected ({:.2} percent), {pairs_total} load-site pairs whose only intervening writes are rotr; by rotr count: {:?}", 100.0 * affected as f64 / count as f64, by_k);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Q1b: the predictable prefix. In iteration 0, a load site whose source register has no dataflow path from an earlier
|
||||||
|
/// load's result can be addressed from (I, n) with ALU work alone. Taint: a load taints its dst; add/sub/xor/mad/shfl/
|
||||||
|
/// or/mul/mulhi/rotr propagate taint from any operand (shfl from the lane group); rotl keeps the dst's taint. Reported
|
||||||
|
/// per program: the count of load-independent sites in iteration 0 (every later iteration is fully tainted by the
|
||||||
|
/// shuffles and the 64-instruction mixing, checked by the same walk).
|
||||||
|
fn cmd_prefix(sels: &[String]) {
|
||||||
|
for sel in sels {
|
||||||
|
let (p, _) = program_of(sel);
|
||||||
|
let mut tainted = [false; 8];
|
||||||
|
let mut free_sites = Vec::new();
|
||||||
|
let mut site = 0usize;
|
||||||
|
for it in 0..ITERATIONS {
|
||||||
|
let run = p.instrs.iter().chain((0..p.shadow_reps()).flat_map(|_| p.shadow.iter()));
|
||||||
|
for (k, ins) in run.enumerate() {
|
||||||
|
let (d, a) = (ins.dst as usize, ins.src as usize);
|
||||||
|
match ins.op {
|
||||||
|
Op::Load => { if !tainted[a] { free_sites.push((it, k, site)); } tainted[d] = true; site += 1; }
|
||||||
|
Op::Rotl => {}
|
||||||
|
Op::Mad => { tainted[d] = tainted[d] || tainted[a] || tainted[ins.src2 as usize]; }
|
||||||
|
_ => { tainted[d] = tainted[d] || tainted[a]; }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let it0 = free_sites.iter().filter(|(it, _, _)| *it == 0).count();
|
||||||
|
log!("prefix: prog={sel} id={:#018x}: load-independent sites in iteration 0 = {it0} of 16 (instructions {:?}); in later iterations = {}",
|
||||||
|
p.program_id(), free_sites.iter().filter(|(it, _, _)| *it == 0).map(|(_, k, _)| *k).collect::<Vec<_>>(), free_sites.len() - it0);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn cmd_price() {
|
||||||
|
println!("Q3 price model (internal adversarial pass, not an independent review)");
|
||||||
|
println!("A card is bound by random 4-byte reads: rate R_hash = (reads/s) / loads_per_hash, loads_per_hash = 128.");
|
||||||
|
println!("A grind finds a header whose hash saves dL loads (distinct items below 128). On the card the found hash");
|
||||||
|
println!("then costs 128 - dL useful reads, but every searched header is itself a full 128-read hash evaluation.");
|
||||||
|
println!("If the saving dL >= t appears with probability 1/S (the tail rate), one found hash costs S search hashes,");
|
||||||
|
println!("each 128 reads, and yields one useful hash of 128 - dL reads. Net rate vs honest:");
|
||||||
|
println!(" gain = 128 / ((128 - dL) + S * 128) (the search reads amortise over one found hash only; a found");
|
||||||
|
println!(" header mines ONE 32-lane group, not a stream, because the address set is fixed by (program, I, g)).");
|
||||||
|
println!();
|
||||||
|
println!("tail rate 1/S dL saved useful reads net rate vs honest over 1%?");
|
||||||
|
for (s, dl) in [(1e3, 8.0), (1e4, 16.0), (1e5, 32.0), (1e4, 2.0), (1e5, 4.0)] {
|
||||||
|
let gain = 128.0 / ((128.0 - dl) + s * 128.0);
|
||||||
|
println!(" {:>8.0} {:>5.0} {:>6.0} {:.6}x {}", s, dl, 128.0 - dl, gain, if gain > 1.01 { "YES" } else { "no" });
|
||||||
|
}
|
||||||
|
println!();
|
||||||
|
println!("Even a 32-load saving at the 1e-5 tail nets 128 / (96 + 1e5*128) = 1.0e-5x: the search cost dwarfs the");
|
||||||
|
println!("saving by five orders. A found header mines one group, so the search never amortises. GPU confirmation");
|
||||||
|
println!("is BLOCKED tonight (no card); the bound is analytic from the read counts and holds for any dL < 128.");
|
||||||
|
}
|
||||||
|
|
||||||
|
fn usage() -> ! { eprintln!("adv-accept-2 draw-check | diffuse [--prog s] [--pairs N] | rows --prog s --hashes N [--shard k/of] [--plant P] [--live] | price"); std::process::exit(2); }
|
||||||
|
|
||||||
|
fn main() {
|
||||||
|
let args: Vec<String> = std::env::args().collect();
|
||||||
|
if args.len() < 2 { usage(); }
|
||||||
|
if args[1] == "prefix" { cmd_prefix(&args[2..]); return; }
|
||||||
|
let mut prog = "devnet".to_string(); let mut hashes = 1_000_000u64; let mut shard = (0u64, 1u64);
|
||||||
|
let mut plant = Plant::None; let mut live = false; let mut pairs = 4096u64;
|
||||||
|
let mut units_arg = 4096usize; let mut best = false; let mut out_arg = "units.bin".to_string();
|
||||||
|
let mut i = 2;
|
||||||
|
while i < args.len() {
|
||||||
|
match args[i].as_str() {
|
||||||
|
"--prog" => { i += 1; prog = args[i].clone(); }
|
||||||
|
"--hashes" => { i += 1; hashes = args[i].parse().unwrap(); }
|
||||||
|
"--pairs" => { i += 1; pairs = args[i].parse().unwrap(); }
|
||||||
|
"--shard" => { i += 1; let (a, b) = args[i].split_once('/').unwrap(); shard = (a.parse().unwrap(), b.parse().unwrap()); }
|
||||||
|
"--plant" => { i += 1; plant = Plant::parse(&args[i]); }
|
||||||
|
"--live" => { live = true; }
|
||||||
|
"--units" => { i += 1; units_arg = args[i].parse().unwrap(); }
|
||||||
|
"--best" => { best = true; }
|
||||||
|
"--out" => { i += 1; out_arg = args[i].clone(); }
|
||||||
|
_ => usage(),
|
||||||
|
}
|
||||||
|
i += 1;
|
||||||
|
}
|
||||||
|
match args[1].as_str() {
|
||||||
|
"draw-check" => cmd_draw_check(),
|
||||||
|
"diffuse" => cmd_diffuse(&prog, pairs, plant),
|
||||||
|
"rows" => cmd_rows(&prog, hashes, shard, plant, live),
|
||||||
|
"price" => cmd_price(),
|
||||||
|
"prefix" => cmd_prefix(&args[2..]),
|
||||||
|
"repeats" => cmd_repeats(&prog, hashes),
|
||||||
|
"dump" => cmd_dump(&prog, 64),
|
||||||
|
"rotclass" => cmd_rotclass(hashes),
|
||||||
|
"export" => cmd_export(&prog, hashes, units_arg, plant, best, &out_arg),
|
||||||
|
"idcheck" => cmd_idcheck(hashes, pairs),
|
||||||
|
_ => usage(),
|
||||||
|
}
|
||||||
|
}
|
||||||
Loading…
Reference in a new issue