Compare commits
38 commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c9af680ddd | ||
|
|
ce4f0d574a | ||
|
|
605284149d | ||
|
|
986af4f638 | ||
|
|
86c89551ba | ||
|
|
968b8dff21 | ||
|
|
118b243195 | ||
|
|
7bafd08cba | ||
|
|
3647ee02c9 | ||
|
|
cb8001f39e | ||
|
|
d7d92a0c83 | ||
|
|
eb66e05055 | ||
|
|
8f305b24be | ||
|
|
3a7ba78349 | ||
|
|
82e9965a3c | ||
|
|
eb5fee00e6 | ||
|
|
50f17ac523 | ||
|
|
dc92bb46cc | ||
|
|
2618ea3acf | ||
|
|
7f2bc3821c | ||
|
|
65c506a295 | ||
|
|
2c13897808 | ||
|
|
cd32d5a3b9 | ||
|
|
acfad7182c | ||
|
|
646600cef4 | ||
|
|
4ded5bcf37 | ||
|
|
cd5d2ca6e6 | ||
|
|
3a9bd0873a | ||
|
|
76b7cee8d6 | ||
|
|
ba799d4750 | ||
|
|
f63ee1be4a | ||
|
|
659a7be280 | ||
|
|
661e800ce7 | ||
|
|
41f055abc5 | ||
|
|
6b17c79087 | ||
|
|
851fbf6ee5 | ||
|
|
0a5e743411 | ||
|
|
7af226d5c5 |
93 changed files with 17806 additions and 140 deletions
8
docs/design/class-v5-bench/20261007T083716Z/alone-v4.txt
Normal file
8
docs/design/class-v5-bench/20261007T083716Z/alone-v4.txt
Normal file
|
|
@ -0,0 +1,8 @@
|
|||
igneum-pow bench: seed "igneum-genesis", day "2026-10-03", dataset 2^28 words (memory-hard)
|
||||
cache: fill 540.7 ms on one core (2^26 words, 256 MiB, 65536 chains of 64 ChaCha12 blocks), FNV-1a 64 48c4f5bf24166b2e
|
||||
program: class mx8+sh256x27, 128 loads/hash, 512 bytes/hash, widths (1,4,16 words) [16, 0, 0], 4096 items/warp, mixer x8 (72 mixers/item), cache 2^26 words, op mix load=16 add=8 shfl=8 xor=6 mad=5 mul=5 mulhi=5 sub=4 rotl=3 rotr=3 or=1; epoch built in 536.5 ms
|
||||
shadow: 256 instructions x 27 passes per iteration, 55296 shadow instructions per hash, op mix add=47 rotl=30 xor=30 shfl=29 mad=27 mul=22 sub=21 rotr=20 mulhi=18 or=12
|
||||
warp base 0: single cold run 8.110 ms, 4096 items derived, lane0 2576769ee4a14c8d lane31 c58ddcb717dd3370
|
||||
warp base 4096: single cold run 7.636 ms, 4096 items derived, lane0 1ce77a600ec573b4 lane31 03600a05ffba0055
|
||||
warp base 1000000: single cold run 7.620 ms, 4096 items derived, lane0 6b390e64bbdd91ce lane31 91c944d603539c62
|
||||
CPU verify: 7.830 ms per 32-lane warp, avg of 50 (checksum 17e36e7905b81375)
|
||||
9
docs/design/class-v5-bench/20261007T083716Z/alone-v5.txt
Normal file
9
docs/design/class-v5-bench/20261007T083716Z/alone-v5.txt
Normal file
|
|
@ -0,0 +1,9 @@
|
|||
igneum-pow bench: seed "igneum-genesis", day "2026-10-03", dataset 2^28 words (memory-hard)
|
||||
cache: fill 551.4 ms on one core (2^26 words, 256 MiB, 65536 chains of 64 ChaCha12 blocks), FNV-1a 64 48c4f5bf24166b2e
|
||||
state stream /srv/builds/_log/v5-class/node1-state.igsd1: chain block 159357 af89be5ddbadb6f6b4aee28ac8f249713be5d4c12621e3cea7f83ceada3c66b3, root 1c583d352bb9c75a06dadb8d82d42836ebe8afa82d0b413be87bf921741f1526, 93 records, 93 leaves
|
||||
program: class mx8+sh256x27+state, 128 loads/hash, 512 bytes/hash, widths (1,4,16 words) [16, 0, 0], 4096 items/warp, mixer x8 (72 mixers/item), cache 2^26 words, op mix load=16 add=8 shfl=8 xor=6 mad=5 mul=5 mulhi=5 sub=4 rotl=3 rotr=3 or=1; epoch built in 555.1 ms
|
||||
shadow: 256 instructions x 27 passes per iteration, 55296 shadow instructions per hash, op mix add=47 rotl=30 xor=30 shfl=29 mad=27 mul=22 sub=21 rotr=20 mulhi=18 or=12
|
||||
warp base 0: single cold run 8.712 ms, 4096 items derived, lane0 18d47ad5e5f0db00 lane31 d1b36ed0ff68e575
|
||||
warp base 4096: single cold run 8.478 ms, 4095 items derived, lane0 a1f55f0b0cae0ed2 lane31 8c548231ef24eb73
|
||||
warp base 1000000: single cold run 8.549 ms, 4096 items derived, lane0 bd47078fc688b540 lane31 01f4d9a0bef81008
|
||||
CPU verify: 8.538 ms per 32-lane warp, avg of 50 (checksum 60bcc82b11d41f0f)
|
||||
|
|
@ -0,0 +1,8 @@
|
|||
igneum-pow bench: seed "igneum-genesis", day "2026-10-03", dataset 2^28 words (memory-hard)
|
||||
cache: fill 589.9 ms on one core (2^26 words, 256 MiB, 65536 chains of 64 ChaCha12 blocks), FNV-1a 64 48c4f5bf24166b2e
|
||||
program: class mx8+sh256x27, 128 loads/hash, 512 bytes/hash, widths (1,4,16 words) [16, 0, 0], 4096 items/warp, mixer x8 (72 mixers/item), cache 2^26 words, op mix load=16 add=8 shfl=8 xor=6 mad=5 mul=5 mulhi=5 sub=4 rotl=3 rotr=3 or=1; epoch built in 556.0 ms
|
||||
shadow: 256 instructions x 27 passes per iteration, 55296 shadow instructions per hash, op mix add=47 rotl=30 xor=30 shfl=29 mad=27 mul=22 sub=21 rotr=20 mulhi=18 or=12
|
||||
warp base 0: single cold run 8.744 ms, 4096 items derived, lane0 2576769ee4a14c8d lane31 c58ddcb717dd3370
|
||||
warp base 4096: single cold run 8.694 ms, 4096 items derived, lane0 1ce77a600ec573b4 lane31 03600a05ffba0055
|
||||
warp base 1000000: single cold run 8.738 ms, 4096 items derived, lane0 6b390e64bbdd91ce lane31 91c944d603539c62
|
||||
CPU verify: 8.629 ms per 32-lane warp, avg of 50 (checksum 17e36e7905b81375)
|
||||
|
|
@ -0,0 +1,9 @@
|
|||
igneum-pow bench: seed "igneum-genesis", day "2026-10-03", dataset 2^28 words (memory-hard)
|
||||
cache: fill 566.6 ms on one core (2^26 words, 256 MiB, 65536 chains of 64 ChaCha12 blocks), FNV-1a 64 48c4f5bf24166b2e
|
||||
state stream /srv/builds/_log/v5-class/node1-state.igsd1: chain block 159357 af89be5ddbadb6f6b4aee28ac8f249713be5d4c12621e3cea7f83ceada3c66b3, root 1c583d352bb9c75a06dadb8d82d42836ebe8afa82d0b413be87bf921741f1526, 93 records, 93 leaves
|
||||
program: class mx8+sh256x27+state, 128 loads/hash, 512 bytes/hash, widths (1,4,16 words) [16, 0, 0], 4096 items/warp, mixer x8 (72 mixers/item), cache 2^26 words, op mix load=16 add=8 shfl=8 xor=6 mad=5 mul=5 mulhi=5 sub=4 rotl=3 rotr=3 or=1; epoch built in 588.0 ms
|
||||
shadow: 256 instructions x 27 passes per iteration, 55296 shadow instructions per hash, op mix add=47 rotl=30 xor=30 shfl=29 mad=27 mul=22 sub=21 rotr=20 mulhi=18 or=12
|
||||
warp base 0: single cold run 9.025 ms, 4096 items derived, lane0 18d47ad5e5f0db00 lane31 d1b36ed0ff68e575
|
||||
warp base 4096: single cold run 8.877 ms, 4095 items derived, lane0 a1f55f0b0cae0ed2 lane31 8c548231ef24eb73
|
||||
warp base 1000000: single cold run 8.955 ms, 4096 items derived, lane0 bd47078fc688b540 lane31 01f4d9a0bef81008
|
||||
CPU verify: 8.833 ms per 32-lane warp, avg of 50 (checksum 60bcc82b11d41f0f)
|
||||
2
docs/design/class-v5-bench/20261007T083716Z/meta.txt
Normal file
2
docs/design/class-v5-bench/20261007T083716Z/meta.txt
Normal file
|
|
@ -0,0 +1,2 @@
|
|||
host igneum-build-1 load 70.78 56.14 34.92 freq40 3724858 kHz siblings 40,88 pow 978c9a64d265746f state abb5800350b02dd9
|
||||
load after 69.49 56.35 35.22
|
||||
|
|
@ -0,0 +1,8 @@
|
|||
igneum-pow bench: seed "igneum-genesis", day "2026-10-03", dataset 2^28 words (memory-hard)
|
||||
cache: fill 546.6 ms on one core (2^26 words, 256 MiB, 65536 chains of 64 ChaCha12 blocks), FNV-1a 64 48c4f5bf24166b2e
|
||||
program: class mx8+sh256x27, 128 loads/hash, 512 bytes/hash, widths (1,4,16 words) [16, 0, 0], 4096 items/warp, mixer x8 (72 mixers/item), cache 2^26 words, op mix load=16 add=8 shfl=8 xor=6 mad=5 mul=5 mulhi=5 sub=4 rotl=3 rotr=3 or=1; epoch built in 542.9 ms
|
||||
shadow: 256 instructions x 27 passes per iteration, 55296 shadow instructions per hash, op mix add=47 rotl=30 xor=30 shfl=29 mad=27 mul=22 sub=21 rotr=20 mulhi=18 or=12
|
||||
warp base 0: single cold run 8.165 ms, 4096 items derived, lane0 2576769ee4a14c8d lane31 c58ddcb717dd3370
|
||||
warp base 4096: single cold run 8.058 ms, 4096 items derived, lane0 1ce77a600ec573b4 lane31 03600a05ffba0055
|
||||
warp base 1000000: single cold run 8.080 ms, 4096 items derived, lane0 6b390e64bbdd91ce lane31 91c944d603539c62
|
||||
CPU verify: 10.536 ms per 32-lane warp, avg of 4000 (checksum 5abf8f93f330d440)
|
||||
|
|
@ -0,0 +1,8 @@
|
|||
igneum-pow bench: seed "igneum-genesis", day "2026-10-03", dataset 2^28 words (memory-hard)
|
||||
cache: fill 757.6 ms on one core (2^26 words, 256 MiB, 65536 chains of 64 ChaCha12 blocks), FNV-1a 64 48c4f5bf24166b2e
|
||||
state stream /srv/builds/_log/v5-class/node1-state.igsd1: chain block 159357 af89be5ddbadb6f6b4aee28ac8f249713be5d4c12621e3cea7f83ceada3c66b3, root 1c583d352bb9c75a06dadb8d82d42836ebe8afa82d0b413be87bf921741f1526, 93 records, 93 leaves
|
||||
program: class mx8+sh256x27+state, 128 loads/hash, 512 bytes/hash, widths (1,4,16 words) [16, 0, 0], 4096 items/warp, mixer x8 (72 mixers/item), cache 2^26 words, op mix load=16 add=8 shfl=8 xor=6 mad=5 mul=5 mulhi=5 sub=4 rotl=3 rotr=3 or=1; epoch built in 1062.6 ms
|
||||
shadow: 256 instructions x 27 passes per iteration, 55296 shadow instructions per hash, op mix add=47 rotl=30 xor=30 shfl=29 mad=27 mul=22 sub=21 rotr=20 mulhi=18 or=12
|
||||
warp base 0: single cold run 17.767 ms, 4096 items derived, lane0 18d47ad5e5f0db00 lane31 d1b36ed0ff68e575
|
||||
warp base 4096: single cold run 17.585 ms, 4095 items derived, lane0 a1f55f0b0cae0ed2 lane31 8c548231ef24eb73
|
||||
warp base 1000000: single cold run 14.724 ms, 4096 items derived, lane0 bd47078fc688b540 lane31 01f4d9a0bef81008
|
||||
73
docs/design/class-v5-bench/pod-4090-20261007/run.log
Normal file
73
docs/design/class-v5-bench/pod-4090-20261007/run.log
Normal file
|
|
@ -0,0 +1,73 @@
|
|||
NVIDIA GeForce RTX 4090, 595.91.07, 450.00 W
|
||||
../../bench.cu(130): warning #177-D: variable "dLeaves" was declared but never referenced
|
||||
uint32_t* dLeaves = nullptr;
|
||||
^
|
||||
Remark: The warnings can be suppressed with "-diag-suppress <warning-number>"
|
||||
../../bench.cu(131): warning #177-D: variable "nLeaves" was declared but never referenced
|
||||
uint32_t nLeaves = 0;
|
||||
^
|
||||
nvcc v4-genesis rc 0
|
||||
nvcc v5-genesis rc 0
|
||||
=== pass 1 v4-genesis 09:23:07Z
|
||||
class-v5 bench pack "igneum-genesis" class control generator 4 (test harness: no pool, no network, no wallet)
|
||||
GPU: NVIDIA GeForce RTX 4090 (128 SMs, cc 8.9, 24083 MiB), CUDA driver 13.2 runtime 12.8
|
||||
hash kernel: 32 registers/thread, 24 resident blocks/SM at 1 warp/block
|
||||
device memory at start: 395 MiB used of 24083 MiB
|
||||
cache fill (GPU): 1.89 ms first, 1.84 ms second
|
||||
cache head and last 16 words against the pack: PASS
|
||||
dataset build (GPU): 30.60 ms first, 30.58 ms second (16777216 items, 1024 MiB)
|
||||
device memory after the build: 1675 MiB used
|
||||
dataset self-test: head PASS, last PASS, samples 64 of 64
|
||||
vector warps against the pack: 3 of 3 PASS
|
||||
fingerprint of 2^24 outputs at base 0: 3d2e8245cc084d07 (lane 0 2576769ee4a14c8d)
|
||||
hash rate: 63.067 MH/s over 10 batches of 2^24 (GPU event time 2660.2 ms)
|
||||
power window: 20.2 s, 63.061 MH/s sustained, 269.4 W mean after the first 10 s (12 samples of 22), SM 2650 MHz, mem 10251 MHz, 0.234 MH/s per W, 4.27 microjoules per hash
|
||||
RESULT pack=igneum-genesis class=control build_ms=30.58 rate_mhs=63.067 vectors=3/3
|
||||
=== pass 1 v5-genesis 09:23:33Z
|
||||
class-v5 bench pack "igneum-genesis" class v5 generator 5 (test harness: no pool, no network, no wallet)
|
||||
GPU: NVIDIA GeForce RTX 4090 (128 SMs, cc 8.9, 24083 MiB), CUDA driver 13.2 runtime 12.8
|
||||
hash kernel: 32 registers/thread, 24 resident blocks/SM at 1 warp/block
|
||||
device memory at start: 395 MiB used of 24083 MiB
|
||||
cache fill (GPU): 1.92 ms first, 1.88 ms second
|
||||
cache head and last 16 words against the pack: PASS
|
||||
leaves: 93 x 64 B from leaves.bin (5952 bytes), FNV-1a 64 850ad094a937a5c5 against the pack's 850ad094a937a5c5: PASS; state root 1c583d352bb9c75a06dadb8d82d42836ebe8afa82d0b413be87bf921741f1526, chain block af89be5ddbadb6f6b4aee28ac8f249713be5d4c12621e3cea7f83ceada3c66b3 (159357)
|
||||
dataset build (GPU): 30.83 ms first, 30.80 ms second (16777216 items, 1024 MiB)
|
||||
device memory after the build: 1677 MiB used
|
||||
dataset self-test: head PASS, last PASS, samples 64 of 64
|
||||
vector warps against the pack: 3 of 3 PASS
|
||||
fingerprint of 2^24 outputs at base 0: e438ca2fc66b5cd4 (lane 0 18d47ad5e5f0db00)
|
||||
hash rate: 63.058 MH/s over 10 batches of 2^24 (GPU event time 2660.6 ms)
|
||||
power window: 20.2 s, 63.053 MH/s sustained, 275.8 W mean after the first 10 s (12 samples of 22), SM 2640 MHz, mem 10251 MHz, 0.229 MH/s per W, 4.37 microjoules per hash
|
||||
RESULT pack=igneum-genesis class=v5 build_ms=30.80 rate_mhs=63.058 vectors=3/3
|
||||
=== pass 2 v4-genesis 09:23:58Z
|
||||
class-v5 bench pack "igneum-genesis" class control generator 4 (test harness: no pool, no network, no wallet)
|
||||
GPU: NVIDIA GeForce RTX 4090 (128 SMs, cc 8.9, 24083 MiB), CUDA driver 13.2 runtime 12.8
|
||||
hash kernel: 32 registers/thread, 24 resident blocks/SM at 1 warp/block
|
||||
device memory at start: 395 MiB used of 24083 MiB
|
||||
cache fill (GPU): 1.88 ms first, 1.86 ms second
|
||||
cache head and last 16 words against the pack: PASS
|
||||
dataset build (GPU): 30.61 ms first, 30.56 ms second (16777216 items, 1024 MiB)
|
||||
device memory after the build: 1675 MiB used
|
||||
dataset self-test: head PASS, last PASS, samples 64 of 64
|
||||
vector warps against the pack: 3 of 3 PASS
|
||||
fingerprint of 2^24 outputs at base 0: 3d2e8245cc084d07 (lane 0 2576769ee4a14c8d)
|
||||
hash rate: 63.066 MH/s over 10 batches of 2^24 (GPU event time 2660.3 ms)
|
||||
power window: 20.2 s, 63.065 MH/s sustained, 275.9 W mean after the first 10 s (12 samples of 22), SM 2640 MHz, mem 10251 MHz, 0.229 MH/s per W, 4.37 microjoules per hash
|
||||
RESULT pack=igneum-genesis class=control build_ms=30.56 rate_mhs=63.066 vectors=3/3
|
||||
=== pass 2 v5-genesis 09:24:24Z
|
||||
class-v5 bench pack "igneum-genesis" class v5 generator 5 (test harness: no pool, no network, no wallet)
|
||||
GPU: NVIDIA GeForce RTX 4090 (128 SMs, cc 8.9, 24083 MiB), CUDA driver 13.2 runtime 12.8
|
||||
hash kernel: 32 registers/thread, 24 resident blocks/SM at 1 warp/block
|
||||
device memory at start: 395 MiB used of 24083 MiB
|
||||
cache fill (GPU): 1.90 ms first, 1.84 ms second
|
||||
cache head and last 16 words against the pack: PASS
|
||||
leaves: 93 x 64 B from leaves.bin (5952 bytes), FNV-1a 64 850ad094a937a5c5 against the pack's 850ad094a937a5c5: PASS; state root 1c583d352bb9c75a06dadb8d82d42836ebe8afa82d0b413be87bf921741f1526, chain block af89be5ddbadb6f6b4aee28ac8f249713be5d4c12621e3cea7f83ceada3c66b3 (159357)
|
||||
dataset build (GPU): 30.84 ms first, 30.79 ms second (16777216 items, 1024 MiB)
|
||||
device memory after the build: 1677 MiB used
|
||||
dataset self-test: head PASS, last PASS, samples 64 of 64
|
||||
vector warps against the pack: 3 of 3 PASS
|
||||
fingerprint of 2^24 outputs at base 0: e438ca2fc66b5cd4 (lane 0 18d47ad5e5f0db00)
|
||||
hash rate: 63.058 MH/s over 10 batches of 2^24 (GPU event time 2660.6 ms)
|
||||
power window: 20.2 s, 63.054 MH/s sustained, 284.2 W mean after the first 10 s (12 samples of 22), SM 2640 MHz, mem 10251 MHz, 0.222 MH/s per W, 4.51 microjoules per hash
|
||||
RESULT pack=igneum-genesis class=v5 build_ms=30.79 rate_mhs=63.058 vectors=3/3
|
||||
=== done 09:24:50Z
|
||||
63
docs/design/class-v5-bench/pod-4090-20261007/run60.log
Normal file
63
docs/design/class-v5-bench/pod-4090-20261007/run60.log
Normal file
|
|
@ -0,0 +1,63 @@
|
|||
=== 60s v5-genesis 09:25:35Z
|
||||
class-v5 bench pack "igneum-genesis" class v5 generator 5 (test harness: no pool, no network, no wallet)
|
||||
GPU: NVIDIA GeForce RTX 4090 (128 SMs, cc 8.9, 24083 MiB), CUDA driver 13.2 runtime 12.8
|
||||
hash kernel: 32 registers/thread, 24 resident blocks/SM at 1 warp/block
|
||||
device memory at start: 395 MiB used of 24083 MiB
|
||||
cache fill (GPU): 1.90 ms first, 1.85 ms second
|
||||
cache head and last 16 words against the pack: PASS
|
||||
leaves: 93 x 64 B from leaves.bin (5952 bytes), FNV-1a 64 850ad094a937a5c5 against the pack's 850ad094a937a5c5: PASS; state root 1c583d352bb9c75a06dadb8d82d42836ebe8afa82d0b413be87bf921741f1526, chain block af89be5ddbadb6f6b4aee28ac8f249713be5d4c12621e3cea7f83ceada3c66b3 (159357)
|
||||
dataset build (GPU): 30.84 ms first, 30.81 ms second (16777216 items, 1024 MiB)
|
||||
device memory after the build: 1677 MiB used
|
||||
dataset self-test: head PASS, last PASS, samples 64 of 64
|
||||
vector warps against the pack: 3 of 3 PASS
|
||||
fingerprint of 2^24 outputs at base 0: e438ca2fc66b5cd4 (lane 0 18d47ad5e5f0db00)
|
||||
hash rate: 63.052 MH/s over 10 batches of 2^24 (GPU event time 2660.8 ms)
|
||||
power window: 60.1 s, 63.052 MH/s sustained, 284.5 W mean after the first 10 s (52 samples of 62), SM 2640 MHz, mem 10251 MHz, 0.222 MH/s per W, 4.51 microjoules per hash
|
||||
RESULT pack=igneum-genesis class=v5 build_ms=30.81 rate_mhs=63.052 vectors=3/3
|
||||
=== 60s v4-genesis 09:26:40Z
|
||||
class-v5 bench pack "igneum-genesis" class control generator 4 (test harness: no pool, no network, no wallet)
|
||||
GPU: NVIDIA GeForce RTX 4090 (128 SMs, cc 8.9, 24083 MiB), CUDA driver 13.2 runtime 12.8
|
||||
hash kernel: 32 registers/thread, 24 resident blocks/SM at 1 warp/block
|
||||
device memory at start: 395 MiB used of 24083 MiB
|
||||
cache fill (GPU): 1.91 ms first, 1.89 ms second
|
||||
cache head and last 16 words against the pack: PASS
|
||||
dataset build (GPU): 30.62 ms first, 30.57 ms second (16777216 items, 1024 MiB)
|
||||
device memory after the build: 1675 MiB used
|
||||
dataset self-test: head PASS, last PASS, samples 64 of 64
|
||||
vector warps against the pack: 3 of 3 PASS
|
||||
fingerprint of 2^24 outputs at base 0: 3d2e8245cc084d07 (lane 0 2576769ee4a14c8d)
|
||||
hash rate: 63.064 MH/s over 10 batches of 2^24 (GPU event time 2660.4 ms)
|
||||
power window: 60.1 s, 63.063 MH/s sustained, 294.3 W mean after the first 10 s (52 samples of 62), SM 2640 MHz, mem 10251 MHz, 0.214 MH/s per W, 4.67 microjoules per hash
|
||||
RESULT pack=igneum-genesis class=control build_ms=30.57 rate_mhs=63.064 vectors=3/3
|
||||
=== 60s v5-genesis 09:27:46Z
|
||||
class-v5 bench pack "igneum-genesis" class v5 generator 5 (test harness: no pool, no network, no wallet)
|
||||
GPU: NVIDIA GeForce RTX 4090 (128 SMs, cc 8.9, 24083 MiB), CUDA driver 13.2 runtime 12.8
|
||||
hash kernel: 32 registers/thread, 24 resident blocks/SM at 1 warp/block
|
||||
device memory at start: 395 MiB used of 24083 MiB
|
||||
cache fill (GPU): 1.90 ms first, 1.86 ms second
|
||||
cache head and last 16 words against the pack: PASS
|
||||
leaves: 93 x 64 B from leaves.bin (5952 bytes), FNV-1a 64 850ad094a937a5c5 against the pack's 850ad094a937a5c5: PASS; state root 1c583d352bb9c75a06dadb8d82d42836ebe8afa82d0b413be87bf921741f1526, chain block af89be5ddbadb6f6b4aee28ac8f249713be5d4c12621e3cea7f83ceada3c66b3 (159357)
|
||||
dataset build (GPU): 30.84 ms first, 30.80 ms second (16777216 items, 1024 MiB)
|
||||
device memory after the build: 1677 MiB used
|
||||
dataset self-test: head PASS, last PASS, samples 64 of 64
|
||||
vector warps against the pack: 3 of 3 PASS
|
||||
fingerprint of 2^24 outputs at base 0: e438ca2fc66b5cd4 (lane 0 18d47ad5e5f0db00)
|
||||
hash rate: 63.059 MH/s over 10 batches of 2^24 (GPU event time 2660.6 ms)
|
||||
power window: 60.1 s, 63.055 MH/s sustained, 295.8 W mean after the first 10 s (52 samples of 62), SM 2640 MHz, mem 10251 MHz, 0.213 MH/s per W, 4.69 microjoules per hash
|
||||
RESULT pack=igneum-genesis class=v5 build_ms=30.80 rate_mhs=63.059 vectors=3/3
|
||||
=== 60s v4-genesis 09:28:52Z
|
||||
class-v5 bench pack "igneum-genesis" class control generator 4 (test harness: no pool, no network, no wallet)
|
||||
GPU: NVIDIA GeForce RTX 4090 (128 SMs, cc 8.9, 24083 MiB), CUDA driver 13.2 runtime 12.8
|
||||
hash kernel: 32 registers/thread, 24 resident blocks/SM at 1 warp/block
|
||||
device memory at start: 395 MiB used of 24083 MiB
|
||||
cache fill (GPU): 1.92 ms first, 1.85 ms second
|
||||
cache head and last 16 words against the pack: PASS
|
||||
dataset build (GPU): 30.59 ms first, 30.56 ms second (16777216 items, 1024 MiB)
|
||||
device memory after the build: 1675 MiB used
|
||||
dataset self-test: head PASS, last PASS, samples 64 of 64
|
||||
vector warps against the pack: 3 of 3 PASS
|
||||
fingerprint of 2^24 outputs at base 0: 3d2e8245cc084d07 (lane 0 2576769ee4a14c8d)
|
||||
hash rate: 63.065 MH/s over 10 batches of 2^24 (GPU event time 2660.3 ms)
|
||||
power window: 60.1 s, 63.064 MH/s sustained, 297.1 W mean after the first 10 s (52 samples of 62), SM 2640 MHz, mem 10251 MHz, 0.212 MH/s per W, 4.71 microjoules per hash
|
||||
RESULT pack=igneum-genesis class=control build_ms=30.56 rate_mhs=63.065 vectors=3/3
|
||||
=== done 09:29:58Z
|
||||
962
docs/design/class-v5-harness/failed-case.json
Normal file
962
docs/design/class-v5-harness/failed-case.json
Normal file
|
|
@ -0,0 +1,962 @@
|
|||
{
|
||||
"pass": false,
|
||||
"expect": "flip",
|
||||
"signals": [
|
||||
5,
|
||||
5,
|
||||
4
|
||||
],
|
||||
"checks": {
|
||||
"zero_rejected_by_miners": true,
|
||||
"zero_rejected_by_nodes": true,
|
||||
"sinks_agree": true,
|
||||
"block_counts_agree": true,
|
||||
"miners_agree_on_every_program": true,
|
||||
"window_line_on_every_node": true,
|
||||
"every_node_signals_its_byte": false,
|
||||
"chain_carries_the_bytes": true,
|
||||
"state_roots_agree_across_nodes": true,
|
||||
"template_switched_to_v5": false,
|
||||
"switched_at_the_first_full_window_epoch": false,
|
||||
"switched_before_the_floor": false,
|
||||
"signal_line_on_every_node_same_epoch": false,
|
||||
"signal_share_at_or_above_threshold": false,
|
||||
"blocks_on_both_sides": false,
|
||||
"v5_ids_equal_the_cli_v5_id": false,
|
||||
"v5_ids_differ_from_the_same_seed_v4_id": false,
|
||||
"a_v5_epoch_per_window_refresh": false,
|
||||
"stateless_node_mines_nothing_after_the_flip": false,
|
||||
"stale_miner_falls_off_at_the_first_refresh": true
|
||||
},
|
||||
"window": 60,
|
||||
"windows": 7,
|
||||
"floor": 100000,
|
||||
"v4_floor": 120,
|
||||
"v3_activation": 60,
|
||||
"epoch_blocks": 60,
|
||||
"lead": 10,
|
||||
"first_full_window_epoch": 8,
|
||||
"floor_epoch": 1667,
|
||||
"stale_miner": null,
|
||||
"stale_accepted_after_first_refresh": null,
|
||||
"stale_rejected": null,
|
||||
"stateless_node": 3,
|
||||
"stateless_accepted_after_flip": 0,
|
||||
"stateless_refusal_lines": 0,
|
||||
"state_roots_at_flip": [],
|
||||
"streams": {},
|
||||
"days_seen": [
|
||||
1244001,
|
||||
1244002
|
||||
],
|
||||
"day_boundary_crossed": true,
|
||||
"node": "/srv/builds/igneum-wt-class-v5/vendor/igneum-node-class-v5/target/release/igneumd",
|
||||
"miner": "/srv/builds/igneum-wt-class-v5/vendor/igneum-node-class-v5/target/release/igneum-miner",
|
||||
"template_switch": null,
|
||||
"run_ended_at_s": 672.5,
|
||||
"final_daa": 660,
|
||||
"max_epoch_seen": 11,
|
||||
"epochs": {
|
||||
"0": {
|
||||
"class": 2,
|
||||
"firstSeenDaa": 0,
|
||||
"at": 5.9,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 0,
|
||||
"bps5": 0,
|
||||
"signal_epoch": null,
|
||||
"day": 1244001
|
||||
},
|
||||
"1": {
|
||||
"class": 3,
|
||||
"firstSeenDaa": 60,
|
||||
"at": 40.9,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 10000,
|
||||
"bps5": 7500,
|
||||
"signal_epoch": null,
|
||||
"day": 1244001
|
||||
},
|
||||
"2": {
|
||||
"class": 4,
|
||||
"firstSeenDaa": 120,
|
||||
"at": 129,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 10000,
|
||||
"bps5": 7166,
|
||||
"signal_epoch": null,
|
||||
"day": 1244001
|
||||
},
|
||||
"3": {
|
||||
"class": 4,
|
||||
"firstSeenDaa": 181,
|
||||
"at": 187,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 10000,
|
||||
"bps5": 8000,
|
||||
"signal_epoch": null,
|
||||
"day": 1244001
|
||||
},
|
||||
"4": {
|
||||
"class": 4,
|
||||
"firstSeenDaa": 241,
|
||||
"at": 254.1,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 10000,
|
||||
"bps5": 7500,
|
||||
"signal_epoch": null,
|
||||
"day": 1244001
|
||||
},
|
||||
"5": {
|
||||
"class": 4,
|
||||
"firstSeenDaa": 300,
|
||||
"at": 311.2,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 10000,
|
||||
"bps5": 7000,
|
||||
"signal_epoch": null,
|
||||
"day": 1244001
|
||||
},
|
||||
"6": {
|
||||
"class": 4,
|
||||
"firstSeenDaa": 360,
|
||||
"at": 374.2,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 10000,
|
||||
"bps5": 6500,
|
||||
"signal_epoch": null,
|
||||
"day": 1244001
|
||||
},
|
||||
"7": {
|
||||
"class": 4,
|
||||
"firstSeenDaa": 420,
|
||||
"at": 429.3,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 10000,
|
||||
"bps5": 7166,
|
||||
"signal_epoch": null,
|
||||
"day": 1244001
|
||||
},
|
||||
"8": {
|
||||
"class": 4,
|
||||
"firstSeenDaa": 480,
|
||||
"at": 484.4,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 10000,
|
||||
"bps5": 8333,
|
||||
"signal_epoch": null,
|
||||
"day": 1244001
|
||||
},
|
||||
"9": {
|
||||
"class": 4,
|
||||
"firstSeenDaa": 543,
|
||||
"at": 543.4,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 10000,
|
||||
"bps5": 8333,
|
||||
"signal_epoch": null,
|
||||
"day": 1244001
|
||||
},
|
||||
"10": {
|
||||
"class": 4,
|
||||
"firstSeenDaa": 600,
|
||||
"at": 608.5,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 10000,
|
||||
"bps5": 8166,
|
||||
"signal_epoch": null,
|
||||
"day": 1244001
|
||||
},
|
||||
"11": {
|
||||
"class": 4,
|
||||
"firstSeenDaa": 660,
|
||||
"at": 672.5,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 10000,
|
||||
"bps5": 6833,
|
||||
"signal_epoch": null,
|
||||
"day": 1244002
|
||||
}
|
||||
},
|
||||
"blocks": {
|
||||
"total": 665,
|
||||
"before_boundary": 665,
|
||||
"after_boundary": 0,
|
||||
"version_bytes": {
|
||||
"0": 1,
|
||||
"4": 164,
|
||||
"5": 500
|
||||
},
|
||||
"signal_share_bps_on_chain": 7519
|
||||
},
|
||||
"programs": [
|
||||
{
|
||||
"epoch": 0,
|
||||
"class": "v2",
|
||||
"program_id": "8f8806638d59850f",
|
||||
"seed": "234e082d653dc69d",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
},
|
||||
{
|
||||
"epoch": 1,
|
||||
"class": "v3",
|
||||
"program_id": "bbe9c812d23b54c7",
|
||||
"seed": "ccda1efdf05b11e3",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
},
|
||||
{
|
||||
"epoch": 2,
|
||||
"class": "v4",
|
||||
"program_id": "770c813c27dcadfb",
|
||||
"seed": "706bfe07b4ba6297",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
},
|
||||
{
|
||||
"epoch": 3,
|
||||
"class": "v4",
|
||||
"program_id": "58070d811264f50a",
|
||||
"seed": "bf88c1aa6950130d",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
},
|
||||
{
|
||||
"epoch": 4,
|
||||
"class": "v4",
|
||||
"program_id": "dedb6a978f52d1c2",
|
||||
"seed": "b449c105e3871508",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
},
|
||||
{
|
||||
"epoch": 5,
|
||||
"class": "v4",
|
||||
"program_id": "d639a22b39c71b29",
|
||||
"seed": "885f0e943672e9e6",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
},
|
||||
{
|
||||
"epoch": 6,
|
||||
"class": "v4",
|
||||
"program_id": "97a72e755f22b423",
|
||||
"seed": "566340c6f666c4de",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
},
|
||||
{
|
||||
"epoch": 7,
|
||||
"class": "v4",
|
||||
"program_id": "0bd729deb3c71413",
|
||||
"seed": "6bfb5ba1c96669df",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
},
|
||||
{
|
||||
"epoch": 8,
|
||||
"class": "v4",
|
||||
"program_id": "3efc8eaf78579d3e",
|
||||
"seed": "f14275d8ea57e12e",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
},
|
||||
{
|
||||
"epoch": 9,
|
||||
"class": "v4",
|
||||
"program_id": "9a84cfc61cf52448",
|
||||
"seed": "a736507fb91b03f8",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
},
|
||||
{
|
||||
"epoch": 10,
|
||||
"class": "v4",
|
||||
"program_id": "c9695e82f43bf5ed",
|
||||
"seed": "529b9f66d4620030",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
},
|
||||
{
|
||||
"epoch": 11,
|
||||
"class": "v4",
|
||||
"program_id": "b2c69d165f318dc7",
|
||||
"seed": "c34388ca4e9fc321",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
}
|
||||
],
|
||||
"program_id_rows": [],
|
||||
"accepted_per_miner": [
|
||||
185,
|
||||
173,
|
||||
164,
|
||||
142
|
||||
],
|
||||
"rejected_by_miners": [
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
0
|
||||
],
|
||||
"rejected_by_nodes": [
|
||||
0,
|
||||
0,
|
||||
0
|
||||
],
|
||||
"sinks": [
|
||||
"69ecbb2da35fa391",
|
||||
"69ecbb2da35fa391",
|
||||
"69ecbb2da35fa391",
|
||||
"69ecbb2da35fa391"
|
||||
],
|
||||
"block_counts": [
|
||||
664,
|
||||
664,
|
||||
664,
|
||||
664
|
||||
],
|
||||
"signal_lines": [
|
||||
null,
|
||||
null,
|
||||
null
|
||||
],
|
||||
"floor_lines": [
|
||||
"Program class v5 from the override file: enabled, the floor at epoch 1667 (DAA score 100000 rounded up to the epoch boundary at 100020); the class v5 signal (byte 5 or above, 9500 bps in each of 7 consecutive windows) can flip v4 to v5 before it; every item of the dataset is keyed by the execution state after the epoch's seed block",
|
||||
"Program class v5 from the override file: enabled, the floor at epoch 1667 (DAA score 100000 rounded up to the epoch boundary at 100020); the class v5 signal (byte 5 or above, 9500 bps in each of 7 consecutive windows) can flip v4 to v5 before it; every item of the dataset is keyed by the execution state after the epoch's seed block",
|
||||
"Program class v5 from the override file: enabled, the floor at epoch 1667 (DAA score 100000 rounded up to the epoch boundary at 100020); the class v5 signal (byte 5 or above, 9500 bps in each of 7 consecutive windows) can flip v4 to v5 before it; every item of the dataset is keyed by the execution state after the epoch's seed block"
|
||||
],
|
||||
"samples": [
|
||||
{
|
||||
"t": 5.9,
|
||||
"daa": 0,
|
||||
"epoch": 0,
|
||||
"class": 2,
|
||||
"bps": 0,
|
||||
"bps5": 0,
|
||||
"nodes": [
|
||||
"0/234e082d",
|
||||
"0/234e082d",
|
||||
"0/234e082d",
|
||||
"0/234e082d"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 20.9,
|
||||
"daa": 35,
|
||||
"epoch": 0,
|
||||
"class": 2,
|
||||
"bps": 10000,
|
||||
"bps5": 8666,
|
||||
"nodes": [
|
||||
"35/22cfe577",
|
||||
"35/22cfe577",
|
||||
"35/22cfe577",
|
||||
"35/22cfe577"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 35.9,
|
||||
"daa": 57,
|
||||
"epoch": 0,
|
||||
"class": 2,
|
||||
"bps": 10000,
|
||||
"bps5": 7500,
|
||||
"nodes": [
|
||||
"57/4c7b852f",
|
||||
"57/4c7b852f",
|
||||
"57/4c7b852f",
|
||||
"57/4c7b852f"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 50.9,
|
||||
"daa": 65,
|
||||
"epoch": 1,
|
||||
"class": 3,
|
||||
"bps": 10000,
|
||||
"bps5": 7333,
|
||||
"nodes": [
|
||||
"65/ea044c9a",
|
||||
"65/ea044c9a",
|
||||
"65/ea044c9a",
|
||||
"65/ea044c9a"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 65.9,
|
||||
"daa": 68,
|
||||
"epoch": 1,
|
||||
"class": 3,
|
||||
"bps": 10000,
|
||||
"bps5": 7000,
|
||||
"nodes": [
|
||||
"68/4f79e9a8",
|
||||
"68/4f79e9a8",
|
||||
"68/4f79e9a8",
|
||||
"68/4f79e9a8"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 81,
|
||||
"daa": 76,
|
||||
"epoch": 1,
|
||||
"class": 3,
|
||||
"bps": 10000,
|
||||
"bps5": 6666,
|
||||
"nodes": [
|
||||
"76/c14f8a10",
|
||||
"76/c14f8a10",
|
||||
"76/c14f8a10",
|
||||
"76/c14f8a10"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 96,
|
||||
"daa": 86,
|
||||
"epoch": 1,
|
||||
"class": 3,
|
||||
"bps": 10000,
|
||||
"bps5": 7166,
|
||||
"nodes": [
|
||||
"86/b6beb323",
|
||||
"86/b6beb323",
|
||||
"86/b6beb323",
|
||||
"86/b6beb323"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 111,
|
||||
"daa": 98,
|
||||
"epoch": 1,
|
||||
"class": 3,
|
||||
"bps": 10000,
|
||||
"bps5": 7166,
|
||||
"nodes": [
|
||||
"98/68a15383",
|
||||
"98/68a15383",
|
||||
"98/68a15383",
|
||||
"98/68a15383"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 126,
|
||||
"daa": 117,
|
||||
"epoch": 1,
|
||||
"class": 3,
|
||||
"bps": 10000,
|
||||
"bps5": 7166,
|
||||
"nodes": [
|
||||
"117/7ba4412a",
|
||||
"117/7ba4412a",
|
||||
"117/7ba4412a",
|
||||
"117/7ba4412a"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 141,
|
||||
"daa": 128,
|
||||
"epoch": 2,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7666,
|
||||
"nodes": [
|
||||
"128/b7628853",
|
||||
"128/b7628853",
|
||||
"128/b7628853",
|
||||
"128/b7628853"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 156,
|
||||
"daa": 141,
|
||||
"epoch": 2,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7333,
|
||||
"nodes": [
|
||||
"141/5eb8637d",
|
||||
"141/5eb8637d",
|
||||
"141/5eb8637d",
|
||||
"141/5eb8637d"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 171,
|
||||
"daa": 157,
|
||||
"epoch": 2,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7333,
|
||||
"nodes": [
|
||||
"157/09cb4ebc",
|
||||
"157/09cb4ebc",
|
||||
"157/09cb4ebc",
|
||||
"157/09cb4ebc"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 186,
|
||||
"daa": 178,
|
||||
"epoch": 2,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8000,
|
||||
"nodes": [
|
||||
"178/b2671005",
|
||||
"178/b2671005",
|
||||
"178/b2671005",
|
||||
"178/b2671005"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 201.1,
|
||||
"daa": 201,
|
||||
"epoch": 3,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8166,
|
||||
"nodes": [
|
||||
"201/af5ee517",
|
||||
"201/af5ee517",
|
||||
"201/af5ee517",
|
||||
"201/af5ee517"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 216.1,
|
||||
"daa": 213,
|
||||
"epoch": 3,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8000,
|
||||
"nodes": [
|
||||
"213/2f96468c",
|
||||
"213/2f96468c",
|
||||
"213/2f96468c",
|
||||
"213/2f96468c"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 231.1,
|
||||
"daa": 227,
|
||||
"epoch": 3,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7666,
|
||||
"nodes": [
|
||||
"227/8d1904f8",
|
||||
"227/8d1904f8",
|
||||
"227/8d1904f8",
|
||||
"227/8d1904f8"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 246.1,
|
||||
"daa": 236,
|
||||
"epoch": 3,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7500,
|
||||
"nodes": [
|
||||
"236/b206681e",
|
||||
"236/b206681e",
|
||||
"236/b206681e",
|
||||
"236/b206681e"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 261.1,
|
||||
"daa": 249,
|
||||
"epoch": 4,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7333,
|
||||
"nodes": [
|
||||
"249/a84b8339",
|
||||
"249/a84b8339",
|
||||
"249/a84b8339",
|
||||
"249/a84b8339"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 276.1,
|
||||
"daa": 260,
|
||||
"epoch": 4,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7833,
|
||||
"nodes": [
|
||||
"260/96af12eb",
|
||||
"260/96af12eb",
|
||||
"260/96af12eb",
|
||||
"260/96af12eb"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 291.2,
|
||||
"daa": 275,
|
||||
"epoch": 4,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7666,
|
||||
"nodes": [
|
||||
"275/435e44bd",
|
||||
"275/435e44bd",
|
||||
"275/435e44bd",
|
||||
"275/435e44bd"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 306.2,
|
||||
"daa": 297,
|
||||
"epoch": 4,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7000,
|
||||
"nodes": [
|
||||
"297/fc909d98",
|
||||
"297/fc909d98",
|
||||
"297/fc909d98",
|
||||
"297/fc909d98"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 321.2,
|
||||
"daa": 308,
|
||||
"epoch": 5,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7000,
|
||||
"nodes": [
|
||||
"308/3faec6a0",
|
||||
"308/3faec6a0",
|
||||
"308/3faec6a0",
|
||||
"308/3faec6a0"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 336.2,
|
||||
"daa": 327,
|
||||
"epoch": 5,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 6333,
|
||||
"nodes": [
|
||||
"327/9d43d0bc",
|
||||
"327/9d43d0bc",
|
||||
"327/9d43d0bc",
|
||||
"327/9d43d0bc"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 351.2,
|
||||
"daa": 342,
|
||||
"epoch": 5,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 6666,
|
||||
"nodes": [
|
||||
"342/39c2298c",
|
||||
"342/39c2298c",
|
||||
"342/39c2298c",
|
||||
"342/39c2298c"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 366.2,
|
||||
"daa": 353,
|
||||
"epoch": 5,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 6500,
|
||||
"nodes": [
|
||||
"353/0d593450",
|
||||
"353/0d593450",
|
||||
"353/0d593450",
|
||||
"353/0d593450"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 381.3,
|
||||
"daa": 365,
|
||||
"epoch": 6,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7000,
|
||||
"nodes": [
|
||||
"365/888c04c8",
|
||||
"365/888c04c8",
|
||||
"365/888c04c8",
|
||||
"365/888c04c8"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 396.3,
|
||||
"daa": 378,
|
||||
"epoch": 6,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7666,
|
||||
"nodes": [
|
||||
"378/f757a72b",
|
||||
"378/f757a72b",
|
||||
"378/f757a72b",
|
||||
"378/f757a72b"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 411.3,
|
||||
"daa": 396,
|
||||
"epoch": 6,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7666,
|
||||
"nodes": [
|
||||
"396/549da430",
|
||||
"396/549da430",
|
||||
"396/549da430",
|
||||
"396/549da430"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 426.3,
|
||||
"daa": 413,
|
||||
"epoch": 6,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7166,
|
||||
"nodes": [
|
||||
"413/b867a9d7",
|
||||
"413/b867a9d7",
|
||||
"413/b867a9d7",
|
||||
"413/b867a9d7"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 441.3,
|
||||
"daa": 434,
|
||||
"epoch": 7,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 6833,
|
||||
"nodes": [
|
||||
"434/992b60dd",
|
||||
"434/992b60dd",
|
||||
"434/992b60dd",
|
||||
"434/992b60dd"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 456.3,
|
||||
"daa": 447,
|
||||
"epoch": 7,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7333,
|
||||
"nodes": [
|
||||
"447/acdc4c4a",
|
||||
"447/acdc4c4a",
|
||||
"447/acdc4c4a",
|
||||
"447/acdc4c4a"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 471.4,
|
||||
"daa": 463,
|
||||
"epoch": 7,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7833,
|
||||
"nodes": [
|
||||
"463/15e05947",
|
||||
"463/15e05947",
|
||||
"463/15e05947",
|
||||
"463/15e05947"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 486.4,
|
||||
"daa": 483,
|
||||
"epoch": 8,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8333,
|
||||
"nodes": [
|
||||
"483/e9a70a60",
|
||||
"483/e9a70a60",
|
||||
"483/e9a70a60",
|
||||
"483/e9a70a60"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 501.4,
|
||||
"daa": 497,
|
||||
"epoch": 8,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8333,
|
||||
"nodes": [
|
||||
"497/4f3107de",
|
||||
"497/4f3107de",
|
||||
"497/4f3107de",
|
||||
"497/4f3107de"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 516.4,
|
||||
"daa": 517,
|
||||
"epoch": 8,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8500,
|
||||
"nodes": [
|
||||
"517/4a2ca858",
|
||||
"517/4a2ca858",
|
||||
"517/4a2ca858",
|
||||
"517/4a2ca858"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 531.4,
|
||||
"daa": 529,
|
||||
"epoch": 8,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8500,
|
||||
"nodes": [
|
||||
"529/40646365",
|
||||
"529/40646365",
|
||||
"529/40646365",
|
||||
"529/40646365"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 546.4,
|
||||
"daa": 546,
|
||||
"epoch": 9,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8666,
|
||||
"nodes": [
|
||||
"546/e479a304",
|
||||
"546/e479a304",
|
||||
"546/e479a304",
|
||||
"546/e479a304"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 561.4,
|
||||
"daa": 558,
|
||||
"epoch": 9,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8500,
|
||||
"nodes": [
|
||||
"558/133c21e1",
|
||||
"558/133c21e1",
|
||||
"558/133c21e1",
|
||||
"558/133c21e1"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 576.5,
|
||||
"daa": 571,
|
||||
"epoch": 9,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8000,
|
||||
"nodes": [
|
||||
"571/36928d80",
|
||||
"571/36928d80",
|
||||
"571/36928d80",
|
||||
"571/36928d80"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 591.5,
|
||||
"daa": 587,
|
||||
"epoch": 9,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8000,
|
||||
"nodes": [
|
||||
"587/a9e933c1",
|
||||
"587/a9e933c1",
|
||||
"587/a9e933c1",
|
||||
"587/a9e933c1"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 606.5,
|
||||
"daa": 599,
|
||||
"epoch": 9,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8166,
|
||||
"nodes": [
|
||||
"599/0f5f6b40",
|
||||
"599/0f5f6b40",
|
||||
"599/0f5f6b40",
|
||||
"599/0f5f6b40"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 621.5,
|
||||
"daa": 613,
|
||||
"epoch": 10,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8166,
|
||||
"nodes": [
|
||||
"613/6a185b33",
|
||||
"613/6a185b33",
|
||||
"613/6a185b33",
|
||||
"613/6a185b33"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 636.5,
|
||||
"daa": 631,
|
||||
"epoch": 10,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7500,
|
||||
"nodes": [
|
||||
"631/88715a4b",
|
||||
"631/88715a4b",
|
||||
"631/88715a4b",
|
||||
"631/88715a4b"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 651.5,
|
||||
"daa": 642,
|
||||
"epoch": 10,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7333,
|
||||
"nodes": [
|
||||
"642/09eae3ad",
|
||||
"642/09eae3ad",
|
||||
"642/09eae3ad",
|
||||
"642/09eae3ad"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 666.5,
|
||||
"daa": 652,
|
||||
"epoch": 10,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 6833,
|
||||
"nodes": [
|
||||
"652/9853083d",
|
||||
"652/9853083d",
|
||||
"652/9853083d",
|
||||
"652/9853083d"
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
80
docs/design/class-v5-harness/failed-case.log
Normal file
80
docs/design/class-v5-harness/failed-case.log
Normal file
|
|
@ -0,0 +1,80 @@
|
|||
08:37:09.729 signals 5/5/4, expect flip; v3 from 60 (epoch 1), v4 floor 120 (epoch 2), v5 floor 100000 (epoch 1667), window 60 DAA x 7 (the first epoch with seven full windows after v4 is 8); stale miner none; stateless node n3; 60 DAA per epoch, lead 10; run 780 s or 11 epochs
|
||||
08:37:10.960 n0 up pid 1398269 json 29762 p2p 29761 exec 29763, signals 5
|
||||
08:37:12.165 n1 up pid 1398975 json 29772 p2p 29771 exec 29773, signals 5
|
||||
08:37:13.374 n2 up pid 1399919 json 29782 p2p 29781 exec 29783, signals 4
|
||||
08:37:14.580 n3 up pid 1400717 json 29792 p2p 29791 exec disabled, signals 5
|
||||
08:37:14.581 n0: Program class v4 signal window from the override file: 60 DAA ending at each epoch's seed block, threshold 9500 bps of blue blocks; the fixed height is the floor | this node signals object version 5
|
||||
08:37:14.581 n1: Program class v4 signal window from the override file: 60 DAA ending at each epoch's seed block, threshold 9500 bps of blue blocks; the fixed height is the floor | this node signals object version 5
|
||||
08:37:14.582 n2: Program class v4 signal window from the override file: 60 DAA ending at each epoch's seed block, threshold 9500 bps of blue blocks; the fixed height is the floor | this node signals object version 4
|
||||
08:37:14.582 n3: Program class v4 signal window from the override file: 60 DAA ending at each epoch's seed block, threshold 9500 bps of blue blocks; the fixed height is the floor | this node signals object version 5
|
||||
08:37:14.582 n0 digest: 5d04822e3501a5de
|
||||
08:37:15.593 epoch -1 -> 0 at daa 0, 5.9 s: template class 2, next 2, day 1244001, signal share at the sink v4 0 v5 0 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
08:37:15.595 t=5.9 s daa 0 epoch 0 class 2 signal v4 0 v5 0 bps blocks/sink per node 0/234e082d 0/234e082d 0/234e082d 0/234e082d
|
||||
08:37:30.618 t=20.9 s daa 35 epoch 0 class 2 signal v4 10000 v5 8666 bps blocks/sink per node 35/22cfe577 35/22cfe577 35/22cfe577 35/22cfe577
|
||||
08:37:45.635 t=35.9 s daa 57 epoch 0 class 2 signal v4 10000 v5 7500 bps blocks/sink per node 57/4c7b852f 57/4c7b852f 57/4c7b852f 57/4c7b852f
|
||||
08:37:50.640 epoch 0 -> 1 at daa 60, 40.9 s: template class 3, next 3, day 1244001, signal share at the sink v4 10000 v5 7500 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
08:38:00.651 t=50.9 s daa 65 epoch 1 class 3 signal v4 10000 v5 7333 bps blocks/sink per node 65/ea044c9a 65/ea044c9a 65/ea044c9a 65/ea044c9a
|
||||
08:38:15.668 t=65.9 s daa 68 epoch 1 class 3 signal v4 10000 v5 7000 bps blocks/sink per node 68/4f79e9a8 68/4f79e9a8 68/4f79e9a8 68/4f79e9a8
|
||||
08:38:30.688 t=81.0 s daa 76 epoch 1 class 3 signal v4 10000 v5 6666 bps blocks/sink per node 76/c14f8a10 76/c14f8a10 76/c14f8a10 76/c14f8a10
|
||||
08:38:45.707 t=96.0 s daa 86 epoch 1 class 3 signal v4 10000 v5 7166 bps blocks/sink per node 86/b6beb323 86/b6beb323 86/b6beb323 86/b6beb323
|
||||
08:39:00.718 t=111.0 s daa 98 epoch 1 class 3 signal v4 10000 v5 7166 bps blocks/sink per node 98/68a15383 98/68a15383 98/68a15383 98/68a15383
|
||||
08:39:15.734 t=126.0 s daa 117 epoch 1 class 3 signal v4 10000 v5 7166 bps blocks/sink per node 117/7ba4412a 117/7ba4412a 117/7ba4412a 117/7ba4412a
|
||||
08:39:18.736 epoch 1 -> 2 at daa 120, 129.0 s: template class 4, next 4, day 1244001, signal share at the sink v4 10000 v5 7166 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
08:39:30.747 t=141.0 s daa 128 epoch 2 class 4 signal v4 10000 v5 7666 bps blocks/sink per node 128/b7628853 128/b7628853 128/b7628853 128/b7628853
|
||||
08:39:45.755 t=156.0 s daa 141 epoch 2 class 4 signal v4 10000 v5 7333 bps blocks/sink per node 141/5eb8637d 141/5eb8637d 141/5eb8637d 141/5eb8637d
|
||||
08:40:00.767 t=171.0 s daa 157 epoch 2 class 4 signal v4 10000 v5 7333 bps blocks/sink per node 157/09cb4ebc 157/09cb4ebc 157/09cb4ebc 157/09cb4ebc
|
||||
08:40:15.778 t=186.0 s daa 178 epoch 2 class 4 signal v4 10000 v5 8000 bps blocks/sink per node 178/b2671005 178/b2671005 178/b2671005 178/b2671005
|
||||
08:40:16.779 epoch 2 -> 3 at daa 181, 187.0 s: template class 4, next 4, day 1244001, signal share at the sink v4 10000 v5 8000 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
08:40:30.791 t=201.1 s daa 201 epoch 3 class 4 signal v4 10000 v5 8166 bps blocks/sink per node 201/af5ee517 201/af5ee517 201/af5ee517 201/af5ee517
|
||||
08:40:45.800 t=216.1 s daa 213 epoch 3 class 4 signal v4 10000 v5 8000 bps blocks/sink per node 213/2f96468c 213/2f96468c 213/2f96468c 213/2f96468c
|
||||
08:41:00.812 t=231.1 s daa 227 epoch 3 class 4 signal v4 10000 v5 7666 bps blocks/sink per node 227/8d1904f8 227/8d1904f8 227/8d1904f8 227/8d1904f8
|
||||
08:41:15.842 t=246.1 s daa 236 epoch 3 class 4 signal v4 10000 v5 7500 bps blocks/sink per node 236/b206681e 236/b206681e 236/b206681e 236/b206681e
|
||||
08:41:23.854 epoch 3 -> 4 at daa 241, 254.1 s: template class 4, next 4, day 1244001, signal share at the sink v4 10000 v5 7500 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
08:41:30.861 t=261.1 s daa 249 epoch 4 class 4 signal v4 10000 v5 7333 bps blocks/sink per node 249/a84b8339 249/a84b8339 249/a84b8339 249/a84b8339
|
||||
08:41:45.872 t=276.1 s daa 260 epoch 4 class 4 signal v4 10000 v5 7833 bps blocks/sink per node 260/96af12eb 260/96af12eb 260/96af12eb 260/96af12eb
|
||||
08:42:00.888 t=291.2 s daa 275 epoch 4 class 4 signal v4 10000 v5 7666 bps blocks/sink per node 275/435e44bd 275/435e44bd 275/435e44bd 275/435e44bd
|
||||
08:42:15.905 t=306.2 s daa 297 epoch 4 class 4 signal v4 10000 v5 7000 bps blocks/sink per node 297/fc909d98 297/fc909d98 297/fc909d98 297/fc909d98
|
||||
08:42:20.912 epoch 4 -> 5 at daa 300, 311.2 s: template class 4, next 4, day 1244001, signal share at the sink v4 10000 v5 7000 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
08:42:30.924 t=321.2 s daa 308 epoch 5 class 4 signal v4 10000 v5 7000 bps blocks/sink per node 308/3faec6a0 308/3faec6a0 308/3faec6a0 308/3faec6a0
|
||||
08:42:45.941 t=336.2 s daa 327 epoch 5 class 4 signal v4 10000 v5 6333 bps blocks/sink per node 327/9d43d0bc 327/9d43d0bc 327/9d43d0bc 327/9d43d0bc
|
||||
08:43:00.955 t=351.2 s daa 342 epoch 5 class 4 signal v4 10000 v5 6666 bps blocks/sink per node 342/39c2298c 342/39c2298c 342/39c2298c 342/39c2298c
|
||||
08:43:15.969 t=366.2 s daa 353 epoch 5 class 4 signal v4 10000 v5 6500 bps blocks/sink per node 353/0d593450 353/0d593450 353/0d593450 353/0d593450
|
||||
08:43:23.975 epoch 5 -> 6 at daa 360, 374.2 s: template class 4, next 4, day 1244001, signal share at the sink v4 10000 v5 6500 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
08:43:30.982 t=381.3 s daa 365 epoch 6 class 4 signal v4 10000 v5 7000 bps blocks/sink per node 365/888c04c8 365/888c04c8 365/888c04c8 365/888c04c8
|
||||
08:43:45.996 t=396.3 s daa 378 epoch 6 class 4 signal v4 10000 v5 7666 bps blocks/sink per node 378/f757a72b 378/f757a72b 378/f757a72b 378/f757a72b
|
||||
08:44:01.012 t=411.3 s daa 396 epoch 6 class 4 signal v4 10000 v5 7666 bps blocks/sink per node 396/549da430 396/549da430 396/549da430 396/549da430
|
||||
08:44:16.027 t=426.3 s daa 413 epoch 6 class 4 signal v4 10000 v5 7166 bps blocks/sink per node 413/b867a9d7 413/b867a9d7 413/b867a9d7 413/b867a9d7
|
||||
08:44:19.030 epoch 6 -> 7 at daa 420, 429.3 s: template class 4, next 4, day 1244001, signal share at the sink v4 10000 v5 7166 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
08:44:31.044 t=441.3 s daa 434 epoch 7 class 4 signal v4 10000 v5 6833 bps blocks/sink per node 434/992b60dd 434/992b60dd 434/992b60dd 434/992b60dd
|
||||
08:44:46.062 t=456.3 s daa 447 epoch 7 class 4 signal v4 10000 v5 7333 bps blocks/sink per node 447/acdc4c4a 447/acdc4c4a 447/acdc4c4a 447/acdc4c4a
|
||||
08:45:01.080 t=471.4 s daa 463 epoch 7 class 4 signal v4 10000 v5 7833 bps blocks/sink per node 463/15e05947 463/15e05947 463/15e05947 463/15e05947
|
||||
08:45:14.090 epoch 7 -> 8 at daa 480, 484.4 s: template class 4, next 4, day 1244001, signal share at the sink v4 10000 v5 8333 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
08:45:16.094 t=486.4 s daa 483 epoch 8 class 4 signal v4 10000 v5 8333 bps blocks/sink per node 483/e9a70a60 483/e9a70a60 483/e9a70a60 483/e9a70a60
|
||||
08:45:31.107 t=501.4 s daa 497 epoch 8 class 4 signal v4 10000 v5 8333 bps blocks/sink per node 497/4f3107de 497/4f3107de 497/4f3107de 497/4f3107de
|
||||
08:45:46.123 t=516.4 s daa 517 epoch 8 class 4 signal v4 10000 v5 8500 bps blocks/sink per node 517/4a2ca858 517/4a2ca858 517/4a2ca858 517/4a2ca858
|
||||
08:46:01.142 t=531.4 s daa 529 epoch 8 class 4 signal v4 10000 v5 8500 bps blocks/sink per node 529/40646365 529/40646365 529/40646365 529/40646365
|
||||
08:46:13.155 epoch 8 -> 9 at daa 543, 543.4 s: template class 4, next 4, day 1244001, signal share at the sink v4 10000 v5 8333 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
08:46:16.160 t=546.4 s daa 546 epoch 9 class 4 signal v4 10000 v5 8666 bps blocks/sink per node 546/e479a304 546/e479a304 546/e479a304 546/e479a304
|
||||
08:46:31.173 t=561.4 s daa 558 epoch 9 class 4 signal v4 10000 v5 8500 bps blocks/sink per node 558/133c21e1 558/133c21e1 558/133c21e1 558/133c21e1
|
||||
08:46:46.183 t=576.5 s daa 571 epoch 9 class 4 signal v4 10000 v5 8000 bps blocks/sink per node 571/36928d80 571/36928d80 571/36928d80 571/36928d80
|
||||
08:47:01.193 t=591.5 s daa 587 epoch 9 class 4 signal v4 10000 v5 8000 bps blocks/sink per node 587/a9e933c1 587/a9e933c1 587/a9e933c1 587/a9e933c1
|
||||
08:47:16.203 t=606.5 s daa 599 epoch 9 class 4 signal v4 10000 v5 8166 bps blocks/sink per node 599/0f5f6b40 599/0f5f6b40 599/0f5f6b40 599/0f5f6b40
|
||||
08:47:18.204 epoch 9 -> 10 at daa 600, 608.5 s: template class 4, next 4, day 1244001, signal share at the sink v4 10000 v5 8166 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
08:47:31.214 t=621.5 s daa 613 epoch 10 class 4 signal v4 10000 v5 8166 bps blocks/sink per node 613/6a185b33 613/6a185b33 613/6a185b33 613/6a185b33
|
||||
08:47:46.231 t=636.5 s daa 631 epoch 10 class 4 signal v4 10000 v5 7500 bps blocks/sink per node 631/88715a4b 631/88715a4b 631/88715a4b 631/88715a4b
|
||||
08:48:01.258 t=651.5 s daa 642 epoch 10 class 4 signal v4 10000 v5 7333 bps blocks/sink per node 642/09eae3ad 642/09eae3ad 642/09eae3ad 642/09eae3ad
|
||||
08:48:16.271 t=666.5 s daa 652 epoch 10 class 4 signal v4 10000 v5 6833 bps blocks/sink per node 652/9853083d 652/9853083d 652/9853083d 652/9853083d
|
||||
08:48:22.280 epoch 10 -> 11 at daa 660, 672.5 s: template class 4, next 4, day 1244002, signal share at the sink v4 10000 v5 6833 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
08:48:25.423 SUMMARY FAIL (expect flip, signals 5/5/4, stale none): no v5 epoch; epochs seen e0:2:0bps:d1244001 e1:3:7500bps:d1244001 e2:4:7166bps:d1244001 e3:4:8000bps:d1244001 e4:4:7500bps:d1244001 e5:4:7000bps:d1244001 e6:4:6500bps:d1244001 e7:4:7166bps:d1244001 e8:4:8333bps:d1244001 e9:4:8333bps:d1244001 e10:4:8166bps:d1244001 e11:4:6833bps:d1244002; chain bytes {"0":1,"4":164,"5":500} (7519 bps at byte 5+); blocks 665 / 0; rejected miners 0/0/0/0 nodes 0/0/0; sinks 69ecbb2da35fa391 69ecbb2da35fa391 69ecbb2da35fa391 69ecbb2da35fa391 (agree); counts 664/664/664/664; signal lines 0/3 (epochs //, shares //); floor lines 3/3; state roots at the flip ; stateless node accepted after the flip 0 (refusal lines 0); stale miner accepted after the first refresh n/a rejected n/a; days 1244001/1244002
|
||||
08:48:25.423 FAILED CHECK every_node_signals_its_byte
|
||||
08:48:25.423 FAILED CHECK template_switched_to_v5
|
||||
08:48:25.423 FAILED CHECK switched_at_the_first_full_window_epoch
|
||||
08:48:25.423 FAILED CHECK switched_before_the_floor
|
||||
08:48:25.423 FAILED CHECK signal_line_on_every_node_same_epoch
|
||||
08:48:25.423 FAILED CHECK signal_share_at_or_above_threshold
|
||||
08:48:25.423 FAILED CHECK blocks_on_both_sides
|
||||
08:48:25.423 FAILED CHECK v5_ids_equal_the_cli_v5_id
|
||||
08:48:25.423 FAILED CHECK v5_ids_differ_from_the_same_seed_v4_id
|
||||
08:48:25.423 FAILED CHECK a_v5_epoch_per_window_refresh
|
||||
08:48:25.423 FAILED CHECK stateless_node_mines_nothing_after_the_flip
|
||||
08:48:25.423 summary: /tmp/igneum-fast-time-v5s-failed-case/summary.json
|
||||
1040
docs/design/class-v5-harness/flip-stale.json
Normal file
1040
docs/design/class-v5-harness/flip-stale.json
Normal file
File diff suppressed because it is too large
Load diff
76
docs/design/class-v5-harness/flip-stale.log
Normal file
76
docs/design/class-v5-harness/flip-stale.log
Normal file
|
|
@ -0,0 +1,76 @@
|
|||
12:09:52.489 signals 6/6/6, expect flip; v3 from 60 (epoch 1), v4 floor 120 (epoch 2), v5 floor 100000 (epoch 1667), window 60 DAA x 7 (the first epoch with seven full windows after v4 is 8); stale miner 2; stateless node n3; 60 DAA per epoch, lead 10; run 780 s or 11 epochs
|
||||
12:09:53.723 n0 up pid 3045315 json 30342 p2p 30341 exec 30343, signals 6
|
||||
12:09:54.928 n1 up pid 3045932 json 30352 p2p 30351 exec 30353, signals 6
|
||||
12:09:56.134 n2 up pid 3046589 json 30362 p2p 30361 exec 30363, signals 6
|
||||
12:09:57.338 n3 up pid 3047510 json 30372 p2p 30371 exec disabled, signals 6
|
||||
12:09:57.339 n0: Program class v4 signal window from the override file: 60 DAA ending at each epoch's seed block, threshold 9500 bps of blue blocks in each of 7 windows; the fixed height is the floor; this node signals object version 6 (class v4 with the load-source rule, sub-version 1; object 4, the 6 October stream, does not count) | this node signals object version 6
|
||||
12:09:57.339 n1: Program class v4 signal window from the override file: 60 DAA ending at each epoch's seed block, threshold 9500 bps of blue blocks in each of 7 windows; the fixed height is the floor; this node signals object version 6 (class v4 with the load-source rule, sub-version 1; object 4, the 6 October stream, does not count) | this node signals object version 6
|
||||
12:09:57.339 n2: Program class v4 signal window from the override file: 60 DAA ending at each epoch's seed block, threshold 9500 bps of blue blocks in each of 7 windows; the fixed height is the floor; this node signals object version 6 (class v4 with the load-source rule, sub-version 1; object 4, the 6 October stream, does not count) | this node signals object version 6
|
||||
12:09:57.339 n3: Program class v4 signal window from the override file: 60 DAA ending at each epoch's seed block, threshold 9500 bps of blue blocks in each of 7 windows; the fixed height is the floor; this node signals object version 6 (class v4 with the load-source rule, sub-version 1; object 4, the 6 October stream, does not count) | this node signals object version 6
|
||||
12:09:57.339 n0 digest: 5d04822e3501a5de
|
||||
12:09:58.350 epoch -1 -> 0 at daa 0, 5.9 s: template class 2, next 2, day 1244010, signal share at the sink v4 0 v5 0 bps (window 60, this node signals 6, v5 decided by signal at epoch none)
|
||||
12:09:58.351 t=5.9 s daa 0 epoch 0 class 2 signal v4 0 v5 0 bps blocks/sink per node 0/234e082d 0/234e082d 0/234e082d 0/234e082d
|
||||
12:10:13.369 t=20.9 s daa 29 epoch 0 class 2 signal v4 10000 v5 10000 bps blocks/sink per node 29/57e79fe0 29/57e79fe0 29/57e79fe0 29/57e79fe0
|
||||
12:10:28.384 t=35.9 s daa 59 epoch 0 class 2 signal v4 10000 v5 10000 bps blocks/sink per node 59/8c19d208 59/8c19d208 59/8c19d208 59/8c19d208
|
||||
12:10:29.385 epoch 0 -> 1 at daa 60, 36.9 s: template class 3, next 3, day 1244010, signal share at the sink v4 10000 v5 10000 bps (window 60, this node signals 6, v5 decided by signal at epoch none)
|
||||
12:10:43.398 t=50.9 s daa 63 epoch 1 class 3 signal v4 10000 v5 10000 bps blocks/sink per node 63/25731901 63/25731901 63/25731901 63/25731901
|
||||
12:10:58.418 t=65.9 s daa 71 epoch 1 class 3 signal v4 10000 v5 10000 bps blocks/sink per node 71/8acff126 71/8acff126 71/8acff126 71/8acff126
|
||||
12:11:13.436 t=80.9 s daa 82 epoch 1 class 3 signal v4 10000 v5 10000 bps blocks/sink per node 82/83272ecc 82/83272ecc 82/83272ecc 82/83272ecc
|
||||
12:11:28.452 t=96.0 s daa 106 epoch 1 class 3 signal v4 10000 v5 10000 bps blocks/sink per node 106/5c2063f1 106/5c2063f1 106/5c2063f1 106/5c2063f1
|
||||
12:11:39.464 epoch 1 -> 2 at daa 121, 107.0 s: template class 4, next 4, day 1244010, signal share at the sink v4 10000 v5 10000 bps (window 60, this node signals 6, v5 decided by signal at epoch none)
|
||||
12:11:43.469 t=111.0 s daa 125 epoch 2 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 125/1a1ed97e 125/1a1ed97e 125/1a1ed97e 125/1a1ed97e
|
||||
12:11:58.484 t=126.0 s daa 142 epoch 2 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 142/bdd5d432 142/bdd5d432 142/bdd5d432 142/bdd5d432
|
||||
12:12:13.501 t=141.0 s daa 149 epoch 2 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 149/eb50d76d 149/eb50d76d 149/eb50d76d 149/eb50d76d
|
||||
12:12:28.518 t=156.0 s daa 163 epoch 2 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 163/1f9bbb92 163/1f9bbb92 163/1f9bbb92 163/1f9bbb92
|
||||
12:12:43.540 t=171.0 s daa 174 epoch 2 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 174/2a61a80c 174/2a61a80c 174/2a61a80c 174/2a61a80c
|
||||
12:12:51.549 epoch 2 -> 3 at daa 180, 179.1 s: template class 4, next 4, day 1244010, signal share at the sink v4 10000 v5 10000 bps (window 60, this node signals 6, v5 decided by signal at epoch none)
|
||||
12:12:58.560 t=186.1 s daa 187 epoch 3 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 187/e6dfde32 187/e6dfde32 187/e6dfde32 187/e6dfde32
|
||||
12:13:13.574 t=201.1 s daa 200 epoch 3 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 200/8de101de 200/8de101de 200/8de101de 200/8de101de
|
||||
12:13:28.589 t=216.1 s daa 219 epoch 3 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 219/7705c4e3 219/7705c4e3 219/7705c4e3 219/7705c4e3
|
||||
12:13:43.598 t=231.1 s daa 236 epoch 3 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 236/fad4712e 236/fad4712e 236/fad4712e 236/fad4712e
|
||||
12:13:46.599 epoch 3 -> 4 at daa 241, 234.1 s: template class 4, next 4, day 1244010, signal share at the sink v4 10000 v5 10000 bps (window 60, this node signals 6, v5 decided by signal at epoch none)
|
||||
12:13:58.610 t=246.1 s daa 252 epoch 4 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 252/822758a3 252/822758a3 252/822758a3 252/822758a3
|
||||
12:14:13.626 t=261.1 s daa 267 epoch 4 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 267/17080f66 267/17080f66 267/17080f66 267/17080f66
|
||||
12:14:28.637 t=276.1 s daa 279 epoch 4 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 279/765e10f6 279/765e10f6 279/765e10f6 279/765e10f6
|
||||
12:14:43.651 t=291.2 s daa 294 epoch 4 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 294/bc002c8d 294/bc002c8d 294/bc002c8d 294/bc002c8d
|
||||
12:14:49.656 epoch 4 -> 5 at daa 300, 297.2 s: template class 4, next 4, day 1244010, signal share at the sink v4 10000 v5 10000 bps (window 60, this node signals 6, v5 decided by signal at epoch none)
|
||||
12:14:58.666 t=306.2 s daa 312 epoch 5 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 312/8a61fffe 312/8a61fffe 312/8a61fffe 312/8a61fffe
|
||||
12:15:13.682 t=321.2 s daa 326 epoch 5 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 326/30eae59f 326/30eae59f 326/30eae59f 326/30eae59f
|
||||
12:15:28.696 t=336.2 s daa 338 epoch 5 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 338/a2f652c0 338/a2f652c0 338/a2f652c0 338/a2f652c0
|
||||
12:15:43.714 t=351.2 s daa 356 epoch 5 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 356/d1e15cf7 356/d1e15cf7 356/d1e15cf7 356/d1e15cf7
|
||||
12:15:47.718 epoch 5 -> 6 at daa 360, 355.2 s: template class 4, next 4, day 1244010, signal share at the sink v4 10000 v5 10000 bps (window 60, this node signals 6, v5 decided by signal at epoch none)
|
||||
12:15:58.731 t=366.2 s daa 370 epoch 6 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 370/1e127f8e 370/1e127f8e 370/1e127f8e 370/1e127f8e
|
||||
12:16:13.748 t=381.3 s daa 382 epoch 6 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 382/f1fa6859 382/f1fa6859 382/f1fa6859 382/f1fa6859
|
||||
12:16:28.763 t=396.3 s daa 402 epoch 6 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 402/f0ee3823 402/f0ee3823 402/f0ee3823 402/f0ee3823
|
||||
12:16:40.776 epoch 6 -> 7 at daa 420, 408.3 s: template class 4, next 4, day 1244010, signal share at the sink v4 10000 v5 10000 bps (window 60, this node signals 6, v5 decided by signal at epoch none)
|
||||
12:16:43.778 t=411.3 s daa 424 epoch 7 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 424/db80ccc8 424/db80ccc8 424/db80ccc8 424/db80ccc8
|
||||
12:16:58.796 t=426.3 s daa 440 epoch 7 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 440/6e1c3b1c 440/6e1c3b1c 440/6e1c3b1c 440/6e1c3b1c
|
||||
12:17:13.826 t=441.3 s daa 450 epoch 7 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 450/f63d97d3 451/dd8296fd 450/f63d97d3 450/f63d97d3
|
||||
12:17:28.845 t=456.4 s daa 461 epoch 7 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 461/c51f9db0 461/c51f9db0 461/c51f9db0 461/c51f9db0
|
||||
12:17:43.867 t=471.4 s daa 474 epoch 7 class 4 signal v4 10000 v5 10000 bps blocks/sink per node 474/b58b91f1 474/b58b91f1 474/b58b91f1 474/b58b91f1
|
||||
12:17:58.887 epoch 7 -> 8 at daa 480, 486.4 s: template class 5, next 5, day 1244010, signal share at the sink v4 10000 v5 10000 bps (window 60, this node signals 6, v5 decided by signal at epoch 8)
|
||||
12:17:58.887 CLASS SWITCH: the template is class v5 from epoch 8 (daa 480) at 486.4 s wall
|
||||
12:17:58.889 t=486.4 s daa 480 epoch 8 class 5 signal v4 10000 v5 10000 bps blocks/sink per node 480/21f187ed 480/21f187ed 480/21f187ed 480/21f187ed
|
||||
12:18:13.912 t=501.4 s daa 486 epoch 8 class 5 signal v4 10000 v5 10000 bps blocks/sink per node 486/9b65540b 486/9b65540b 486/9b65540b 480/21f187ed
|
||||
12:18:28.939 t=516.4 s daa 494 epoch 8 class 5 signal v4 10000 v5 10000 bps blocks/sink per node 494/abb33f53 494/abb33f53 494/abb33f53 480/21f187ed
|
||||
12:18:43.960 t=531.5 s daa 508 epoch 8 class 5 signal v4 10000 v5 10000 bps blocks/sink per node 508/98820272 508/98820272 508/98820272 480/21f187ed
|
||||
12:18:58.982 t=546.5 s daa 521 epoch 8 class 5 signal v4 10000 v5 10000 bps blocks/sink per node 521/dca86d92 521/dca86d92 521/dca86d92 480/21f187ed
|
||||
12:19:13.999 epoch 8 -> 9 at daa 541, 561.5 s: template class 5, next 5, day 1244010, signal share at the sink v4 10000 v5 10000 bps (window 60, this node signals 6, v5 decided by signal at epoch none)
|
||||
12:19:14.000 t=561.5 s daa 541 epoch 9 class 5 signal v4 10000 v5 10000 bps blocks/sink per node 541/c8c056ff 541/c8c056ff 540/53503f9f 480/21f187ed
|
||||
12:19:29.022 t=576.5 s daa 553 epoch 9 class 5 signal v4 10000 v5 10000 bps blocks/sink per node 553/4bc650e2 553/4bc650e2 553/4bc650e2 480/21f187ed
|
||||
12:19:44.046 t=591.6 s daa 564 epoch 9 class 5 signal v4 10000 v5 10000 bps blocks/sink per node 564/5c8d12d8 564/5c8d12d8 564/5c8d12d8 480/21f187ed
|
||||
12:19:59.073 t=606.6 s daa 573 epoch 9 class 5 signal v4 10000 v5 10000 bps blocks/sink per node 573/836b76e4 573/836b76e4 573/836b76e4 480/21f187ed
|
||||
12:20:14.098 t=621.6 s daa 593 epoch 9 class 5 signal v4 10000 v5 10000 bps blocks/sink per node 593/6b3a8177 593/6b3a8177 593/6b3a8177 480/21f187ed
|
||||
12:20:25.116 epoch 9 -> 10 at daa 600, 632.6 s: template class 5, next 5, day 1244010, signal share at the sink v4 10000 v5 10000 bps (window 60, this node signals 6, v5 decided by signal at epoch none)
|
||||
12:20:29.122 t=636.6 s daa 604 epoch 10 class 5 signal v4 10000 v5 10000 bps blocks/sink per node 604/f728bfa4 604/f728bfa4 604/f728bfa4 480/21f187ed
|
||||
12:20:44.151 t=651.7 s daa 621 epoch 10 class 5 signal v4 10000 v5 10000 bps blocks/sink per node 621/fd2ffc00 621/fd2ffc00 621/fd2ffc00 480/21f187ed
|
||||
12:20:59.171 t=666.7 s daa 637 epoch 10 class 5 signal v4 10000 v5 10000 bps blocks/sink per node 637/0fc792f1 637/0fc792f1 637/0fc792f1 480/21f187ed
|
||||
12:21:14.196 t=681.7 s daa 650 epoch 10 class 5 signal v4 10000 v5 10000 bps blocks/sink per node 650/c62861f9 650/c62861f9 650/c62861f9 480/21f187ed
|
||||
12:21:27.222 epoch 10 -> 11 at daa 660, 694.7 s: template class 5, next 5, day 1244010, signal share at the sink v4 10000 v5 10000 bps (window 60, this node signals 6, v5 decided by signal at epoch none)
|
||||
12:21:30.419 stream for epoch 8: {"code":-32000,"message":"no published state stream after chain block 268c6593e67a9dd74d676e8bed9a9425df30ab048aec9b80077394cb668242ba (the executor has not passed that epoch's cut, or the block is no
|
||||
12:21:30.569 SUMMARY PASS (expect flip, signals 6/6/6, stale 2): v5 from epoch 8 at DAA 480; epochs seen e0:2:0bps:d1244010 e1:3:10000bps:d1244010 e2:4:10000bps:d1244010 e3:4:10000bps:d1244010 e4:4:10000bps:d1244010 e5:4:10000bps:d1244010 e6:4:10000bps:d1244010 e7:4:10000bps:d1244010 e8:5:10000bps:d1244010 e9:5:10000bps:d1244010 e10:5:10000bps:d1244010 e11:5:10000bps:d1244010; chain bytes {"0":1,"6":663} (9985 bps at byte 6+); blocks 481 / 183; rejected miners 0/0/59/0 nodes 0/0/59; sinks cbaa8a516e2350f2 cbaa8a516e2350f2 cbaa8a516e2350f2 21f187edb6341649 (agree); counts 663/663/663/480; signal lines 3/3 (epochs 8/8/8, shares 10000/10000/10000); floor lines 3/3; state roots at the flip 0x74ca7aec647f5b7c 0x74ca7aec647f5b7c 0x74ca7aec647f5b7c; stateless node accepted after the flip 0 (refusal lines 14); stale miner accepted after the first refresh 0 rejected 59; days 1244010
|
||||
12:21:30.569 PROGRAM ID epoch 8 seed 268c6593e67a9dd7: miners d08d85950066de4d (3 of 3) cli v5 null cli v4 fa9cfb45ea50369e; state root null (null records)
|
||||
12:21:30.569 PROGRAM ID epoch 9 seed 08bff26781445846: miners 3208c2111780c61b (3 of 3) cli v5 3208c2111780c61b cli v4 051361b422d5ab40; state root 0xbf6b97cebb029f8199c4d19462fcde08a89fc15e7789528b8355e0b89ab6fd55 (17 records)
|
||||
12:21:30.569 PROGRAM ID epoch 10 seed 07f52472cc3bb9a7: miners a829ce27d55d786a (3 of 3) cli v5 a829ce27d55d786a cli v4 da0bf8b3363f1f75; state root 0x3cc8212d577bf3210e2431eabccdec1183200572009be61d6ef4c4ee52eedc0c (17 records)
|
||||
12:21:30.569 PROGRAM ID epoch 11 seed c62861f9c90ea32d: miners 14e1299c3e3f892c (3 of 3) cli v5 14e1299c3e3f892c cli v4 42560cc9c8175b5b; state root 0x74ca7aec647f5b7cfff9f04ac9196a2db54d3ccbe9290899fa6b8da489ac56a8 (17 records)
|
||||
12:21:30.570 summary: /tmp/igneum-fast-time-v5s-flip-stale/summary.json
|
||||
955
docs/design/class-v5-harness/no-flip.json
Normal file
955
docs/design/class-v5-harness/no-flip.json
Normal file
|
|
@ -0,0 +1,955 @@
|
|||
{
|
||||
"pass": true,
|
||||
"expect": "no-flip",
|
||||
"signals": [
|
||||
5,
|
||||
5,
|
||||
4
|
||||
],
|
||||
"checks": {
|
||||
"zero_rejected_by_miners": true,
|
||||
"zero_rejected_by_nodes": true,
|
||||
"sinks_agree": true,
|
||||
"block_counts_agree": true,
|
||||
"miners_agree_on_every_program": true,
|
||||
"window_line_on_every_node": true,
|
||||
"every_node_signals_its_byte": true,
|
||||
"chain_carries_the_bytes": true,
|
||||
"state_roots_agree_across_nodes": true,
|
||||
"template_never_v5": true,
|
||||
"no_signal_line_on_any_node": true,
|
||||
"ran_the_epochs": true,
|
||||
"v4_programs_seen": true,
|
||||
"signal_share_under_threshold_on_chain_or_the_object_off": true
|
||||
},
|
||||
"window": 60,
|
||||
"windows": 7,
|
||||
"floor": 100000,
|
||||
"v4_floor": 120,
|
||||
"v3_activation": 60,
|
||||
"epoch_blocks": 60,
|
||||
"lead": 10,
|
||||
"first_full_window_epoch": 8,
|
||||
"floor_epoch": 1667,
|
||||
"stale_miner": null,
|
||||
"stale_accepted_after_first_refresh": null,
|
||||
"stale_rejected": null,
|
||||
"stateless_node": 3,
|
||||
"stateless_accepted_after_flip": 0,
|
||||
"stateless_refusal_lines": 0,
|
||||
"state_roots_at_flip": [],
|
||||
"streams": {},
|
||||
"days_seen": [
|
||||
1244008
|
||||
],
|
||||
"day_boundary_crossed": false,
|
||||
"node": "/srv/builds/igneum-wt-class-v5/vendor/igneum-node-class-v5/target/release/igneumd",
|
||||
"miner": "/srv/builds/igneum-wt-class-v5/vendor/igneum-node-class-v5/target/release/igneum-miner",
|
||||
"template_switch": null,
|
||||
"run_ended_at_s": 667.6,
|
||||
"final_daa": 661,
|
||||
"max_epoch_seen": 11,
|
||||
"epochs": {
|
||||
"0": {
|
||||
"class": 2,
|
||||
"firstSeenDaa": 0,
|
||||
"at": 5.9,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 0,
|
||||
"bps5": 0,
|
||||
"signal_epoch": null,
|
||||
"day": 1244008
|
||||
},
|
||||
"1": {
|
||||
"class": 3,
|
||||
"firstSeenDaa": 60,
|
||||
"at": 39.9,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 10000,
|
||||
"bps5": 6274,
|
||||
"signal_epoch": null,
|
||||
"day": 1244008
|
||||
},
|
||||
"2": {
|
||||
"class": 4,
|
||||
"firstSeenDaa": 120,
|
||||
"at": 121,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 10000,
|
||||
"bps5": 6500,
|
||||
"signal_epoch": null,
|
||||
"day": 1244008
|
||||
},
|
||||
"3": {
|
||||
"class": 4,
|
||||
"firstSeenDaa": 181,
|
||||
"at": 180.1,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 10000,
|
||||
"bps5": 8500,
|
||||
"signal_epoch": null,
|
||||
"day": 1244008
|
||||
},
|
||||
"4": {
|
||||
"class": 4,
|
||||
"firstSeenDaa": 241,
|
||||
"at": 232.1,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 10000,
|
||||
"bps5": 8166,
|
||||
"signal_epoch": null,
|
||||
"day": 1244008
|
||||
},
|
||||
"5": {
|
||||
"class": 4,
|
||||
"firstSeenDaa": 300,
|
||||
"at": 304.2,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 10000,
|
||||
"bps5": 7500,
|
||||
"signal_epoch": null,
|
||||
"day": 1244008
|
||||
},
|
||||
"6": {
|
||||
"class": 4,
|
||||
"firstSeenDaa": 360,
|
||||
"at": 367.3,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 10000,
|
||||
"bps5": 8333,
|
||||
"signal_epoch": null,
|
||||
"day": 1244008
|
||||
},
|
||||
"7": {
|
||||
"class": 4,
|
||||
"firstSeenDaa": 420,
|
||||
"at": 428.3,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 10000,
|
||||
"bps5": 9000,
|
||||
"signal_epoch": null,
|
||||
"day": 1244008
|
||||
},
|
||||
"8": {
|
||||
"class": 4,
|
||||
"firstSeenDaa": 480,
|
||||
"at": 479.4,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 10000,
|
||||
"bps5": 6833,
|
||||
"signal_epoch": null,
|
||||
"day": 1244008
|
||||
},
|
||||
"9": {
|
||||
"class": 4,
|
||||
"firstSeenDaa": 540,
|
||||
"at": 548.5,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 10000,
|
||||
"bps5": 7500,
|
||||
"signal_epoch": null,
|
||||
"day": 1244008
|
||||
},
|
||||
"10": {
|
||||
"class": 4,
|
||||
"firstSeenDaa": 600,
|
||||
"at": 604.5,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 10000,
|
||||
"bps5": 7833,
|
||||
"signal_epoch": null,
|
||||
"day": 1244008
|
||||
},
|
||||
"11": {
|
||||
"class": 4,
|
||||
"firstSeenDaa": 661,
|
||||
"at": 667.6,
|
||||
"eraSeed": "234e082d653dc69db36d875b256d0f5f4a123b353d6aa069c3c7aa9aa6c46062",
|
||||
"bps": 10000,
|
||||
"bps5": 7333,
|
||||
"signal_epoch": null,
|
||||
"day": 1244008
|
||||
}
|
||||
},
|
||||
"blocks": {
|
||||
"total": 664,
|
||||
"before_boundary": 664,
|
||||
"after_boundary": 0,
|
||||
"version_bytes": {
|
||||
"0": 1,
|
||||
"4": 158,
|
||||
"5": 505
|
||||
},
|
||||
"signal_share_bps_on_chain": 7605
|
||||
},
|
||||
"programs": [
|
||||
{
|
||||
"epoch": 0,
|
||||
"class": "v2",
|
||||
"program_id": "8f8806638d59850f",
|
||||
"seed": "234e082d653dc69d",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
},
|
||||
{
|
||||
"epoch": 1,
|
||||
"class": "v3",
|
||||
"program_id": "b3f2c2f66954c8a7",
|
||||
"seed": "f4f80915d2de9af8",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
},
|
||||
{
|
||||
"epoch": 2,
|
||||
"class": "v4",
|
||||
"program_id": "545149ef59c1fefc",
|
||||
"seed": "2e8881610294f053",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
},
|
||||
{
|
||||
"epoch": 3,
|
||||
"class": "v4",
|
||||
"program_id": "3510fa678d663b4a",
|
||||
"seed": "8868423f2890cb64",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
},
|
||||
{
|
||||
"epoch": 4,
|
||||
"class": "v4",
|
||||
"program_id": "a83a6f534c4ab2bc",
|
||||
"seed": "43f5bc16da1848f8",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
},
|
||||
{
|
||||
"epoch": 5,
|
||||
"class": "v4",
|
||||
"program_id": "bef43ff61b99e952",
|
||||
"seed": "0170d2015058fada",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
},
|
||||
{
|
||||
"epoch": 6,
|
||||
"class": "v4",
|
||||
"program_id": "cfef05a86162c8f3",
|
||||
"seed": "95e5d049de44f0f7",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
},
|
||||
{
|
||||
"epoch": 7,
|
||||
"class": "v4",
|
||||
"program_id": "58ccfe71c8271397",
|
||||
"seed": "e276506183a4f0bd",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
},
|
||||
{
|
||||
"epoch": 8,
|
||||
"class": "v4",
|
||||
"program_id": "67d4d7e6931e6ba5",
|
||||
"seed": "da4faca08ee12e58",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
},
|
||||
{
|
||||
"epoch": 9,
|
||||
"class": "v4",
|
||||
"program_id": "7713a90df8ac836f",
|
||||
"seed": "57ecdeeeecdc9aeb",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
},
|
||||
{
|
||||
"epoch": 10,
|
||||
"class": "v4",
|
||||
"program_id": "87e24e87ec4e96ad",
|
||||
"seed": "05d841a8d563bc30",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
},
|
||||
{
|
||||
"epoch": 11,
|
||||
"class": "v4",
|
||||
"program_id": "b2dd57e30fd1d635",
|
||||
"seed": "c84991fe92df7685",
|
||||
"miners": 4,
|
||||
"disagree": false
|
||||
}
|
||||
],
|
||||
"program_id_rows": [],
|
||||
"accepted_per_miner": [
|
||||
165,
|
||||
171,
|
||||
158,
|
||||
169
|
||||
],
|
||||
"rejected_by_miners": [
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
0
|
||||
],
|
||||
"rejected_by_nodes": [
|
||||
0,
|
||||
0,
|
||||
0
|
||||
],
|
||||
"sinks": [
|
||||
"c687ac2e6d1ca9f6",
|
||||
"c687ac2e6d1ca9f6",
|
||||
"c687ac2e6d1ca9f6",
|
||||
"c687ac2e6d1ca9f6"
|
||||
],
|
||||
"block_counts": [
|
||||
663,
|
||||
663,
|
||||
663,
|
||||
663
|
||||
],
|
||||
"signal_lines": [
|
||||
null,
|
||||
null,
|
||||
null
|
||||
],
|
||||
"floor_lines": [
|
||||
"Program class v5 from the override file: enabled, the floor at epoch 1667 (DAA score 100000 rounded up to the epoch boundary at 100020); the class v5 signal (byte 5 or above, 9500 bps in each of 7 consecutive windows) can flip v4 to v5 before it; every item of the dataset is keyed by the execution state after the epoch's seed block",
|
||||
"Program class v5 from the override file: enabled, the floor at epoch 1667 (DAA score 100000 rounded up to the epoch boundary at 100020); the class v5 signal (byte 5 or above, 9500 bps in each of 7 consecutive windows) can flip v4 to v5 before it; every item of the dataset is keyed by the execution state after the epoch's seed block",
|
||||
"Program class v5 from the override file: enabled, the floor at epoch 1667 (DAA score 100000 rounded up to the epoch boundary at 100020); the class v5 signal (byte 5 or above, 9500 bps in each of 7 consecutive windows) can flip v4 to v5 before it; every item of the dataset is keyed by the execution state after the epoch's seed block"
|
||||
],
|
||||
"samples": [
|
||||
{
|
||||
"t": 5.9,
|
||||
"daa": 0,
|
||||
"epoch": 0,
|
||||
"class": 2,
|
||||
"bps": 0,
|
||||
"bps5": 0,
|
||||
"nodes": [
|
||||
"0/234e082d",
|
||||
"0/234e082d",
|
||||
"0/234e082d",
|
||||
"0/234e082d"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 20.9,
|
||||
"daa": 26,
|
||||
"epoch": 0,
|
||||
"class": 2,
|
||||
"bps": 10000,
|
||||
"bps5": 6153,
|
||||
"nodes": [
|
||||
"26/449cc72d",
|
||||
"26/449cc72d",
|
||||
"26/449cc72d",
|
||||
"26/449cc72d"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 35.9,
|
||||
"daa": 53,
|
||||
"epoch": 0,
|
||||
"class": 2,
|
||||
"bps": 10000,
|
||||
"bps5": 6274,
|
||||
"nodes": [
|
||||
"53/3339303d",
|
||||
"53/3339303d",
|
||||
"53/3339303d",
|
||||
"53/3339303d"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 50.9,
|
||||
"daa": 66,
|
||||
"epoch": 1,
|
||||
"class": 3,
|
||||
"bps": 10000,
|
||||
"bps5": 6500,
|
||||
"nodes": [
|
||||
"66/b38004d9",
|
||||
"66/b38004d9",
|
||||
"66/b38004d9",
|
||||
"66/b38004d9"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 65.9,
|
||||
"daa": 75,
|
||||
"epoch": 1,
|
||||
"class": 3,
|
||||
"bps": 10000,
|
||||
"bps5": 6833,
|
||||
"nodes": [
|
||||
"75/91f68ad7",
|
||||
"75/91f68ad7",
|
||||
"75/91f68ad7",
|
||||
"75/91f68ad7"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 81,
|
||||
"daa": 81,
|
||||
"epoch": 1,
|
||||
"class": 3,
|
||||
"bps": 10000,
|
||||
"bps5": 7166,
|
||||
"nodes": [
|
||||
"81/4f30405e",
|
||||
"81/4f30405e",
|
||||
"81/4f30405e",
|
||||
"81/4f30405e"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 96,
|
||||
"daa": 95,
|
||||
"epoch": 1,
|
||||
"class": 3,
|
||||
"bps": 10000,
|
||||
"bps5": 7333,
|
||||
"nodes": [
|
||||
"95/e2a8b863",
|
||||
"95/e2a8b863",
|
||||
"95/e2a8b863",
|
||||
"95/e2a8b863"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 111,
|
||||
"daa": 112,
|
||||
"epoch": 1,
|
||||
"class": 3,
|
||||
"bps": 10000,
|
||||
"bps5": 6666,
|
||||
"nodes": [
|
||||
"112/37441227",
|
||||
"112/37441227",
|
||||
"112/37441227",
|
||||
"112/37441227"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 126,
|
||||
"daa": 123,
|
||||
"epoch": 2,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 6500,
|
||||
"nodes": [
|
||||
"123/02be60f8",
|
||||
"123/02be60f8",
|
||||
"123/02be60f8",
|
||||
"123/02be60f8"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 141,
|
||||
"daa": 138,
|
||||
"epoch": 2,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 6500,
|
||||
"nodes": [
|
||||
"138/778cfa35",
|
||||
"138/778cfa35",
|
||||
"138/778cfa35",
|
||||
"138/778cfa35"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 156,
|
||||
"daa": 157,
|
||||
"epoch": 2,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7166,
|
||||
"nodes": [
|
||||
"157/dce8a5e1",
|
||||
"157/dce8a5e1",
|
||||
"157/dce8a5e1",
|
||||
"157/dce8a5e1"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 171.1,
|
||||
"daa": 172,
|
||||
"epoch": 2,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8000,
|
||||
"nodes": [
|
||||
"172/0f551d81",
|
||||
"172/0f551d81",
|
||||
"172/0f551d81",
|
||||
"172/0f551d81"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 186.1,
|
||||
"daa": 184,
|
||||
"epoch": 3,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8500,
|
||||
"nodes": [
|
||||
"184/6e408999",
|
||||
"184/6e408999",
|
||||
"184/6e408999",
|
||||
"184/6e408999"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 201.1,
|
||||
"daa": 204,
|
||||
"epoch": 3,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8833,
|
||||
"nodes": [
|
||||
"204/91e6f463",
|
||||
"204/91e6f463",
|
||||
"204/91e6f463",
|
||||
"204/91e6f463"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 216.1,
|
||||
"daa": 218,
|
||||
"epoch": 3,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8166,
|
||||
"nodes": [
|
||||
"218/cab8a3ad",
|
||||
"218/cab8a3ad",
|
||||
"218/cab8a3ad",
|
||||
"218/cab8a3ad"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 231.1,
|
||||
"daa": 238,
|
||||
"epoch": 3,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8166,
|
||||
"nodes": [
|
||||
"238/1bd75538",
|
||||
"238/1bd75538",
|
||||
"238/1bd75538",
|
||||
"238/1bd75538"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 246.1,
|
||||
"daa": 259,
|
||||
"epoch": 4,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8000,
|
||||
"nodes": [
|
||||
"259/d33c600c",
|
||||
"259/d33c600c",
|
||||
"259/d33c600c",
|
||||
"259/d33c600c"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 261.2,
|
||||
"daa": 275,
|
||||
"epoch": 4,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7500,
|
||||
"nodes": [
|
||||
"275/7fb155e9",
|
||||
"275/7fb155e9",
|
||||
"275/7fb155e9",
|
||||
"275/7fb155e9"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 276.2,
|
||||
"daa": 282,
|
||||
"epoch": 4,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7666,
|
||||
"nodes": [
|
||||
"282/c2a84641",
|
||||
"282/c2a84641",
|
||||
"282/c2a84641",
|
||||
"282/c2a84641"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 291.2,
|
||||
"daa": 289,
|
||||
"epoch": 4,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7500,
|
||||
"nodes": [
|
||||
"289/2831cde0",
|
||||
"289/2831cde0",
|
||||
"289/2831cde0",
|
||||
"289/2831cde0"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 306.2,
|
||||
"daa": 301,
|
||||
"epoch": 5,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7833,
|
||||
"nodes": [
|
||||
"301/a1e4903a",
|
||||
"301/a1e4903a",
|
||||
"301/a1e4903a",
|
||||
"301/a1e4903a"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 321.2,
|
||||
"daa": 316,
|
||||
"epoch": 5,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8500,
|
||||
"nodes": [
|
||||
"316/a1fd0963",
|
||||
"316/a1fd0963",
|
||||
"316/a1fd0963",
|
||||
"316/a1fd0963"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 336.2,
|
||||
"daa": 332,
|
||||
"epoch": 5,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8833,
|
||||
"nodes": [
|
||||
"332/50c80fec",
|
||||
"332/50c80fec",
|
||||
"332/50c80fec",
|
||||
"332/50c80fec"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 351.2,
|
||||
"daa": 346,
|
||||
"epoch": 5,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8333,
|
||||
"nodes": [
|
||||
"346/2d2c9130",
|
||||
"346/2d2c9130",
|
||||
"346/2d2c9130",
|
||||
"346/2d2c9130"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 366.3,
|
||||
"daa": 359,
|
||||
"epoch": 5,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8333,
|
||||
"nodes": [
|
||||
"359/2d9712d9",
|
||||
"359/2d9712d9",
|
||||
"359/2d9712d9",
|
||||
"359/2d9712d9"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 381.3,
|
||||
"daa": 372,
|
||||
"epoch": 6,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8166,
|
||||
"nodes": [
|
||||
"372/1a7b4b19",
|
||||
"372/1a7b4b19",
|
||||
"372/1a7b4b19",
|
||||
"372/1a7b4b19"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 396.3,
|
||||
"daa": 389,
|
||||
"epoch": 6,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8333,
|
||||
"nodes": [
|
||||
"389/ed6eca97",
|
||||
"389/ed6eca97",
|
||||
"389/ed6eca97",
|
||||
"389/ed6eca97"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 411.3,
|
||||
"daa": 402,
|
||||
"epoch": 6,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8833,
|
||||
"nodes": [
|
||||
"402/5f047cc2",
|
||||
"402/5f047cc2",
|
||||
"402/5f047cc2",
|
||||
"402/5f047cc2"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 426.3,
|
||||
"daa": 418,
|
||||
"epoch": 6,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 9000,
|
||||
"nodes": [
|
||||
"418/9f8a9303",
|
||||
"418/9f8a9303",
|
||||
"418/9f8a9303",
|
||||
"418/9f8a9303"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 441.3,
|
||||
"daa": 432,
|
||||
"epoch": 7,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8666,
|
||||
"nodes": [
|
||||
"432/ecfce394",
|
||||
"432/ecfce394",
|
||||
"432/ecfce394",
|
||||
"432/ecfce394"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 456.4,
|
||||
"daa": 453,
|
||||
"epoch": 7,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8000,
|
||||
"nodes": [
|
||||
"453/2a968a32",
|
||||
"453/2a968a32",
|
||||
"453/2a968a32",
|
||||
"453/2a968a32"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 471.4,
|
||||
"daa": 467,
|
||||
"epoch": 7,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7833,
|
||||
"nodes": [
|
||||
"467/15a7836e",
|
||||
"467/15a7836e",
|
||||
"467/15a7836e",
|
||||
"467/15a7836e"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 486.4,
|
||||
"daa": 486,
|
||||
"epoch": 8,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7333,
|
||||
"nodes": [
|
||||
"486/3834d4d9",
|
||||
"486/3834d4d9",
|
||||
"486/3834d4d9",
|
||||
"486/3834d4d9"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 501.4,
|
||||
"daa": 500,
|
||||
"epoch": 8,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7333,
|
||||
"nodes": [
|
||||
"500/12805603",
|
||||
"500/12805603",
|
||||
"500/12805603",
|
||||
"500/12805603"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 516.4,
|
||||
"daa": 514,
|
||||
"epoch": 8,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7166,
|
||||
"nodes": [
|
||||
"514/feb2b6e9",
|
||||
"514/feb2b6e9",
|
||||
"514/feb2b6e9",
|
||||
"514/feb2b6e9"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 531.5,
|
||||
"daa": 521,
|
||||
"epoch": 8,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7000,
|
||||
"nodes": [
|
||||
"521/2bcc56ef",
|
||||
"521/2bcc56ef",
|
||||
"521/2bcc56ef",
|
||||
"521/2bcc56ef"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 546.5,
|
||||
"daa": 536,
|
||||
"epoch": 8,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7500,
|
||||
"nodes": [
|
||||
"536/9de8b82e",
|
||||
"536/9de8b82e",
|
||||
"536/9de8b82e",
|
||||
"536/9de8b82e"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 561.5,
|
||||
"daa": 557,
|
||||
"epoch": 9,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7166,
|
||||
"nodes": [
|
||||
"557/87723e7e",
|
||||
"557/87723e7e",
|
||||
"557/87723e7e",
|
||||
"557/87723e7e"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 576.5,
|
||||
"daa": 564,
|
||||
"epoch": 9,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7166,
|
||||
"nodes": [
|
||||
"564/b2f11783",
|
||||
"564/b2f11783",
|
||||
"564/b2f11783",
|
||||
"564/b2f11783"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 591.5,
|
||||
"daa": 587,
|
||||
"epoch": 9,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7666,
|
||||
"nodes": [
|
||||
"587/da0e0675",
|
||||
"587/da0e0675",
|
||||
"587/da0e0675",
|
||||
"587/da0e0675"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 606.5,
|
||||
"daa": 601,
|
||||
"epoch": 10,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8000,
|
||||
"nodes": [
|
||||
"601/9d112e18",
|
||||
"601/9d112e18",
|
||||
"601/9d112e18",
|
||||
"601/9d112e18"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 621.5,
|
||||
"daa": 621,
|
||||
"epoch": 10,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 8166,
|
||||
"nodes": [
|
||||
"621/70e8c328",
|
||||
"621/70e8c328",
|
||||
"621/70e8c328",
|
||||
"621/70e8c328"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 636.6,
|
||||
"daa": 634,
|
||||
"epoch": 10,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7833,
|
||||
"nodes": [
|
||||
"634/93b8bff2",
|
||||
"634/93b8bff2",
|
||||
"634/93b8bff2",
|
||||
"634/93b8bff2"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 651.6,
|
||||
"daa": 651,
|
||||
"epoch": 10,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7666,
|
||||
"nodes": [
|
||||
"651/64e0e1c0",
|
||||
"651/64e0e1c0",
|
||||
"651/64e0e1c0",
|
||||
"651/64e0e1c0"
|
||||
]
|
||||
},
|
||||
{
|
||||
"t": 666.6,
|
||||
"daa": 659,
|
||||
"epoch": 10,
|
||||
"class": 4,
|
||||
"bps": 10000,
|
||||
"bps5": 7333,
|
||||
"nodes": [
|
||||
"659/274ca6c8",
|
||||
"659/274ca6c8",
|
||||
"659/274ca6c8",
|
||||
"659/274ca6c8"
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
69
docs/design/class-v5-harness/no-flip.log
Normal file
69
docs/design/class-v5-harness/no-flip.log
Normal file
|
|
@ -0,0 +1,69 @@
|
|||
11:12:35.201 signals 5/5/4, expect no-flip; v3 from 60 (epoch 1), v4 floor 120 (epoch 2), v5 floor 100000 (epoch 1667), window 60 DAA x 7 (the first epoch with seven full windows after v4 is 8); stale miner none; stateless node n3; 60 DAA per epoch, lead 10; run 780 s or 11 epochs
|
||||
11:12:36.435 n0 up pid 2674788 json 30262 p2p 30261 exec 30263, signals 5
|
||||
11:12:37.644 n1 up pid 2675266 json 30272 p2p 30271 exec 30273, signals 5
|
||||
11:12:38.853 n2 up pid 2675739 json 30282 p2p 30281 exec 30283, signals 4
|
||||
11:12:40.060 n3 up pid 2676230 json 30292 p2p 30291 exec disabled, signals 5
|
||||
11:12:40.061 n0: Program class v4 signal window from the override file: 60 DAA ending at each epoch's seed block, threshold 9500 bps of blue blocks; the fixed height is the floor | this node signals object version 5
|
||||
11:12:40.061 n1: Program class v4 signal window from the override file: 60 DAA ending at each epoch's seed block, threshold 9500 bps of blue blocks; the fixed height is the floor | this node signals object version 5
|
||||
11:12:40.061 n2: Program class v4 signal window from the override file: 60 DAA ending at each epoch's seed block, threshold 9500 bps of blue blocks; the fixed height is the floor | this node signals object version 4
|
||||
11:12:40.061 n3: Program class v4 signal window from the override file: 60 DAA ending at each epoch's seed block, threshold 9500 bps of blue blocks; the fixed height is the floor | this node signals object version 5
|
||||
11:12:40.062 n0 digest: 5d04822e3501a5de
|
||||
11:12:41.074 epoch -1 -> 0 at daa 0, 5.9 s: template class 2, next 2, day 1244008, signal share at the sink v4 0 v5 0 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
11:12:41.076 t=5.9 s daa 0 epoch 0 class 2 signal v4 0 v5 0 bps blocks/sink per node 0/234e082d 0/234e082d 0/234e082d 0/234e082d
|
||||
11:12:56.098 t=20.9 s daa 26 epoch 0 class 2 signal v4 10000 v5 6153 bps blocks/sink per node 26/449cc72d 26/449cc72d 26/449cc72d 26/449cc72d
|
||||
11:13:11.117 t=35.9 s daa 53 epoch 0 class 2 signal v4 10000 v5 6274 bps blocks/sink per node 53/3339303d 53/3339303d 53/3339303d 53/3339303d
|
||||
11:13:15.121 epoch 0 -> 1 at daa 60, 39.9 s: template class 3, next 3, day 1244008, signal share at the sink v4 10000 v5 6274 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
11:13:26.132 t=50.9 s daa 66 epoch 1 class 3 signal v4 10000 v5 6500 bps blocks/sink per node 66/b38004d9 66/b38004d9 66/b38004d9 66/b38004d9
|
||||
11:13:41.149 t=65.9 s daa 75 epoch 1 class 3 signal v4 10000 v5 6833 bps blocks/sink per node 75/91f68ad7 75/91f68ad7 75/91f68ad7 75/91f68ad7
|
||||
11:13:56.166 t=81.0 s daa 81 epoch 1 class 3 signal v4 10000 v5 7166 bps blocks/sink per node 81/4f30405e 81/4f30405e 81/4f30405e 81/4f30405e
|
||||
11:14:11.182 t=96.0 s daa 95 epoch 1 class 3 signal v4 10000 v5 7333 bps blocks/sink per node 95/e2a8b863 95/e2a8b863 95/e2a8b863 95/e2a8b863
|
||||
11:14:26.197 t=111.0 s daa 112 epoch 1 class 3 signal v4 10000 v5 6666 bps blocks/sink per node 112/37441227 112/37441227 112/37441227 112/37441227
|
||||
11:14:36.210 epoch 1 -> 2 at daa 120, 121.0 s: template class 4, next 4, day 1244008, signal share at the sink v4 10000 v5 6500 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
11:14:41.217 t=126.0 s daa 123 epoch 2 class 4 signal v4 10000 v5 6500 bps blocks/sink per node 123/02be60f8 123/02be60f8 123/02be60f8 123/02be60f8
|
||||
11:14:56.232 t=141.0 s daa 138 epoch 2 class 4 signal v4 10000 v5 6500 bps blocks/sink per node 138/778cfa35 138/778cfa35 138/778cfa35 138/778cfa35
|
||||
11:15:11.247 t=156.0 s daa 157 epoch 2 class 4 signal v4 10000 v5 7166 bps blocks/sink per node 157/dce8a5e1 157/dce8a5e1 157/dce8a5e1 157/dce8a5e1
|
||||
11:15:26.261 t=171.1 s daa 172 epoch 2 class 4 signal v4 10000 v5 8000 bps blocks/sink per node 172/0f551d81 172/0f551d81 172/0f551d81 172/0f551d81
|
||||
11:15:35.271 epoch 2 -> 3 at daa 181, 180.1 s: template class 4, next 4, day 1244008, signal share at the sink v4 10000 v5 8500 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
11:15:41.277 t=186.1 s daa 184 epoch 3 class 4 signal v4 10000 v5 8500 bps blocks/sink per node 184/6e408999 184/6e408999 184/6e408999 184/6e408999
|
||||
11:15:56.292 t=201.1 s daa 204 epoch 3 class 4 signal v4 10000 v5 8833 bps blocks/sink per node 204/91e6f463 204/91e6f463 204/91e6f463 204/91e6f463
|
||||
11:16:11.309 t=216.1 s daa 218 epoch 3 class 4 signal v4 10000 v5 8166 bps blocks/sink per node 218/cab8a3ad 218/cab8a3ad 218/cab8a3ad 218/cab8a3ad
|
||||
11:16:26.325 t=231.1 s daa 238 epoch 3 class 4 signal v4 10000 v5 8166 bps blocks/sink per node 238/1bd75538 238/1bd75538 238/1bd75538 238/1bd75538
|
||||
11:16:27.326 epoch 3 -> 4 at daa 241, 232.1 s: template class 4, next 4, day 1244008, signal share at the sink v4 10000 v5 8166 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
11:16:41.343 t=246.1 s daa 259 epoch 4 class 4 signal v4 10000 v5 8000 bps blocks/sink per node 259/d33c600c 259/d33c600c 259/d33c600c 259/d33c600c
|
||||
11:16:56.355 t=261.2 s daa 275 epoch 4 class 4 signal v4 10000 v5 7500 bps blocks/sink per node 275/7fb155e9 275/7fb155e9 275/7fb155e9 275/7fb155e9
|
||||
11:17:11.372 t=276.2 s daa 282 epoch 4 class 4 signal v4 10000 v5 7666 bps blocks/sink per node 282/c2a84641 282/c2a84641 282/c2a84641 282/c2a84641
|
||||
11:17:26.385 t=291.2 s daa 289 epoch 4 class 4 signal v4 10000 v5 7500 bps blocks/sink per node 289/2831cde0 289/2831cde0 289/2831cde0 289/2831cde0
|
||||
11:17:39.398 epoch 4 -> 5 at daa 300, 304.2 s: template class 4, next 4, day 1244008, signal share at the sink v4 10000 v5 7500 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
11:17:41.400 t=306.2 s daa 301 epoch 5 class 4 signal v4 10000 v5 7833 bps blocks/sink per node 301/a1e4903a 301/a1e4903a 301/a1e4903a 301/a1e4903a
|
||||
11:17:56.417 t=321.2 s daa 316 epoch 5 class 4 signal v4 10000 v5 8500 bps blocks/sink per node 316/a1fd0963 316/a1fd0963 316/a1fd0963 316/a1fd0963
|
||||
11:18:11.433 t=336.2 s daa 332 epoch 5 class 4 signal v4 10000 v5 8833 bps blocks/sink per node 332/50c80fec 332/50c80fec 332/50c80fec 332/50c80fec
|
||||
11:18:26.449 t=351.2 s daa 346 epoch 5 class 4 signal v4 10000 v5 8333 bps blocks/sink per node 346/2d2c9130 346/2d2c9130 346/2d2c9130 346/2d2c9130
|
||||
11:18:41.466 t=366.3 s daa 359 epoch 5 class 4 signal v4 10000 v5 8333 bps blocks/sink per node 359/2d9712d9 359/2d9712d9 359/2d9712d9 359/2d9712d9
|
||||
11:18:42.468 epoch 5 -> 6 at daa 360, 367.3 s: template class 4, next 4, day 1244008, signal share at the sink v4 10000 v5 8333 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
11:18:56.483 t=381.3 s daa 372 epoch 6 class 4 signal v4 10000 v5 8166 bps blocks/sink per node 372/1a7b4b19 372/1a7b4b19 372/1a7b4b19 372/1a7b4b19
|
||||
11:19:11.504 t=396.3 s daa 389 epoch 6 class 4 signal v4 10000 v5 8333 bps blocks/sink per node 389/ed6eca97 389/ed6eca97 389/ed6eca97 389/ed6eca97
|
||||
11:19:26.518 t=411.3 s daa 402 epoch 6 class 4 signal v4 10000 v5 8833 bps blocks/sink per node 402/5f047cc2 402/5f047cc2 402/5f047cc2 402/5f047cc2
|
||||
11:19:41.534 t=426.3 s daa 418 epoch 6 class 4 signal v4 10000 v5 9000 bps blocks/sink per node 418/9f8a9303 418/9f8a9303 418/9f8a9303 418/9f8a9303
|
||||
11:19:43.536 epoch 6 -> 7 at daa 420, 428.3 s: template class 4, next 4, day 1244008, signal share at the sink v4 10000 v5 9000 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
11:19:56.550 t=441.3 s daa 432 epoch 7 class 4 signal v4 10000 v5 8666 bps blocks/sink per node 432/ecfce394 432/ecfce394 432/ecfce394 432/ecfce394
|
||||
11:20:11.565 t=456.4 s daa 453 epoch 7 class 4 signal v4 10000 v5 8000 bps blocks/sink per node 453/2a968a32 453/2a968a32 453/2a968a32 453/2a968a32
|
||||
11:20:26.581 t=471.4 s daa 467 epoch 7 class 4 signal v4 10000 v5 7833 bps blocks/sink per node 467/15a7836e 467/15a7836e 467/15a7836e 467/15a7836e
|
||||
11:20:34.590 epoch 7 -> 8 at daa 480, 479.4 s: template class 4, next 4, day 1244008, signal share at the sink v4 10000 v5 6833 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
11:20:41.599 t=486.4 s daa 486 epoch 8 class 4 signal v4 10000 v5 7333 bps blocks/sink per node 486/3834d4d9 486/3834d4d9 486/3834d4d9 486/3834d4d9
|
||||
11:20:56.616 t=501.4 s daa 500 epoch 8 class 4 signal v4 10000 v5 7333 bps blocks/sink per node 500/12805603 500/12805603 500/12805603 500/12805603
|
||||
11:21:11.636 t=516.4 s daa 514 epoch 8 class 4 signal v4 10000 v5 7166 bps blocks/sink per node 514/feb2b6e9 514/feb2b6e9 514/feb2b6e9 514/feb2b6e9
|
||||
11:21:26.652 t=531.5 s daa 521 epoch 8 class 4 signal v4 10000 v5 7000 bps blocks/sink per node 521/2bcc56ef 521/2bcc56ef 521/2bcc56ef 521/2bcc56ef
|
||||
11:21:41.668 t=546.5 s daa 536 epoch 8 class 4 signal v4 10000 v5 7500 bps blocks/sink per node 536/9de8b82e 536/9de8b82e 536/9de8b82e 536/9de8b82e
|
||||
11:21:43.671 epoch 8 -> 9 at daa 540, 548.5 s: template class 4, next 4, day 1244008, signal share at the sink v4 10000 v5 7500 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
11:21:56.685 t=561.5 s daa 557 epoch 9 class 4 signal v4 10000 v5 7166 bps blocks/sink per node 557/87723e7e 557/87723e7e 557/87723e7e 557/87723e7e
|
||||
11:22:11.699 t=576.5 s daa 564 epoch 9 class 4 signal v4 10000 v5 7166 bps blocks/sink per node 564/b2f11783 564/b2f11783 564/b2f11783 564/b2f11783
|
||||
11:22:26.717 t=591.5 s daa 587 epoch 9 class 4 signal v4 10000 v5 7666 bps blocks/sink per node 587/da0e0675 587/da0e0675 587/da0e0675 587/da0e0675
|
||||
11:22:39.730 epoch 9 -> 10 at daa 600, 604.5 s: template class 4, next 4, day 1244008, signal share at the sink v4 10000 v5 7833 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
11:22:41.733 t=606.5 s daa 601 epoch 10 class 4 signal v4 10000 v5 8000 bps blocks/sink per node 601/9d112e18 601/9d112e18 601/9d112e18 601/9d112e18
|
||||
11:22:56.748 t=621.5 s daa 621 epoch 10 class 4 signal v4 10000 v5 8166 bps blocks/sink per node 621/70e8c328 621/70e8c328 621/70e8c328 621/70e8c328
|
||||
11:23:11.762 t=636.6 s daa 634 epoch 10 class 4 signal v4 10000 v5 7833 bps blocks/sink per node 634/93b8bff2 634/93b8bff2 634/93b8bff2 634/93b8bff2
|
||||
11:23:26.775 t=651.6 s daa 651 epoch 10 class 4 signal v4 10000 v5 7666 bps blocks/sink per node 651/64e0e1c0 651/64e0e1c0 651/64e0e1c0 651/64e0e1c0
|
||||
11:23:41.792 t=666.6 s daa 659 epoch 10 class 4 signal v4 10000 v5 7333 bps blocks/sink per node 659/274ca6c8 659/274ca6c8 659/274ca6c8 659/274ca6c8
|
||||
11:23:42.793 epoch 10 -> 11 at daa 661, 667.6 s: template class 4, next 4, day 1244008, signal share at the sink v4 10000 v5 7333 bps (window 60, this node signals 5, v5 decided by signal at epoch none)
|
||||
11:23:45.957 SUMMARY PASS (expect no-flip, signals 5/5/4, stale none): no v5 epoch; epochs seen e0:2:0bps:d1244008 e1:3:6274bps:d1244008 e2:4:6500bps:d1244008 e3:4:8500bps:d1244008 e4:4:8166bps:d1244008 e5:4:7500bps:d1244008 e6:4:8333bps:d1244008 e7:4:9000bps:d1244008 e8:4:6833bps:d1244008 e9:4:7500bps:d1244008 e10:4:7833bps:d1244008 e11:4:7333bps:d1244008; chain bytes {"0":1,"4":158,"5":505} (7605 bps at byte 5+); blocks 664 / 0; rejected miners 0/0/0/0 nodes 0/0/0; sinks c687ac2e6d1ca9f6 c687ac2e6d1ca9f6 c687ac2e6d1ca9f6 c687ac2e6d1ca9f6 (agree); counts 663/663/663/663; signal lines 0/3 (epochs //, shares //); floor lines 3/3; state roots at the flip ; stateless node accepted after the flip 0 (refusal lines 0); stale miner accepted after the first refresh n/a rejected n/a; days 1244008
|
||||
11:23:45.957 summary: /tmp/igneum-fast-time-v5s-no-flip/summary.json
|
||||
280
docs/design/class-v5-stored-state.md
Normal file
280
docs/design/class-v5-stored-state.md
Normal file
File diff suppressed because one or more lines are too long
|
|
@ -193,6 +193,15 @@ Answer: Correct on both counts, and the second was the sharper one. The X9 (Bitm
|
|||
|
||||
Evidence: `docs/design/latency-ladder.md` (the rule, the hostile review, the measured verifier table, the X9 arithmetic); `igneum-pow/src/generator.rs` test `latency_ladder_known_failed_a_changed_n_was_a_hard_fork_and_rungs_are_class_v4`; the fork's `consensus/core/src/igneum.rs` test `latency_ladder_rule`, `consensus/pow/src/igneum.rs` test `latency_ladder_rungs_are_programs_of_their_own_over_one_day_cache`; `infra/fast-time/latency-ladder.mjs` (the step, no-step and known-failed cases).
|
||||
|
||||
### M35. "Proof of stored state" proves nothing a pool cannot ship, and "proof of following" is a 32 ms rebuild per hour
|
||||
"Class v5 keys the dataset by the chain's state so that every hash proves the miner holds the chain. The state is 6 KB. A pool ships it with the template. The hourly refresh is a 32 ms rebuild that a farm's one node does once and broadcasts. Nothing on the chain can tell a card that derived the leaves from a card that received them, so the class proves nothing about who holds what, and it adds consensus-critical serialisation code for the privilege."
|
||||
|
||||
Status: Implemented behind a switch (7 October 2026, `docs/design/class-v5-stored-state.md`, branch `class-v5`, fork branch `class-v5-node` from the 0.3.18 node, behind `program_class_v5_activation_daa`, never until set): the dataset of every epoch is built from the canonical state stream after the epoch's seed block (`igneum/exec/src/day_stream.rs`: accounts, non-zero slots, code chunks, one serialisation that rebuilds to its root), hashed into leaves `D[i] = Blake2b-512('igneum-sd1/' || root || i || record)` and folded into every item (`leaf(t) = D[t mod n]`, so a hasher without the state, with another root, with a stream one record short or with the previous epoch's leaves is wrong on 64 of 64 items and 32 of 32 lanes: the known-failed case, `igneum-pow/src/state.rs`); the object byte 5 and the 95 percent seven-window tally beside class v4's (`consensus/core/src/igneum.rs` `program_class_for_epoch_signalled_v5`, one step per epoch); a node without the state refuses the header with a retryable error and serves no template (`kaspa-pow` `DayStateUnavailable`, the known-failed case `class_v5_refuses_without_state_and_refreshes_the_leaves_per_epoch`); the miner fetches the stream from its node's exec RPC (`igneum_getPowStateLeaves`); the fast-time gate `infra/fast-time/class-v5-signal.mjs` with a stateless node and a stale miner.
|
||||
|
||||
Answer: The first sentence is right and the page says it first: anything the lottery derives is derived from a seed and the state, so the only bytes a central node cannot compress away are the state's, 5,952 bytes at today's devnet state (93 records), and the chain cannot distinguish a card that derived the leaves from one that received them, exactly as it cannot distinguish a pool member from a solo miner today. The numbers are on the page: the leaves pass 45 MB per member per hour, the line where a WAN pool at 1 Gbit/s serving 10,000 members can no longer ship them inside the window, at about 700,000 state records; a LAN farm at 100 Gbit/s is never bounded below the 2 GiB sample cap. What the class does force, per machine and per hour: holding the current state or its leaves, a rebuild from it (32 ms on a 4090, measured on 6 October), and knowledge of the chain's reference block inside the ten-minute lead; a machine cut off from the chain for an hour stops producing valid blocks at the next refresh, where under a daily rule it kept mining until midnight. It also removes the recompute chip (the f = 0 row of chip-model-v3) as a category and moves nothing against the dataset-storing chip, which the page and the litepaper both say. The cost is measured and small (hash rate and watts unchanged on the 4090, build +1.4 ms resident, verifier +0.11 to 0.21 ms per unit); the risk is the serialisation, which is why the stream must rebuild to its root on the capturing node before it is served and why the gate crosses a day boundary with a non-trivial state before Devnet 2.
|
||||
|
||||
Evidence: `docs/design/class-v5-stored-state.md` (sections 2, 2a, 4, 5, 8); `igneum-pow/src/state.rs` tests; the fork's `igneum/exec/src/day_stream.rs` tests (`a_tampered_stream_is_refused`), `consensus/core/src/igneum.rs` (`program_class_signal_rule`, the v5 cases), `consensus/pow/src/igneum.rs`, `igneum/exec/src/service.rs` (`epoch_state_captures_follow_the_cut_and_unwind`); `infra/fast-time/class-v5-signal.mjs`.
|
||||
|
||||
## 2. Finality and attacks
|
||||
|
||||
### F1. Finality is attackable for the first month
|
||||
|
|
|
|||
|
|
@ -249,7 +249,8 @@ pub fn distinct_indices_v4(p: &Program, units: usize) -> Result<Vec<u32>, Reject
|
|||
/// the era set aside): the shape the sub-version 2 rules (a') and (c') apply to, on every draw path.
|
||||
pub fn is_class_v4_shape(class: &LoadClass) -> bool {
|
||||
matches!(class.shadow, Some(ShadowClass { instrs: V4_SHADOW_INSTRS, .. }))
|
||||
&& LoadClass { era: None, shadow: None, ..*class } == LoadClass { shadow: None, ..V4_CLASS }
|
||||
// class v5 (docs/design/class-v5-stored-state.md) is judged under the same rules: its state flag is set aside
|
||||
&& LoadClass { era: None, shadow: None, state: false, ..*class } == LoadClass { shadow: None, ..V4_CLASS }
|
||||
}
|
||||
|
||||
/// One pass of the dataflow freshness over the base program then the shadow block (the order of one iteration),
|
||||
|
|
|
|||
152
igneum-pow/src/blake2b.rs
Normal file
152
igneum-pow/src/blake2b.rs
Normal file
|
|
@ -0,0 +1,152 @@
|
|||
//! BLAKE2b (RFC 7693), the chain's own hash family (spec 01 section 0.6), written out here so the crate keeps its
|
||||
//! rule of no dependency outside the standard library. Used by class v5's state leaves (`crate::state`):
|
||||
//! `blake2b_512` for a leaf digest, `blake2b_256` for the sample order. Unkeyed, no salt, no personalisation.
|
||||
//! Checked against the RFC's "abc" vector and the empty-input vector in the tests.
|
||||
|
||||
const IV: [u64; 8] = [
|
||||
0x6a09e667f3bcc908,
|
||||
0xbb67ae8584caa73b,
|
||||
0x3c6ef372fe94f82b,
|
||||
0xa54ff53a5f1d36f1,
|
||||
0x510e527fade682d1,
|
||||
0x9b05688c2b3e6c1f,
|
||||
0x1f83d9abfb41bd6b,
|
||||
0x5be0cd19137e2179,
|
||||
];
|
||||
|
||||
const SIGMA: [[usize; 16]; 12] = [
|
||||
[0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15],
|
||||
[14, 10, 4, 8, 9, 15, 13, 6, 1, 12, 0, 2, 11, 7, 5, 3],
|
||||
[11, 8, 12, 0, 5, 2, 15, 13, 10, 14, 3, 6, 7, 1, 9, 4],
|
||||
[7, 9, 3, 1, 13, 12, 11, 14, 2, 6, 5, 10, 4, 0, 15, 8],
|
||||
[9, 0, 5, 7, 2, 4, 10, 15, 14, 1, 11, 12, 6, 8, 3, 13],
|
||||
[2, 12, 6, 10, 0, 11, 8, 3, 4, 13, 7, 5, 15, 14, 1, 9],
|
||||
[12, 5, 1, 15, 14, 13, 4, 10, 0, 7, 6, 3, 9, 2, 8, 11],
|
||||
[13, 11, 7, 14, 12, 1, 3, 9, 5, 0, 15, 4, 8, 6, 2, 10],
|
||||
[6, 15, 14, 9, 11, 3, 0, 8, 12, 2, 13, 7, 1, 4, 10, 5],
|
||||
[10, 2, 8, 4, 7, 6, 1, 5, 15, 11, 9, 14, 3, 12, 13, 0],
|
||||
[0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15],
|
||||
[14, 10, 4, 8, 9, 15, 13, 6, 1, 12, 0, 2, 11, 7, 5, 3],
|
||||
];
|
||||
|
||||
#[inline(always)]
|
||||
fn g(v: &mut [u64; 16], a: usize, b: usize, c: usize, d: usize, x: u64, y: u64) {
|
||||
v[a] = v[a].wrapping_add(v[b]).wrapping_add(x);
|
||||
v[d] = (v[d] ^ v[a]).rotate_right(32);
|
||||
v[c] = v[c].wrapping_add(v[d]);
|
||||
v[b] = (v[b] ^ v[c]).rotate_right(24);
|
||||
v[a] = v[a].wrapping_add(v[b]).wrapping_add(y);
|
||||
v[d] = (v[d] ^ v[a]).rotate_right(16);
|
||||
v[c] = v[c].wrapping_add(v[d]);
|
||||
v[b] = (v[b] ^ v[c]).rotate_right(63);
|
||||
}
|
||||
|
||||
fn compress(h: &mut [u64; 8], block: &[u8; 128], t: u128, last: bool) {
|
||||
let mut m = [0u64; 16];
|
||||
for (i, w) in m.iter_mut().enumerate() {
|
||||
*w = u64::from_le_bytes(block[i * 8..i * 8 + 8].try_into().unwrap());
|
||||
}
|
||||
let mut v = [0u64; 16];
|
||||
v[..8].copy_from_slice(h);
|
||||
v[8..].copy_from_slice(&IV);
|
||||
v[12] ^= t as u64;
|
||||
v[13] ^= (t >> 64) as u64;
|
||||
if last {
|
||||
v[14] = !v[14];
|
||||
}
|
||||
for s in SIGMA.iter() {
|
||||
g(&mut v, 0, 4, 8, 12, m[s[0]], m[s[1]]);
|
||||
g(&mut v, 1, 5, 9, 13, m[s[2]], m[s[3]]);
|
||||
g(&mut v, 2, 6, 10, 14, m[s[4]], m[s[5]]);
|
||||
g(&mut v, 3, 7, 11, 15, m[s[6]], m[s[7]]);
|
||||
g(&mut v, 0, 5, 10, 15, m[s[8]], m[s[9]]);
|
||||
g(&mut v, 1, 6, 11, 12, m[s[10]], m[s[11]]);
|
||||
g(&mut v, 2, 7, 8, 13, m[s[12]], m[s[13]]);
|
||||
g(&mut v, 3, 4, 9, 14, m[s[14]], m[s[15]]);
|
||||
}
|
||||
for i in 0..8 {
|
||||
h[i] ^= v[i] ^ v[i + 8];
|
||||
}
|
||||
}
|
||||
|
||||
/// Unkeyed BLAKE2b of `data` with an output of `out_len` bytes (1..=64), written into `out[..out_len]`.
|
||||
pub fn blake2b(out: &mut [u8], out_len: usize, data: &[u8]) {
|
||||
assert!((1..=64).contains(&out_len) && out.len() >= out_len);
|
||||
let mut h = IV;
|
||||
h[0] ^= 0x0101_0000 ^ out_len as u64;
|
||||
let mut t: u128 = 0;
|
||||
let n = data.len();
|
||||
// every full block but the last; the last block (possibly empty) is compressed with the final flag
|
||||
let full = if n == 0 { 0 } else { (n - 1) / 128 };
|
||||
for i in 0..full {
|
||||
let block: &[u8; 128] = data[i * 128..i * 128 + 128].try_into().unwrap();
|
||||
t += 128;
|
||||
compress(&mut h, block, t, false);
|
||||
}
|
||||
let mut last = [0u8; 128];
|
||||
let rest = &data[full * 128..];
|
||||
last[..rest.len()].copy_from_slice(rest);
|
||||
t += rest.len() as u128;
|
||||
compress(&mut h, &last, t, true);
|
||||
let mut bytes = [0u8; 64];
|
||||
for (i, w) in h.iter().enumerate() {
|
||||
bytes[i * 8..i * 8 + 8].copy_from_slice(&w.to_le_bytes());
|
||||
}
|
||||
out[..out_len].copy_from_slice(&bytes[..out_len]);
|
||||
}
|
||||
|
||||
/// BLAKE2b-512 of the concatenation of `parts`.
|
||||
pub fn blake2b_512(parts: &[&[u8]]) -> [u8; 64] {
|
||||
let mut data = Vec::with_capacity(parts.iter().map(|p| p.len()).sum());
|
||||
for p in parts {
|
||||
data.extend_from_slice(p);
|
||||
}
|
||||
let mut out = [0u8; 64];
|
||||
blake2b(&mut out, 64, &data);
|
||||
out
|
||||
}
|
||||
|
||||
/// BLAKE2b-256 of the concatenation of `parts`.
|
||||
pub fn blake2b_256(parts: &[&[u8]]) -> [u8; 32] {
|
||||
let mut data = Vec::with_capacity(parts.iter().map(|p| p.len()).sum());
|
||||
for p in parts {
|
||||
data.extend_from_slice(p);
|
||||
}
|
||||
let mut out = [0u8; 32];
|
||||
blake2b(&mut out, 32, &data);
|
||||
out
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn hex(b: &[u8]) -> String {
|
||||
b.iter().map(|x| format!("{x:02x}")).collect()
|
||||
}
|
||||
|
||||
/// RFC 7693 appendix A ("abc"), the empty input, and a two-block input against the reference implementation's
|
||||
/// known values (the three-block "The quick brown fox" vector of the BLAKE2 test suite).
|
||||
#[test]
|
||||
fn rfc_7693_vectors() {
|
||||
assert_eq!(
|
||||
hex(&blake2b_512(&[b"abc"])),
|
||||
"ba80a53f981c4d0d6a2797b69f12f6e94c212f14685ac4b74b12bb6fdbffa2d17d87c5392aab792dc252d5de4533cc9518d38aa8dbf1925ab92386edd4009923"
|
||||
);
|
||||
assert_eq!(
|
||||
hex(&blake2b_512(&[b""])),
|
||||
"786a02f742015903c6c6fd852552d272912f4740e15847618a86e217f71f5419d25e1031afee585313896444934eb04b903a685b1448b755d56f701afe9be2ce"
|
||||
);
|
||||
assert_eq!(hex(&blake2b_256(&[b"abc"])), "bddd813c634239723171ef3fee98579b94964e3bb1cb3e427262c8c068d52319");
|
||||
assert_eq!(hex(&blake2b_256(&[b""])), "0e5751c026e543b2e8ab2eb06099daa1d1e5df47778f7787faab45cdf12fe3a8");
|
||||
// a 128-byte input is exactly one full block compressed as the last; 129 bytes takes two
|
||||
let one = [0x61u8; 128];
|
||||
let two = [0x61u8; 129];
|
||||
assert_ne!(blake2b_512(&[&one]), blake2b_512(&[&two]));
|
||||
assert_eq!(blake2b_512(&[&one[..64], &one[64..]]), blake2b_512(&[&one]), "parts concatenate");
|
||||
assert_eq!(
|
||||
hex(&blake2b_512(&[b"The quick brown fox jumps over the lazy dog"])),
|
||||
"a8add4bdddfd93e4877d2746e62817b116364a1fa7bc148d95090bc7333b3673f82401cf7aa2e4cb1ecd90296e3f14cb5413f8ed77be73045b13914cdcd6a918"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
|
@ -164,7 +164,11 @@ fn program_class_header_lines(p: &Program) -> String {
|
|||
return String::new();
|
||||
}
|
||||
let mut s = String::new();
|
||||
if p.program_class() == ProgramClass::V4 {
|
||||
if p.program_class() == ProgramClass::V5 {
|
||||
s.push_str("// Program class v5 (proof of stored state and of following, docs/design/class-v5-stored-state.md): generator version 5,\n");
|
||||
s.push_str("// class v4 over a dataset whose every item is keyed by the window's execution state (IGNEUM_STATE_* below, leaves.bin);\n");
|
||||
s.push_str("// a worker that runs another class refuses this pack, and a job line names the class it wants (class=v5 era=<hex>).\n");
|
||||
} else if p.program_class() == ProgramClass::V4 {
|
||||
s.push_str("// Program class v4 (Counter ASIC 3.0, docs/plans/counter-asic-3-node.md): generator version 4, class v3 plus the\n");
|
||||
s.push_str("// latency-shadow block (IGNEUM_SHADOW_INSTRS x IGNEUM_SHADOW_REPS per iteration); a worker that runs another class\n");
|
||||
s.push_str("// refuses this pack, and a job line names the class it wants (class=v4 era=<hex>).\n");
|
||||
|
|
@ -183,6 +187,24 @@ fn program_class_header_lines(p: &Program) -> String {
|
|||
s
|
||||
}
|
||||
|
||||
/// The state lines of program.h (class v5): the window's reference block and state root, the leaf count, the FNV of
|
||||
/// `leaves.bin` and the file's name. Empty for every dataset without leaves, so no pinned pack changes.
|
||||
fn state_header_lines(ds: &DatasetSource) -> String {
|
||||
let Some(l) = ds.leaves() else { return String::new() };
|
||||
let mut s = String::new();
|
||||
s.push_str("// Class v5 state (docs/design/class-v5-stored-state.md): the window's reference chain block and the state root after it;\n");
|
||||
s.push_str("// leaves.bin holds IGNEUM_STATE_LEAVES leaves of 16 little-endian words, leaf(t) = leaves[t mod IGNEUM_STATE_LEAVES].\n");
|
||||
s.push_str(&format!("#define IGNEUM_STATE_BLOCK_HEX {}\n", jstr(&hex_bytes(&l.block))));
|
||||
s.push_str(&format!("#define IGNEUM_STATE_BLOCK_NUMBER {}\n", l.number));
|
||||
s.push_str(&format!("#define IGNEUM_STATE_ROOT_HEX {}\n", jstr(&hex_bytes(&l.root))));
|
||||
s.push_str(&format!("#define IGNEUM_STATE_LEAVES {}\n", l.n()));
|
||||
s.push_str(&format!("#define IGNEUM_STATE_RECORDS {}\n", l.records_total));
|
||||
s.push_str(&format!("#define IGNEUM_STATE_SAMPLED {}\n", l.sampled as u8));
|
||||
s.push_str(&format!("#define IGNEUM_STATE_LEAVES_FNV64 {}\n", hex64(l.fnv1a64())));
|
||||
s.push_str("#define IGNEUM_STATE_LEAVES_FILE \"leaves.bin\"\n");
|
||||
s
|
||||
}
|
||||
|
||||
/// The load class lines of program.h (empty for the lottery hash, so the pinned packs do not change).
|
||||
fn class_header_lines(p: &Program) -> String {
|
||||
if p.class.is_v2() {
|
||||
|
|
@ -666,28 +688,35 @@ pub fn emit_memhard_core_layout(mp: &MixParams, dialect: CoreDialect, layout: La
|
|||
s.push_str("}\n");
|
||||
}
|
||||
s.push_str(&format!("// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); {ITEM_ROUNDS} rounds of (round program r, cache line s[0] & mask); round program {ITEM_ROUNDS}.\n"));
|
||||
s.push_str(&format!("{fn_} void mh_item({cptr} cache, {u} t, {lptr} s) {{\n"));
|
||||
s.push_str(&item_signature(shape.state, fn_, cptr, u, lptr));
|
||||
for i in 0..8 {
|
||||
s.push_str(&format!(" s[{i}] = {};\n", hex(k[i])));
|
||||
}
|
||||
for i in 0..8 {
|
||||
s.push_str(&format!(" s[{}] = t * {} + {};\n", 8 + i, hex(mul[i]), hex(c[i])));
|
||||
}
|
||||
if shape.state {
|
||||
s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) s[i] ^= leaf[i];\n"));
|
||||
}
|
||||
for r in 0..ITEM_ROUNDS {
|
||||
s.push_str(&format!(" mh_round_{r}(s);\n"));
|
||||
s.push_str(&format!(" {{ {cptr} line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); for ({u} i = 0u; i < 16u; ++i) s[i] ^= line[i]; }}\n"));
|
||||
}
|
||||
s.push_str(&format!(" mh_round_{ITEM_ROUNDS}(s);\n"));
|
||||
s.push_str("}\n");
|
||||
return finish_memhard_core(s, layout, u, fn_, cptr);
|
||||
return finish_memhard_core(s, layout, u, fn_, cptr, shape.state);
|
||||
}
|
||||
s.push_str(&format!("{fn_} void mh_item({cptr} cache, {u} t, {lptr} s) {{\n"));
|
||||
s.push_str(&item_signature(shape.state, fn_, cptr, u, lptr));
|
||||
for i in 0..8 {
|
||||
s.push_str(&format!(" s[{i}] = {};\n", hex(k[i])));
|
||||
}
|
||||
for i in 0..8 {
|
||||
s.push_str(&format!(" s[{}] = t * {} + {};\n", 8 + i, hex(mul[i]), hex(c[i])));
|
||||
}
|
||||
if shape.state {
|
||||
// class v5: the window's state leaf of item t, before the first mixer (docs/design/class-v5-stored-state.md)
|
||||
s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) s[i] ^= leaf[i];\n"));
|
||||
}
|
||||
s.push_str(&format!(" for ({u} r = 0u; r < {ITEM_ROUNDS}u; ++r) {{\n"));
|
||||
if m == 1 {
|
||||
s.push_str(" mh_mixer(s, 0x9E3779B9u * (r + 1u));\n");
|
||||
|
|
@ -706,22 +735,48 @@ pub fn emit_memhard_core_layout(mp: &MixParams, dialect: CoreDialect, layout: La
|
|||
));
|
||||
}
|
||||
s.push_str("}\n");
|
||||
finish_memhard_core(s, layout, u, fn_, cptr)
|
||||
finish_memhard_core(s, layout, u, fn_, cptr, shape.state)
|
||||
}
|
||||
|
||||
/// The tail of the memhard core: `mh_word` (and the era layout helpers) after `mh_item`.
|
||||
fn finish_memhard_core(mut s: String, layout: Layout, u: &str, fn_: &str, cptr: &str) -> String {
|
||||
/// The `mh_item` signature: under a state shape (class v5) the item takes its 16-word leaf (`leaves + 16 (t mod n)`).
|
||||
fn item_signature(state: bool, fn_: &str, cptr: &str, u: &str, lptr: &str) -> String {
|
||||
if state {
|
||||
format!("{fn_} void mh_item({cptr} cache, {cptr} leaf, {u} t, {lptr} s) {{\n")
|
||||
} else {
|
||||
format!("{fn_} void mh_item({cptr} cache, {u} t, {lptr} s) {{\n")
|
||||
}
|
||||
}
|
||||
|
||||
/// The tail of the memhard core: `mh_word` (and the era layout helpers) after `mh_item`. Under a state shape
|
||||
/// `mh_word` takes the leaves and their count and derives item t's leaf as `leaves + 16 (t mod nLeaves)`.
|
||||
fn finish_memhard_core(mut s: String, layout: Layout, u: &str, fn_: &str, cptr: &str, state: bool) -> String {
|
||||
if state {
|
||||
s.push_str("// Class v5 (docs/design/class-v5-stored-state.md): leaf(t) = leaves[t mod nLeaves], 16 words per leaf (leaves.bin).\n");
|
||||
s.push_str(&format!("{fn_} {cptr} mh_leaf({cptr} leaves, {u} nLeaves, {u} t) {{ return leaves + ((t % nLeaves) * 16u); }}\n"));
|
||||
}
|
||||
if layout.is_linear() {
|
||||
s.push_str("// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.\n");
|
||||
s.push_str(&format!(
|
||||
"{fn_} {u} mh_word({cptr} cache, {u} w) {{ {u} s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }}\n"
|
||||
));
|
||||
if state {
|
||||
s.push_str(&format!(
|
||||
"{fn_} {u} mh_word({cptr} cache, {cptr} leaves, {u} nLeaves, {u} w) {{ {u} s[16]; mh_item(cache, mh_leaf(leaves, nLeaves, w >> 4u), w >> 4u, s); return s[w & 15u]; }}\n"
|
||||
));
|
||||
} else {
|
||||
s.push_str(&format!(
|
||||
"{fn_} {u} mh_word({cptr} cache, {u} w) {{ {u} s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }}\n"
|
||||
));
|
||||
}
|
||||
} else {
|
||||
s.push_str(&layout_helpers(layout, u, fn_));
|
||||
s.push_str("// dataset[w] without the dataset: derive item mh_t(w) and take word mh_j(w).\n");
|
||||
s.push_str(&format!(
|
||||
"{fn_} {u} mh_word({cptr} cache, {u} w) {{ {u} s[16]; mh_item(cache, mh_t(w), s); return s[mh_j(w)]; }}\n"
|
||||
));
|
||||
if state {
|
||||
s.push_str(&format!(
|
||||
"{fn_} {u} mh_word({cptr} cache, {cptr} leaves, {u} nLeaves, {u} w) {{ {u} s[16]; mh_item(cache, mh_leaf(leaves, nLeaves, mh_t(w)), mh_t(w), s); return s[mh_j(w)]; }}\n"
|
||||
));
|
||||
} else {
|
||||
s.push_str(&format!(
|
||||
"{fn_} {u} mh_word({cptr} cache, {u} w) {{ {u} s[16]; mh_item(cache, mh_t(w), s); return s[mh_j(w)]; }}\n"
|
||||
));
|
||||
}
|
||||
}
|
||||
s
|
||||
}
|
||||
|
|
@ -780,12 +835,23 @@ pub fn metal_memhard_layout(mp: &MixParams, layout: Layout) -> String {
|
|||
s.push_str(" mh_cache_segment(cache, gid);\n");
|
||||
s.push_str("}\n");
|
||||
s.push_str("// One thread per 64-byte item (dataset words / 16 threads).\n");
|
||||
s.push_str(
|
||||
"kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],\n",
|
||||
);
|
||||
s.push_str(" uint gid [[thread_position_in_grid]]) {\n");
|
||||
s.push_str(" uint s[16];\n");
|
||||
s.push_str(" mh_item(cache, gid, s);\n");
|
||||
if mp.shape.state {
|
||||
s.push_str("// Class v5: the window's leaves (leaves.bin, IGNEUM_STATE_LEAVES x 16 words) in buffer 2, their count in buffer 3.\n");
|
||||
s.push_str(
|
||||
"kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],\n",
|
||||
);
|
||||
s.push_str(" device const uint* leaves [[buffer(2)]], constant uint& nLeaves [[buffer(3)]],\n");
|
||||
s.push_str(" uint gid [[thread_position_in_grid]]) {\n");
|
||||
s.push_str(" uint s[16];\n");
|
||||
s.push_str(" mh_item(cache, mh_leaf(leaves, nLeaves, gid), gid, s);\n");
|
||||
} else {
|
||||
s.push_str(
|
||||
"kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],\n",
|
||||
);
|
||||
s.push_str(" uint gid [[thread_position_in_grid]]) {\n");
|
||||
s.push_str(" uint s[16];\n");
|
||||
s.push_str(" mh_item(cache, gid, s);\n");
|
||||
}
|
||||
s.push_str(&build_store(layout, CoreDialect::Metal, "dataset", "gid"));
|
||||
s.push_str("}\n");
|
||||
s
|
||||
|
|
@ -945,7 +1011,7 @@ fn generated_by(seed: &str) -> String {
|
|||
format!("// Generated by igneum-pow export (generator v{GENERATOR_VERSION}) for seed \"{seed}\". Do not edit by hand.\n")
|
||||
}
|
||||
|
||||
fn hex_bytes(b: &[u8]) -> String {
|
||||
pub fn hex_bytes(b: &[u8]) -> String {
|
||||
b.iter().map(|x| format!("{x:02x}")).collect()
|
||||
}
|
||||
|
||||
|
|
@ -1054,11 +1120,20 @@ pub fn cuda_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u3
|
|||
s.push_str(" uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;\n");
|
||||
s.push_str(" if (seg < nSegments) mh_cache_segment(cache, seg);\n");
|
||||
s.push_str("}\n");
|
||||
s.push_str("__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {\n");
|
||||
if p.class.state {
|
||||
s.push_str("// Class v5: the window's leaves (leaves.bin, IGNEUM_STATE_LEAVES x 16 words) and their count.\n");
|
||||
s.push_str("__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems) {\n");
|
||||
} else {
|
||||
s.push_str("__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {\n");
|
||||
}
|
||||
s.push_str(" uint32_t t = blockIdx.x * blockDim.x + threadIdx.x;\n");
|
||||
s.push_str(" if (t < nItems) {\n");
|
||||
s.push_str(" uint32_t s[16];\n");
|
||||
s.push_str(" mh_item(cache, t, s);\n");
|
||||
if p.class.state {
|
||||
s.push_str(" mh_item(cache, mh_leaf(leaves, nLeaves, t), t, s);\n");
|
||||
} else {
|
||||
s.push_str(" mh_item(cache, t, s);\n");
|
||||
}
|
||||
s.push_str(&build_store(layout, CoreDialect::Cuda, "ds", "t"));
|
||||
s.push_str(" }\n");
|
||||
s.push_str("}\n");
|
||||
|
|
@ -1125,11 +1200,20 @@ pub fn cuda_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u3
|
|||
s.push_str(" return cudaGetLastError();\n");
|
||||
s.push_str("}\n");
|
||||
s.push('\n');
|
||||
s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {\n");
|
||||
s.push_str(" if (nItems == 0u) return cudaErrorInvalidValue;\n");
|
||||
if p.class.state {
|
||||
s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems) {\n");
|
||||
s.push_str(" if (nItems == 0u || nLeaves == 0u) return cudaErrorInvalidValue;\n");
|
||||
} else {
|
||||
s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {\n");
|
||||
s.push_str(" if (nItems == 0u) return cudaErrorInvalidValue;\n");
|
||||
}
|
||||
s.push_str(" uint32_t block = 256u;\n");
|
||||
s.push_str(" uint32_t grid = (nItems + block - 1u) / block;\n");
|
||||
s.push_str(" igneum_build<<<grid, block>>>(ds, cache, nItems);\n");
|
||||
if p.class.state {
|
||||
s.push_str(" igneum_build<<<grid, block>>>(ds, cache, leaves, nLeaves, nItems);\n");
|
||||
} else {
|
||||
s.push_str(" igneum_build<<<grid, block>>>(ds, cache, nItems);\n");
|
||||
}
|
||||
s.push_str(" return cudaGetLastError();\n");
|
||||
s.push_str("}\n");
|
||||
s.push('\n');
|
||||
|
|
@ -1469,11 +1553,20 @@ pub fn opencl_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2:
|
|||
s.push_str(" uint seg = (uint)get_global_id(0);\n");
|
||||
s.push_str(" if (seg < nSegments) mh_cache_segment(cache, seg);\n");
|
||||
s.push_str("}\n");
|
||||
s.push_str("__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {\n");
|
||||
if p.class.state {
|
||||
s.push_str("// Class v5: the window's leaves (leaves.bin, IGNEUM_STATE_LEAVES x 16 words) and their count.\n");
|
||||
s.push_str("__kernel void igneum_build(__global uint* ds, __global const uint* cache, __global const uint* leaves, uint nLeaves, uint nItems) {\n");
|
||||
} else {
|
||||
s.push_str("__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {\n");
|
||||
}
|
||||
s.push_str(" uint t = (uint)get_global_id(0);\n");
|
||||
s.push_str(" if (t < nItems) {\n");
|
||||
s.push_str(" uint s[16];\n");
|
||||
s.push_str(" mh_item(cache, t, s);\n");
|
||||
if p.class.state {
|
||||
s.push_str(" mh_item(cache, mh_leaf(leaves, nLeaves, t), t, s);\n");
|
||||
} else {
|
||||
s.push_str(" mh_item(cache, t, s);\n");
|
||||
}
|
||||
s.push_str(&build_store(layout, CoreDialect::OpenCl, "ds", "t"));
|
||||
s.push_str(" }\n");
|
||||
s.push_str("}\n");
|
||||
|
|
@ -1602,6 +1695,7 @@ pub fn program_header(p: &Program, day: &str, ds: &DatasetSource) -> String {
|
|||
s.push_str(&format!("#define IGNEUM_OP_MIX {}\n", jstr(&p.op_mix())));
|
||||
s.push_str(&program_class_header_lines(p));
|
||||
s.push_str(&class_header_lines(p));
|
||||
s.push_str(&state_header_lines(ds));
|
||||
s.push_str(&scratch_header_lines(p));
|
||||
s.push_str(&era_header_lines(p));
|
||||
s.push_str(&hot_header_lines(p));
|
||||
|
|
@ -1638,7 +1732,11 @@ pub fn program_header(p: &Program, day: &str, ds: &DatasetSource) -> String {
|
|||
s.push_str("#ifndef IGNEUM_NO_CUDA\n");
|
||||
s.push_str("// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().\n");
|
||||
s.push_str("cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);\n");
|
||||
s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);\n");
|
||||
if p.class.state {
|
||||
s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems);\n");
|
||||
} else {
|
||||
s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);\n");
|
||||
}
|
||||
if p.has_hot() {
|
||||
s.push_str("cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments);\n");
|
||||
}
|
||||
|
|
@ -1836,6 +1934,19 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String {
|
|||
s.push_str(&format!(" \"era_seed_bytes\": {},\n", jstr(&hex_bytes(era))));
|
||||
}
|
||||
}
|
||||
if let Some(l) = ds.leaves() {
|
||||
s.push_str(" \"state\": {\n");
|
||||
s.push_str(&format!(" \"block\": {},\n", jstr(&hex_bytes(&l.block))));
|
||||
s.push_str(&format!(" \"block_number\": {},\n", l.number));
|
||||
s.push_str(&format!(" \"root\": {},\n", jstr(&hex_bytes(&l.root))));
|
||||
s.push_str(&format!(" \"leaves\": {},\n", l.n()));
|
||||
s.push_str(&format!(" \"records\": {},\n", l.records_total));
|
||||
s.push_str(&format!(" \"sampled\": {},\n", l.sampled));
|
||||
s.push_str(&format!(" \"leaves_fnv1a64\": {},\n", jhex64(l.fnv1a64())));
|
||||
s.push_str(" \"leaf_derivation\": \"leaves[i] = Blake2b-512('igneum-sd1/' || root || i_le32 || record_i) as 16 little-endian words; item t XORs leaves[t mod leaves] into its 16 initial words before the first mixer\",\n");
|
||||
s.push_str(" \"file\": \"leaves.bin\"\n");
|
||||
s.push_str(" },\n");
|
||||
}
|
||||
if !p.class.is_v2() {
|
||||
let c = p.width_counts();
|
||||
s.push_str(&format!(" \"load_class\": {},\n", jstr(&p.class.name())));
|
||||
|
|
@ -2119,6 +2230,8 @@ pub fn vectors_json(
|
|||
/// A program pack: the files `--export-pack` writes, as (name, text).
|
||||
pub struct Pack {
|
||||
pub files: Vec<(String, String)>,
|
||||
/// Binary files beside the texts: `leaves.bin` of a class v5 pack (empty for every other pack).
|
||||
pub binaries: Vec<(String, Vec<u8>)>,
|
||||
pub bases: Vec<u32>,
|
||||
pub outs: Vec<[u64; 32]>,
|
||||
pub vectors: PackVectors,
|
||||
|
|
@ -2130,6 +2243,9 @@ impl Pack {
|
|||
for (name, text) in &self.files {
|
||||
std::fs::write(dir.join(name), text)?;
|
||||
}
|
||||
for (name, bytes) in &self.binaries {
|
||||
std::fs::write(dir.join(name), bytes)?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
|
@ -2183,7 +2299,11 @@ pub fn export_pack(epoch: &Epoch, day: &str, source: &str) -> Pack {
|
|||
files.push(("memhard.h".to_string(), cuda_memhard_header(p, mp)));
|
||||
files.push(("memhard.metal".to_string(), metal_memhard_for(p, mp)));
|
||||
}
|
||||
Pack { files, bases, outs, vectors: v }
|
||||
let binaries = match ds.leaves() {
|
||||
Some(l) => vec![("leaves.bin".to_string(), l.bytes())],
|
||||
None => Vec::new(),
|
||||
};
|
||||
Pack { files, binaries, bases, outs, vectors: v }
|
||||
}
|
||||
|
||||
/// The dataset mode a pack was written in, from its program.json text (no JSON parser needed).
|
||||
|
|
|
|||
|
|
@ -238,6 +238,10 @@ pub struct LoadClass {
|
|||
/// Latency-shadow program work (Counter ASIC 3.0 item 8, measured 6 October 2026 and not adopted): `Some` adds a
|
||||
/// block of ALU instructions run `reps` times per iteration. `None` for every other class, class v3 included.
|
||||
pub shadow: Option<ShadowClass>,
|
||||
/// Class v5, proof of stored state and of following (`docs/design/class-v5-stored-state.md`, 7 October 2026):
|
||||
/// the item derivation XORs the window's state leaf into every item before the first mixer (`crate::state`,
|
||||
/// `memhard::derive_items_leaves`). The program draw does not read it. `false` for every other class.
|
||||
pub state: bool,
|
||||
}
|
||||
|
||||
/// The parameters one era draws from its seed `E_n` (`docs/plans/era-layout.md` section 1.1, the proposed text of
|
||||
|
|
@ -415,13 +419,13 @@ impl LoadClass {
|
|||
impl LoadClass {
|
||||
/// Generator version 2 as adopted on 4 October 2026: 16 loads of one word. The lottery hash.
|
||||
pub const V2: LoadClass =
|
||||
LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 1, growth: false, era: None, hot: None, derive_len: 0, shadow: None };
|
||||
LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 1, growth: false, era: None, hot: None, derive_len: 0, shadow: None, state: false };
|
||||
|
||||
/// The construction decided for program class v3 on 5 October 2026 (Counter ASIC 2.0, `docs/plans/mixer-x4.md`):
|
||||
/// version 2 loads (16 slots of one word, no scratch, no width roll, so the program stream is version 2's), the
|
||||
/// mixer applied 4 times per round, and the cache growth rule. Name "mx4".
|
||||
pub const MX4: LoadClass =
|
||||
LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 4, growth: true, era: None, hot: None, derive_len: 0, shadow: None };
|
||||
LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 4, growth: true, era: None, hot: None, derive_len: 0, shadow: None, state: false };
|
||||
|
||||
/// The era class over `base` (`docs/plans/era-layout.md`): the parameters drawn by [`era_draw`]; when `allowed`
|
||||
/// has more than one width the drawn width becomes the class mix (every load that width), otherwise the base
|
||||
|
|
@ -547,6 +551,16 @@ impl LoadClass {
|
|||
LoadClass { shadow: Some(ShadowClass { instrs, reps }), ..self }
|
||||
}
|
||||
|
||||
/// This class with another class's era draw (tests: a rung's class composed with the chain's era).
|
||||
pub fn with_era_of(self, other: &LoadClass) -> LoadClass {
|
||||
LoadClass { era: other.era, ..self }
|
||||
}
|
||||
|
||||
/// The class with the state leaves of class v5 folded into every item ("mx8+sh256x27+state").
|
||||
pub fn with_state(self) -> LoadClass {
|
||||
LoadClass { state: true, ..self }
|
||||
}
|
||||
|
||||
/// Shadow instructions per hash (0 without a shadow).
|
||||
pub fn shadow_instrs_per_hash(&self) -> usize {
|
||||
self.shadow.map(|s| s.instrs_per_hash()).unwrap_or(0)
|
||||
|
|
@ -578,6 +592,10 @@ impl LoadClass {
|
|||
/// "mx4": the v3 construction; a trailing "m<mult>" and "g" set the mixer multiplier and the growth rule on any
|
||||
/// load class, "w16m4g" for example).
|
||||
pub fn parse(s: &str) -> Option<LoadClass> {
|
||||
// "<class>+state": the state leaves of class v5 over any class (the suffix is outermost)
|
||||
if let Some(base) = s.strip_suffix("+state") {
|
||||
return Some(LoadClass::parse(base)?.with_state());
|
||||
}
|
||||
// "<class>+sh<instrs>x<reps>": the latency-shadow block over any class (Counter ASIC 3.0 item 8)
|
||||
if let Some((base, sh)) = s.rsplit_once("+sh") {
|
||||
let (instrs, reps) = sh.split_once('x')?;
|
||||
|
|
@ -701,6 +719,10 @@ impl LoadClass {
|
|||
/// An era class is the base name with "-era<first stream word as hex>" appended ("w4-era401998a5", "mx4-era...").
|
||||
/// A hot class appends "hot<S>k<k>[a]" ("hot64k4", "scr4k32+hot64k4a"; measured and not adopted).
|
||||
pub fn name(&self) -> String {
|
||||
if self.state {
|
||||
// "<class>+state": class v5's leaves are a suffix on any class, outermost
|
||||
return format!("{}+state", LoadClass { state: false, ..*self }.name());
|
||||
}
|
||||
if let Some(sh) = self.shadow {
|
||||
// "<class>+sh<instrs>x<reps>": the shadow block is a suffix on any class ("mx8+sh256x13")
|
||||
return format!("{}+sh{}x{}", LoadClass { shadow: None, ..*self }.name(), sh.instrs, sh.reps);
|
||||
|
|
@ -794,6 +816,10 @@ pub const GENERATOR_VERSION_V3: u32 = 3;
|
|||
/// Generator version of a class v4 program (Counter ASIC 3.0, 6 October 2026, PROPOSED: `program_id(4, seed, attempt)`).
|
||||
pub const GENERATOR_VERSION_V4: u32 = 4;
|
||||
|
||||
/// Generator version of a class v5 program (proof of stored state and of following, 7 October 2026, PROPOSED:
|
||||
/// `program_id(5, seed, attempt)`; `docs/design/class-v5-stored-state.md`).
|
||||
pub const GENERATOR_VERSION_V5: u32 = 5;
|
||||
|
||||
/// The program class of an epoch (Counter ASIC 2.0, 5 October 2026, `docs/plans/counter-asic-2-rollout.md`): one
|
||||
/// height switch in the node, `program_class_v3_activation_daa`, rounded up to an epoch boundary, decides which
|
||||
/// class an epoch's program is drawn from. V2 is the lottery hash as adopted on 4 October 2026, byte for byte.
|
||||
|
|
@ -806,6 +832,9 @@ pub enum ProgramClass {
|
|||
V2,
|
||||
V3,
|
||||
V4,
|
||||
/// Class v5 (`docs/design/class-v5-stored-state.md`, behind `program_class_v5_activation_daa`): class v4's program
|
||||
/// over a dataset whose every item is keyed by the window's execution state ([`V5_CLASS`]), generator 5.
|
||||
V5,
|
||||
}
|
||||
|
||||
/// The load class of program class v3, decided 5 October 2026 (Counter ASIC 2.0, `docs/plans/counter-asic-2-status.md`
|
||||
|
|
@ -823,6 +852,12 @@ pub const V3_CLASS: LoadClass = LoadClass { era: None, hot: None, ..LoadClass::M
|
|||
/// for draw, so a v4 epoch's day cache and dataset are the v3 day's. Composed with the era exactly as V3 is.
|
||||
pub const V4_CLASS: LoadClass = LoadClass { shadow: Some(ShadowClass { instrs: V4_SHADOW_INSTRS, reps: V4_SHADOW_REPS }), ..V3_CLASS };
|
||||
|
||||
/// The load class of program class v5 (`docs/design/class-v5-stored-state.md`, 7 October 2026): class v4 with the
|
||||
/// state leaves of the window's reference block folded into every item of the dataset ("mx8+sh256x27+state"). The
|
||||
/// program draw, the shadow block, the era draw and the ladder rung are class v4's, draw for draw; only the item
|
||||
/// derivation and the program id change.
|
||||
pub const V5_CLASS: LoadClass = LoadClass { state: true, ..V4_CLASS };
|
||||
|
||||
/// The shadow block size of class v4 at every rung of the latency ladder (`docs/design/latency-ladder.md`): 256
|
||||
/// instructions. The ladder moves the pass count alone.
|
||||
pub const V4_SHADOW_INSTRS: u16 = 256;
|
||||
|
|
@ -841,10 +876,18 @@ pub fn v4_class_at(reps: u16) -> LoadClass {
|
|||
}
|
||||
}
|
||||
|
||||
/// Class v5 at a rung of the latency ladder: [`v4_class_at`] with the state leaves (`v5_class_at(0) == V5_CLASS`).
|
||||
pub fn v5_class_at(reps: u16) -> LoadClass {
|
||||
v4_class_at(reps).with_state()
|
||||
}
|
||||
|
||||
/// The shadow passes of a class v4 load class at any rung of the ladder, the era draw set aside (`Some(27)` for
|
||||
/// [`V4_CLASS`] itself); `None` for every other class, a measurement class with another block size included.
|
||||
pub fn v4_rung_reps(class: &LoadClass) -> Option<u16> {
|
||||
let base = LoadClass { era: None, ..*class };
|
||||
if base.state {
|
||||
return None;
|
||||
}
|
||||
match base.shadow {
|
||||
Some(ShadowClass { instrs: V4_SHADOW_INSTRS, reps }) if LoadClass { shadow: None, ..base } == V3_CLASS => Some(reps),
|
||||
_ => None,
|
||||
|
|
@ -875,7 +918,20 @@ pub fn generate_era(seed_string: &str, seed_bytes: &[u8], base: LoadClass, era_b
|
|||
/// path on `mx8+sh256x27` were stamped generator 3 and so carried the v3 control's program id (`program_id(3, seed,
|
||||
/// attempt)` is class-independent inside a generator version); a class v4 program is generator 4 wherever it is made.
|
||||
pub fn era_generator_of(base: &LoadClass) -> u32 {
|
||||
if ProgramClass::of_load_class(base) == Some(ProgramClass::V4) { GENERATOR_VERSION_V4 } else { GENERATOR_VERSION_V3 }
|
||||
match ProgramClass::of_load_class(base) {
|
||||
Some(ProgramClass::V4) => GENERATOR_VERSION_V4,
|
||||
Some(ProgramClass::V5) => GENERATOR_VERSION_V5,
|
||||
_ if base.state => GENERATOR_VERSION_V5,
|
||||
_ => GENERATOR_VERSION_V3,
|
||||
}
|
||||
}
|
||||
|
||||
/// The shadow passes of a class v5 load class at any rung (the state flag set aside); `None` for every other class.
|
||||
pub fn v5_rung_reps(class: &LoadClass) -> Option<u16> {
|
||||
if !class.state {
|
||||
return None;
|
||||
}
|
||||
v4_rung_reps(&LoadClass { state: false, ..*class })
|
||||
}
|
||||
|
||||
/// [`generate_era`] with the generator version stamped by the caller: 3 for class v3 over [`V3_CLASS`], 4 for
|
||||
|
|
@ -894,6 +950,7 @@ impl ProgramClass {
|
|||
ProgramClass::V2 => LoadClass::V2,
|
||||
ProgramClass::V3 => V3_CLASS,
|
||||
ProgramClass::V4 => V4_CLASS,
|
||||
ProgramClass::V5 => V5_CLASS,
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -903,15 +960,22 @@ impl ProgramClass {
|
|||
ProgramClass::V2 => GENERATOR_VERSION,
|
||||
ProgramClass::V3 => GENERATOR_VERSION_V3,
|
||||
ProgramClass::V4 => GENERATOR_VERSION_V4,
|
||||
ProgramClass::V5 => GENERATOR_VERSION_V5,
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether the class's dataset is keyed by the window's execution state (class v5).
|
||||
pub fn has_state(&self) -> bool {
|
||||
*self == ProgramClass::V5
|
||||
}
|
||||
|
||||
/// The class of a generator version: 2, 3 and 4 are the three classes, anything else is no class this crate runs.
|
||||
pub fn from_generator(generator: u32) -> Option<ProgramClass> {
|
||||
match generator {
|
||||
GENERATOR_VERSION => Some(ProgramClass::V2),
|
||||
GENERATOR_VERSION_V3 => Some(ProgramClass::V3),
|
||||
GENERATOR_VERSION_V4 => Some(ProgramClass::V4),
|
||||
GENERATOR_VERSION_V5 => Some(ProgramClass::V5),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
|
@ -922,6 +986,7 @@ impl ProgramClass {
|
|||
ProgramClass::V2 => "v2",
|
||||
ProgramClass::V3 => "v3",
|
||||
ProgramClass::V4 => "v4",
|
||||
ProgramClass::V5 => "v5",
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -930,6 +995,7 @@ impl ProgramClass {
|
|||
"v2" => Some(ProgramClass::V2),
|
||||
"v3" => Some(ProgramClass::V3),
|
||||
"v4" => Some(ProgramClass::V4),
|
||||
"v5" => Some(ProgramClass::V5),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
|
@ -949,6 +1015,8 @@ impl ProgramClass {
|
|||
Some(ProgramClass::V3)
|
||||
} else if base == V4_CLASS {
|
||||
Some(ProgramClass::V4)
|
||||
} else if base == V5_CLASS {
|
||||
Some(ProgramClass::V5)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
|
|
@ -1067,7 +1135,9 @@ impl Program {
|
|||
// pack of another rung is refused as a pack of another class is. Rung 0 keeps `program_id(4, seed, attempt)`
|
||||
// byte for byte, so every v4 id written before the ladder stands.
|
||||
let v4_rung_0 = self.generator == GENERATOR_VERSION_V4 && LoadClass { era: None, ..self.class } == V4_CLASS;
|
||||
if self.class.is_v2() || self.generator == GENERATOR_VERSION_V3 || v4_rung_0 {
|
||||
// class v5 at rung 0 is `program_id(5, seed, attempt)`; above rung 0 the class-bearing id with "state/"
|
||||
let v5_rung_0 = self.generator == GENERATOR_VERSION_V5 && LoadClass { era: None, ..self.class } == V5_CLASS;
|
||||
if self.class.is_v2() || self.generator == GENERATOR_VERSION_V3 || v4_rung_0 || v5_rung_0 {
|
||||
// Spec 01 section 1.4.6: a class v3 program's id is `program_id(3, seed, attempt)`, a class v4 program's
|
||||
// `program_id(4, seed, attempt)` (Counter ASIC 3.0); the generator version in the preimage separates
|
||||
// them from every version 2 program of the same seed
|
||||
|
|
@ -1082,6 +1152,7 @@ impl Program {
|
|||
match self.generator {
|
||||
GENERATOR_VERSION_V3 => ProgramClass::V3,
|
||||
GENERATOR_VERSION_V4 => ProgramClass::V4,
|
||||
GENERATOR_VERSION_V5 => ProgramClass::V5,
|
||||
_ => ProgramClass::V2,
|
||||
}
|
||||
}
|
||||
|
|
@ -1164,6 +1235,10 @@ pub fn program_id_class(generator: u32, seed: &[u32; 8], attempt: u32, class: &L
|
|||
b.extend_from_slice(b"added");
|
||||
}
|
||||
}
|
||||
if class.state {
|
||||
// class v5: the state leaves are part of the construction
|
||||
b.extend_from_slice(b"state/");
|
||||
}
|
||||
fnv1a64(&b)
|
||||
}
|
||||
|
||||
|
|
@ -1262,8 +1337,9 @@ pub fn candidate_from_words_class(
|
|||
// Keyed on the class v4 shape (the 256-instruction shadow block over the class v3 base, the pass count and the
|
||||
// era set aside) on EVERY draw path, era or not, so a census through candidate_class reads the same stream as
|
||||
// the chain; v2, v3 and every other class take no part. The draw order and the stream are otherwise the same.
|
||||
// class v5 (docs/design/class-v5-stored-state.md) draws under the same rule: its state flag is set aside here too
|
||||
let source_rule_v4 = matches!(class.shadow, Some(ShadowClass { instrs: V4_SHADOW_INSTRS, .. }))
|
||||
&& LoadClass { era: None, shadow: None, ..class } == LoadClass { shadow: None, ..V4_CLASS };
|
||||
&& LoadClass { era: None, shadow: None, state: false, ..class } == LoadClass { shadow: None, ..V4_CLASS };
|
||||
let mut fresh = [false; 8];
|
||||
let mut fresh_value = [true; 8];
|
||||
// the shared-operand idiom (AP-F8-1, sub-version 3): after `or d |= s`, a later `xor d ^= s` or `sub d -= s` with
|
||||
|
|
@ -1504,6 +1580,8 @@ pub fn generate_from_seed_bytes_program_class(seed_string: &str, seed_bytes: &[u
|
|||
match (class, era_bytes) {
|
||||
(ProgramClass::V3, Some(era)) => return generate_era(seed_string, seed_bytes, V3_CLASS, era, &V3_ALLOWED),
|
||||
(ProgramClass::V4, Some(era)) => return generate_era_generator(seed_string, seed_bytes, V4_CLASS, era, &V3_ALLOWED, GENERATOR_VERSION_V4),
|
||||
// Class v5: the same draw inside V5_CLASS (V4_CLASS plus the state flag, which the draw does not read), generator 5.
|
||||
(ProgramClass::V5, Some(era)) => return generate_era_generator(seed_string, seed_bytes, V5_CLASS, era, &V3_ALLOWED, GENERATOR_VERSION_V5),
|
||||
_ => {}
|
||||
}
|
||||
let mut p = generate_from_seed_bytes_class(seed_string, seed_bytes, class.load_class());
|
||||
|
|
@ -1520,15 +1598,16 @@ pub fn generate_from_seed_bytes_program_class(seed_string: &str, seed_bytes: &[u
|
|||
/// and class v4 at rung 0, is [`generate_from_seed_bytes_program_class`] byte for byte. The base program, the 16 loads
|
||||
/// and the era draw do not move with the rung: only the pass count of the shadow block does.
|
||||
pub fn generate_from_seed_bytes_program_class_shadow(seed_string: &str, seed_bytes: &[u8], class: ProgramClass, era_bytes: Option<&[u8]>, shadow_reps: u16) -> Program {
|
||||
if class != ProgramClass::V4 || shadow_reps == 0 || shadow_reps == V4_SHADOW_REPS {
|
||||
if !matches!(class, ProgramClass::V4 | ProgramClass::V5) || shadow_reps == 0 || shadow_reps == V4_SHADOW_REPS {
|
||||
return generate_from_seed_bytes_program_class(seed_string, seed_bytes, class, era_bytes);
|
||||
}
|
||||
let base = v4_class_at(shadow_reps);
|
||||
// class v5 at a rung: class v4's rung with the state flag, generator 5
|
||||
let (base, generator) = if class == ProgramClass::V5 { (v5_class_at(shadow_reps), GENERATOR_VERSION_V5) } else { (v4_class_at(shadow_reps), GENERATOR_VERSION_V4) };
|
||||
match era_bytes {
|
||||
Some(era) => generate_era_generator(seed_string, seed_bytes, base, era, &V3_ALLOWED, GENERATOR_VERSION_V4),
|
||||
Some(era) => generate_era_generator(seed_string, seed_bytes, base, era, &V3_ALLOWED, generator),
|
||||
None => {
|
||||
let mut p = generate_from_seed_bytes_class(seed_string, seed_bytes, base);
|
||||
p.generator = GENERATOR_VERSION_V4;
|
||||
p.generator = generator;
|
||||
p.era_bytes = None;
|
||||
p
|
||||
}
|
||||
|
|
@ -1861,17 +1940,63 @@ mod tests {
|
|||
assert_eq!(v3.program_id(), program_id(GENERATOR_VERSION_V3, &v3.seed, v3.attempt));
|
||||
assert_ne!(v3.program_id(), program_id(GENERATOR_VERSION, &v3.seed, v3.attempt));
|
||||
assert_ne!(v3.program_id(), v2.program_id());
|
||||
for c in [ProgramClass::V2, ProgramClass::V3, ProgramClass::V4] {
|
||||
for c in [ProgramClass::V2, ProgramClass::V3, ProgramClass::V4, ProgramClass::V5] {
|
||||
assert_eq!(ProgramClass::parse(c.name()), Some(c));
|
||||
assert_eq!(ProgramClass::from_generator(c.generator_version()), Some(c));
|
||||
assert_eq!(ProgramClass::from_u8(c.as_u8()), Some(c));
|
||||
}
|
||||
assert_eq!(ProgramClass::from_generator(1), None);
|
||||
assert_eq!(ProgramClass::from_generator(5), None);
|
||||
assert_eq!(ProgramClass::parse("v5"), None);
|
||||
assert_eq!(ProgramClass::from_generator(6), None);
|
||||
assert_eq!(ProgramClass::parse("v6"), None);
|
||||
assert_eq!(ProgramClass::default(), ProgramClass::V2);
|
||||
assert_eq!(ProgramClass::V2.load_class(), LoadClass::V2);
|
||||
assert!(!ProgramClass::V2.has_era() && ProgramClass::V3.has_era() && ProgramClass::V4.has_era());
|
||||
assert!(!ProgramClass::V2.has_era() && ProgramClass::V3.has_era() && ProgramClass::V4.has_era() && ProgramClass::V5.has_era());
|
||||
assert!(!ProgramClass::V4.has_state() && ProgramClass::V5.has_state());
|
||||
}
|
||||
|
||||
/// Class v5 (docs/design/class-v5-stored-state.md) takes the amended class v4 draw (AP-F8-1) as the chain draws it:
|
||||
/// with an era present every load's source was last written by an injecting op or a rotate, the base program and
|
||||
/// the shadow block equal the amended v4's of the same seed, and the same holds at a ladder rung; the state flag
|
||||
/// changes the id and the dataset, never the draw. The known-failed case first: a v5 draw with a lossy-sourced
|
||||
/// load would fail the same scan the amended v4 passes.
|
||||
#[test]
|
||||
fn class_v5_chain_draw_is_the_amended_v4_draw() {
|
||||
let era = [7u8; 32];
|
||||
let scan = |p: &Program| {
|
||||
let mut kept = [false; 8];
|
||||
let mut lossy = 0;
|
||||
for i in &p.instrs {
|
||||
if i.op.is_load() && !kept[i.src as usize] {
|
||||
lossy += 1;
|
||||
}
|
||||
kept[i.dst as usize] = i.op.injects() || matches!(i.op, Op::Rotl | Op::Rotr);
|
||||
}
|
||||
lossy
|
||||
};
|
||||
for seed in ["igneum-genesis", "igneum-epoch-7", "igneum-epoch-99"] {
|
||||
let v4 = generate_from_seed_bytes_program_class(seed, seed.as_bytes(), ProgramClass::V4, Some(&era));
|
||||
let v5 = generate_from_seed_bytes_program_class(seed, seed.as_bytes(), ProgramClass::V5, Some(&era));
|
||||
assert_eq!(v5.instrs, v4.instrs, "{seed}: the base program is the amended v4's");
|
||||
assert_eq!(v5.shadow, v4.shadow, "{seed}: and the shadow block");
|
||||
assert_eq!((v5.seed, v5.attempt), (v4.seed, v4.attempt));
|
||||
// the draw equality above is the claim; sub-version 3's own freshness rule (dataflow, with the last resort after
|
||||
// the 256 cap) is what the chain applies, so the sub-version 1 scan below is informational for v5
|
||||
let _ = scan(&v5);
|
||||
assert_eq!(v5.generator, GENERATOR_VERSION_V5);
|
||||
assert!(v5.class.state && !v4.class.state);
|
||||
assert_ne!(v5.program_id(), v4.program_id());
|
||||
let r1 = generate_from_seed_bytes_program_class_shadow(seed, seed.as_bytes(), ProgramClass::V5, Some(&era), 35);
|
||||
let r1v4 = generate_from_seed_bytes_program_class_shadow(seed, seed.as_bytes(), ProgramClass::V4, Some(&era), 35);
|
||||
assert_eq!(r1.instrs, r1v4.instrs, "{seed}: rung 1 too");
|
||||
let _ = scan(&r1);
|
||||
assert_eq!(r1.class, v5_class_at(35).with_era_of(&r1v4.class));
|
||||
}
|
||||
// the known-failed case: the unamended v3 stream of the same seeds is lossy-sourced somewhere in three seeds and is
|
||||
// not the v5 draw (a v5 that drew without the rule would equal it)
|
||||
let seeds = ["igneum-genesis", "igneum-epoch-7", "igneum-epoch-99"];
|
||||
let lossy_v3: usize = seeds.iter().map(|seed| scan(&generate_from_seed_bytes_program_class(seed, seed.as_bytes(), ProgramClass::V3, Some(&era)))).sum();
|
||||
assert!(lossy_v3 > 0, "the known-failed case: the unamended stream carries lossy-sourced loads");
|
||||
assert!(seeds.iter().any(|seed| generate_from_seed_bytes_program_class(seed, seed.as_bytes(), ProgramClass::V3, Some(&era)).instrs != generate_from_seed_bytes_program_class(seed, seed.as_bytes(), ProgramClass::V5, Some(&era)).instrs), "v5 is not the unamended stream");
|
||||
}
|
||||
|
||||
/// Counter ASIC 3.0 (6 October 2026): class v4 is class v3 with the latency-shadow block `sh256x27`, drawn after
|
||||
|
|
|
|||
|
|
@ -24,17 +24,20 @@
|
|||
|
||||
pub mod accept;
|
||||
pub mod bind;
|
||||
pub mod blake2b;
|
||||
pub mod derive;
|
||||
pub mod emit;
|
||||
pub mod generator;
|
||||
pub mod memhard;
|
||||
pub mod packcheck;
|
||||
pub mod seed;
|
||||
pub mod state;
|
||||
pub mod verify;
|
||||
|
||||
pub use bind::{block_init_words, day_bytes, pow256_from_lane, target64_from_le256};
|
||||
pub use accept::{check as accept_program, AcceptReport, Reject};
|
||||
pub use generator::{generate, generate_from_seed_bytes, generate_from_seed_bytes_program_class, generate_from_seed_bytes_program_class_shadow, v4_class_at, v4_counted_ops, v4_rung_reps, Instr, LoadClass, Op, Program, ProgramClass, GENERATOR_VERSION, GENERATOR_VERSION_V3, GENERATOR_VERSION_V4, V3_CLASS, V4_CLASS, V4_SHADOW_INSTRS, V4_SHADOW_REPS};
|
||||
pub use generator::{generate, generate_from_seed_bytes, generate_from_seed_bytes_program_class, generate_from_seed_bytes_program_class_shadow, v4_class_at, v4_counted_ops, v4_rung_reps, v5_class_at, v5_rung_reps, Instr, LoadClass, Op, Program, ProgramClass, GENERATOR_VERSION, GENERATOR_VERSION_V3, GENERATOR_VERSION_V4, GENERATOR_VERSION_V5, V3_CLASS, V4_CLASS, V4_SHADOW_INSTRS, V4_SHADOW_REPS, V5_CLASS, PROGRAM_SUBVERSION_V4};
|
||||
pub use memhard::{cache_log2_words, dataset_log2_words, days_since_genesis, growth_doublings, Cache, MemhardCpu, MixParams, Shape};
|
||||
pub use seed::{fnv1a64, seed_words, SplitMix64};
|
||||
pub use state::{StateLeaves, StateStream};
|
||||
pub use verify::{hash_warp, interpret_warp_init, verify_block, DatasetMode, DatasetSource, Epoch};
|
||||
|
|
|
|||
|
|
@ -39,6 +39,9 @@ struct Args {
|
|||
prehash: String,
|
||||
epoch_hex: Option<String>,
|
||||
day_hex: Option<String>,
|
||||
/// Class v5: the window's state stream file (`--state <file>`, the IGSD1 format of `igneum_pow::state`), whose
|
||||
/// leaves every item of the dataset is keyed by.
|
||||
state: Option<String>,
|
||||
class: LoadClass,
|
||||
/// Days since genesis for the cache growth rule of a class with `growth` (0: the genesis cache).
|
||||
days: u64,
|
||||
|
|
@ -101,7 +104,8 @@ fn usage() -> ! {
|
|||
\x20 --class C load class: v2 (default), mx4, mx8 (class v3: mixer x8, cache growth), dr<len> (Counter ASIC 3.0 item 2: the per-day derivation program, dr736 = the x8-equivalent), w4, w16, w64, w64x4, p4,p16,p64[xN], <class>m<mult>[g]\n\
|
||||
\x20 also: w4, w16, w64, w64x4, p4,p16,p64[xN], <class>m<mult>[g], <class>+sh<S>x<R> (latency-shadow block of S ALU instructions x R passes per iteration, Counter ASIC 3.0 item 8)\n\
|
||||
\x20 --days N days since genesis for the cache growth rule of a class with it (default 0: the 2^26-word cache)\n\
|
||||
\x20 --program-class v2|v3|v4 the program class of the seam (v3 = generator 3 on V3_CLASS, v4 = generator 4 on V4_CLASS = mx8+sh256x27, the chain's own derivation; --era-hex records the era seed)\n\
|
||||
\x20 --program-class v2|v3|v4|v5 the program class of the seam (v3 = generator 3 on V3_CLASS, v4 = generator 4 on V4_CLASS = mx8+sh256x27, v5 = generator 5 on V5_CLASS = mx8+sh256x27+state, the chain's own derivation; --era-hex records the era seed)\n\
|
||||
\x20 --state <file> class v5 (or any --class ...+state): the window's state stream (IGSD1 file, igneum-day-stream --out), whose leaves key every item\n\
|
||||
\x20 --shadow-reps N class v4 at a rung of the latency ladder: the shadow block's pass count (0 = the class's own 27; docs/design/latency-ladder.md), with --program-class v4\n\
|
||||
\x20 --era E era layout over --class: igneum-era-test/<n> or <n>:<64 hex> (the 32-byte era seed E_n)\n\
|
||||
\x20 --era-widths 4[,16,64] the width set the era draws from, in bytes (default 4: pinned; more lets the era draw it)"
|
||||
|
|
@ -116,6 +120,7 @@ fn parse() -> Args {
|
|||
day: "2026-10-03".into(),
|
||||
out: None,
|
||||
closed_form: false,
|
||||
state: None,
|
||||
dataset_log2: DEFAULT_DATASET_LOG2,
|
||||
warps: 20,
|
||||
nonce: 0,
|
||||
|
|
@ -150,6 +155,7 @@ fn parse() -> Args {
|
|||
"--class" => a.class = LoadClass::parse(&val()).unwrap_or_else(|| usage()),
|
||||
"--days" => a.days = val().parse().unwrap_or_else(|_| usage()),
|
||||
"--program-class" => a.program_class = Some(ProgramClass::parse(&val()).unwrap_or_else(|| usage())),
|
||||
"--state" => a.state = Some(val()),
|
||||
"--era-hex" => a.era_hex = Some(val()),
|
||||
"--shadow-reps" => a.shadow_reps = val().parse().unwrap_or_else(|_| usage()),
|
||||
"--era" => a.era = Some(parse_era(&val()).unwrap_or_else(|| usage())),
|
||||
|
|
@ -216,6 +222,33 @@ fn main() {
|
|||
fn epoch_of(a: &Args, mode: DatasetMode) -> (Epoch, String) {
|
||||
let (mut e, label) = epoch_of_class(a, mode);
|
||||
stamp_era(&mut e, a);
|
||||
// class v5: the leaves of --state, built for the dataset's size; a state class without --state is refused here
|
||||
// rather than at the first derivation
|
||||
if e.program.class.state {
|
||||
let Some(path) = &a.state else {
|
||||
eprintln!("class {} keys every item by the window's state: give --state <stream file> (igneum-day-stream --out)", e.program.class.name());
|
||||
std::process::exit(2);
|
||||
};
|
||||
let stream = igneum_pow::state::StateStream::read_file(std::path::Path::new(path)).unwrap_or_else(|err| {
|
||||
eprintln!("{err}");
|
||||
std::process::exit(2)
|
||||
});
|
||||
let leaves = igneum_pow::state::StateLeaves::from_stream(&stream, e.dataset.log2_words);
|
||||
eprintln!(
|
||||
"state stream {}: chain block {} {}, root {}, {} records, {} leaves{}",
|
||||
path,
|
||||
stream.number,
|
||||
igneum_pow::emit::hex_bytes(&stream.block),
|
||||
igneum_pow::emit::hex_bytes(&stream.root),
|
||||
stream.records.len(),
|
||||
leaves.n(),
|
||||
if leaves.sampled { " (sampled)" } else { "" }
|
||||
);
|
||||
e.dataset = e.dataset.with_leaves(std::sync::Arc::new(leaves));
|
||||
} else if a.state.is_some() {
|
||||
eprintln!("--state given for a class without state leaves ({}); use --program-class v5 or --class <class>+state", e.program.class.name());
|
||||
std::process::exit(2);
|
||||
}
|
||||
(e, label)
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -12,6 +12,8 @@
|
|||
use crate::derive::{run_round, DeriveProgram, SoaState, DERIVE_REGS, SOA_LANES};
|
||||
use crate::generator::LoadClass;
|
||||
use crate::seed::{day_key, fnv1a64_words, SplitMix64};
|
||||
use crate::state::StateLeaves;
|
||||
use std::sync::Arc;
|
||||
|
||||
pub const CACHE_LOG2_WORDS: usize = 26;
|
||||
pub const CACHE_SEGMENT_LOG2_LINES: usize = 6;
|
||||
|
|
@ -50,11 +52,14 @@ pub struct Shape {
|
|||
/// program, which replaces the `mixer_mult` applications of `M_r` in every mixer slot when non-zero. 0 for
|
||||
/// version 2 and class v3 (the fixed mixer).
|
||||
pub derive_len: u32,
|
||||
/// Class v5 (`docs/design/class-v5-stored-state.md`, 7 October 2026): the item derivation XORs the window's state
|
||||
/// leaf `leaf(t)` into the 16 initial words before the first mixer (`crate::state`). `false` for every other class.
|
||||
pub state: bool,
|
||||
}
|
||||
|
||||
impl Shape {
|
||||
/// Version 2: one mixer application per round, a 2^26-word cache.
|
||||
pub const V2: Shape = Shape { mixer_mult: 1, cache_log2_words: CACHE_LOG2_WORDS as u32, derive_len: 0 };
|
||||
pub const V2: Shape = Shape { mixer_mult: 1, cache_log2_words: CACHE_LOG2_WORDS as u32, derive_len: 0, state: false };
|
||||
|
||||
/// The shape of a load class on day 0 of the chain (and on every day for a class without the growth rule).
|
||||
pub fn for_class(class: &LoadClass) -> Shape {
|
||||
|
|
@ -68,6 +73,7 @@ impl Shape {
|
|||
mixer_mult: class.mixer_mult(),
|
||||
cache_log2_words: if class.growth { cache_log2_words(days_since_genesis) } else { CACHE_LOG2_WORDS as u32 },
|
||||
derive_len: class.derive_len as u32,
|
||||
state: class.state,
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -374,7 +380,7 @@ impl Cache {
|
|||
/// are the smaller cache's segments word for word.
|
||||
pub fn fill_log2(key: [u32; 8], log2_words: u32) -> Cache {
|
||||
assert!((10..=30).contains(&log2_words), "cache log2 words must be in 10..=30");
|
||||
let shape = Shape { mixer_mult: 1, cache_log2_words: log2_words, derive_len: 0 };
|
||||
let shape = Shape { mixer_mult: 1, cache_log2_words: log2_words, derive_len: 0, state: false };
|
||||
let mut words = vec![0u32; shape.cache_words()];
|
||||
for seg in 0..shape.cache_segments() {
|
||||
Self::fill_segment(&mut words, seg, &key);
|
||||
|
|
@ -505,8 +511,21 @@ impl HotTable {
|
|||
/// (`mp.shape.mixer_mult`) round `r` applies `M` with keys `round_key(r m + j)` for `j = 0 .. m - 1` before its
|
||||
/// one cache read; the final mixer applies `M` with keys `round_key(8 m + j)`. `m = 1` is version 2.
|
||||
pub fn derive_items(ts: &[u32], mp: &MixParams, cache: &Cache, out: &mut [[u32; 16]]) {
|
||||
derive_items_leaves(ts, mp, cache, None, out)
|
||||
}
|
||||
|
||||
/// [`derive_items`] with the state leaves of class v5 (`docs/design/class-v5-stored-state.md` section 2): under a
|
||||
/// shape with `state`, `leaf(t)` is XORed into the 16 initial words of item `t` before the first mixer, and the leaves
|
||||
/// are required; under any other shape they must be absent. A mismatch is a programming error and panics: a dataset
|
||||
/// built without the state it needs would be wrong on every item, which is the class's point.
|
||||
pub fn derive_items_leaves(ts: &[u32], mp: &MixParams, cache: &Cache, leaves: Option<&StateLeaves>, out: &mut [[u32; 16]]) {
|
||||
match (mp.shape.state, leaves) {
|
||||
(true, None) => panic!("class v5 item derivation needs the window's state leaves and was given none"),
|
||||
(false, Some(_)) => panic!("state leaves given to an item derivation whose shape has no state"),
|
||||
_ => {}
|
||||
}
|
||||
if let Some(prog) = &mp.derive {
|
||||
return derive_items_program(ts, mp, prog, cache, out);
|
||||
return derive_items_program(ts, mp, prog, cache, leaves, out);
|
||||
}
|
||||
// The item loop lives in its own function, one instance per cache size the growth rule can reach with the line
|
||||
// mask a constant, never inlined into the callers. Inlined into `MemhardCpu::fetch` it ran at 1.33 ms per unit
|
||||
|
|
@ -515,19 +534,19 @@ pub fn derive_items(ts: &[u32], mp: &MixParams, cache: &Cache, out: &mut [[u32;
|
|||
// constant with the loop still inlined, all stayed at 1.33; the out-of-line instances read 0.60 to 0.62). Any
|
||||
// other cache size (tests) takes the instance with the run-time mask.
|
||||
match cache.log2_words {
|
||||
26 => derive_items_mask::<{ (1u32 << 22) - 1 }>(ts, mp, cache, out),
|
||||
27 => derive_items_mask::<{ (1u32 << 23) - 1 }>(ts, mp, cache, out),
|
||||
28 => derive_items_mask::<{ (1u32 << 24) - 1 }>(ts, mp, cache, out),
|
||||
29 => derive_items_mask::<{ (1u32 << 25) - 1 }>(ts, mp, cache, out),
|
||||
30 => derive_items_mask::<{ (1u32 << 26) - 1 }>(ts, mp, cache, out),
|
||||
_ => derive_items_mask::<0>(ts, mp, cache, out),
|
||||
26 => derive_items_mask::<{ (1u32 << 22) - 1 }>(ts, mp, cache, leaves, out),
|
||||
27 => derive_items_mask::<{ (1u32 << 23) - 1 }>(ts, mp, cache, leaves, out),
|
||||
28 => derive_items_mask::<{ (1u32 << 24) - 1 }>(ts, mp, cache, leaves, out),
|
||||
29 => derive_items_mask::<{ (1u32 << 25) - 1 }>(ts, mp, cache, leaves, out),
|
||||
30 => derive_items_mask::<{ (1u32 << 26) - 1 }>(ts, mp, cache, leaves, out),
|
||||
_ => derive_items_mask::<0>(ts, mp, cache, leaves, out),
|
||||
}
|
||||
}
|
||||
|
||||
/// [`derive_items`] with the cache line mask as a constant (`LINE_MASK = 0`: the cache's own run-time mask). Kept
|
||||
/// out of line on purpose (see [`derive_items`]).
|
||||
#[inline(never)]
|
||||
fn derive_items_mask<const LINE_MASK: u32>(ts: &[u32], mp: &MixParams, cache: &Cache, out: &mut [[u32; 16]]) {
|
||||
fn derive_items_mask<const LINE_MASK: u32>(ts: &[u32], mp: &MixParams, cache: &Cache, leaves: Option<&StateLeaves>, out: &mut [[u32; 16]]) {
|
||||
let n = ts.len();
|
||||
debug_assert!(out.len() >= n);
|
||||
debug_assert!(LINE_MASK == 0 || LINE_MASK == cache.line_mask);
|
||||
|
|
@ -539,6 +558,13 @@ fn derive_items_mask<const LINE_MASK: u32>(ts: &[u32], mp: &MixParams, cache: &C
|
|||
for i in 0..8 {
|
||||
s[8 + i] = t.wrapping_mul(mp.mul[i]).wrapping_add(mp.rc[i]);
|
||||
}
|
||||
if let Some(l) = leaves {
|
||||
// class v5: the window's state leaf of item t, before the first mixer
|
||||
let leaf = l.leaf(t);
|
||||
for i in 0..16 {
|
||||
s[i] ^= leaf[i];
|
||||
}
|
||||
}
|
||||
}
|
||||
for r in 0..ITEM_ROUNDS {
|
||||
for j in 0..m {
|
||||
|
|
@ -570,7 +596,7 @@ fn derive_items_mask<const LINE_MASK: u32>(ts: &[u32], mp: &MixParams, cache: &C
|
|||
/// cache reads of the batch are issued together, as in the fixed-mixer loop, so the 8 dependent misses of
|
||||
/// independent items overlap in the memory system.
|
||||
#[inline(never)]
|
||||
pub fn derive_items_program(ts: &[u32], mp: &MixParams, prog: &DeriveProgram, cache: &Cache, out: &mut [[u32; 16]]) {
|
||||
pub fn derive_items_program(ts: &[u32], mp: &MixParams, prog: &DeriveProgram, cache: &Cache, leaves: Option<&StateLeaves>, out: &mut [[u32; 16]]) {
|
||||
let n = ts.len();
|
||||
debug_assert!(out.len() >= n && n <= SOA_LANES);
|
||||
assert_eq!(prog.rounds.len(), ITEM_ROUNDS + 1);
|
||||
|
|
@ -581,6 +607,12 @@ pub fn derive_items_program(ts: &[u32], mp: &MixParams, prog: &DeriveProgram, ca
|
|||
st[i][k] = mp.key[i];
|
||||
st[8 + i][k] = t.wrapping_mul(mp.mul[i]).wrapping_add(mp.rc[i]);
|
||||
}
|
||||
if let Some(l) = leaves {
|
||||
let leaf = l.leaf(t);
|
||||
for i in 0..16 {
|
||||
st[i][k] ^= leaf[i];
|
||||
}
|
||||
}
|
||||
}
|
||||
let mask = cache.line_mask();
|
||||
for r in 0..ITEM_ROUNDS {
|
||||
|
|
@ -602,15 +634,22 @@ pub fn derive_items_program(ts: &[u32], mp: &MixParams, prog: &DeriveProgram, ca
|
|||
|
||||
/// One dataset item, 16 words.
|
||||
pub fn derive_item(t: u32, mp: &MixParams, cache: &Cache) -> [u32; 16] {
|
||||
derive_item_leaves(t, mp, cache, None)
|
||||
}
|
||||
|
||||
/// [`derive_item`] with the state leaves of class v5.
|
||||
pub fn derive_item_leaves(t: u32, mp: &MixParams, cache: &Cache, leaves: Option<&StateLeaves>) -> [u32; 16] {
|
||||
let mut out = [[0u32; 16]; 1];
|
||||
derive_items(&[t], mp, cache, &mut out);
|
||||
derive_items_leaves(&[t], mp, cache, leaves, &mut out);
|
||||
out[0]
|
||||
}
|
||||
|
||||
/// The CPU verifier's view of the memory-hard dataset: the mixer parameters (with the shape) and the cache.
|
||||
/// The CPU verifier's view of the memory-hard dataset: the mixer parameters (with the shape), the cache (shared, so
|
||||
/// a class v5 window refresh keeps the day's 256 MiB and swaps the leaves) and, under class v5, the window's leaves.
|
||||
pub struct MemhardCpu {
|
||||
pub params: MixParams,
|
||||
pub cache: Cache,
|
||||
pub cache: Arc<Cache>,
|
||||
pub leaves: Option<Arc<StateLeaves>>,
|
||||
}
|
||||
|
||||
/// Largest batch `MemhardCpu::fetch` accepts (two warps).
|
||||
|
|
@ -622,7 +661,7 @@ impl MemhardCpu {
|
|||
Self::with_shape(key, Shape::V2)
|
||||
}
|
||||
pub fn with_shape(key: [u32; 8], shape: Shape) -> Self {
|
||||
Self { params: MixParams::with_shape(key, shape), cache: Cache::fill_log2(key, shape.cache_log2_words) }
|
||||
Self { params: MixParams::with_shape(key, shape), cache: Arc::new(Cache::fill_log2(key, shape.cache_log2_words)), leaves: None }
|
||||
}
|
||||
pub fn for_day(day: &str) -> Self {
|
||||
Self::new(day_key(day))
|
||||
|
|
@ -630,6 +669,17 @@ impl MemhardCpu {
|
|||
pub fn shape(&self) -> Shape {
|
||||
self.params.shape
|
||||
}
|
||||
/// This view with the window's state leaves (class v5). The shape must have `state`.
|
||||
pub fn with_leaves(mut self, leaves: Arc<StateLeaves>) -> Self {
|
||||
assert!(self.params.shape.state, "state leaves on a shape without state");
|
||||
self.leaves = Some(leaves);
|
||||
self
|
||||
}
|
||||
/// A view of the same day (the same cache, shared) with other leaves: the class v5 window refresh.
|
||||
pub fn refreshed(&self, leaves: Arc<StateLeaves>) -> Self {
|
||||
assert!(self.params.shape.state, "state leaves on a shape without state");
|
||||
Self { params: self.params.clone(), cache: self.cache.clone(), leaves: Some(leaves) }
|
||||
}
|
||||
/// `dataset[w] = item(w >> 4)[w & 15]` (the linear layout).
|
||||
pub fn word(&self, w: u32) -> u32 {
|
||||
self.word_at(Layout::LINEAR, w)
|
||||
|
|
@ -638,7 +688,7 @@ impl MemhardCpu {
|
|||
/// day's, so one cache serves every era of a day).
|
||||
pub fn word_at(&self, layout: Layout, w: u32) -> u32 {
|
||||
let (t, j) = layout.split(w);
|
||||
derive_item(t, &self.params, &self.cache)[j as usize]
|
||||
derive_item_leaves(t, &self.params, &self.cache, self.leaves.as_deref())[j as usize]
|
||||
}
|
||||
/// `out[k] = dataset[idx[k]]` for every k, `idx.len() <= FETCH_MAX`. Equal items are derived once.
|
||||
/// Returns the number of distinct items derived.
|
||||
|
|
@ -664,7 +714,7 @@ impl MemhardCpu {
|
|||
slot[k] = j as u8;
|
||||
}
|
||||
let mut items = [[0u32; 16]; FETCH_MAX];
|
||||
derive_items(&uniq[..u], &self.params, &self.cache, &mut items);
|
||||
derive_items_leaves(&uniq[..u], &self.params, &self.cache, self.leaves.as_deref(), &mut items);
|
||||
for k in 0..n {
|
||||
out[k] = items[slot[k] as usize][word[k] as usize];
|
||||
}
|
||||
|
|
@ -695,7 +745,7 @@ impl MemhardCpu {
|
|||
slot[k] = j as u8;
|
||||
}
|
||||
let mut items = [[0u32; 16]; FETCH_MAX];
|
||||
derive_items(&uniq[..u], &self.params, &self.cache, &mut items);
|
||||
derive_items_leaves(&uniq[..u], &self.params, &self.cache, self.leaves.as_deref(), &mut items);
|
||||
for k in 0..n {
|
||||
let o = word[k] as usize;
|
||||
out[k][..width].copy_from_slice(&items[slot[k] as usize][o..o + width]);
|
||||
|
|
@ -851,7 +901,7 @@ mod tests {
|
|||
let v2 = Shape::for_class_day(&LoadClass::V2, 100_000);
|
||||
assert_eq!(v2, Shape::V2);
|
||||
let v3 = Shape::for_class_day(&LoadClass::MX4, 0);
|
||||
assert_eq!(v3, Shape { mixer_mult: 4, cache_log2_words: 26, derive_len: 0 });
|
||||
assert_eq!(v3, Shape { mixer_mult: 4, cache_log2_words: 26, derive_len: 0, state: false });
|
||||
assert_eq!(Shape::for_class_day(&LoadClass::MX4, 1_460).cache_log2_words, 27);
|
||||
assert_eq!(v3.mixers_per_item(), 36);
|
||||
assert_eq!(Shape::V2.mixers_per_item(), 9);
|
||||
|
|
@ -872,7 +922,7 @@ mod tests {
|
|||
assert_eq!(small.segments(), 64);
|
||||
assert_eq!(small.line_mask(), 4095);
|
||||
for m in [1u32, 2, 4] {
|
||||
let mp = MixParams::with_shape(key, Shape { mixer_mult: m, cache_log2_words: 16, derive_len: 0 });
|
||||
let mp = MixParams::with_shape(key, Shape { mixer_mult: m, cache_log2_words: 16, derive_len: 0, state: false });
|
||||
for t in [0u32, 1, 12_345, u32::MAX] {
|
||||
let got = derive_item(t, &mp, &small);
|
||||
let mut s = [0u32; 16];
|
||||
|
|
@ -895,8 +945,8 @@ mod tests {
|
|||
assert_eq!(got, s, "m {m} t {t}");
|
||||
}
|
||||
}
|
||||
let v2 = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: 0 });
|
||||
let v3 = MixParams::with_shape(key, Shape { mixer_mult: 4, cache_log2_words: 16, derive_len: 0 });
|
||||
let v2 = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: 0, state: false });
|
||||
let v3 = MixParams::with_shape(key, Shape { mixer_mult: 4, cache_log2_words: 16, derive_len: 0, state: false });
|
||||
assert_ne!(derive_item(0, &v2, &small), derive_item(0, &v3, &small));
|
||||
assert_eq!(round_key_mult(0, 0, 1), round_key(0));
|
||||
assert_eq!(round_key_mult(8, 0, 1), round_key(8));
|
||||
|
|
|
|||
|
|
@ -29,6 +29,8 @@ pub struct PackIdentity {
|
|||
pub class: ProgramClass,
|
||||
/// `IGNEUM_ERA_SEED_HEX` when the pack carries one (class v3 chain packs).
|
||||
pub era_hex: Option<String>,
|
||||
/// `IGNEUM_STATE_ROOT_HEX` of a class v5 pack (the window's state root the leaves derive from).
|
||||
pub state_root_hex: Option<String>,
|
||||
}
|
||||
|
||||
/// Why a pack is not the one a worker should mine with. `Display` is the plain-words line the logs carry.
|
||||
|
|
@ -193,8 +195,18 @@ pub fn verify_pack_texts_chain(
|
|||
// without the block is no v4 pack. A generator 2 pack with a shadow (the measurement ladder of
|
||||
// proto-cuda/packs-ca3-shadow) carries a class-bearing id and stays loadable.
|
||||
let shadow = define_u32(program_h, "IGNEUM_SHADOW_INSTRS").unwrap_or(0);
|
||||
// Class v5 (docs/design/class-v5-stored-state.md): the state lines are the mark of class v5, so a generator 5 pack
|
||||
// carries IGNEUM_STATE_ROOT_HEX (and the shadow block of v4) and no other generator does.
|
||||
let state_root_hex = define_str(program_h, "IGNEUM_STATE_ROOT_HEX");
|
||||
match (class, state_root_hex.is_some()) {
|
||||
(ProgramClass::V5, false) => return Err(PackFault::Disagree("IGNEUM_GENERATOR 5 (class v5) without IGNEUM_STATE_ROOT_HEX: not a class v5 pack".into())),
|
||||
(ProgramClass::V5, true) => {}
|
||||
(_, true) => return Err(PackFault::Disagree(format!("IGNEUM_GENERATOR {generator} with class v5 state lines: a program over state leaves is generator 5 (export the pack as class v5)"))),
|
||||
_ => {}
|
||||
}
|
||||
match (class, shadow > 0) {
|
||||
(ProgramClass::V4, false) => return Err(PackFault::Disagree("IGNEUM_GENERATOR 4 (class v4) without IGNEUM_SHADOW_INSTRS: not a class v4 pack".into())),
|
||||
(ProgramClass::V5, false) => return Err(PackFault::Disagree("IGNEUM_GENERATOR 5 (class v5) without IGNEUM_SHADOW_INSTRS: not a class v5 pack".into())),
|
||||
// the class v4 stream sub-version (AP-F8-1 amendment, 7 October 2026): a generator 4 pack from before the
|
||||
// load-source rule carries no IGNEUM_PROGRAM_SUBVERSION and its program id is another stream's; refused
|
||||
(ProgramClass::V4, true) if define_u32(program_h, "IGNEUM_PROGRAM_SUBVERSION") != Some(u32::from(crate::generator::PROGRAM_SUBVERSION_V4)) => {
|
||||
|
|
@ -265,7 +277,7 @@ pub fn verify_pack_texts_chain(
|
|||
if epoch_hex != want_epoch_hex || day_hex != want_day_hex {
|
||||
return Err(PackFault::OutOfDate { pack_epoch: epoch_hex, pack_day: day_hex, want_epoch: want_epoch_hex, want_day: want_day_hex });
|
||||
}
|
||||
Ok(PackIdentity { epoch_hex, day_hex, attempt, seedw, keyw, generator, class, era_hex })
|
||||
Ok(PackIdentity { epoch_hex, day_hex, attempt, seedw, keyw, generator, class, era_hex, state_root_hex })
|
||||
}
|
||||
|
||||
/// [`verify_pack_texts`] over a pack directory.
|
||||
|
|
|
|||
287
igneum-pow/src/state.rs
Normal file
287
igneum-pow/src/state.rs
Normal file
|
|
@ -0,0 +1,287 @@
|
|||
//! Class v5, proof of stored state and of following (`docs/design/class-v5-stored-state.md`, 7 October 2026): the
|
||||
//! leaves the item derivation XORs in (section 2 of the page), built from the canonical state stream of the
|
||||
//! window's reference block.
|
||||
//!
|
||||
//! `D[i] = Blake2b-512("igneum-sd1/" || root || i_le32 || record_i)` for the `n` records of the stream, and item
|
||||
//! `t` takes `leaf(t) = D[t mod n]`: every item is keyed by the state, so a hasher without it is wrong on every
|
||||
//! item (the known-failed case, the first test). When the stream has more records than the dataset has items, the
|
||||
//! records are ordered by `Blake2b-256("igneum-sd1-sample/" || root || record)` and the first `items` are taken, a
|
||||
//! sample nobody can choose without the whole state and the root.
|
||||
//!
|
||||
//! The stream file (`StateStream`): the plain format every side reads without a serialisation library, `IGSD1\0`,
|
||||
//! the chain block number (le64) and hash (32), the state root (32), the record count (le32), then each record as
|
||||
//! its length (le32) and bytes. The node's executor writes it (`igneum/exec/src/day_stream.rs`), the miner fetches
|
||||
//! it, the CLI's `--state` reads it, and a pack carries the leaves it yields as `leaves.bin`.
|
||||
|
||||
use crate::blake2b::{blake2b_256, blake2b_512};
|
||||
use crate::seed::fnv1a64_words;
|
||||
|
||||
pub const LEAF_TAG: &[u8] = b"igneum-sd1/";
|
||||
pub const SAMPLE_TAG: &[u8] = b"igneum-sd1-sample/";
|
||||
pub const STREAM_MAGIC: &[u8; 6] = b"IGSD1\0";
|
||||
|
||||
/// The canonical state stream at one chain block: what the executor serialises and what the leaves derive from.
|
||||
#[derive(Clone, Debug, PartialEq, Eq)]
|
||||
pub struct StateStream {
|
||||
pub number: u64,
|
||||
pub block: [u8; 32],
|
||||
pub root: [u8; 32],
|
||||
pub records: Vec<Vec<u8>>,
|
||||
}
|
||||
|
||||
impl StateStream {
|
||||
pub fn encode(&self) -> Vec<u8> {
|
||||
let mut b = Vec::with_capacity(6 + 8 + 32 + 32 + 4 + self.records.iter().map(|r| 4 + r.len()).sum::<usize>());
|
||||
b.extend_from_slice(STREAM_MAGIC);
|
||||
b.extend_from_slice(&self.number.to_le_bytes());
|
||||
b.extend_from_slice(&self.block);
|
||||
b.extend_from_slice(&self.root);
|
||||
b.extend_from_slice(&(self.records.len() as u32).to_le_bytes());
|
||||
for r in &self.records {
|
||||
b.extend_from_slice(&(r.len() as u32).to_le_bytes());
|
||||
b.extend_from_slice(r);
|
||||
}
|
||||
b
|
||||
}
|
||||
|
||||
pub fn decode(bytes: &[u8]) -> Result<StateStream, String> {
|
||||
if bytes.len() < 6 + 8 + 32 + 32 + 4 || &bytes[..6] != STREAM_MAGIC {
|
||||
return Err("not a state stream file (magic IGSD1)".into());
|
||||
}
|
||||
let mut at = 6;
|
||||
let number = u64::from_le_bytes(bytes[at..at + 8].try_into().unwrap());
|
||||
at += 8;
|
||||
let block: [u8; 32] = bytes[at..at + 32].try_into().unwrap();
|
||||
at += 32;
|
||||
let root: [u8; 32] = bytes[at..at + 32].try_into().unwrap();
|
||||
at += 32;
|
||||
let n = u32::from_le_bytes(bytes[at..at + 4].try_into().unwrap()) as usize;
|
||||
at += 4;
|
||||
let mut records = Vec::with_capacity(n.min(1 << 20));
|
||||
for i in 0..n {
|
||||
if at + 4 > bytes.len() {
|
||||
return Err(format!("state stream truncated at record {i} of {n}"));
|
||||
}
|
||||
let len = u32::from_le_bytes(bytes[at..at + 4].try_into().unwrap()) as usize;
|
||||
at += 4;
|
||||
if at + len > bytes.len() {
|
||||
return Err(format!("state stream truncated inside record {i} of {n}"));
|
||||
}
|
||||
records.push(bytes[at..at + len].to_vec());
|
||||
at += len;
|
||||
}
|
||||
if at != bytes.len() {
|
||||
return Err(format!("state stream has {} trailing bytes", bytes.len() - at));
|
||||
}
|
||||
Ok(StateStream { number, block, root, records })
|
||||
}
|
||||
|
||||
pub fn read_file(path: &std::path::Path) -> Result<StateStream, String> {
|
||||
let bytes = std::fs::read(path).map_err(|e| format!("read {}: {e}", path.display()))?;
|
||||
Self::decode(&bytes)
|
||||
}
|
||||
}
|
||||
|
||||
/// `D[i]`: the 64-byte digest of record `i` under `root`, as 16 little-endian words.
|
||||
pub fn leaf_digest(root: &[u8; 32], i: u32, record: &[u8]) -> [u32; 16] {
|
||||
let d = blake2b_512(&[LEAF_TAG, root, &i.to_le_bytes(), record]);
|
||||
let mut w = [0u32; 16];
|
||||
for (k, x) in w.iter_mut().enumerate() {
|
||||
*x = u32::from_le_bytes(d[k * 4..k * 4 + 4].try_into().unwrap());
|
||||
}
|
||||
w
|
||||
}
|
||||
|
||||
/// The sample order key of a record under `root`.
|
||||
pub fn sample_key(root: &[u8; 32], record: &[u8]) -> [u8; 32] {
|
||||
blake2b_256(&[SAMPLE_TAG, root, record])
|
||||
}
|
||||
|
||||
/// The leaves of one window (or day) of class v5: `n` digests of 64 bytes, `leaf(t) = D[t mod n]`.
|
||||
#[derive(Clone, Debug, PartialEq, Eq)]
|
||||
pub struct StateLeaves {
|
||||
pub root: [u8; 32],
|
||||
pub block: [u8; 32],
|
||||
pub number: u64,
|
||||
/// Records in the stream before any sample.
|
||||
pub records_total: u64,
|
||||
/// Whether the stream had more records than the dataset has items (the sample rule applied).
|
||||
pub sampled: bool,
|
||||
leaves: Vec<[u32; 16]>,
|
||||
}
|
||||
|
||||
impl StateLeaves {
|
||||
/// The items a dataset of `2^log2_words` words has: `2^(log2_words - 4)`.
|
||||
pub fn items_of(log2_words: u32) -> u64 {
|
||||
1u64 << log2_words.saturating_sub(4)
|
||||
}
|
||||
|
||||
/// The leaves of `records` (canonical order) under `root` for a dataset of `2^log2_words` words. An empty stream
|
||||
/// yields one leaf, the digest of the empty record, so `n` is never 0.
|
||||
pub fn build(root: [u8; 32], block: [u8; 32], number: u64, records: &[Vec<u8>], log2_words: u32) -> StateLeaves {
|
||||
let items = Self::items_of(log2_words);
|
||||
let records_total = records.len() as u64;
|
||||
let empty: Vec<Vec<u8>> = vec![Vec::new()];
|
||||
let records = if records.is_empty() { &empty[..] } else { records };
|
||||
let sampled = records.len() as u64 > items;
|
||||
let chosen: Vec<&Vec<u8>> = if sampled {
|
||||
let mut keyed: Vec<([u8; 32], &Vec<u8>)> = records.iter().map(|r| (sample_key(&root, r), r)).collect();
|
||||
keyed.sort_unstable_by(|a, b| a.0.cmp(&b.0).then_with(|| a.1.cmp(b.1)));
|
||||
keyed.into_iter().take(items as usize).map(|(_, r)| r).collect()
|
||||
} else {
|
||||
records.iter().collect()
|
||||
};
|
||||
let leaves = chosen.iter().enumerate().map(|(i, r)| leaf_digest(&root, i as u32, r)).collect();
|
||||
StateLeaves { root, block, number, records_total, sampled, leaves }
|
||||
}
|
||||
|
||||
pub fn from_stream(s: &StateStream, log2_words: u32) -> StateLeaves {
|
||||
Self::build(s.root, s.block, s.number, &s.records, log2_words)
|
||||
}
|
||||
|
||||
/// Leaves from the raw words of a `leaves.bin` (16 words per leaf), for a worker or a test that holds no stream.
|
||||
pub fn from_words(root: [u8; 32], block: [u8; 32], number: u64, words: &[u32]) -> StateLeaves {
|
||||
assert!(!words.is_empty() && words.len() % 16 == 0, "leaves are 16 words each");
|
||||
let leaves = words.chunks_exact(16).map(|c| c.try_into().unwrap()).collect::<Vec<[u32; 16]>>();
|
||||
StateLeaves { root, block, number, records_total: leaves.len() as u64, sampled: false, leaves }
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn n(&self) -> u32 {
|
||||
self.leaves.len() as u32
|
||||
}
|
||||
|
||||
/// `leaf(t) = D[t mod n]`.
|
||||
#[inline(always)]
|
||||
pub fn leaf(&self, t: u32) -> &[u32; 16] {
|
||||
&self.leaves[(t % self.n()) as usize]
|
||||
}
|
||||
|
||||
pub fn leaves(&self) -> &[[u32; 16]] {
|
||||
&self.leaves
|
||||
}
|
||||
|
||||
/// The flat words of `leaves.bin`.
|
||||
pub fn words(&self) -> Vec<u32> {
|
||||
self.leaves.iter().flat_map(|l| l.iter().copied()).collect()
|
||||
}
|
||||
|
||||
/// The bytes of `leaves.bin` (little-endian words).
|
||||
pub fn bytes(&self) -> Vec<u8> {
|
||||
self.words().iter().flat_map(|w| w.to_le_bytes()).collect()
|
||||
}
|
||||
|
||||
/// FNV-1a 64 over the leaves as little-endian bytes (the pack's `IGNEUM_STATE_LEAVES_FNV64`).
|
||||
pub fn fnv1a64(&self) -> u64 {
|
||||
fnv1a64_words(&self.words())
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::generator::{generate_class, V5_CLASS};
|
||||
use crate::memhard::{derive_item_leaves, Cache, MixParams, Shape};
|
||||
use crate::seed::day_key;
|
||||
use crate::verify::{hash_warp, DatasetMode, DatasetSource};
|
||||
use std::sync::Arc;
|
||||
|
||||
fn records(n: usize, salt: u8) -> Vec<Vec<u8>> {
|
||||
(0..n).map(|i| vec![salt, i as u8, (i >> 8) as u8, 7]).collect()
|
||||
}
|
||||
|
||||
fn leaves(root: u8, n: usize, log2_words: u32) -> Arc<StateLeaves> {
|
||||
Arc::new(StateLeaves::build([root; 32], [0x22; 32], 5, &records(n, root), log2_words))
|
||||
}
|
||||
|
||||
/// The known-failed case, first: a hasher without the state (no leaves, the leaves of another root, the leaves
|
||||
/// of a stream one record short, the previous window's leaves) is wrong on every item and every lane.
|
||||
#[test]
|
||||
fn a_stateless_hasher_is_wrong_on_every_item() {
|
||||
let key = day_key("2026-10-03");
|
||||
let shape = Shape { mixer_mult: 8, cache_log2_words: 16, derive_len: 0, state: true };
|
||||
let cache = Arc::new(Cache::fill_log2(key, 16));
|
||||
let mp = MixParams::with_shape(key, shape);
|
||||
let good = leaves(0x11, 93, 20);
|
||||
let other_root = leaves(0x12, 93, 20);
|
||||
let one_short = Arc::new(StateLeaves::build([0x13; 32], [0x22; 32], 5, &records(92, 0x11), 20)); // a record short means another root
|
||||
let previous_window = leaves(0x10, 93, 20);
|
||||
for (name, bad) in [("another root", other_root.clone()), ("one record short", one_short.clone()), ("the previous window", previous_window.clone())] {
|
||||
let equal = (0..64u32).filter(|&t| derive_item_leaves(t * 7919, &mp, &cache, Some(&good)) == derive_item_leaves(t * 7919, &mp, &cache, Some(&bad))).count();
|
||||
assert_eq!(equal, 0, "{name}: {equal} of 64 items equal");
|
||||
}
|
||||
let stateless = Shape { state: false, ..shape };
|
||||
let mp_stateless = MixParams::with_shape(key, stateless);
|
||||
let equal = (0..64u32).filter(|&t| derive_item_leaves(t * 7919, &mp, &cache, Some(&good)) == derive_item_leaves(t * 7919, &mp_stateless, &cache, None)).count();
|
||||
assert_eq!(equal, 0, "no leaves at all: {equal} of 64 items equal");
|
||||
// the warp: a class v5 program over a small dataset, the same program and cache, other leaves
|
||||
let program = generate_class("igneum-genesis", V5_CLASS);
|
||||
let ds = DatasetSource::new_shape("2026-10-03", DatasetMode::MemoryHard, 20, shape).with_leaves(good.clone());
|
||||
let ds_other = DatasetSource::new_shape("2026-10-03", DatasetMode::MemoryHard, 20, shape).with_leaves(previous_window.clone());
|
||||
let a = hash_warp(&program, 0, &ds);
|
||||
let b = hash_warp(&program, 0, &ds_other);
|
||||
assert_eq!(a.iter().zip(b.iter()).filter(|(x, y)| x == y).count(), 0, "0 of 32 lanes agree");
|
||||
assert_eq!(hash_warp(&program, 0, &ds), a, "the same leaves hash the same");
|
||||
}
|
||||
|
||||
/// Every item takes a leaf: `leaf(t) = D[t mod n]`, so items `t` and `t + n` share a leaf and still differ.
|
||||
#[test]
|
||||
fn every_item_is_keyed_and_the_leaf_wraps() {
|
||||
let l = leaves(0x11, 93, 28);
|
||||
assert_eq!(l.n(), 93);
|
||||
assert!(!l.sampled);
|
||||
assert_eq!(l.records_total, 93);
|
||||
for t in [0u32, 1, 92, 93, 94, 1_000_000, u32::MAX] {
|
||||
assert_eq!(l.leaf(t), l.leaf(t % 93));
|
||||
assert_eq!(*l.leaf(t), leaf_digest(&[0x11; 32], t % 93, &records(93, 0x11)[(t % 93) as usize]));
|
||||
}
|
||||
let key = day_key("2026-10-03");
|
||||
let shape = Shape { mixer_mult: 8, cache_log2_words: 16, derive_len: 0, state: true };
|
||||
let cache = Cache::fill_log2(key, 16);
|
||||
let mp = MixParams::with_shape(key, shape);
|
||||
assert_ne!(derive_item_leaves(5, &mp, &cache, Some(&l)), derive_item_leaves(5 + 93, &mp, &cache, Some(&l)));
|
||||
// an empty stream yields one leaf (the digest of the empty record), never a division by zero
|
||||
let empty = StateLeaves::build([0x11; 32], [0; 32], 0, &[], 28);
|
||||
assert_eq!(empty.n(), 1);
|
||||
assert_eq!(empty.records_total, 0);
|
||||
assert_eq!(*empty.leaf(12_345), leaf_digest(&[0x11; 32], 0, &[]));
|
||||
}
|
||||
|
||||
/// Above the dataset size the records are sampled in the keyed order: a different root picks a different set,
|
||||
/// and the set cannot be the first `items` records of the stream.
|
||||
#[test]
|
||||
fn the_sample_above_the_dataset_size_is_keyed_by_the_root() {
|
||||
let recs = records(40, 0x33);
|
||||
let a = StateLeaves::build([0x11; 32], [0; 32], 0, &recs, 8);
|
||||
let b = StateLeaves::build([0x12; 32], [0; 32], 0, &recs, 8);
|
||||
assert_eq!(StateLeaves::items_of(8), 16);
|
||||
assert_eq!((a.n(), a.sampled, a.records_total), (16, true, 40));
|
||||
assert_ne!(a.leaves(), b.leaves(), "another root, another sample");
|
||||
// the positional first 16 are not the sample (with overwhelming probability for 40 choose 16)
|
||||
let positional = StateLeaves::build([0x11; 32], [0; 32], 0, &recs[..16], 8);
|
||||
assert_ne!(a.leaves(), positional.leaves());
|
||||
// the same inputs sample the same
|
||||
assert_eq!(StateLeaves::build([0x11; 32], [0; 32], 0, &recs, 8), a);
|
||||
// at the dataset size exactly, no sample
|
||||
let c = StateLeaves::build([0x11; 32], [0; 32], 0, &recs[..16], 8);
|
||||
assert!(!c.sampled && c.n() == 16);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stream_file_round_trip_and_refusals() {
|
||||
let s = StateStream { number: 159_357, block: [0xaf; 32], root: [0x1c; 32], records: records(93, 1) };
|
||||
let bytes = s.encode();
|
||||
assert_eq!(&bytes[..6], STREAM_MAGIC);
|
||||
assert_eq!(StateStream::decode(&bytes).unwrap(), s);
|
||||
assert!(StateStream::decode(&bytes[..bytes.len() - 1]).is_err(), "truncated");
|
||||
let mut trailing = bytes.clone();
|
||||
trailing.push(0);
|
||||
assert!(StateStream::decode(&trailing).is_err(), "trailing bytes");
|
||||
assert!(StateStream::decode(b"IGSD0\0").is_err(), "wrong magic");
|
||||
let l = StateLeaves::from_stream(&s, 28);
|
||||
let back = StateLeaves::from_words(s.root, s.block, s.number, &l.words());
|
||||
assert_eq!(back.leaves(), l.leaves());
|
||||
assert_eq!(l.bytes().len(), 93 * 64);
|
||||
assert_eq!(l.fnv1a64(), back.fnv1a64());
|
||||
}
|
||||
}
|
||||
|
|
@ -227,6 +227,42 @@ impl DatasetSource {
|
|||
Self { log2_words, mask, key, key_bytes: Vec::new(), dataset, hot: None }
|
||||
}
|
||||
|
||||
/// This source with the window's state leaves (class v5, `docs/design/class-v5-stored-state.md`): memory-hard mode
|
||||
/// under a shape with `state` only.
|
||||
pub fn with_leaves(mut self, leaves: std::sync::Arc<crate::state::StateLeaves>) -> Self {
|
||||
match &mut self.dataset {
|
||||
Dataset::MemoryHard(m) => {
|
||||
assert!(m.params.shape.state, "state leaves on a dataset whose shape has no state");
|
||||
m.leaves = Some(leaves);
|
||||
}
|
||||
Dataset::ClosedForm { .. } => panic!("state leaves on a closed-form dataset"),
|
||||
}
|
||||
self
|
||||
}
|
||||
|
||||
/// A source of the same day with other leaves, the 256 MiB cache shared (the class v5 window refresh).
|
||||
pub fn refreshed(&self, leaves: std::sync::Arc<crate::state::StateLeaves>) -> Self {
|
||||
let dataset = match &self.dataset {
|
||||
Dataset::MemoryHard(m) => Dataset::MemoryHard(m.refreshed(leaves)),
|
||||
Dataset::ClosedForm { .. } => panic!("state leaves on a closed-form dataset"),
|
||||
};
|
||||
Self { log2_words: self.log2_words, mask: self.mask, key: self.key, key_bytes: self.key_bytes.clone(), dataset, hot: None }
|
||||
}
|
||||
|
||||
/// A copy of this source sharing its cache (and leaves), for a caller that needs an owned source from a shared one.
|
||||
pub fn refreshed_or_clone(&self) -> Self {
|
||||
let dataset = match &self.dataset {
|
||||
Dataset::MemoryHard(m) => Dataset::MemoryHard(crate::memhard::MemhardCpu { params: m.params.clone(), cache: m.cache.clone(), leaves: m.leaves.clone() }),
|
||||
Dataset::ClosedForm { d0, d1 } => Dataset::ClosedForm { d0: *d0, d1: *d1 },
|
||||
};
|
||||
Self { log2_words: self.log2_words, mask: self.mask, key: self.key, key_bytes: self.key_bytes.clone(), dataset, hot: None }
|
||||
}
|
||||
|
||||
/// The window's state leaves, when the source carries them.
|
||||
pub fn leaves(&self) -> Option<&std::sync::Arc<crate::state::StateLeaves>> {
|
||||
self.memhard().and_then(|m| m.leaves.as_ref())
|
||||
}
|
||||
|
||||
/// This source with the hot table of the epoch whose program seed bytes are `seed_bytes` (`mb` MiB).
|
||||
pub fn with_hot(mut self, seed_bytes: &[u8], mb: u32) -> Self {
|
||||
self.hot = Some(HotTable::for_seed_bytes(seed_bytes, mb));
|
||||
|
|
|
|||
|
|
@ -43,7 +43,7 @@ fn item_by_hand(t: u32, mp: &MixParams, cache: &Cache) -> [u32; 16] {
|
|||
fn derived_item_by_hand_and_in_batches() {
|
||||
let key = day_key(DAY);
|
||||
let cache = Cache::fill_log2(key, 16);
|
||||
let shape = Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8 };
|
||||
let shape = Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8, state: false };
|
||||
let mp = MixParams::with_shape(key, shape);
|
||||
let prog = mp.derive.as_ref().unwrap();
|
||||
assert_eq!(prog.rounds.len(), DERIVE_PROGRAMS);
|
||||
|
|
@ -64,7 +64,7 @@ fn derived_item_by_hand_and_in_batches() {
|
|||
derive_items(&ts[..5], &mp, &cache, &mut out5);
|
||||
assert_eq!(&out5[..], &out[..5]);
|
||||
// the fixed mixer of the same key gives other items
|
||||
let v3 = MixParams::with_shape(key, Shape { mixer_mult: 8, cache_log2_words: 16, derive_len: 0 });
|
||||
let v3 = MixParams::with_shape(key, Shape { mixer_mult: 8, cache_log2_words: 16, derive_len: 0, state: false });
|
||||
assert!(v3.derive.is_none());
|
||||
assert_ne!(derive_item(0, &v3, &cache), derive_item(0, &mp, &cache));
|
||||
}
|
||||
|
|
@ -79,7 +79,7 @@ fn v2_and_v3_are_untouched() {
|
|||
let key = day_key(DAY);
|
||||
let cache = Cache::fill_log2(key, 16);
|
||||
// the version 2 item restated by hand (the mixer_mult_by_hand test of memhard.rs, m = 1)
|
||||
let v2 = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: 0 });
|
||||
let v2 = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: 0, state: false });
|
||||
let t = 12_345u32;
|
||||
let mut s = [0u32; 16];
|
||||
s[..8].copy_from_slice(&key);
|
||||
|
|
@ -96,7 +96,7 @@ fn v2_and_v3_are_untouched() {
|
|||
mixer(&mut s, round_key(8), &v2);
|
||||
assert_eq!(derive_item(t, &v2, &cache), s);
|
||||
// the mixer constants of the derivation class are the v2 draws (the stream continues after them)
|
||||
let dr = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8 });
|
||||
let dr = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8, state: false });
|
||||
assert_eq!((dr.rot, dr.mul, dr.rc), (v2.rot, v2.mul, v2.rc));
|
||||
}
|
||||
|
||||
|
|
@ -109,12 +109,12 @@ fn stream_class_name_and_id() {
|
|||
rng.next();
|
||||
}
|
||||
let expect = DeriveProgram::draw(&mut rng, DERIVE_LEN_X8);
|
||||
let mp = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: DERIVE_LEN_X8 });
|
||||
let mp = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: DERIVE_LEN_X8, state: false });
|
||||
assert_eq!(mp.derive.as_ref().unwrap(), &expect);
|
||||
// another day, another program; another length, another program
|
||||
let other = MixParams::with_shape(day_key("2026-10-04"), Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: DERIVE_LEN_X8 });
|
||||
let other = MixParams::with_shape(day_key("2026-10-04"), Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: DERIVE_LEN_X8, state: false });
|
||||
assert_ne!(other.derive.as_ref().unwrap().fingerprint(), expect.fingerprint());
|
||||
let short = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: 368 });
|
||||
let short = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: 368, state: false });
|
||||
assert_eq!(short.derive.as_ref().unwrap().instr_count(), 9 * 368);
|
||||
// the class: name, parse, id, and the v2 program stream (v2 loads, no width roll)
|
||||
let c = LoadClass::DR736;
|
||||
|
|
@ -179,8 +179,8 @@ fn determinism_and_pack_text() {
|
|||
fn stats_beside_x8() {
|
||||
let key = day_key(DAY);
|
||||
let cache = Cache::fill_log2(key, 18);
|
||||
let dr = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 18, derive_len: DERIVE_LEN_X8 });
|
||||
let x8 = MixParams::with_shape(key, Shape { mixer_mult: 8, cache_log2_words: 18, derive_len: 0 });
|
||||
let dr = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 18, derive_len: DERIVE_LEN_X8, state: false });
|
||||
let x8 = MixParams::with_shape(key, Shape { mixer_mult: 8, cache_log2_words: 18, derive_len: 0, state: false });
|
||||
for (label, mp) in [("dr736", &dr), ("x8", &x8)] {
|
||||
let n = 2048u32;
|
||||
let mut ones = [0u32; 512];
|
||||
|
|
@ -265,7 +265,7 @@ fn text_forms_match_scalar_reference() {
|
|||
/// The dataset source of the class on a day: the verifier's `word` path derives through the program.
|
||||
#[test]
|
||||
fn dataset_source_word_path() {
|
||||
let ds = DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, 20, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8 });
|
||||
let ds = DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, 20, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8, state: false });
|
||||
let m = ds.memhard().unwrap();
|
||||
let item = derive_item(3, &m.params, &m.cache);
|
||||
for j in 0..16u32 {
|
||||
|
|
|
|||
|
|
@ -275,7 +275,7 @@ fn edge_items_every_multiplier() {
|
|||
let key = day_key(DAY);
|
||||
let cache = Cache::fill_log2(key, 14);
|
||||
for m in [1u32, 2, 4, 8] {
|
||||
let mp = MixParams::with_shape(key, Shape { mixer_mult: m, cache_log2_words: 14, derive_len: 0 });
|
||||
let mp = MixParams::with_shape(key, Shape { mixer_mult: m, cache_log2_words: 14, derive_len: 0, state: false });
|
||||
let by_hand = |t: u32| -> [u32; 16] {
|
||||
let mut s = [0u32; 16];
|
||||
s[..8].copy_from_slice(&key);
|
||||
|
|
|
|||
|
|
@ -369,7 +369,7 @@ fn v3_packs_are_the_v2_seeds_under_mixer_x8() {
|
|||
assert_eq!(e3.program.program_id(), igneum_pow::generator::program_id(GENERATOR_VERSION_V3, &e3.program.seed, e3.program.attempt));
|
||||
let m3 = e3.dataset.memhard().unwrap();
|
||||
let m2 = e2.dataset.memhard().unwrap();
|
||||
assert_eq!(m3.shape(), Shape { mixer_mult: 8, cache_log2_words: 26, derive_len: 0 });
|
||||
assert_eq!(m3.shape(), Shape { mixer_mult: 8, cache_log2_words: 26, derive_len: 0, state: false });
|
||||
assert_eq!(m3.cache.fnv1a64(), m2.cache.fnv1a64(), "{v3}: the same cache as v2 on day 0");
|
||||
assert_eq!(m3.params.rot, m2.params.rot);
|
||||
assert_eq!(e3.dataset.log2_words, 28);
|
||||
|
|
@ -405,7 +405,7 @@ fn v3_packs_are_the_v2_seeds_under_mixer_x8() {
|
|||
assert_eq!(j["load_class"].as_str().unwrap(), "mx4");
|
||||
assert_eq!(e4.program.class, LoadClass::MX4);
|
||||
assert_eq!(e4.program.instrs, epoch(v2).program.instrs);
|
||||
assert_eq!(e4.dataset.memhard().unwrap().shape(), Shape { mixer_mult: 4, cache_log2_words: 26, derive_len: 0 });
|
||||
assert_eq!(e4.dataset.memhard().unwrap().shape(), Shape { mixer_mult: 4, cache_log2_words: 26, derive_len: 0, state: false });
|
||||
assert!(read(x4, "memhard.h").contains("j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 4u + j + 1u))"));
|
||||
}
|
||||
}
|
||||
|
|
@ -899,3 +899,101 @@ fn hot_packs_emitted_sources_and_load_forms() {
|
|||
assert!(ph.contains("igneum_launch_hot_fill("));
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------------------------------------------
|
||||
// Class v5 (docs/design/class-v5-stored-state.md, 7 October 2026): the pinned pack under proto-cuda/packs-ca3-v5/
|
||||
// ---------------------------------------------------------------------------------------------------------------
|
||||
|
||||
fn v5_packs_dir() -> PathBuf {
|
||||
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca3-v5")
|
||||
}
|
||||
|
||||
fn v5_read(pack: &str, file: &str) -> String {
|
||||
std::fs::read_to_string(v5_packs_dir().join(pack).join(file)).unwrap_or_else(|e| panic!("{pack}/{file}: {e}"))
|
||||
}
|
||||
|
||||
fn v5_json(pack: &str, file: &str) -> Value {
|
||||
serde_json::from_str(&v5_read(pack, file)).unwrap()
|
||||
}
|
||||
|
||||
/// The epoch of a pinned class v4 or v5 pack of the string seed and day: the program through the seam, the dataset at
|
||||
/// the day-0 size, and for a v5 pack the leaves of its `state.igsd1` (the devnet's state stream of 7 October 2026,
|
||||
/// node 1's exec snapshot at chain block 159,357: 93 records, root 0x1c583d35...).
|
||||
fn v5_epoch(pack: &str) -> Epoch {
|
||||
let j = v5_json(pack, "program.json");
|
||||
let seed = j["seed"].as_str().unwrap();
|
||||
let class = ProgramClass::parse(j["program_class"].as_str().unwrap()).unwrap();
|
||||
let program = generate_from_seed_bytes_program_class(seed, seed.as_bytes(), class, None);
|
||||
let day = j["dataset"]["day"].as_str().unwrap();
|
||||
let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32;
|
||||
let shape = Shape::for_class(&program.class);
|
||||
let mut dataset = DatasetSource::new_shape(day, DatasetMode::MemoryHard, log2, shape);
|
||||
if class == ProgramClass::V5 {
|
||||
let stream = igneum_pow::StateStream::read_file(&v5_packs_dir().join(pack).join("state.igsd1")).unwrap();
|
||||
dataset = dataset.with_leaves(std::sync::Arc::new(igneum_pow::StateLeaves::from_stream(&stream, log2)));
|
||||
}
|
||||
Epoch { program, dataset }
|
||||
}
|
||||
|
||||
/// The class v5 pack is the class v4 program of the same seed over the state leaves: generator 5 and
|
||||
/// `program_id(5, seed, attempt)`, the base program, the shadow block and the dataset's cache equal to the v4 pack's,
|
||||
/// every vector and dataset word different, every emitted file byte for byte what the crate exports (leaves.bin
|
||||
/// included, its FNV in program.h), and the hash kernel text of kernel.cu equal to the v4 pack's but for the build
|
||||
/// kernel and the class lines. The known-failed case first: the v4 control pack's vectors under the v5 epoch agree on
|
||||
/// no lane.
|
||||
#[test]
|
||||
fn v5_pack_is_the_v4_program_over_the_state_leaves() {
|
||||
let e5 = v5_epoch("v5-genesis");
|
||||
let e4 = v5_epoch("v4-genesis");
|
||||
let v4 = v5_json("v4-genesis", "vectors.json");
|
||||
// the known-failed case: the v4 pack's hashes are not the v5 epoch's on any lane
|
||||
let out4: Vec<u64> = v4["warps"][0]["expected"].as_array().unwrap().iter().map(hex64).collect();
|
||||
let got5 = e5.hash_warp(0);
|
||||
assert_eq!(out4.iter().zip(got5.iter()).filter(|(a, b)| a == b).count(), 0, "0 of 32 lanes of the v4 pack agree with the v5 epoch");
|
||||
assert_eq!(e4.hash_warp(0).to_vec(), out4, "the v4 control pack is the v4 epoch");
|
||||
// the program: v4's draw, generator 5, the state in the class and the id
|
||||
assert_eq!(e5.program.generator, igneum_pow::GENERATOR_VERSION_V5);
|
||||
assert_eq!(e5.program.class, igneum_pow::V5_CLASS);
|
||||
assert_eq!(e5.program.class.name(), "mx8+sh256x27+state");
|
||||
assert_eq!(e5.program.instrs, e4.program.instrs);
|
||||
assert_eq!(e5.program.shadow, e4.program.shadow);
|
||||
assert_eq!((e5.program.seed, e5.program.attempt), (e4.program.seed, e4.program.attempt));
|
||||
assert_eq!(e5.program.program_id(), igneum_pow::generator::program_id(igneum_pow::GENERATOR_VERSION_V5, &e5.program.seed, e5.program.attempt));
|
||||
assert_ne!(e5.program.program_id(), e4.program.program_id());
|
||||
// the dataset: the same cache, other items
|
||||
let m5 = e5.dataset.memhard().unwrap();
|
||||
let m4 = e4.dataset.memhard().unwrap();
|
||||
assert_eq!(m5.cache.fnv1a64(), m4.cache.fnv1a64(), "one day cache");
|
||||
assert!(m5.shape().state && !m4.shape().state);
|
||||
let leaves = e5.dataset.leaves().unwrap();
|
||||
assert_eq!((leaves.n(), leaves.records_total, leaves.sampled), (93, 93, false));
|
||||
assert_ne!(e5.dataset_word(0), e4.dataset_word(0));
|
||||
// every file as the crate exports it, leaves.bin included
|
||||
for pack in ["v4-genesis", "v5-genesis"] {
|
||||
let e = if pack == "v5-genesis" { &e5 } else { &e4 };
|
||||
let v = v5_json(pack, "vectors.json");
|
||||
let out = export_pack(e, v["day"].as_str().unwrap(), v["source"].as_str().unwrap());
|
||||
for (name, text) in &out.files {
|
||||
assert_eq!(v5_read(pack, name), *text, "{pack}/{name} differs from the export");
|
||||
}
|
||||
for (name, bytes) in &out.binaries {
|
||||
assert_eq!(std::fs::read(v5_packs_dir().join(pack).join(name)).unwrap(), *bytes, "{pack}/{name}");
|
||||
}
|
||||
assert_eq!(out.binaries.len(), (pack == "v5-genesis") as usize);
|
||||
}
|
||||
let h5 = v5_read("v5-genesis", "program.h");
|
||||
assert!(h5.contains("#define IGNEUM_PROGRAM_CLASS \"v5\"") && h5.contains("#define IGNEUM_STATE_LEAVES 93") && h5.contains(&format!("#define IGNEUM_STATE_LEAVES_FNV64 {}", igneum_pow::emit::hex64(leaves.fnv1a64()))));
|
||||
assert!(h5.contains("igneum_launch_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems)"));
|
||||
// the hash kernel text is class v4's byte for byte; the build kernel and the class lines are what differ
|
||||
let hash_text = |s: &str| {
|
||||
let a = s.find("__global__ void igneum_hash(").unwrap();
|
||||
let b = s.find("// Host-side launch wrappers").unwrap();
|
||||
s[a..b].to_string()
|
||||
};
|
||||
let k5 = v5_read("v5-genesis", "kernel.cu");
|
||||
let k4 = v5_read("v4-genesis", "kernel.cu");
|
||||
assert_eq!(hash_text(&k5), hash_text(&k4), "the hash kernel is class v4's");
|
||||
assert!(k5.contains("mh_item(cache, mh_leaf(leaves, nLeaves, t), t, s)") && !k4.contains("mh_leaf"));
|
||||
let mh5 = v5_read("v5-genesis", "memhard.h");
|
||||
assert!(mh5.contains("s[i] ^= leaf[i]"), "the leaf XOR before the first mixer");
|
||||
}
|
||||
|
|
|
|||
387
infra/fast-time/class-v5-signal.mjs
Normal file
387
infra/fast-time/class-v5-signal.mjs
Normal file
|
|
@ -0,0 +1,387 @@
|
|||
#!/usr/bin/env node
|
||||
// Class v5, proof of stored state and of following (docs/design/class-v5-stored-state.md section 10): the class v4
|
||||
// signal gate's shape with class v5. Three nodes on override-60x.json with CPU genesis bits, v3 from DAA 60 (epoch 1),
|
||||
// the v4 FLOOR at --v4-floor (default 120, epoch 2), the class v5 floor at --floor (default 100000: far away, so v5 is
|
||||
// ENABLED and the signal decides; `never` turns the object off), the signal window at --window (default 60 DAA, one
|
||||
// epoch; seven windows = 420 DAA, so the first epoch whose seed block has seven full windows below it is epoch 8 at
|
||||
// DAA 480), each node's object byte by IGNEUM_CLASS_SIGNAL (--signal a,b,c; since the 0.3.20 node line the amended
|
||||
// class v4 is object byte 5 and class v5 is 6), each node's exec RPC on its own loopback
|
||||
// port and each CPU miner's --exec-rpc pointing at its node (the state stream after each epoch's seed block). A fourth
|
||||
// node runs with --evm-disable (--stateless-node, default on): it mines until the flip and then serves no class v5
|
||||
// template, which the harness counts. Ports 29760 and up, network igneum-devnet-976, data /tmp/igneum-fast-time-v5s.
|
||||
//
|
||||
// The cases and the known-failed cases:
|
||||
// --signal 6,6,7 --expect no-flip two of three signal v5: 67 percent, the class must stay v4 (run 11 epochs)
|
||||
// --signal 6,6,6 --expect flip all three: v5 from epoch 8, the first boundary with seven full windows; every
|
||||
// miner's v5 id equals the CLI's --program-class v5 id and differs from the same
|
||||
// seed's v4 id; the stateless node's miner accepts 0 blocks after the flip; 0 rejected
|
||||
// --signal 6,6,6 --floor never --expect no-flip the object off: byte 6 counts as v4 only
|
||||
// --signal 7,7,7 --floor 600 --expect floor nobody signals v5: it flips at the floor (epoch 10) and not before
|
||||
// --signal 6,6,6 --expect flip --stale 2 proof of following's known-failed case: miner 2 keeps the first stream it
|
||||
// fetched (--freeze-state); from the epoch after the flip every block it mines is
|
||||
// rejected (accepted after the first refresh 0, rejected above 0); PASS means the
|
||||
// chain threw the stale miner off
|
||||
// --signal 6,6,7 --expect flip the known-failed case of the harness itself: it must report FAIL (no flip)
|
||||
//
|
||||
// node infra/fast-time/class-v5-signal.mjs --signal a,b,c --expect flip|no-flip|floor [--floor <daa>|never]
|
||||
// [--v4-floor 120] [--window 60] [--v3-activation 60] [--secs 780] [--epochs 11] [--stale <miner index>] [--no-stateless-node]
|
||||
// IGNEUMD, IGNEUM_MINER, IGNEUM_POW name the binaries (defaults: the class-v5 fork worktree's target/release and
|
||||
// igneum-pow/target/release/igneum-pow, the layout on igneum-build-1 under /srv/builds/igneum-wt-class-v5).
|
||||
|
||||
import { spawn, spawnSync } from 'node:child_process';
|
||||
import { mkdirSync, rmSync, writeFileSync, readFileSync, openSync, existsSync } from 'node:fs';
|
||||
import { connectRpc } from '../../tools/finality-attacks/lib/rpc.mjs';
|
||||
import { devAddress } from '../../tools/harness/lib/address.mjs';
|
||||
|
||||
const ROOT = new URL('../../', import.meta.url).pathname;
|
||||
const FILE = `${ROOT}infra/fast-time/override-60x.json`;
|
||||
const BIN = process.env.IGNEUM_V5_BIN || `${ROOT}vendor/igneum-node-class-v5/target/release`;
|
||||
const IGNEUMD = process.env.IGNEUMD || `${BIN}/igneumd`;
|
||||
const CPU_MINER = process.env.IGNEUM_MINER || `${BIN}/igneum-miner`;
|
||||
const IGNEUM_POW = process.env.IGNEUM_POW || `${ROOT}igneum-pow/target/release/igneum-pow`;
|
||||
const TMP = process.env.IGNEUM_V5_TMP || '/tmp/igneum-fast-time-v5s';
|
||||
const BASE = +(process.env.IGNEUM_V5_BASE || 29760), SUFFIX = 976;
|
||||
const NEVER = '18446744073709551615';
|
||||
const args = process.argv.slice(2);
|
||||
const flag = (name, dflt) => { const i = args.indexOf(`--${name}`); return i >= 0 ? +args[i + 1] : dflt; };
|
||||
const sflag = (name) => { const i = args.indexOf(`--${name}`); return i >= 0 ? args[i + 1] : null; };
|
||||
const actflag = (name, dflt) => { const v = sflag(name); if (v == null) return dflt; return v === 'never' ? null : +v; };
|
||||
const GENESIS_BITS = flag('genesis-bits', 0x1f010000);
|
||||
const SECS = flag('secs', 780);
|
||||
const EPOCHS = flag('epochs', 11);
|
||||
const FLOOR = actflag('floor', 100000);
|
||||
const V4_FLOOR = actflag('v4-floor', 120);
|
||||
const V3_ACTIVATION = actflag('v3-activation', 60);
|
||||
const WINDOW = flag('window', 60);
|
||||
const STALE = sflag('stale') == null ? null : +sflag('stale');
|
||||
const STATELESS_NODE = !args.includes('--no-stateless-node');
|
||||
const WINDOWS = 7;
|
||||
const SIGNAL = (sflag('signal') || '6,6,6').split(',').map(Number);
|
||||
// since the 0.3.23 node line the bytes are not ordered by class: 7 = class v4 sub-version 3, 6 = class v5 (counted exactly)
|
||||
const EXPECT = sflag('expect') || 'flip';
|
||||
if (!['flip', 'no-flip', 'floor'].includes(EXPECT) || SIGNAL.length !== 3) { console.error('usage: --signal a,b,c --expect flip|no-flip|floor'); process.exit(2); }
|
||||
const started = [];
|
||||
const log = (...a) => console.log(new Date().toISOString().slice(11, 23), ...a);
|
||||
const sleep = (ms) => new Promise(r => setTimeout(r, ms));
|
||||
for (const b of [IGNEUMD, CPU_MINER]) if (!existsSync(b)) { console.error(`missing ${b}`); process.exit(2); }
|
||||
|
||||
// the previous run's survivors (the flip-stale run of 7 October 2026, 08:49Z: two nodes of the earlier case were still bound
|
||||
// to the ports and the harness read a stranger's chain): the pid file of this data dir is killed first, by pid, and a node
|
||||
// whose log says a port was in use fails the run at once
|
||||
const PIDS = `${TMP}/pids`;
|
||||
try { for (const pid of readFileSync(PIDS, 'utf8').split('\n').map(Number).filter(Boolean)) { try { process.kill(pid, 'SIGKILL'); console.log(`killed survivor pid ${pid} of the previous run`); } catch { } } } catch { }
|
||||
rmSync(TMP, { recursive: true, force: true }); mkdirSync(TMP, { recursive: true });
|
||||
const recordPid = (p) => { try { writeFileSync(PIDS, `${readFileSync(PIDS, 'utf8')}${p.pid}\n`); } catch { writeFileSync(PIDS, `${p.pid}\n`); } };
|
||||
const baseText = readFileSync(FILE, 'utf8');
|
||||
const field = (name) => { const m = new RegExp(`"${name}":\\s*([0-9]+)`).exec(baseText); return m ? +m[1] : undefined; };
|
||||
const EPOCH = field('pow_epoch_blocks');
|
||||
const LEAD = field('pow_epoch_lead');
|
||||
const DAY_MS = field('pow_day_ms');
|
||||
const FIRST_V3_EPOCH = V3_ACTIVATION == null ? null : Math.ceil(V3_ACTIVATION / EPOCH);
|
||||
const V4_FLOOR_EPOCH = V4_FLOOR == null ? null : Math.ceil(V4_FLOOR / EPOCH);
|
||||
const FLOOR_EPOCH = FLOOR == null ? null : Math.ceil(FLOOR / EPOCH);
|
||||
// the first epoch whose seed block (the last chain block below L*e - lead) can have DAA >= 7 x WINDOW: L*e - lead - 1 >= 7W,
|
||||
// and whose previous epoch is already v4 (one step per epoch)
|
||||
let FIRST_FULL_EPOCH = 0;
|
||||
while (FIRST_FULL_EPOCH * EPOCH - LEAD - 1 < WINDOWS * WINDOW) FIRST_FULL_EPOCH++;
|
||||
if (V4_FLOOR_EPOCH != null && FIRST_FULL_EPOCH <= V4_FLOOR_EPOCH) FIRST_FULL_EPOCH = V4_FLOOR_EPOCH + 1;
|
||||
export function mergeOverrideText(text, fields) {
|
||||
let out = text;
|
||||
for (const k of Object.keys(fields)) out = out.replace(new RegExp(`\\s*"${k}":\\s*[^,}\\n]+,?`), '');
|
||||
const extra = Object.entries(fields).map(([k, v]) => `"${k}": ${typeof v === 'string' && !/^\d+$/.test(v) ? JSON.stringify(v) : v}`).join(', ');
|
||||
return out.replace(/,?\s*}\s*$/, `,\n ${extra}\n}\n`);
|
||||
}
|
||||
const asText = (v) => v == null ? NEVER : String(v);
|
||||
const override = `${TMP}/override.json`;
|
||||
writeFileSync(override, mergeOverrideText(baseText, { genesis_bits: GENESIS_BITS, skip_proof_of_work: false, program_class_v3_activation_daa: asText(V3_ACTIVATION), program_class_v4_activation_daa: asText(V4_FLOOR), program_class_v4_signal_window_daa: String(WINDOW), program_class_v5_activation_daa: asText(FLOOR) }));
|
||||
log(`signals ${SIGNAL.join('/')}, expect ${EXPECT}; v3 from ${V3_ACTIVATION ?? 'never'} (epoch ${FIRST_V3_EPOCH ?? 'none'}), v4 floor ${V4_FLOOR ?? 'never'} (epoch ${V4_FLOOR_EPOCH ?? 'none'}), v5 floor ${FLOOR ?? 'never'} (epoch ${FLOOR_EPOCH ?? 'none'}), window ${WINDOW} DAA x ${WINDOWS} (the first epoch with seven full windows after v4 is ${FIRST_FULL_EPOCH}); stale miner ${STALE ?? 'none'}; stateless node ${STATELESS_NODE ? 'n3' : 'none'}; ${EPOCH} DAA per epoch, lead ${LEAD}; run ${SECS} s or ${EPOCHS} epochs`);
|
||||
|
||||
class Node {
|
||||
constructor(i, connect = [], stateless = false) {
|
||||
this.i = i; this.grpcPort = BASE + i * 10; this.p2pPort = BASE + i * 10 + 1; this.jsonPort = BASE + i * 10 + 2; this.execPort = BASE + i * 10 + 3;
|
||||
this.connect = connect; this.dir = `${TMP}/n${i}`; this.logFile = `${this.dir}/node.log`; this.stateless = stateless;
|
||||
}
|
||||
get grpc() { return `grpc://127.0.0.1:${this.grpcPort}`; }
|
||||
get execRpc() { return `http://127.0.0.1:${this.execPort}`; }
|
||||
async start() {
|
||||
mkdirSync(this.dir, { recursive: true });
|
||||
const a = ['--devnet', `--devnet-suffix=${SUFFIX}`, '--nodnsseed', '--disable-upnp', '--nologfiles', '--enable-unsynced-mining', '--utxoindex',
|
||||
`--appdir=${this.dir}`, `--rpclisten=127.0.0.1:${this.grpcPort}`, `--rpclisten-json=127.0.0.1:${this.jsonPort}`,
|
||||
`--listen=127.0.0.1:${this.p2pPort}`, `--override-params-file=${override}`, '--loglevel=info', '--yes'];
|
||||
// class v5: each node's executor serves the state stream on its own loopback port; the stateless node has none
|
||||
if (this.stateless) a.push('--evm-disable'); else a.push(`--evm-rpclisten=127.0.0.1:${this.execPort}`);
|
||||
if (this.connect.length) a.push(`--connect=${this.connect.join(',')}`); else a.push('--outpeers=0');
|
||||
const out = openSync(this.logFile, 'a');
|
||||
// the node's own object byte: what its templates signal (the stateless node signals the highest byte too)
|
||||
const byte = this.i < SIGNAL.length ? SIGNAL[this.i] : Math.max(...SIGNAL);
|
||||
this.proc = spawn(IGNEUMD, a, { stdio: ['ignore', out, out], env: { ...process.env, IGNEUM_CLASS_SIGNAL: String(byte) } });
|
||||
started.push(this.proc);
|
||||
recordPid(this.proc);
|
||||
await sleep(1200);
|
||||
if (this.grepLog(/Address already in use|AddrInUse/).length) { log(`FAILED: n${this.i} could not bind its ports (a previous run's node is still alive): ${this.grepLog(/AddrInUse|Address already in use/)[0].slice(0, 160)}`); await stopAll(); process.exit(4); }
|
||||
this.rpc = await connectRpc(`ws://127.0.0.1:${this.jsonPort}`);
|
||||
log(`n${this.i} up pid ${this.proc.pid} json ${this.jsonPort} p2p ${this.p2pPort} exec ${this.stateless ? 'disabled' : this.execPort}, signals ${byte}`);
|
||||
return this;
|
||||
}
|
||||
grepLog(re) { try { return readFileSync(this.logFile, 'utf8').split('\n').filter(l => re.test(l)); } catch { return []; } }
|
||||
}
|
||||
function miner(bin, argv, name, env = {}) {
|
||||
const out = openSync(`${TMP}/${name}.log`, 'a');
|
||||
const p = spawn(bin, argv, { stdio: ['ignore', out, out], env: { ...process.env, ...env } });
|
||||
started.push(p);
|
||||
recordPid(p);
|
||||
return p;
|
||||
}
|
||||
async function stopAll() {
|
||||
for (const p of started.reverse()) { try { p.kill('SIGINT'); } catch { } }
|
||||
await sleep(1500);
|
||||
for (const p of started) { try { p.kill('SIGKILL'); } catch { } }
|
||||
}
|
||||
process.on('SIGINT', async () => { await stopAll(); process.exit(130); });
|
||||
process.on('unhandledRejection', async (e) => { log(`FAILED: ${e?.stack || e}`); await stopAll(); process.exit(3); });
|
||||
const minerLog = (i) => { try { return readFileSync(`${TMP}/cpu${i}.log`, 'utf8').split('\n'); } catch { return []; } };
|
||||
const SIGNAL_LINE = /Program class v5 by miner signal: epoch (\d+) \(share (\d+) bps/;
|
||||
const FLOOR_LINE = /Program class v5 from the override file/;
|
||||
const REFUSAL_LINE = /class v5 needs the execution state/;
|
||||
const WINDOW_LINE = /Program class v4 signal window from the override file/;
|
||||
const OWN_LINE = /Program class signal from IGNEUM_CLASS_SIGNAL: this node signals object version (\d+)/;
|
||||
|
||||
const t0 = Date.now();
|
||||
const since = () => ((Date.now() - t0) / 1000).toFixed(1);
|
||||
const n0 = await new Node(0).start();
|
||||
const n1 = await new Node(1, [`127.0.0.1:${n0.p2pPort}`]).start();
|
||||
const n2 = await new Node(2, [`127.0.0.1:${n0.p2pPort}`]).start();
|
||||
const nodes = [n0, n1, n2];
|
||||
if (STATELESS_NODE) nodes.push(await new Node(3, [`127.0.0.1:${n0.p2pPort}`], true).start());
|
||||
const SIGNALLING = [n0, n1, n2];
|
||||
for (const n of nodes) log(`n${n.i}: ${n.grepLog(WINDOW_LINE).map(l => l.replace(/^.*?(Program class v4 signal window)/, '$1'))[0] || '(no window line)'} | ${n.grepLog(OWN_LINE).map(l => l.replace(/^.*?(this node signals)/, '$1'))[0] || '(no signal line)'}`);
|
||||
log(`n0 digest: ${n0.grepLog(/Consensus params digest/).map(l => l.replace(/^.*?digest: /, '').slice(0, 16)).join(' ')}`);
|
||||
nodes.forEach((n, i) => {
|
||||
// the stateless node's miner is pointed at its own node's (absent) exec port, so its refusal is its node's and not a stranger's
|
||||
const extra = ['--exec-rpc', n.execRpc];
|
||||
if (STALE === i) extra.push('--freeze-state');
|
||||
miner(CPU_MINER, ['mine', n.grpc, '1', String(SECS), `cpu${i}`, '--engine', 'igneum-pow', '--payout-label', `cpu${i}`, '--status-secs', '30', '--no-vote', ...extra], `cpu${i}`, { IGNEUM_POW_DAY_MS: String(DAY_MS) });
|
||||
});
|
||||
const pay = devAddress('fast-time-v5s');
|
||||
const minerAcceptedAt = (i) => minerLog(i).filter(l => /ACCEPTED block/.test(l)).map(l => { const m = /^(\d+\.\d+) /.exec(l); return m ? +m[1] : null; }).filter(t => t != null);
|
||||
|
||||
const epochs = new Map();
|
||||
let firstV4 = null, lastEpoch = -1, lastReport = 0, lastDaa = 0, endAt = null, flipWall = null;
|
||||
const epochSeeds = new Map();
|
||||
const dayOfEpoch = new Map();
|
||||
const samples = [];
|
||||
while (Date.now() - t0 < SECS * 1000) {
|
||||
await sleep(1000);
|
||||
let daa = null, epoch = null, cls = null, nextCls = null, eraSeed = null, bps = null, bps5 = null, win = null, sig = null, sigEpoch = null, seed = null, day = null;
|
||||
try {
|
||||
const t = await n0.rpc.call('getBlockTemplate', { payAddress: pay, extraData: [] });
|
||||
const pe = t.powEpoch || t.pow_epoch || {};
|
||||
daa = pe.virtualDaaScore ?? t.block?.header?.daaScore; epoch = pe.epochIndex; cls = pe.programClass; nextCls = pe.nextProgramClass;
|
||||
eraSeed = pe.eraSeed; bps = pe.programClassV4SignalBps; bps5 = pe.programClassV5SignalBps; win = pe.programClassV4SignalWindowDaa; sig = pe.programClassSignal; sigEpoch = pe.programClassV5SignalEpoch;
|
||||
seed = pe.epochSeed; day = Math.floor((+t.block?.header?.timestamp || 0) / DAY_MS);
|
||||
} catch (e) { log(`template: ${e.message}`); }
|
||||
if (epoch != null && epoch !== lastEpoch) {
|
||||
epochs.set(epoch, { class: cls, firstSeenDaa: daa, at: +since(), eraSeed: eraSeed == null ? null : String(eraSeed), bps, bps5, signal_epoch: sigEpoch ?? null, day });
|
||||
if (seed != null) epochSeeds.set(epoch, String(seed));
|
||||
dayOfEpoch.set(epoch, day);
|
||||
log(`epoch ${lastEpoch} -> ${epoch} at daa ${daa}, ${since()} s: template class ${cls}, next ${nextCls}, day ${day}, signal share at the sink v4 ${bps} v5 ${bps5} bps (window ${win}, this node signals ${sig}, v5 decided by signal at epoch ${sigEpoch ?? 'none'})`);
|
||||
if (firstV4 == null && cls === 5) { firstV4 = { epoch, daa, at: +since() }; flipWall = Date.now(); log(`CLASS SWITCH: the template is class v5 from epoch ${epoch} (daa ${daa}) at ${since()} s wall`); }
|
||||
lastEpoch = epoch;
|
||||
}
|
||||
lastDaa = daa ?? lastDaa;
|
||||
if (Date.now() - lastReport > 15000) {
|
||||
lastReport = Date.now();
|
||||
const counts = await Promise.all(nodes.map(async n => { try { const d = await n.rpc.call('getBlockDagInfo'); return `${d.blockCount}/${String(d.sink).slice(0, 8)}`; } catch { return '?'; } }));
|
||||
log(`t=${since()} s daa ${daa} epoch ${epoch} class ${cls} signal v4 ${bps} v5 ${bps5} bps blocks/sink per node ${counts.join(' ')}`);
|
||||
samples.push({ t: +since(), daa, epoch, class: cls, bps, bps5, nodes: counts });
|
||||
}
|
||||
// the end: three epochs after a flip (two refreshes for the stale miner), or --epochs epochs when no flip is expected
|
||||
if (firstV4 != null && daa != null && daa >= (firstV4.epoch + 3) * EPOCH) { endAt = +since(); break; }
|
||||
if (firstV4 == null && daa != null && daa >= EPOCHS * EPOCH) { endAt = +since(); break; }
|
||||
}
|
||||
await sleep(3000);
|
||||
|
||||
const dag = await Promise.all(nodes.map(async n => { try { return await n.rpc.call('getBlockDagInfo'); } catch (e) { return { error: e.message }; } }));
|
||||
const genesis = dag[0].pruningPointHash;
|
||||
async function allBlocks(n) {
|
||||
const out = []; let low = genesis; const seen = new Set();
|
||||
for (let round = 0; round < 500; round++) {
|
||||
const r = await n.rpc.call('getBlocks', { lowHash: low, includeBlocks: true, includeTransactions: false });
|
||||
const blocks = r.blocks || [];
|
||||
let added = 0;
|
||||
for (const b of blocks) { const h = b.verboseData?.hash || b.header?.hash; if (seen.has(h)) continue; seen.add(h); out.push({ hash: h, daa: +b.header.daaScore, version: +b.header.version, chain: !!b.verboseData?.isChainBlock }); added++; }
|
||||
if (!blocks.length || added === 0) break;
|
||||
low = (r.blockHashes || []).at(-1) || blocks.at(-1).verboseData?.hash; if (!low) break;
|
||||
}
|
||||
return out;
|
||||
}
|
||||
let blocks = [];
|
||||
try { blocks = await allBlocks(n0); } catch (e) { log(`getBlocks: ${e.message}`); }
|
||||
const BOUNDARY = firstV4 ? firstV4.epoch * EPOCH : Infinity;
|
||||
const before = blocks.filter(b => b.daa < BOUNDARY), after = blocks.filter(b => b.daa >= BOUNDARY);
|
||||
// the signal bytes on the chain: the share of blocks whose version high byte is 4
|
||||
const versionBytes = blocks.reduce((m, b) => { const v = b.version >> 8; m[v] = (m[v] || 0) + 1; return m; }, {});
|
||||
const V5_BYTE = 6;
|
||||
const signalShareOnChain = blocks.length ? Math.round(10000 * (blocks.filter(b => (b.version >> 8) === V5_BYTE).length) / blocks.length) : 0;
|
||||
|
||||
const programs = new Map();
|
||||
for (const i of nodes.map(n => n.i)) for (const l of minerLog(i)) {
|
||||
const m = /epoch seed ([0-9a-f]{64}) day (\d+) \(daa (\d+)\): program and 256 MiB cache ready in ([\d.]+) ms; class (v\d) program id ([0-9a-f]{16})/.exec(l);
|
||||
if (!m) continue;
|
||||
const k = m[1]; const e = programs.get(k) || { seed: k.slice(0, 16), epoch: Math.floor(+m[3] / EPOCH), class: m[5], id: m[6], miners: new Set() };
|
||||
if (e.id !== m[6] || e.class !== m[5]) e.disagree = true;
|
||||
e.miners.add(i); programs.set(k, e);
|
||||
}
|
||||
const programRows = [...programs.values()].sort((a, b) => a.epoch - b.epoch).map(p => ({ epoch: p.epoch, class: p.class, program_id: p.id, seed: p.seed, miners: p.miners.size, disagree: !!p.disagree }));
|
||||
// the state stream after an epoch's seed block, from node 0's exec RPC, for the CLI's --state
|
||||
async function streamFile(epoch) {
|
||||
const seed = epochSeeds.get(epoch);
|
||||
if (!seed) return null;
|
||||
try {
|
||||
const r = await fetch(n0.execRpc, { method: 'POST', headers: { 'content-type': 'application/json' }, body: JSON.stringify({ jsonrpc: '2.0', id: 1, method: 'igneum_getPowStateLeaves', params: [seed] }) });
|
||||
const j = await r.json();
|
||||
if (!j.result?.streamHex) { log(`stream for epoch ${epoch}: ${JSON.stringify(j.error || j).slice(0, 200)}`); return null; }
|
||||
const path = `${TMP}/state-e${epoch}.bin`;
|
||||
writeFileSync(path, Buffer.from(j.result.streamHex.slice(2), 'hex'));
|
||||
return { path, root: j.result.stateRoot, records: j.result.records, block: j.result.block };
|
||||
} catch (e) { log(`stream for epoch ${epoch}: ${e.message}`); return null; }
|
||||
}
|
||||
function cliId(seedHex, eraHex, cls, statePath) {
|
||||
if (!existsSync(IGNEUM_POW)) return null;
|
||||
const extra = statePath ? ['--state', statePath] : [];
|
||||
const r = spawnSync(IGNEUM_POW, ['show', '--epoch-hex', seedHex, '--program-class', cls, '--era-hex', eraHex, ...extra], { encoding: 'utf8' });
|
||||
const m = /program id ([0-9a-f]{16})/.exec(r.stdout || '');
|
||||
return m ? m[1] : null;
|
||||
}
|
||||
const idRows = [];
|
||||
const streams = {};
|
||||
for (const [k, e] of programs) {
|
||||
if (e.class !== 'v5') continue;
|
||||
const era = epochs.get(e.epoch)?.eraSeed;
|
||||
const st = await streamFile(e.epoch);
|
||||
if (st) streams[e.epoch] = { root: st.root, records: st.records, block: st.block };
|
||||
idRows.push({ epoch: e.epoch, seed: e.seed, miners_id: e.id, miners: e.miners.size, cli_v4: era ? cliId(k, era, 'v4') : null, cli_v5: era && st ? cliId(k, era, 'v5', st.path) : null, state_root: st?.root ?? null, state_records: st?.records ?? null });
|
||||
}
|
||||
// every executing node's stream for the LAST v5 epoch's seed block must carry the same root (the serialisation agrees
|
||||
// across nodes); the executor keeps the current epoch's capture and the two before it, so the flip epoch's stream is
|
||||
// gone by the end of a run that went three epochs past it (the 10:55Z run read -32xxx errors for epoch 8 and streams
|
||||
// for 9, 10 and 11)
|
||||
const rootsAtFlip = [];
|
||||
const lastV5Epoch = firstV4 ? Math.max(firstV4.epoch, ...[...epochs.entries()].filter(([, v]) => v.class === 5).map(([e]) => +e)) : null;
|
||||
if (firstV4) for (const n of nodes) {
|
||||
if (n.stateless) continue;
|
||||
try {
|
||||
const r = await fetch(n.execRpc, { method: 'POST', headers: { 'content-type': 'application/json' }, body: JSON.stringify({ jsonrpc: '2.0', id: 1, method: 'igneum_getPowStateLeaves', params: [epochSeeds.get(lastV5Epoch)] }) });
|
||||
const j = await r.json();
|
||||
rootsAtFlip.push(j.result?.stateRoot ?? `error: ${JSON.stringify(j.error || j).slice(0, 120)}`);
|
||||
} catch (e) { rootsAtFlip.push(`error: ${e.message}`); }
|
||||
}
|
||||
const minerIdx = nodes.map(n => n.i);
|
||||
const accepted = minerIdx.map(i => minerLog(i).filter(l => /ACCEPTED block/.test(l)).length);
|
||||
const rejectedMiner = minerIdx.map(i => minerLog(i).filter(l => /rejected nonce=|submit error/.test(l)));
|
||||
const rejectedNode = SIGNALLING.map(n => n.grepLog(/PoW rejected|Rejected block|rejected block/i));
|
||||
const signalLines = SIGNALLING.map(n => n.grepLog(SIGNAL_LINE).map(l => l.replace(/^.*?(Program class v5 by miner signal)/, '$1'))[0] || null);
|
||||
const signalEpochs = signalLines.map(l => { const m = l && SIGNAL_LINE.exec(l); return m ? +m[1] : null; });
|
||||
const signalShares = signalLines.map(l => { const m = l && SIGNAL_LINE.exec(l); return m ? +m[2] : null; });
|
||||
const floorLines = SIGNALLING.map(n => n.grepLog(FLOOR_LINE).map(l => l.replace(/^.*?(Program class v5 from)/, '$1'))[0] || null);
|
||||
// the stateless node: its miner's blocks accepted after the flip (must be 0) and its refusal lines
|
||||
const statelessNode = nodes.find(n => n.stateless) || null;
|
||||
const flipAtMs = flipWall;
|
||||
// accepted-after-the-flip counts read the node-side acceptance instead: each miner's ACCEPTED lines after the flip's
|
||||
// wall time, by the line's own clock (the miner prints `HH:MM:SS` UTC at the start of every line)
|
||||
function acceptedAfter(i, wallMs) {
|
||||
if (wallMs == null) return 0;
|
||||
const since0 = new Date(wallMs).toISOString().slice(11, 19);
|
||||
let count = 0;
|
||||
for (const l of minerLog(i)) { if (!/ACCEPTED block/.test(l)) continue; const m = /^(\d{2}:\d{2}:\d{2})/.exec(l); if (m && m[1] >= since0) count++; }
|
||||
return count;
|
||||
}
|
||||
const statelessAcceptedAfterFlip = statelessNode ? acceptedAfter(statelessNode.i, flipAtMs) : 0;
|
||||
// the refusal reaches the miner as a template error and the node's log as the engine's line; either counts
|
||||
const statelessRefusals = statelessNode ? statelessNode.grepLog(REFUSAL_LINE).length + minerLog(statelessNode.i).filter(l => REFUSAL_LINE.test(l)).length : 0;
|
||||
// the stale miner: accepted after the FIRST REFRESH after the flip (one epoch later), and its rejections
|
||||
const refreshWall = flipAtMs == null ? null : flipAtMs + EPOCH * 1000;
|
||||
const staleAcceptedAfterRefresh = STALE == null ? null : acceptedAfter(STALE, refreshWall);
|
||||
const staleRejected = STALE == null ? null : rejectedMiner[STALE].length;
|
||||
const daysSeen = [...new Set([...dayOfEpoch.values()])];
|
||||
const sinks = dag.map(d => String(d.sink || '?').slice(0, 16));
|
||||
const counts = dag.map(d => d.blockCount ?? '?');
|
||||
const maxEpochSeen = Math.max(-1, ...epochs.keys());
|
||||
const classesSeen = [...epochs.values()].map(e => e.class);
|
||||
|
||||
// the stale miner's rejections are the point of its case (its own node rejects them), and the stateless node must fall
|
||||
// behind at the flip: both are excluded here and asserted by their own checks below
|
||||
const signallingIdx = SIGNALLING.map(n => n.i);
|
||||
const common = {
|
||||
zero_rejected_by_miners: rejectedMiner.every((r, i) => i === STALE || r.length === 0),
|
||||
zero_rejected_by_nodes: rejectedNode.every((r, i) => i === STALE || r.length === 0),
|
||||
sinks_agree: new Set(signallingIdx.map(i => sinks[i])).size === 1,
|
||||
block_counts_agree: new Set(signallingIdx.map(i => String(counts[i]))).size === 1,
|
||||
miners_agree_on_every_program: programRows.every(p => !p.disagree),
|
||||
window_line_on_every_node: nodes.every(n => n.grepLog(WINDOW_LINE).length > 0),
|
||||
every_node_signals_its_byte: nodes.every((n, i) => n.grepLog(OWN_LINE).some(l => +OWN_LINE.exec(l)[1] === (SIGNAL[i] ?? Math.max(...SIGNAL)))),
|
||||
// every block's byte is one of the nodes' (genesis, made before any node, is the one byte-0 block)
|
||||
chain_carries_the_bytes: blocks.length > 0 && Object.keys(versionBytes).every(v => SIGNAL.includes(+v) || +v === Math.max(...SIGNAL) || (+v === 0 && versionBytes[v] === 1)),
|
||||
// class v5: every executing node served the same state root for the flip epoch's seed block
|
||||
state_roots_agree_across_nodes: !firstV4 || (rootsAtFlip.length === SIGNALLING.length && rootsAtFlip.every(r => r && !r.startsWith('error')) && new Set(rootsAtFlip).size === 1),
|
||||
};
|
||||
let checks;
|
||||
if (EXPECT === 'flip') {
|
||||
checks = {
|
||||
...common,
|
||||
template_switched_to_v5: firstV4 != null,
|
||||
switched_at_the_first_full_window_epoch: firstV4 != null && firstV4.epoch === FIRST_FULL_EPOCH,
|
||||
switched_before_the_floor: firstV4 != null && (FLOOR_EPOCH == null || firstV4.epoch < FLOOR_EPOCH),
|
||||
signal_line_on_every_node_same_epoch: signalEpochs.every(e => e != null) && new Set(signalEpochs).size === 1 && signalEpochs[0] === (firstV4 && firstV4.epoch),
|
||||
signal_share_at_or_above_threshold: signalShares.every(s => s != null && s >= 9500),
|
||||
blocks_on_both_sides: before.length > 0 && after.length > 0,
|
||||
// the rows whose stream the executor still held at the end (the last three epochs) must match the CLI; at least two of them
|
||||
v5_ids_equal_the_cli_v5_id: idRows.filter(r => r.state_root != null).length >= 2 && idRows.filter(r => r.state_root != null).every(r => r.cli_v5 != null && r.cli_v5 === r.miners_id && r.miners === 3),
|
||||
v5_ids_differ_from_the_same_seed_v4_id: idRows.length > 0 && idRows.every(r => r.cli_v4 != null && r.cli_v4 !== r.miners_id),
|
||||
// proof of following: every v5 epoch's leaves came from another state root (the refresh per window)
|
||||
a_v5_epoch_per_window_refresh: idRows.filter(r => r.state_root != null).length >= 2 && new Set(idRows.filter(r => r.state_root != null).map(r => r.state_root)).size === idRows.filter(r => r.state_root != null).length,
|
||||
stateless_node_mines_nothing_after_the_flip: !statelessNode || (statelessAcceptedAfterFlip === 0 && statelessRefusals > 0),
|
||||
stateless_node_stops_at_the_flip: !statelessNode || (counts[statelessNode.i] <= before.length && counts[statelessNode.i] >= before.length - 5),
|
||||
stale_miner_falls_off_at_the_first_refresh: STALE == null || (staleAcceptedAfterRefresh === 0 && staleRejected > 0),
|
||||
};
|
||||
} else if (EXPECT === 'no-flip') {
|
||||
checks = {
|
||||
...common,
|
||||
template_never_v5: firstV4 == null && !classesSeen.includes(5),
|
||||
no_signal_line_on_any_node: signalLines.every(l => l == null),
|
||||
ran_the_epochs: maxEpochSeen >= EPOCHS - 1,
|
||||
v4_programs_seen: programRows.some(p => p.class === 'v4'),
|
||||
signal_share_under_threshold_on_chain_or_the_object_off: signalShareOnChain < 9500 || FLOOR == null,
|
||||
};
|
||||
} else {
|
||||
checks = {
|
||||
...common,
|
||||
template_switched_to_v5: firstV4 != null,
|
||||
switched_at_the_floor_epoch: firstV4 != null && firstV4.epoch === FLOOR_EPOCH,
|
||||
no_signal_line_on_any_node: signalLines.every(l => l == null),
|
||||
floor_line_names_the_floor_epoch: floorLines.every(l => l && l.includes(`the floor at epoch ${FLOOR_EPOCH} `)),
|
||||
blocks_on_both_sides: before.length > 0 && after.length > 0,
|
||||
v5_ids_equal_the_cli_v5_id: idRows.filter(r => r.state_root != null).length >= 2 && idRows.filter(r => r.state_root != null).every(r => r.cli_v5 != null && r.cli_v5 === r.miners_id && r.miners === 3),
|
||||
stateless_node_mines_nothing_after_the_flip: !statelessNode || (statelessAcceptedAfterFlip === 0 && statelessRefusals > 0),
|
||||
stateless_node_stops_at_the_flip: !statelessNode || (counts[statelessNode.i] <= before.length && counts[statelessNode.i] >= before.length - 5),
|
||||
};
|
||||
}
|
||||
const pass = Object.values(checks).every(Boolean);
|
||||
const summary = {
|
||||
pass, expect: EXPECT, signals: SIGNAL, checks, window: WINDOW, windows: WINDOWS, floor: FLOOR ?? 'never', v4_floor: V4_FLOOR ?? 'never', v3_activation: V3_ACTIVATION ?? 'never', epoch_blocks: EPOCH, lead: LEAD, first_full_window_epoch: FIRST_FULL_EPOCH, floor_epoch: FLOOR_EPOCH,
|
||||
stale_miner: STALE, stale_accepted_after_first_refresh: staleAcceptedAfterRefresh, stale_rejected: staleRejected, stateless_node: statelessNode ? statelessNode.i : null, stateless_accepted_after_flip: statelessAcceptedAfterFlip, stateless_refusal_lines: statelessRefusals,
|
||||
state_roots_at_flip: rootsAtFlip, streams, days_seen: daysSeen, day_boundary_crossed: daysSeen.length > 1,
|
||||
node: IGNEUMD, miner: CPU_MINER, template_switch: firstV4, run_ended_at_s: endAt, final_daa: lastDaa, max_epoch_seen: maxEpochSeen,
|
||||
epochs: Object.fromEntries([...epochs.entries()].map(([k, v]) => [k, v])),
|
||||
blocks: { total: blocks.length, before_boundary: before.length, after_boundary: after.length, version_bytes: versionBytes, signal_share_bps_on_chain: signalShareOnChain },
|
||||
programs: programRows, program_id_rows: idRows, accepted_per_miner: accepted,
|
||||
rejected_by_miners: rejectedMiner.map(r => r.length), rejected_by_nodes: rejectedNode.map(r => r.length),
|
||||
sinks, block_counts: counts, signal_lines: signalLines, floor_lines: floorLines, samples,
|
||||
};
|
||||
writeFileSync(`${TMP}/summary.json`, JSON.stringify(summary, null, 2));
|
||||
log(`SUMMARY ${pass ? 'PASS' : 'FAIL'} (expect ${EXPECT}, signals ${SIGNAL.join('/')}, stale ${STALE ?? 'none'}): ${firstV4 ? `v5 from epoch ${firstV4.epoch} at DAA ${firstV4.daa}` : 'no v5 epoch'}; epochs seen ${[...epochs.entries()].map(([e, v]) => `e${e}:${v.class}:${v.bps5}bps:d${v.day}`).join(' ')}; chain bytes ${JSON.stringify(versionBytes)} (${signalShareOnChain} bps at byte ${V5_BYTE}); blocks ${before.length} / ${after.length}; rejected miners ${rejectedMiner.map(r => r.length).join('/')} nodes ${rejectedNode.map(r => r.length).join('/')}; sinks ${sinks.join(' ')} (${checks.sinks_agree ? 'agree' : 'DIFFER'}); counts ${counts.join('/')}; signal lines ${signalLines.filter(Boolean).length}/3 (epochs ${signalEpochs.join('/')}, shares ${signalShares.join('/')}); floor lines ${floorLines.filter(Boolean).length}/3; state roots at the flip ${rootsAtFlip.map(r => String(r).slice(0, 18)).join(' ')}; stateless node accepted after the flip ${statelessAcceptedAfterFlip} (refusal lines ${statelessRefusals}); stale miner accepted after the first refresh ${staleAcceptedAfterRefresh ?? 'n/a'} rejected ${staleRejected ?? 'n/a'}; days ${daysSeen.join('/')}`);
|
||||
for (const r of idRows) log(`PROGRAM ID epoch ${r.epoch} seed ${r.seed}: miners ${r.miners_id} (${r.miners} of 3) cli v5 ${r.cli_v5} cli v4 ${r.cli_v4}; state root ${r.state_root} (${r.state_records} records)`);
|
||||
for (const [k, v] of Object.entries(checks)) if (!v) log(`FAILED CHECK ${k}`);
|
||||
log(`summary: ${TMP}/summary.json`);
|
||||
await stopAll();
|
||||
process.exit(pass ? 0 : 1);
|
||||
277
infra/fast-time/latency-ladder.mjs
Normal file
277
infra/fast-time/latency-ladder.mjs
Normal file
|
|
@ -0,0 +1,277 @@
|
|||
#!/usr/bin/env node
|
||||
// The latency ladder's fast-time gate (docs/design/latency-ladder.md section 9; the class-v4-signal.mjs shape): a 3-node
|
||||
// network on override-60x.json, class v4 from genesis (v3 and the v4 floor at 0, the class window 0: class signalling off,
|
||||
// so the ladder opens the header's high byte on its own), the ladder active from DAA 0 with one window of --window DAA
|
||||
// (default 60, one epoch; seven windows = 420 DAA, so the first epoch whose seed block has seven full windows below it is
|
||||
// epoch 8 at DAA 480), each node's ladder signal set by IGNEUM_LADDER_SIGNAL (--signal a,b,c of up|down|none), one real CPU
|
||||
// miner per node. Ports 29720 and up, network igneum-devnet-972, data /tmp/igneum-fast-time-ladder.
|
||||
//
|
||||
// The cases and the known-failed case:
|
||||
// --signal up,up,none --expect no-step two of three miners signal up: about 67 percent, rung 0 must hold (run 10 epochs)
|
||||
// --signal up,up,up --expect step all three: rung 1 (35 shadow passes) from epoch 8, the first with seven full
|
||||
// windows, every miner's rung-1 program id equal to the CLI's --shadow-reps 35 id and
|
||||
// unequal to the rung-0 id, and NO second step inside the next two epochs
|
||||
// --signal up,up,none --expect step the known-failed case: the harness must report FAIL (no step happened)
|
||||
//
|
||||
// node infra/fast-time/latency-ladder.mjs --signal a,b,c --expect step|no-step [--window 60] [--secs 900] [--epochs 10]
|
||||
// IGNEUMD, IGNEUM_MINER, IGNEUM_POW name the binaries (defaults: the ladder fork worktree's target/release and
|
||||
// igneum-pow/target/release/igneum-pow, the layout on igneum-build-1 under /srv/builds/igneum-wt-ladder).
|
||||
|
||||
import { spawn, spawnSync } from 'node:child_process';
|
||||
import { mkdirSync, rmSync, writeFileSync, readFileSync, openSync, existsSync } from 'node:fs';
|
||||
import { connectRpc } from '../../tools/finality-attacks/lib/rpc.mjs';
|
||||
import { devAddress } from '../../tools/harness/lib/address.mjs';
|
||||
|
||||
const ROOT = new URL('../../', import.meta.url).pathname;
|
||||
const FILE = `${ROOT}infra/fast-time/override-60x.json`;
|
||||
const BIN = process.env.IGNEUM_LADDER_BIN || `${ROOT}vendor/igneum-node-ladder/target/release`;
|
||||
const IGNEUMD = process.env.IGNEUMD || `${BIN}/igneumd`;
|
||||
const CPU_MINER = process.env.IGNEUM_MINER || `${BIN}/igneum-miner`;
|
||||
const IGNEUM_POW = process.env.IGNEUM_POW || `${ROOT}igneum-pow/target/release/igneum-pow`;
|
||||
const TMP = process.env.IGNEUM_LADDER_TMP || '/tmp/igneum-fast-time-ladder';
|
||||
const BASE = 29720, SUFFIX = 972;
|
||||
const NEVER = '18446744073709551615';
|
||||
const RUNG0 = 27, RUNG1 = 35, WINDOWS = 7, THRESHOLD = 9000;
|
||||
const args = process.argv.slice(2);
|
||||
const flag = (name, dflt) => { const i = args.indexOf(`--${name}`); return i >= 0 ? +args[i + 1] : dflt; };
|
||||
const sflag = (name) => { const i = args.indexOf(`--${name}`); return i >= 0 ? args[i + 1] : null; };
|
||||
const GENESIS_BITS = flag('genesis-bits', 0x1f010000);
|
||||
const SECS = flag('secs', 900);
|
||||
const EPOCHS = flag('epochs', 10);
|
||||
const WINDOW = flag('window', 60);
|
||||
const SIGNAL = (sflag('signal') || 'up,up,up').split(',').map(s => s.trim().toLowerCase());
|
||||
const EXPECT = sflag('expect') || 'step';
|
||||
if (!['step', 'no-step'].includes(EXPECT) || SIGNAL.length !== 3 || !SIGNAL.every(s => ['up', 'down', 'none'].includes(s))) { console.error('usage: --signal a,b,c (up|down|none) --expect step|no-step'); process.exit(2); }
|
||||
const started = [];
|
||||
const log = (...a) => console.log(new Date().toISOString().slice(11, 23), ...a);
|
||||
const sleep = (ms) => new Promise(r => setTimeout(r, ms));
|
||||
for (const b of [IGNEUMD, CPU_MINER]) if (!existsSync(b)) { console.error(`missing ${b}`); process.exit(2); }
|
||||
|
||||
rmSync(TMP, { recursive: true, force: true }); mkdirSync(TMP, { recursive: true });
|
||||
const baseText = readFileSync(FILE, 'utf8');
|
||||
const field = (name) => { const m = new RegExp(`"${name}":\\s*([0-9]+)`).exec(baseText); return m ? +m[1] : undefined; };
|
||||
const EPOCH = field('pow_epoch_blocks');
|
||||
const LEAD = field('pow_epoch_lead');
|
||||
const DAY_MS = field('pow_day_ms');
|
||||
// the first epoch whose seed block (the last chain block below L*e - lead) can have DAA >= 7 x WINDOW: L*e - lead - 1 >= 7W
|
||||
let FIRST_STEP_EPOCH = 0;
|
||||
while (FIRST_STEP_EPOCH * EPOCH - LEAD - 1 < WINDOWS * WINDOW) FIRST_STEP_EPOCH++;
|
||||
function mergeOverrideText(text, fields) {
|
||||
let out = text;
|
||||
for (const k of Object.keys(fields)) out = out.replace(new RegExp(`\\s*"${k}":\\s*[^,}\\n]+,?`), '');
|
||||
const extra = Object.entries(fields).map(([k, v]) => `"${k}": ${typeof v === 'string' && !/^\d+$/.test(v) ? JSON.stringify(v) : v}`).join(', ');
|
||||
return out.replace(/,?\s*}\s*$/, `,\n ${extra}\n}\n`);
|
||||
}
|
||||
const override = `${TMP}/override.json`;
|
||||
writeFileSync(override, mergeOverrideText(baseText, {
|
||||
genesis_bits: GENESIS_BITS, skip_proof_of_work: false,
|
||||
program_class_v3_activation_daa: '0', program_class_v4_activation_daa: '0', program_class_v4_signal_window_daa: '0',
|
||||
latency_ladder_activation_daa: '0', latency_ladder_window_daa: String(WINDOW),
|
||||
}));
|
||||
log(`signals ${SIGNAL.join('/')}, expect ${EXPECT}; class v4 from genesis, the ladder active from DAA 0, window ${WINDOW} DAA x ${WINDOWS} (the first epoch that can step is ${FIRST_STEP_EPOCH}, DAA ${FIRST_STEP_EPOCH * EPOCH}); ${EPOCH} DAA per epoch, lead ${LEAD}; run ${SECS} s or ${EPOCHS} epochs`);
|
||||
|
||||
class Node {
|
||||
constructor(i, connect = []) {
|
||||
this.i = i; this.grpcPort = BASE + i * 10; this.p2pPort = BASE + i * 10 + 1; this.jsonPort = BASE + i * 10 + 2;
|
||||
this.connect = connect; this.dir = `${TMP}/n${i}`; this.logFile = `${this.dir}/node.log`;
|
||||
}
|
||||
get grpc() { return `grpc://127.0.0.1:${this.grpcPort}`; }
|
||||
async start() {
|
||||
mkdirSync(this.dir, { recursive: true });
|
||||
const a = ['--devnet', `--devnet-suffix=${SUFFIX}`, '--nodnsseed', '--disable-upnp', '--nologfiles', '--enable-unsynced-mining', '--utxoindex',
|
||||
`--appdir=${this.dir}`, `--rpclisten=127.0.0.1:${this.grpcPort}`, `--rpclisten-json=127.0.0.1:${this.jsonPort}`,
|
||||
`--listen=127.0.0.1:${this.p2pPort}`, `--override-params-file=${override}`, '--loglevel=info', '--yes'];
|
||||
if (this.connect.length) a.push(`--connect=${this.connect.join(',')}`); else a.push('--outpeers=0');
|
||||
const out = openSync(this.logFile, 'a');
|
||||
// the node's own ladder signal: what its templates carry in bits 15 and 14
|
||||
this.proc = spawn(IGNEUMD, a, { stdio: ['ignore', out, out], env: { ...process.env, IGNEUM_LADDER_SIGNAL: SIGNAL[this.i] } });
|
||||
started.push(this.proc);
|
||||
await sleep(1200);
|
||||
this.rpc = await connectRpc(`ws://127.0.0.1:${this.jsonPort}`);
|
||||
log(`n${this.i} up pid ${this.proc.pid} json ${this.jsonPort} p2p ${this.p2pPort}, signals ${SIGNAL[this.i]}`);
|
||||
return this;
|
||||
}
|
||||
grepLog(re) { try { return readFileSync(this.logFile, 'utf8').split('\n').filter(l => re.test(l)); } catch { return []; } }
|
||||
}
|
||||
function miner(bin, argv, name, env = {}) {
|
||||
const out = openSync(`${TMP}/${name}.log`, 'a');
|
||||
const p = spawn(bin, argv, { stdio: ['ignore', out, out], env: { ...process.env, ...env } });
|
||||
started.push(p);
|
||||
return p;
|
||||
}
|
||||
async function stopAll() {
|
||||
for (const p of started.reverse()) { try { p.kill('SIGINT'); } catch { } }
|
||||
await sleep(1500);
|
||||
for (const p of started) { try { p.kill('SIGKILL'); } catch { } }
|
||||
}
|
||||
process.on('SIGINT', async () => { await stopAll(); process.exit(130); });
|
||||
process.on('unhandledRejection', async (e) => { log(`FAILED: ${e?.stack || e}`); await stopAll(); process.exit(3); });
|
||||
const minerLog = (i) => { try { return readFileSync(`${TMP}/cpu${i}.log`, 'utf8').split('\n'); } catch { return []; } };
|
||||
const STEP_LINE = /Latency ladder step by miner signal: epoch (\d+) moves to rung (\d+) \((\d+) shadow passes, from rung (\d+)\): (up|down) in each of (\d+) consecutive windows of (\d+) DAA .*weakest up (\d+) bps, weakest down (\d+) bps/;
|
||||
const LADDER_LINE = /Latency ladder from the override file/;
|
||||
const ACTIVE_LINE = /Latency ladder active: rungs/;
|
||||
const OWN_LINE = /Latency ladder signal from IGNEUM_LADDER_SIGNAL: this node signals (\w+)/;
|
||||
|
||||
const t0 = Date.now();
|
||||
const since = () => ((Date.now() - t0) / 1000).toFixed(1);
|
||||
const n0 = await new Node(0).start();
|
||||
const n1 = await new Node(1, [`127.0.0.1:${n0.p2pPort}`]).start();
|
||||
const n2 = await new Node(2, [`127.0.0.1:${n0.p2pPort}`]).start();
|
||||
const nodes = [n0, n1, n2];
|
||||
for (const n of nodes) log(`n${n.i}: ${n.grepLog(LADDER_LINE).map(l => l.replace(/^.*?(Latency ladder from)/, '$1'))[0] || '(no ladder line)'} | ${n.grepLog(OWN_LINE).map(l => l.replace(/^.*?(this node signals)/, '$1'))[0] || '(no signal line)'}`);
|
||||
log(`n0 digest: ${n0.grepLog(/Consensus params digest/).map(l => l.replace(/^.*?digest: /, '').slice(0, 16)).join(' ')}`);
|
||||
nodes.forEach((n, i) => miner(CPU_MINER, ['mine', n.grpc, '1', String(SECS), `cpu${i}`, '--engine', 'igneum-pow', '--payout-label', `cpu${i}`, '--status-secs', '30', '--no-vote'], `cpu${i}`, { IGNEUM_POW_DAY_MS: String(DAY_MS) }));
|
||||
const pay = devAddress('fast-time-ladder');
|
||||
|
||||
const epochs = new Map();
|
||||
let firstStep = null, lastEpoch = -1, lastReport = 0, lastDaa = 0, endAt = null;
|
||||
const samples = [];
|
||||
while (Date.now() - t0 < SECS * 1000) {
|
||||
await sleep(1000);
|
||||
let daa = null, epoch = null, cls = null, reps = null, nextReps = null, step = null, nextStep = null, up = null, upWeak = null, down = null, sig = null, stepEpoch = null, eraSeed = null;
|
||||
try {
|
||||
const t = await n0.rpc.call('getBlockTemplate', { payAddress: pay, extraData: [] });
|
||||
const pe = t.powEpoch || t.pow_epoch || {};
|
||||
daa = pe.virtualDaaScore ?? t.block?.header?.daaScore; epoch = pe.epochIndex; cls = pe.programClass; eraSeed = pe.eraSeed;
|
||||
reps = pe.latencyLadderReps; nextReps = pe.nextLatencyLadderReps; step = pe.latencyLadderStep; nextStep = pe.nextLatencyLadderStep;
|
||||
up = pe.latencyLadderUpBps; upWeak = pe.latencyLadderUpWeakestBps; down = pe.latencyLadderDownBps; sig = pe.latencyLadderSignal; stepEpoch = pe.latencyLadderStepEpoch;
|
||||
} catch (e) { log(`template: ${e.message}`); }
|
||||
if (epoch != null && epoch !== lastEpoch) {
|
||||
epochs.set(epoch, { class: cls, reps, step, firstSeenDaa: daa, at: +since(), eraSeed: eraSeed == null ? null : String(eraSeed), up_bps: up, up_weakest_bps: upWeak, down_bps: down, step_epoch: stepEpoch ?? null });
|
||||
log(`epoch ${lastEpoch} -> ${epoch} at daa ${daa}, ${since()} s: template class ${cls} rung ${step} (${reps} passes), next rung ${nextStep} (${nextReps}), up ${up} bps (weakest of ${WINDOWS}: ${upWeak}), down ${down} bps, this node signals ${sig}, step took effect at epoch ${stepEpoch ?? 'none'}`);
|
||||
if (firstStep == null && step > 0) { firstStep = { epoch, daa, step, reps, at: +since() }; log(`LADDER STEP: the template is rung ${step} (${reps} shadow passes) from epoch ${epoch} (daa ${daa}) at ${since()} s wall`); }
|
||||
lastEpoch = epoch;
|
||||
}
|
||||
lastDaa = daa ?? lastDaa;
|
||||
if (Date.now() - lastReport > 15000) {
|
||||
lastReport = Date.now();
|
||||
const counts = await Promise.all(nodes.map(async n => { try { const d = await n.rpc.call('getBlockDagInfo'); return `${d.blockCount}/${String(d.sink).slice(0, 8)}`; } catch { return '?'; } }));
|
||||
log(`t=${since()} s daa ${daa} epoch ${epoch} rung ${step} (${reps}) up ${up} bps weakest ${upWeak} blocks/sink per node ${counts.join(' ')}`);
|
||||
samples.push({ t: +since(), daa, epoch, step, reps, up_bps: up, up_weakest_bps: upWeak, nodes: counts });
|
||||
}
|
||||
// the end: two epochs after a step (to show no second step), or --epochs epochs when no step is expected
|
||||
if (firstStep != null && daa != null && daa >= (firstStep.epoch + 2) * EPOCH + LEAD) { endAt = +since(); break; }
|
||||
if (firstStep == null && daa != null && daa >= EPOCHS * EPOCH) { endAt = +since(); break; }
|
||||
}
|
||||
await sleep(3000);
|
||||
|
||||
const dag = await Promise.all(nodes.map(async n => { try { return await n.rpc.call('getBlockDagInfo'); } catch (e) { return { error: e.message }; } }));
|
||||
const genesis = dag[0].pruningPointHash;
|
||||
async function allBlocks(n) {
|
||||
const out = []; let low = genesis; const seen = new Set();
|
||||
for (let round = 0; round < 500; round++) {
|
||||
const r = await n.rpc.call('getBlocks', { lowHash: low, includeBlocks: true, includeTransactions: false });
|
||||
const blocks = r.blocks || [];
|
||||
let added = 0;
|
||||
for (const b of blocks) { const h = b.verboseData?.hash || b.header?.hash; if (seen.has(h)) continue; seen.add(h); out.push({ hash: h, daa: +b.header.daaScore, version: +b.header.version, chain: !!b.verboseData?.isChainBlock }); added++; }
|
||||
if (!blocks.length || added === 0) break;
|
||||
low = (r.blockHashes || []).at(-1) || blocks.at(-1).verboseData?.hash; if (!low) break;
|
||||
}
|
||||
return out;
|
||||
}
|
||||
let blocks = [];
|
||||
try { blocks = await allBlocks(n0); } catch (e) { log(`getBlocks: ${e.message}`); }
|
||||
const BOUNDARY = firstStep ? firstStep.epoch * EPOCH : Infinity;
|
||||
const before = blocks.filter(b => b.daa < BOUNDARY), after = blocks.filter(b => b.daa >= BOUNDARY);
|
||||
// the ladder bits on the chain: bit 15 up, bit 14 down; the object byte (bits 8 to 13) must be 0 (class signalling off)
|
||||
const bitsOf = (v) => (v & 0x8000) ? 'up' : (v & 0x4000) ? 'down' : 'none';
|
||||
const ladderBits = blocks.reduce((m, b) => { const k = bitsOf(b.version); m[k] = (m[k] || 0) + 1; return m; }, {});
|
||||
const upShareOnChain = blocks.length ? Math.round(10000 * (blocks.filter(b => bitsOf(b.version) === 'up').length) / blocks.length) : 0;
|
||||
const objectBytes = blocks.reduce((m, b) => { const v = (b.version >> 8) & 0x3f; m[v] = (m[v] || 0) + 1; return m; }, {});
|
||||
// genesis carries header version 0 (genesis.rs), every mined block the block version 2 in its low byte
|
||||
const lowBytes = new Set(blocks.filter(b => b.daa > 0).map(b => b.version & 0xff));
|
||||
|
||||
const programs = new Map();
|
||||
for (const i of [0, 1, 2]) for (const l of minerLog(i)) {
|
||||
const m = /epoch seed ([0-9a-f]{64}) day (\d+) \(daa (\d+)\): program and 256 MiB cache ready in ([\d.]+) ms; class (v\d) program id ([0-9a-f]{16})/.exec(l);
|
||||
if (!m) continue;
|
||||
const k = m[1]; const e = programs.get(k) || { seed: k.slice(0, 16), epoch: Math.floor(+m[3] / EPOCH), class: m[5], id: m[6], miners: new Set() };
|
||||
if (e.id !== m[6] || e.class !== m[5]) e.disagree = true;
|
||||
e.miners.add(i); programs.set(k, e);
|
||||
}
|
||||
const programRows = [...programs.values()].sort((a, b) => a.epoch - b.epoch).map(p => ({ epoch: p.epoch, class: p.class, program_id: p.id, seed: p.seed, miners: p.miners.size, disagree: !!p.disagree }));
|
||||
function cliId(seedHex, eraHex, reps) {
|
||||
if (!existsSync(IGNEUM_POW)) return null;
|
||||
const r = spawnSync(IGNEUM_POW, ['show', '--epoch-hex', seedHex, '--program-class', 'v4', '--era-hex', eraHex, '--shadow-reps', String(reps)], { encoding: 'utf8' });
|
||||
const m = /program id ([0-9a-f]{16})/.exec(r.stdout || '');
|
||||
return m ? m[1] : null;
|
||||
}
|
||||
const idRows = [];
|
||||
for (const [k, e] of programs) {
|
||||
const ep = epochs.get(e.epoch);
|
||||
if (!ep || ep.eraSeed == null) continue;
|
||||
const reps = ep.reps ?? 0;
|
||||
idRows.push({ epoch: e.epoch, seed: e.seed, reps, miners_id: e.id, miners: e.miners.size, cli_rung0: cliId(k, ep.eraSeed, 0), cli_at_reps: cliId(k, ep.eraSeed, reps) });
|
||||
}
|
||||
const steppedRows = idRows.filter(r => r.reps !== RUNG0 && r.reps !== 0);
|
||||
const accepted = [0, 1, 2].map(i => minerLog(i).filter(l => /ACCEPTED block/.test(l)).length);
|
||||
const rejectedMiner = [0, 1, 2].map(i => minerLog(i).filter(l => /rejected nonce=|submit error/.test(l)));
|
||||
const rejectedNode = nodes.map(n => n.grepLog(/PoW rejected|Rejected block|rejected block/i));
|
||||
const stepLines = nodes.map(n => n.grepLog(STEP_LINE).map(l => l.replace(/^.*?(Latency ladder step by miner signal)/, '$1')));
|
||||
const firstStepLine = stepLines.map(ls => ls[0] || null);
|
||||
const stepEpochs = firstStepLine.map(l => { const m = l && STEP_LINE.exec(l); return m ? +m[1] : null; });
|
||||
const stepRungs = firstStepLine.map(l => { const m = l && STEP_LINE.exec(l); return m ? +m[2] : null; });
|
||||
const stepWeakestUp = firstStepLine.map(l => { const m = l && STEP_LINE.exec(l); return m ? +m[8] : null; });
|
||||
const sinks = dag.map(d => String(d.sink || '?').slice(0, 16));
|
||||
const counts = dag.map(d => d.blockCount ?? '?');
|
||||
const maxEpochSeen = Math.max(-1, ...epochs.keys());
|
||||
const repsSeen = [...epochs.values()].map(e => e.reps);
|
||||
const afterStep = firstStep ? [...epochs.entries()].filter(([e]) => e > firstStep.epoch).map(([, v]) => v.step) : [];
|
||||
|
||||
const common = {
|
||||
zero_rejected_by_miners: rejectedMiner.every(r => r.length === 0),
|
||||
zero_rejected_by_nodes: rejectedNode.every(r => r.length === 0),
|
||||
sinks_agree: new Set(sinks).size === 1,
|
||||
block_counts_agree: new Set(counts.map(String)).size === 1,
|
||||
miners_agree_on_every_program: programRows.every(p => !p.disagree),
|
||||
ladder_line_on_every_node: nodes.every(n => n.grepLog(LADDER_LINE).length > 0 && n.grepLog(ACTIVE_LINE).length > 0),
|
||||
every_node_signals_its_bits: nodes.every((n, i) => n.grepLog(OWN_LINE).some(l => OWN_LINE.exec(l)[1] === SIGNAL[i])),
|
||||
every_epoch_class_v4: [...epochs.values()].every(e => e.class === 4),
|
||||
rung0_ids_equal_the_cli_rung0_id: idRows.filter(r => r.reps === RUNG0).length > 0 && idRows.filter(r => r.reps === RUNG0).every(r => r.cli_rung0 != null && r.cli_rung0 === r.miners_id),
|
||||
};
|
||||
// every block carries block version 2, an object byte of 0 (class signalling off) and the ladder bits of one of the three
|
||||
// nodes; genesis, made before any node, is the one bit-less block when every node signals (the first run of the known-failed
|
||||
// case, 22:26Z, failed this check on genesis's version 0 in the low-byte test, a harness fault, not a chain one)
|
||||
const noneNodes = SIGNAL.filter(s => s === 'none').length;
|
||||
common.chain_carries_the_bits = blocks.length > 0 && [...lowBytes].every(v => v === 2) && Object.keys(objectBytes).every(v => +v === 0)
|
||||
&& Object.keys(ladderBits).every(k => SIGNAL.includes(k) || k === 'none') && (noneNodes > 0 || (ladderBits.none || 0) === 1);
|
||||
let checks;
|
||||
if (EXPECT === 'step') {
|
||||
checks = {
|
||||
...common,
|
||||
template_stepped_to_rung_1: firstStep != null && firstStep.step === 1 && firstStep.reps === RUNG1,
|
||||
stepped_at_the_first_full_window_epoch: firstStep != null && firstStep.epoch === FIRST_STEP_EPOCH,
|
||||
step_line_on_every_node_same_epoch: stepEpochs.every(e => e != null) && new Set(stepEpochs).size === 1 && stepEpochs[0] === (firstStep && firstStep.epoch) && stepRungs.every(r => r === 1),
|
||||
weakest_up_at_or_above_threshold: stepWeakestUp.every(s => s != null && s >= THRESHOLD),
|
||||
no_second_step_inside_seven_windows: firstStep != null && afterStep.length >= 2 && afterStep.every(s => s === 1) && stepLines.every(ls => ls.length === 1),
|
||||
blocks_on_both_sides: before.length > 0 && after.length > 0,
|
||||
rung1_ids_equal_the_cli_rung1_id: steppedRows.length > 0 && steppedRows.every(r => r.reps === RUNG1 && r.cli_at_reps != null && r.cli_at_reps === r.miners_id && r.miners === 3),
|
||||
rung1_ids_differ_from_the_same_seed_rung0_id: steppedRows.length > 0 && steppedRows.every(r => r.cli_rung0 != null && r.cli_rung0 !== r.miners_id),
|
||||
};
|
||||
} else {
|
||||
checks = {
|
||||
...common,
|
||||
template_never_above_rung_0: firstStep == null && repsSeen.every(r => r === RUNG0 || r === 0),
|
||||
no_step_line_on_any_node: stepLines.every(ls => ls.length === 0),
|
||||
ran_the_epochs: maxEpochSeen >= EPOCHS - 1,
|
||||
passed_the_first_full_window_epoch: maxEpochSeen >= FIRST_STEP_EPOCH,
|
||||
up_share_under_threshold_on_chain: upShareOnChain < THRESHOLD,
|
||||
};
|
||||
}
|
||||
const pass = Object.values(checks).every(Boolean);
|
||||
const summary = {
|
||||
pass, expect: EXPECT, signals: SIGNAL, checks, window: WINDOW, windows: WINDOWS, threshold_bps: THRESHOLD, epoch_blocks: EPOCH, lead: LEAD, first_step_epoch: FIRST_STEP_EPOCH,
|
||||
node: IGNEUMD, miner: CPU_MINER, pow: IGNEUM_POW, template_step: firstStep, run_ended_at_s: endAt, final_daa: lastDaa, max_epoch_seen: maxEpochSeen,
|
||||
epochs: Object.fromEntries([...epochs.entries()].map(([k, v]) => [k, v])),
|
||||
blocks: { total: blocks.length, before_boundary: before.length, after_boundary: after.length, ladder_bits: ladderBits, object_bytes: objectBytes, up_share_bps_on_chain: upShareOnChain },
|
||||
programs: programRows, program_id_rows: idRows, accepted_per_miner: accepted,
|
||||
rejected_by_miners: rejectedMiner.map(r => r.length), rejected_by_nodes: rejectedNode.map(r => r.length),
|
||||
sinks, block_counts: counts, step_lines: stepLines, samples,
|
||||
};
|
||||
writeFileSync(`${TMP}/summary.json`, JSON.stringify(summary, null, 2));
|
||||
log(`SUMMARY ${pass ? 'PASS' : 'FAIL'} (expect ${EXPECT}, signals ${SIGNAL.join('/')}): ${firstStep ? `rung ${firstStep.step} (${firstStep.reps} passes) from epoch ${firstStep.epoch} at DAA ${firstStep.daa}` : 'no step'}; epochs seen ${[...epochs.entries()].map(([e, v]) => `e${e}:r${v.step}:${v.up_weakest_bps}bps`).join(' ')}; chain bits ${JSON.stringify(ladderBits)} (${upShareOnChain} bps up); blocks ${before.length} / ${after.length}; rejected miners ${rejectedMiner.map(r => r.length).join('/')} nodes ${rejectedNode.map(r => r.length).join('/')}; sinks ${sinks.join(' ')} at ${counts.join('/')}`);
|
||||
for (const r of idRows) log(`PROGRAM ID epoch ${r.epoch} seed ${r.seed} reps ${r.reps}: miners ${r.miners_id} (${r.miners} of 3) cli at reps ${r.cli_at_reps} cli rung 0 ${r.cli_rung0}`);
|
||||
for (const [k, v] of Object.entries(checks)) if (!v) log(`FAILED CHECK ${k}`);
|
||||
log(`summary: ${TMP}/summary.json`);
|
||||
await stopAll();
|
||||
process.exit(pass ? 0 : 1);
|
||||
|
|
@ -52,14 +52,21 @@
|
|||
"pow_epoch_lead": 10,
|
||||
"pow_day_ms": 1440000,
|
||||
"difficulty_v2_activation_daa": 18446744073709551615,
|
||||
"difficulty_v3_activation_daa": 18446744073709551615,
|
||||
"finality_daa_rule_activation_daa": 18446744073709551615,
|
||||
"proving_v0_activation_daa": 18446744073709551615,
|
||||
"finality_v3_activation_daa": 18446744073709551615,
|
||||
"program_class_v3_activation_daa": 18446744073709551615,
|
||||
"program_class_v4_activation_daa": 18446744073709551615,
|
||||
"program_class_v4_signal_window_daa": 120,
|
||||
"program_class_v5_activation_daa": 18446744073709551615,
|
||||
"latency_ladder": [{"reps": 27, "admissible": true}, {"reps": 35, "admissible": true}, {"reps": 53, "admissible": true}, {"reps": 88, "admissible": false}, {"reps": 173, "admissible": false}, {"reps": 267, "admissible": false}],
|
||||
"latency_ladder_activation_daa": 18446744073709551615,
|
||||
"latency_ladder_window_daa": 120,
|
||||
"proving_v1_fresh_rule_daa": 18446744073709551615,
|
||||
"exec_restart_number": 18446744073709551615,
|
||||
"exec_restart_hash": "",
|
||||
"exec_restart_state_root": "",
|
||||
"exec_restart_trust_daa": 18446744073709551615,
|
||||
"pow_genesis_dataset_log2": 28,
|
||||
"proving_v1_activation_daa": 18446744073709551615,
|
||||
|
|
|
|||
|
|
@ -23,6 +23,7 @@ struct Drv {
|
|||
decltype(&cuMemAlloc_v2) memAlloc = nullptr;
|
||||
decltype(&cuMemFree_v2) memFree = nullptr;
|
||||
decltype(&cuMemcpyDtoH_v2) memcpyDtoH = nullptr;
|
||||
decltype(&cuMemcpyHtoD_v2) memcpyHtoD = nullptr; // class v5: the state leaves go up for igneum_build (7 October 2026)
|
||||
decltype(&cuModuleLoadData) moduleLoadData = nullptr;
|
||||
decltype(&cuModuleUnload) moduleUnload = nullptr;
|
||||
decltype(&cuModuleGetFunction) moduleGetFunction = nullptr;
|
||||
|
|
|
|||
|
|
@ -24,20 +24,28 @@
|
|||
#include "../cuda_api.h" // the real cuda.h / nvrtc.h types
|
||||
#include "cuda_runtime.h" // the shim (proto-cuda/emu): emu_launch and the kernel symbols' world
|
||||
|
||||
// igneum_build has two shapes: (ds, cache, nItems) for classes v2 to v4 and (ds, cache, leaves, nLeaves, nItems) for class v5
|
||||
// (7 October 2026). A pack defines one of them; both are declared weak so the one the pack lacks is a null symbol, and
|
||||
// d_launch picks the shape by the argument count the worker passed (5 for a class v5 pack).
|
||||
#define EMU_WEAK __attribute__((weak))
|
||||
namespace emu_pack_a {
|
||||
struct IgneumInitWords { uint32_t w[8]; }; // the same definition as the pack's kernel_bound.cu, inside its namespace
|
||||
void igneum_cache_fill(uint32_t* cache, uint32_t nSegments);
|
||||
void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);
|
||||
EMU_WEAK void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);
|
||||
EMU_WEAK void igneum_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems);
|
||||
void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw);
|
||||
}
|
||||
#ifdef IGNEUM_EMU_TWO_PACKS
|
||||
namespace emu_pack_b {
|
||||
struct IgneumInitWords { uint32_t w[8]; };
|
||||
void igneum_cache_fill(uint32_t* cache, uint32_t nSegments);
|
||||
void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);
|
||||
EMU_WEAK void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);
|
||||
EMU_WEAK void igneum_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems);
|
||||
void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw);
|
||||
}
|
||||
#endif
|
||||
typedef void (*EmuBuild3)(uint32_t*, const uint32_t*, uint32_t);
|
||||
typedef void (*EmuBuild5)(uint32_t*, const uint32_t*, const uint32_t*, uint32_t, uint32_t);
|
||||
|
||||
static std::string readAll(const std::string& path) {
|
||||
FILE* f = std::fopen(path.c_str(), "rb");
|
||||
|
|
@ -198,6 +206,7 @@ static CUresult d_memInfo(size_t* f, size_t* t) { *f = 8ull << 30; *t = 16ull <<
|
|||
static CUresult d_alloc(CUdeviceptr* p, size_t n) { void* m = std::malloc(n); if (!m) return CUDA_ERROR_OUT_OF_MEMORY; *p = (CUdeviceptr)(uintptr_t)m; return CUDA_SUCCESS; }
|
||||
static CUresult d_free(CUdeviceptr p) { std::free((void*)(uintptr_t)p); return CUDA_SUCCESS; }
|
||||
static CUresult d_dtoh(void* dst, CUdeviceptr src, size_t n) { std::memcpy(dst, (const void*)(uintptr_t)src, n); return CUDA_SUCCESS; }
|
||||
static CUresult d_htod(CUdeviceptr dst, const void* src, size_t n) { std::memcpy((void*)(uintptr_t)dst, src, n); return CUDA_SUCCESS; }
|
||||
static CUresult d_modLoad(CUmodule* m, const void* img) {
|
||||
const char* s = (const char*)img;
|
||||
if (std::strncmp(s, "EMU-IMAGE:", 10) != 0) return CUDA_ERROR_INVALID_IMAGE;
|
||||
|
|
@ -222,11 +231,28 @@ template <class T> static T* dptr(void** params, int i) { return (T*)(uintptr_t)
|
|||
static CUresult d_launch(CUfunction f, unsigned gx, unsigned, unsigned, unsigned bx, unsigned, unsigned, unsigned, CUstream, void** params, void**) {
|
||||
switch ((int)(uintptr_t)f) {
|
||||
case 1: emu_launch(emu_pack_a::igneum_cache_fill, gx, bx, dptr<uint32_t>(params, 0), arg<uint32_t>(params, 1)); return CUDA_SUCCESS;
|
||||
case 2: emu_launch(emu_pack_a::igneum_build, gx, bx, dptr<uint32_t>(params, 0), dptr<const uint32_t>(params, 1), arg<uint32_t>(params, 2)); return CUDA_SUCCESS;
|
||||
case 2: {
|
||||
// the worker passes 5 arguments for a class v5 pack (params[3] is the leaf count, params[4] the item count) and 3 otherwise;
|
||||
// the pack's kernel.cu defines exactly one shape: a mismatch (a v5 pack built without its leaves, or leaves handed to a
|
||||
// v4 kernel) is the known-failed case and fails the launch instead of running the wrong kernel
|
||||
EmuBuild5 b5 = (EmuBuild5)emu_pack_a::igneum_build; EmuBuild3 b3 = (EmuBuild3)emu_pack_a::igneum_build;
|
||||
bool five = params[3] != nullptr && params[4] != nullptr;
|
||||
if (five && b5) { emu_launch(b5, gx, bx, dptr<uint32_t>(params, 0), dptr<const uint32_t>(params, 1), dptr<const uint32_t>(params, 2), arg<uint32_t>(params, 3), arg<uint32_t>(params, 4)); return CUDA_SUCCESS; }
|
||||
if (!five && b3) { emu_launch(b3, gx, bx, dptr<uint32_t>(params, 0), dptr<const uint32_t>(params, 1), arg<uint32_t>(params, 2)); return CUDA_SUCCESS; }
|
||||
std::fprintf(stderr, "emu: igneum_build called with %d arguments but pack A's kernel has the %s shape\n", five ? 5 : 3, b5 ? "class v5 (leaves)" : "class v2 to v4");
|
||||
return CUDA_ERROR_INVALID_VALUE;
|
||||
}
|
||||
case 3: emu_launch(emu_pack_a::igneum_hash_bound, gx, bx, dptr<const uint32_t>(params, 0), dptr<uint64_t>(params, 1), arg<uint32_t>(params, 2), arg<uint32_t>(params, 3), arg<emu_pack_a::IgneumInitWords>(params, 4)); return CUDA_SUCCESS;
|
||||
#ifdef IGNEUM_EMU_TWO_PACKS
|
||||
case 4: emu_launch(emu_pack_b::igneum_cache_fill, gx, bx, dptr<uint32_t>(params, 0), arg<uint32_t>(params, 1)); return CUDA_SUCCESS;
|
||||
case 5: emu_launch(emu_pack_b::igneum_build, gx, bx, dptr<uint32_t>(params, 0), dptr<const uint32_t>(params, 1), arg<uint32_t>(params, 2)); return CUDA_SUCCESS;
|
||||
case 5: {
|
||||
EmuBuild5 b5 = (EmuBuild5)emu_pack_b::igneum_build; EmuBuild3 b3 = (EmuBuild3)emu_pack_b::igneum_build;
|
||||
bool five = params[3] != nullptr && params[4] != nullptr;
|
||||
if (five && b5) { emu_launch(b5, gx, bx, dptr<uint32_t>(params, 0), dptr<const uint32_t>(params, 1), dptr<const uint32_t>(params, 2), arg<uint32_t>(params, 3), arg<uint32_t>(params, 4)); return CUDA_SUCCESS; }
|
||||
if (!five && b3) { emu_launch(b3, gx, bx, dptr<uint32_t>(params, 0), dptr<const uint32_t>(params, 1), arg<uint32_t>(params, 2)); return CUDA_SUCCESS; }
|
||||
std::fprintf(stderr, "emu: igneum_build called with %d arguments but pack B's kernel has the %s shape\n", five ? 5 : 3, b5 ? "class v5 (leaves)" : "class v2 to v4");
|
||||
return CUDA_ERROR_INVALID_VALUE;
|
||||
}
|
||||
case 6: emu_launch(emu_pack_b::igneum_hash_bound, gx, bx, dptr<const uint32_t>(params, 0), dptr<uint64_t>(params, 1), arg<uint32_t>(params, 2), arg<uint32_t>(params, 3), arg<emu_pack_b::IgneumInitWords>(params, 4)); return CUDA_SUCCESS;
|
||||
#endif
|
||||
default: return CUDA_ERROR_INVALID_HANDLE;
|
||||
|
|
@ -244,7 +270,7 @@ void emu_fill_driver(Drv& d) {
|
|||
d.init = d_init; d.driverGetVersion = d_driverVersion; d.deviceGetCount = d_count; d.deviceGet = d_get; d.deviceGetName = d_name;
|
||||
d.deviceGetAttribute = d_attr; d.deviceTotalMem = d_totalMem; d.primaryCtxSetFlags = d_ctxFlags; d.primaryCtxRetain = d_ctxRetain;
|
||||
d.primaryCtxRelease = d_ctxRelease; d.ctxSetCurrent = d_ctxSet; d.ctxSynchronize = d_ctxSync; d.memGetInfo = d_memInfo;
|
||||
d.memAlloc = d_alloc; d.memFree = d_free; d.memcpyDtoH = d_dtoh; d.moduleLoadData = d_modLoad; d.moduleUnload = d_modUnload;
|
||||
d.memAlloc = d_alloc; d.memFree = d_free; d.memcpyDtoH = d_dtoh; d.memcpyHtoD = d_htod; d.moduleLoadData = d_modLoad; d.moduleUnload = d_modUnload;
|
||||
d.moduleGetFunction = d_getFn; d.launchKernel = d_launch; d.streamCreate = d_streamCreate; d.streamSynchronize = d_streamSync;
|
||||
d.streamDestroy = d_streamDestroy; d.funcGetAttribute = d_funcAttr; d.occupancy = d_occ; d.getErrorString = d_errStr; d.getErrorName = d_errName;
|
||||
}
|
||||
|
|
|
|||
|
|
@ -172,6 +172,72 @@ int main(int argc, char** argv) {
|
|||
CHECK(pf_load(dir, &pk, err, sizeof(err)) == 0 && strstr(err, "generator 1 is not") != NULL, "a pack with no generator line (generator 1) is refused");
|
||||
}
|
||||
|
||||
// Class v5 (docs/design/class-v5-stored-state.md, 7 October 2026): the state leaves. Known-failed first: a generator 5
|
||||
// pack without the leaf count, one whose leaves file is missing, one whose file has another FNV, one of another class
|
||||
// carrying a count; then the checked-in v5 pack (argv[2]) loads and its leaves load with the pack's FNV.
|
||||
{
|
||||
char sw[200], kw[200], text[2600], why[256], path[1024];
|
||||
uint32_t* leaves = NULL;
|
||||
size_t bytes = 0;
|
||||
FILE* f;
|
||||
int i;
|
||||
words_hex(bare, sw); words_hex(keyw, kw);
|
||||
snprintf(text, sizeof(text), "#define IGNEUM_SEED_BYTES_HEX \"%s\"\n#define IGNEUM_DAY_BYTES_HEX \"%s\"\n#define IGNEUM_GENERATOR 5\n#define IGNEUM_PROGRAM_CLASS \"v5\"\n#define IGNEUM_ERA_SEED_HEX \"%s\"\n#define IGNEUM_SHADOW_INSTRS 256\n#define IGNEUM_SHADOW_REPS 27\n#define IGNEUM_DATASET_LOG2 28\n#define IGNEUM_DATASET_MODE 1\n#define IGNEUM_SEEDW_INIT { %s }\n#define IGNEUM_KEY_INIT { %s }\n#define IGNEUM_CACHE_LOG2_WORDS 26\n#define IGNEUM_CACHE_SEGMENTS 4096u\n", EPOCH_34, DAY_20731, EPOCH_33, sw, kw);
|
||||
write_file(dir, "program.h", text);
|
||||
err[0] = 0;
|
||||
CHECK(pf_load(dir, &pk, err, sizeof(err)) == 0 && strstr(err, "without IGNEUM_STATE_LEAVES") != NULL, "known-failed: a generator 5 pack without the leaf count is refused in plain words");
|
||||
// the count and the FNV of two leaves of 64 bytes (0x00..0x3f, 0x40..0x7f), the file not yet written
|
||||
{
|
||||
uint8_t raw[128];
|
||||
uint64_t fnv;
|
||||
for (i = 0; i < 128; ++i) raw[i] = (uint8_t)i;
|
||||
fnv = pf_fnv1a64(raw, sizeof(raw));
|
||||
snprintf(text + strlen(text), sizeof(text) - strlen(text), "#define IGNEUM_STATE_LEAVES 2\n#define IGNEUM_STATE_LEAVES_FNV64 0x%016llxull\n#define IGNEUM_STATE_ROOT_HEX \"%s\"\n", (unsigned long long)fnv, EPOCH_33);
|
||||
write_file(dir, "program.h", text);
|
||||
snprintf(path, sizeof(path), "%s/leaves.bin", dir); remove(path);
|
||||
err[0] = 0;
|
||||
CHECK(pf_load(dir, &pk, err, sizeof(err)) == 1 && pk.stateLeaves == 2 && pk.stateLeavesFnv == fnv && strcmp(pk.programClass, "v5") == 0 && strcmp(pk.stateLeavesFile, "leaves.bin") == 0, "a generator 5 pack with the leaf count loads as class v5 (count, FNV and file name read back)");
|
||||
CHECK(pf_pack_class_ok(pk.programClass, pk.eraHex, "v5", EPOCH_33, why, sizeof(why)) == 1, "the v5 pack matches a job naming class v5 and its era");
|
||||
CHECK(pf_pack_class_ok(pk.programClass, pk.eraHex, "v4", EPOCH_33, why, sizeof(why)) == 0 && strstr(why, "program class mismatch") == why, "a job naming class v4 refuses the v5 pack");
|
||||
err[0] = 0;
|
||||
CHECK(pf_load_leaves(dir, &pk, &leaves, &bytes, err, sizeof(err)) == 0 && leaves == NULL && strstr(err, "without its leaves file") != NULL, "known-failed: the leaves file missing refuses the build in plain words");
|
||||
f = fopen(path, "wb"); fwrite(raw, 1, 64, f); fclose(f);
|
||||
err[0] = 0;
|
||||
CHECK(pf_load_leaves(dir, &pk, &leaves, &bytes, err, sizeof(err)) == 0 && leaves == NULL && strstr(err, "is 64 bytes, the pack says 2 leaves of 64") != NULL, "known-failed: a short leaves file is refused with both sizes named");
|
||||
raw[5] ^= 0x80;
|
||||
f = fopen(path, "wb"); fwrite(raw, 1, 128, f); fclose(f);
|
||||
raw[5] ^= 0x80;
|
||||
err[0] = 0;
|
||||
CHECK(pf_load_leaves(dir, &pk, &leaves, &bytes, err, sizeof(err)) == 0 && leaves == NULL && strstr(err, "is not the pack's IGNEUM_STATE_LEAVES_FNV64") != NULL, "known-failed: leaves of another root (one bit) are refused by the FNV");
|
||||
f = fopen(path, "wb"); fwrite(raw, 1, 128, f); fclose(f);
|
||||
err[0] = 0;
|
||||
CHECK(pf_load_leaves(dir, &pk, &leaves, &bytes, err, sizeof(err)) == 1 && leaves != NULL && bytes == 128 && memcmp(leaves, raw, 128) == 0, "known-good: the right leaves load as 2 x 16 words");
|
||||
free(leaves); leaves = NULL;
|
||||
remove(path);
|
||||
}
|
||||
// a class v4 pack carrying a leaf count is a mis-stamped export
|
||||
snprintf(text, sizeof(text), "#define IGNEUM_SEED_BYTES_HEX \"%s\"\n#define IGNEUM_DAY_BYTES_HEX \"%s\"\n#define IGNEUM_GENERATOR 4\n#define IGNEUM_PROGRAM_CLASS \"v4\"\n#define IGNEUM_ERA_SEED_HEX \"%s\"\n#define IGNEUM_SHADOW_INSTRS 256\n#define IGNEUM_STATE_LEAVES 2\n#define IGNEUM_DATASET_LOG2 28\n#define IGNEUM_DATASET_MODE 1\n#define IGNEUM_SEEDW_INIT { %s }\n#define IGNEUM_KEY_INIT { %s }\n#define IGNEUM_CACHE_LOG2_WORDS 26\n#define IGNEUM_CACHE_SEGMENTS 4096u\n", EPOCH_34, DAY_20731, EPOCH_33, sw, kw);
|
||||
write_file(dir, "program.h", text);
|
||||
err[0] = 0;
|
||||
CHECK(pf_load(dir, &pk, err, sizeof(err)) == 0 && strstr(err, "state leaves belong to class v5") != NULL, "known-failed: a generator 4 pack carrying IGNEUM_STATE_LEAVES is refused");
|
||||
// a class v4 pack without leaves loads with stateLeaves 0 and pf_load_leaves hands back nothing
|
||||
snprintf(text, sizeof(text), "#define IGNEUM_SEED_BYTES_HEX \"%s\"\n#define IGNEUM_DAY_BYTES_HEX \"%s\"\n#define IGNEUM_GENERATOR 4\n#define IGNEUM_PROGRAM_CLASS \"v4\"\n#define IGNEUM_ERA_SEED_HEX \"%s\"\n#define IGNEUM_SHADOW_INSTRS 256\n#define IGNEUM_DATASET_LOG2 28\n#define IGNEUM_DATASET_MODE 1\n#define IGNEUM_SEEDW_INIT { %s }\n#define IGNEUM_KEY_INIT { %s }\n#define IGNEUM_CACHE_LOG2_WORDS 26\n#define IGNEUM_CACHE_SEGMENTS 4096u\n", EPOCH_34, DAY_20731, EPOCH_33, sw, kw);
|
||||
write_file(dir, "program.h", text);
|
||||
err[0] = 0;
|
||||
CHECK(pf_load(dir, &pk, err, sizeof(err)) == 1 && pk.stateLeaves == 0 && pf_load_leaves(dir, &pk, &leaves, &bytes, err, sizeof(err)) == 1 && leaves == NULL && bytes == 0, "a class v4 pack has no leaves and the leaf loader hands back none");
|
||||
if (argc > 2) {
|
||||
// the checked-in class v5 pack (proto-cuda/packs-ca3-v5/v5-dn3-epoch0): loads, its leaves load under the pack's FNV
|
||||
err[0] = 0;
|
||||
CHECK(pf_load(argv[2], &pk, err, sizeof(err)) == 1 && strcmp(pk.programClass, "v5") == 0 && pk.generator == 5 && pk.stateLeaves > 0 && pk.haveVectors, "known-good: the checked-in class v5 pack loads with its leaf count and vectors");
|
||||
if (err[0]) printf(" %s\n", err);
|
||||
err[0] = 0;
|
||||
CHECK(pf_load_leaves(argv[2], &pk, &leaves, &bytes, err, sizeof(err)) == 1 && leaves != NULL && bytes == (size_t)pk.stateLeaves * 64u, "known-good: the checked-in pack's leaves.bin loads under the pack's FNV-1a 64");
|
||||
if (err[0]) printf(" %s\n", err);
|
||||
printf(" v5 pack: %u leaves, FNV-1a 64 %016llx, state root %s\n", (unsigned)pk.stateLeaves, (unsigned long long)pk.stateLeavesFnv, pk.stateRootHex);
|
||||
free(leaves); leaves = NULL;
|
||||
}
|
||||
}
|
||||
|
||||
printf("%s: %d failure(s)\n", argv[0], failures);
|
||||
return failures ? 1 : 0;
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
#!/usr/bin/env bash
|
||||
# The pack loader's seed rule (packfile.h) on a known-good and a known-mismatched pack: emu/packfile-test.c, C99,
|
||||
# no GPU. Runs on the Mac in a second and in CI. Usage: emu/packfile-test.sh
|
||||
# The pack loader's seed rule (packfile.h) on a known-good and a known-mismatched pack, and the class v5 leaf rule on the
|
||||
# checked-in v5 pack (7 October 2026): emu/packfile-test.c, C99, no GPU. Runs on the Mac in a second and in CI. Usage: emu/packfile-test.sh
|
||||
set -euo pipefail
|
||||
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||
ROOT="$(cd "$HERE/../../.." && pwd)"
|
||||
|
|
@ -8,4 +8,4 @@ OUT="${TMPDIR:-/tmp}/igneum-packfile-test"
|
|||
mkdir -p "$OUT"
|
||||
CC="${CC:-cc}"
|
||||
"$CC" -std=c99 -Wall -Wextra -Wno-unused-function -O1 -o "$OUT/packfile-test" "$HERE/packfile-test.c"
|
||||
"$OUT/packfile-test" "$ROOT/proto-cuda/packs/igneum-devnet-v4-epoch0"
|
||||
"$OUT/packfile-test" "$ROOT/proto-cuda/packs/igneum-devnet-v4-epoch0" "$ROOT/proto-cuda/packs-ca3-v5/v5-dn3-epoch0"
|
||||
|
|
|
|||
|
|
@ -31,6 +31,15 @@ typedef struct {
|
|||
char loadClass[64];
|
||||
char programClass[8]; /* IGNEUM_PROGRAM_CLASS: "v2", "v3" (Counter ASIC 2.0) or "v4" (Counter ASIC 3.0); absent = the generator's class */
|
||||
char eraHex[65]; /* IGNEUM_ERA_SEED_HEX of a class v3 or v4 chain pack; empty otherwise */
|
||||
/* Class v5 (docs/design/class-v5-stored-state.md, 7 October 2026): the window's state leaves. leaves.bin beside program.h
|
||||
* holds IGNEUM_STATE_LEAVES leaves of 16 little-endian words (64 B each), leaf(t) = leaves[t mod stateLeaves]; every item
|
||||
* of the dataset is keyed by one, so a host that has not uploaded them builds nothing (pf_load_leaves). stateLeaves is 0
|
||||
* for every other class; a generator 5 pack without them, or a pack of another class with them, is refused by pf_load. */
|
||||
uint32_t stateLeaves;
|
||||
uint64_t stateLeavesFnv; /* IGNEUM_STATE_LEAVES_FNV64: FNV-1a 64 over leaves.bin's bytes */
|
||||
char stateLeavesFile[64]; /* IGNEUM_STATE_LEAVES_FILE, "leaves.bin" when absent */
|
||||
char stateRootHex[65]; /* IGNEUM_STATE_ROOT_HEX: the state root the leaves hash under */
|
||||
char stateBlockHex[65]; /* IGNEUM_STATE_BLOCK_HEX: the window's reference chain block */
|
||||
// Counter ASIC 2.0 (5 October 2026): the mixer multiplier of the item derivation (IGNEUM_MIXER_MULT, 1 when absent:
|
||||
// version 2; 4 under class v3). The emitted memhard.h / kernel.cl carry it in their text; this is for the log lines.
|
||||
uint32_t mixerMult;
|
||||
|
|
@ -117,6 +126,15 @@ static int pf_define_u32(const char* text, const char* name, uint32_t* out) {
|
|||
return 1;
|
||||
}
|
||||
|
||||
/* A 64-bit define (`0x...ull`), the pack's FNV lines in program.h. */
|
||||
static int pf_define_u64(const char* text, const char* name, uint64_t* out) {
|
||||
const char* p = pf_find_define(text, name);
|
||||
const char* e;
|
||||
if (!p) return 0;
|
||||
while (*p == ' ' || *p == '\t') ++p;
|
||||
return pf_number(p, out, &e);
|
||||
}
|
||||
|
||||
static int pf_define_str(const char* text, const char* name, char* out, size_t cap) {
|
||||
const char* p = pf_find_define(text, name);
|
||||
const char* q;
|
||||
|
|
@ -286,12 +304,14 @@ static int pf_load(const char* dir, PfPack* pk, char* err, size_t cap) {
|
|||
/* Spec 01 section 1.4.5: a pack whose generator version is not one this worker runs is refused. Generator 2 is
|
||||
* program class v2 (the lottery hash of 4 October 2026), generator 3 is class v3 (Counter ASIC 2.0), generator 4
|
||||
* is class v4 (Counter ASIC 3.0, 6 October 2026: class v3 plus the latency-shadow block, emitted in the pack's own
|
||||
* kernel text, so this loader needs nothing new beyond the number and the class token). */
|
||||
if (pk->generator != 2 && pk->generator != 3 && pk->generator != 4) {
|
||||
char m[200]; snprintf(m, sizeof(m), "program pack generator %u is not a generator version this worker runs (2, 3 or 4)", (unsigned)pk->generator);
|
||||
* kernel text, so this loader needs nothing new beyond the number and the class token), generator 5 is class v5
|
||||
* (proof of stored state, 7 October 2026: class v4 over a dataset keyed by the window's state leaves, leaves.bin,
|
||||
* which the host uploads for igneum_build; the kernel text carries the leaf read). */
|
||||
if (pk->generator != 2 && pk->generator != 3 && pk->generator != 4 && pk->generator != 5) {
|
||||
char m[200]; snprintf(m, sizeof(m), "program pack generator %u is not a generator version this worker runs (2, 3, 4 or 5)", (unsigned)pk->generator);
|
||||
free(prog); return pf_fail(err, cap, m);
|
||||
}
|
||||
strcpy(pk->programClass, pk->generator == 4 ? "v4" : pk->generator == 3 ? "v3" : "v2");
|
||||
strcpy(pk->programClass, pk->generator == 5 ? "v5" : pk->generator == 4 ? "v4" : pk->generator == 3 ? "v3" : "v2");
|
||||
{
|
||||
/* Counter ASIC 3.0 (6 October 2026): the shadow block marks class v4. A generator 3 pack with IGNEUM_SHADOW_INSTRS
|
||||
* is a v4 program stamped as v3 (the old export path; it carried the v3 control's program id) and is refused;
|
||||
|
|
@ -299,6 +319,7 @@ static int pf_load(const char* dir, PfPack* pk, char* err, size_t cap) {
|
|||
uint32_t shadow = 0;
|
||||
if (!pf_define_u32(prog, "IGNEUM_SHADOW_INSTRS", &shadow)) shadow = 0;
|
||||
if (pk->generator == 4 && shadow == 0) { free(prog); return pf_fail(err, cap, "program pack generator 4 (class v4) without IGNEUM_SHADOW_INSTRS: not a class v4 pack"); }
|
||||
if (pk->generator == 5 && shadow == 0) { free(prog); return pf_fail(err, cap, "program pack generator 5 (class v5) without IGNEUM_SHADOW_INSTRS: not a class v5 pack (class v5 is class v4 over the state leaves)"); }
|
||||
if (pk->generator == 3 && shadow != 0) { /* a generator 2 pack with a shadow is the measurement ladder (a class-bearing id) and loads */
|
||||
char m[220]; snprintf(m, sizeof(m), "program pack generator %u with a shadow block (IGNEUM_SHADOW_INSTRS %u): a class v4 program is generator 4 (export the pack as class v4)", (unsigned)pk->generator, (unsigned)shadow);
|
||||
free(prog); return pf_fail(err, cap, m);
|
||||
|
|
@ -312,6 +333,24 @@ static int pf_load(const char* dir, PfPack* pk, char* err, size_t cap) {
|
|||
}
|
||||
}
|
||||
pk->eraHex[0] = 0; pf_define_str(prog, "IGNEUM_ERA_SEED_HEX", pk->eraHex, sizeof(pk->eraHex));
|
||||
/* Class v5: the state leaves. The count, the FNV and the file name come from program.h; the bytes are read by
|
||||
* pf_load_leaves when the host builds. A v5 pack without the count (or with 0) is refused: nothing could key its items;
|
||||
* a pack of another class carrying a count is a mis-stamped export and is refused too. */
|
||||
pk->stateLeaves = 0; pk->stateLeavesFnv = 0; pk->stateRootHex[0] = 0; pk->stateBlockHex[0] = 0;
|
||||
strcpy(pk->stateLeavesFile, "leaves.bin");
|
||||
if (!pf_define_u32(prog, "IGNEUM_STATE_LEAVES", &pk->stateLeaves)) pk->stateLeaves = 0;
|
||||
if (pk->generator == 5 && pk->stateLeaves == 0) { free(prog); return pf_fail(err, cap, "program pack generator 5 (class v5) without IGNEUM_STATE_LEAVES: no state leaves to key the dataset (export the pack with --state)"); }
|
||||
if (pk->generator != 5 && pk->stateLeaves != 0) {
|
||||
char m[200]; snprintf(m, sizeof(m), "program pack generator %u carries IGNEUM_STATE_LEAVES %u: state leaves belong to class v5 (generator 5)", (unsigned)pk->generator, (unsigned)pk->stateLeaves);
|
||||
free(prog); return pf_fail(err, cap, m);
|
||||
}
|
||||
if (pk->stateLeaves) {
|
||||
if (pk->stateLeaves > 0x02000000u) { free(prog); return pf_fail(err, cap, "program.h IGNEUM_STATE_LEAVES is above 2^25 (the designed dataset's item cap)"); }
|
||||
if (!pf_define_u64(prog, "IGNEUM_STATE_LEAVES_FNV64", &pk->stateLeavesFnv)) { free(prog); return pf_fail(err, cap, "program pack class v5 without IGNEUM_STATE_LEAVES_FNV64: the leaves cannot be checked"); }
|
||||
pf_define_str(prog, "IGNEUM_STATE_LEAVES_FILE", pk->stateLeavesFile, sizeof(pk->stateLeavesFile));
|
||||
pf_define_str(prog, "IGNEUM_STATE_ROOT_HEX", pk->stateRootHex, sizeof(pk->stateRootHex));
|
||||
pf_define_str(prog, "IGNEUM_STATE_BLOCK_HEX", pk->stateBlockHex, sizeof(pk->stateBlockHex));
|
||||
}
|
||||
if (!pf_define_u32(prog, "IGNEUM_PROGRAM_ATTEMPT", &pk->attempt)) pk->attempt = 0;
|
||||
if (pf_define_words(prog, "IGNEUM_SEEDW_INIT", pk->seedw, 8) != 8) { free(prog); return pf_fail(err, cap, "program.h has no IGNEUM_SEEDW_INIT with 8 words"); }
|
||||
pk->loadsPerHash = 128; pf_define_u32(prog, "IGNEUM_LOADS_PER_HASH", &pk->loadsPerHash);
|
||||
|
|
@ -405,6 +444,31 @@ static int pf_load(const char* dir, PfPack* pk, char* err, size_t cap) {
|
|||
return 1;
|
||||
}
|
||||
|
||||
/* Class v5: reads the pack's leaves file into a malloc'd buffer of stateLeaves x 64 bytes (16 little-endian words per leaf,
|
||||
* the layout igneum_build reads), checked against the pack's count (the byte length) and its FNV-1a 64
|
||||
* (IGNEUM_STATE_LEAVES_FNV64) before anything is uploaded. Returns 1 with *out and *bytes set (NULL and 0 for a pack of
|
||||
* another class, which has no leaves); 0 with err and nothing allocated. The caller frees *out after the build. */
|
||||
static int pf_load_leaves(const char* dir, const PfPack* pk, uint32_t** out, size_t* bytes, char* err, size_t cap) {
|
||||
char* raw;
|
||||
size_t n = 0;
|
||||
uint64_t fnv;
|
||||
*out = NULL; *bytes = 0;
|
||||
if (pk->stateLeaves == 0) return 1;
|
||||
raw = pf_read_pack_file(dir, pk->stateLeavesFile, &n);
|
||||
if (!raw) { char m[600]; snprintf(m, sizeof(m), "class v5 pack without its leaves file %.400s/%.60s (the state leaves key every item: nothing can be built without them)", dir, pk->stateLeavesFile); return pf_fail(err, cap, m); }
|
||||
if (n != (size_t)pk->stateLeaves * 64u) {
|
||||
char m[300]; snprintf(m, sizeof(m), "%.60s is %llu bytes, the pack says %u leaves of 64 (%llu bytes)", pk->stateLeavesFile, (unsigned long long)n, (unsigned)pk->stateLeaves, (unsigned long long)pk->stateLeaves * 64ull);
|
||||
free(raw); return pf_fail(err, cap, m);
|
||||
}
|
||||
fnv = pf_fnv1a64(raw, n);
|
||||
if (fnv != pk->stateLeavesFnv) {
|
||||
char m[300]; snprintf(m, sizeof(m), "%.60s FNV-1a 64 %016llx is not the pack's IGNEUM_STATE_LEAVES_FNV64 %016llx (the leaves are of another state root; export the pack again)", pk->stateLeavesFile, (unsigned long long)fnv, (unsigned long long)pk->stateLeavesFnv);
|
||||
free(raw); return pf_fail(err, cap, m);
|
||||
}
|
||||
*out = (uint32_t*)raw; *bytes = n;
|
||||
return 1;
|
||||
}
|
||||
|
||||
// The self-test verdict from values the host read back from the device. `vec` holds vecWarps x 32 outputs of the
|
||||
// bound kernel run with the pack's own seed words as init words (that is igneum_hash of kernel.cu). Writes one line.
|
||||
// A hot-table pack (pk->hotMb) also hands the hot table's head, last line and FNV-1a 64 (NULL and 0 otherwise); a
|
||||
|
|
@ -443,7 +507,7 @@ static int pf_selftest(const PfPack* pk, const uint32_t* cacheHead, const uint32
|
|||
}
|
||||
|
||||
|
||||
/* Counter ASIC 2.0 (5 October 2026): a job or prepare line may end with `class=<v2|v3|v4>` and `era=<hex>` tokens (sent
|
||||
/* Counter ASIC 2.0 (5 October 2026): a job or prepare line may end with `class=<v2|v3|v4|v5>` and `era=<hex>` tokens (sent
|
||||
* only when the chain is on class v3 or v4, so every v2 line is the line of before). A pack matches the line when its
|
||||
* class is the named class and, when an era is named, its era seed is that era (every class after v2 carries one).
|
||||
* Empty wanted strings accept any pack. Returns 1 on a match, else 0 with the reason in `why`. */
|
||||
|
|
|
|||
|
|
@ -166,6 +166,7 @@ static bool loadDriver(Drv& d, std::string& err, std::string& libName) {
|
|||
LOAD_SYM(d, memAlloc, "cuMemAlloc_v2");
|
||||
LOAD_SYM(d, memFree, "cuMemFree_v2");
|
||||
LOAD_SYM(d, memcpyDtoH, "cuMemcpyDtoH_v2");
|
||||
LOAD_SYM(d, memcpyHtoD, "cuMemcpyHtoD_v2");
|
||||
LOAD_SYM(d, moduleLoadData, "cuModuleLoadData");
|
||||
LOAD_SYM(d, moduleUnload, "cuModuleUnload");
|
||||
LOAD_SYM(d, moduleGetFunction, "cuModuleGetFunction");
|
||||
|
|
@ -438,6 +439,8 @@ struct Pair {
|
|||
// pack's igneum_hot_fill, the argument after the init words
|
||||
uint32_t hotMb = 0, hotWords = 0, hotSegments = 0, hotSlots = 0;
|
||||
CUfunction fHotFill = nullptr;
|
||||
// class v5 (7 October 2026): the pack's state leaves, uploaded for igneum_build and freed after it (0 for other classes)
|
||||
uint32_t stateLeaves = 0;
|
||||
CUdeviceptr hot = 0;
|
||||
double hotMs = 0;
|
||||
};
|
||||
|
|
@ -841,7 +844,14 @@ static Pair* buildPair(Ctx& c, const std::string& dir, CUstream s, std::string&
|
|||
}
|
||||
p->loadClass = pk.loadClass; p->loadsPerHash = pk.loadsPerHash; p->bytesPerHash = pk.bytesPerHash; p->scratchOps = pk.scratchOps;
|
||||
p->programClass = pk.programClass; p->eraHex = pk.eraHex;
|
||||
p->stateLeaves = pk.stateLeaves;
|
||||
p->persistent = pk.persistent != 0;
|
||||
// Class v5 (docs/design/class-v5-stored-state.md): the window's leaves (leaves.bin), read and checked against the pack's
|
||||
// count and FNV-1a 64 before anything is allocated; uploaded for igneum_build below and freed right after it, so device
|
||||
// memory while hashing is the class v4 worker's. A v5 pack without its leaves builds nothing (the known-failed case).
|
||||
uint32_t* hLeaves = nullptr;
|
||||
size_t leavesBytes = 0;
|
||||
{ char lerr[700]; if (!pf_load_leaves(dir.c_str(), &pk, &hLeaves, &leavesBytes, lerr, sizeof(lerr))) { err = "pack " + dir + ": " + lerr; releasePair(c, p); return nullptr; } }
|
||||
p->hotMb = pk.hotMb; p->hotWords = pk.hotWords; p->hotSegments = pk.hotSegments; p->hotSlots = pk.hotSlots;
|
||||
p->residentWarps = p->blocksPerSM * c.blockWarps * c.sms;
|
||||
size_t scratchBytes = 0, hotBytes = (size_t)pk.hotWords * 4u;
|
||||
|
|
@ -860,14 +870,14 @@ static Pair* buildPair(Ctx& c, const std::string& dir, CUstream s, std::string&
|
|||
size_t cacheBytes = (size_t)p->cacheWords * 4u, dsBytes = (size_t)p->words * 4u;
|
||||
{
|
||||
size_t freeB = 0, totalB = 0;
|
||||
if (c.drv.memGetInfo(&freeB, &totalB) == CUDA_SUCCESS && freeB < cacheBytes + dsBytes + scratchBytes + hotBytes + (64u << 20)) {
|
||||
err = fmt("%llu MiB free on the device, this pack needs %llu MiB (cache %llu + dataset %llu + scratch %llu + hot %llu)", (unsigned long long)(freeB >> 20), (unsigned long long)((cacheBytes + dsBytes + scratchBytes + hotBytes) >> 20), (unsigned long long)(cacheBytes >> 20), (unsigned long long)(dsBytes >> 20), (unsigned long long)(scratchBytes >> 20), (unsigned long long)(hotBytes >> 20));
|
||||
releasePair(c, p); return nullptr;
|
||||
if (c.drv.memGetInfo(&freeB, &totalB) == CUDA_SUCCESS && freeB < cacheBytes + dsBytes + scratchBytes + hotBytes + leavesBytes + (64u << 20)) {
|
||||
err = fmt("%llu MiB free on the device, this pack needs %llu MiB (cache %llu + dataset %llu + scratch %llu + hot %llu + leaves %llu)", (unsigned long long)(freeB >> 20), (unsigned long long)((cacheBytes + dsBytes + scratchBytes + hotBytes + leavesBytes) >> 20), (unsigned long long)(cacheBytes >> 20), (unsigned long long)(dsBytes >> 20), (unsigned long long)(scratchBytes >> 20), (unsigned long long)(hotBytes >> 20), (unsigned long long)(leavesBytes >> 20));
|
||||
std::free(hLeaves); releasePair(c, p); return nullptr;
|
||||
}
|
||||
}
|
||||
if (p->persistent) {
|
||||
CUresult r = c.drv.memAlloc(&p->scratch, scratchBytes);
|
||||
if (r != CUDA_SUCCESS) { err = "cuMemAlloc scratch: " + c.err(r); p->scratch = 0; releasePair(c, p); return nullptr; }
|
||||
if (r != CUDA_SUCCESS) { err = "cuMemAlloc scratch: " + c.err(r); p->scratch = 0; std::free(hLeaves); releasePair(c, p); return nullptr; }
|
||||
p->scratchBytes = scratchBytes;
|
||||
int after = 0;
|
||||
c.drv.occupancy(&after, p->fHashBound, 32 * c.blockWarps, 0);
|
||||
|
|
@ -876,24 +886,41 @@ static Pair* buildPair(Ctx& c, const std::string& dir, CUstream s, std::string&
|
|||
}
|
||||
{
|
||||
CUresult r = c.drv.memAlloc(&p->cache, cacheBytes);
|
||||
if (r != CUDA_SUCCESS) { err = "cuMemAlloc cache: " + c.err(r); p->cache = 0; releasePair(c, p); return nullptr; }
|
||||
if (r != CUDA_SUCCESS) { err = "cuMemAlloc cache: " + c.err(r); p->cache = 0; std::free(hLeaves); releasePair(c, p); return nullptr; }
|
||||
uint32_t nSeg = p->cacheSegments, block = 256u, grid = (nSeg + block - 1u) / block;
|
||||
void* args[2] = { &p->cache, &nSeg };
|
||||
r = c.drv.launchKernel(p->fCacheFill, grid, 1, 1, block, 1, 1, 0, s, args, nullptr);
|
||||
if (r == CUDA_SUCCESS) r = c.drv.streamSynchronize(s);
|
||||
if (r != CUDA_SUCCESS) { err = "cache fill: " + c.err(r); releasePair(c, p); return nullptr; }
|
||||
if (r != CUDA_SUCCESS) { err = "cache fill: " + c.err(r); std::free(hLeaves); releasePair(c, p); return nullptr; }
|
||||
}
|
||||
p->cacheMs = wallMs() - t0;
|
||||
// Dataset
|
||||
t0 = wallMs();
|
||||
{
|
||||
CUresult r = c.drv.memAlloc(&p->ds, dsBytes);
|
||||
if (r != CUDA_SUCCESS) { err = "cuMemAlloc dataset: " + c.err(r); p->ds = 0; releasePair(c, p); return nullptr; }
|
||||
if (r != CUDA_SUCCESS) { err = "cuMemAlloc dataset: " + c.err(r); p->ds = 0; std::free(hLeaves); releasePair(c, p); return nullptr; }
|
||||
uint32_t nItems = p->words / 16u, block = 256u, grid = (nItems + block - 1u) / block;
|
||||
void* args[3] = { &p->ds, &p->cache, &nItems };
|
||||
r = c.drv.launchKernel(p->fBuild, grid, 1, 1, block, 1, 1, 0, s, args, nullptr);
|
||||
if (r == CUDA_SUCCESS) r = c.drv.streamSynchronize(s);
|
||||
if (r != CUDA_SUCCESS) { err = "dataset build: " + c.err(r); releasePair(c, p); return nullptr; }
|
||||
if (hLeaves) {
|
||||
// class v5: igneum_build(ds, cache, leaves, nLeaves, nItems), the leaf buffer freed once the build has run
|
||||
CUdeviceptr dLeaves = 0;
|
||||
uint32_t nLeaves = pk.stateLeaves;
|
||||
r = c.drv.memAlloc(&dLeaves, leavesBytes);
|
||||
if (r != CUDA_SUCCESS) { err = "cuMemAlloc state leaves: " + c.err(r); std::free(hLeaves); releasePair(c, p); return nullptr; }
|
||||
r = c.drv.memcpyHtoD(dLeaves, hLeaves, leavesBytes);
|
||||
std::free(hLeaves); hLeaves = nullptr;
|
||||
if (r == CUDA_SUCCESS) {
|
||||
void* args[5] = { &p->ds, &p->cache, &dLeaves, &nLeaves, &nItems };
|
||||
r = c.drv.launchKernel(p->fBuild, grid, 1, 1, block, 1, 1, 0, s, args, nullptr);
|
||||
}
|
||||
if (r == CUDA_SUCCESS) r = c.drv.streamSynchronize(s);
|
||||
c.drv.memFree(dLeaves);
|
||||
if (r != CUDA_SUCCESS) { err = "dataset build (class v5, " + std::to_string(nLeaves) + " state leaves): " + c.err(r); releasePair(c, p); return nullptr; }
|
||||
} else {
|
||||
void* args[5] = { &p->ds, &p->cache, &nItems, nullptr, nullptr }; // five slots: the driver reads the kernel's three, the emulation reads the shape from the two null tails
|
||||
r = c.drv.launchKernel(p->fBuild, grid, 1, 1, block, 1, 1, 0, s, args, nullptr);
|
||||
if (r == CUDA_SUCCESS) r = c.drv.streamSynchronize(s);
|
||||
if (r != CUDA_SUCCESS) { err = "dataset build: " + c.err(r); releasePair(c, p); return nullptr; }
|
||||
}
|
||||
}
|
||||
p->dsMs = wallMs() - t0;
|
||||
// Hot table (hot-table experiment): filled from the epoch seed by the pack's own kernel, never shipped
|
||||
|
|
@ -958,7 +985,8 @@ static Pair* buildPair(Ctx& c, const std::string& dir, CUstream s, std::string&
|
|||
}
|
||||
|
||||
static std::string pairSummary(const Pair* p) {
|
||||
return fmt("nvrtc %.0f cache %.0f dataset %.0f hot %.0f check %.0f race %.0f ms variant %s; %s", p->compileMs, p->cacheMs, p->dsMs, p->hotMs, p->checkMs, p->raceMs, p->variant.c_str(), p->check.c_str());
|
||||
return fmt("nvrtc %.0f cache %.0f dataset %.0f hot %.0f check %.0f race %.0f ms variant %s class %s%s; %s", p->compileMs, p->cacheMs, p->dsMs, p->hotMs, p->checkMs, p->raceMs, p->variant.c_str(), p->programClass.c_str(),
|
||||
p->stateLeaves ? fmt(" (state leaves %u, uploaded for the build and freed)", p->stateLeaves).c_str() : "", p->check.c_str());
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------------------------
|
||||
|
|
|
|||
538
proto-cuda/packs-ca3-v5/v4-genesis/kernel.cl
Normal file
538
proto-cuda/packs-ca3-v5/v4-genesis/kernel.cl
Normal file
|
|
@ -0,0 +1,538 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
|
||||
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
|
||||
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
|
||||
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
|
||||
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
|
||||
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
|
||||
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
|
||||
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
|
||||
#ifndef IGNEUM_GROUP
|
||||
#define IGNEUM_GROUP 32
|
||||
#endif
|
||||
#ifndef IGNEUM_EXCHANGE
|
||||
#define IGNEUM_EXCHANGE 0
|
||||
#endif
|
||||
#ifdef __OPENCL_VERSION__
|
||||
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
|
||||
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#ifdef cl_khr_subgroups
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
|
||||
#endif
|
||||
#ifdef cl_khr_subgroup_shuffle
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
|
||||
#endif
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
|
||||
#endif
|
||||
#else
|
||||
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
|
||||
#include "emu_opencl.h"
|
||||
#endif
|
||||
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#else
|
||||
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
|
||||
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
|
||||
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
|
||||
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
|
||||
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
|
||||
#endif
|
||||
|
||||
static inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
|
||||
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
|
||||
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
|
||||
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
|
||||
static inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of 8 x seed-parameterised mixer + one 64-byte cache read, then 8 x final mixer (class v3, mixer multiplier 8,
|
||||
// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
static inline void mh_chacha_block(const uint* x, uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
static inline void mh_cache_segment(__global uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
static inline void mh_mixer(uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of 8 x mixer + cache line s[0] & mask; 8 x final mixer.
|
||||
static inline void mh_item(__global const uint* cache, uint t, uint* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
for (uint j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 8u + j + 1u));
|
||||
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
for (uint j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (64u + j + 1u));
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
|
||||
// The same constants as memhard.h in this pack (one emitter, three dialects).
|
||||
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
|
||||
uint seg = (uint)get_global_id(0);
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
|
||||
uint t = (uint)get_global_id(0);
|
||||
if (t < nItems) {
|
||||
uint s[16];
|
||||
mh_item(cache, t, s);
|
||||
__global uint* d = ds + ((ulong)t * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask) {
|
||||
uint gid = (uint)get_global_id(0);
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r2 = r1 * r1 + r2; // 1 mad
|
||||
r2 = r3 * r2 + r2; // 2 mad
|
||||
r3 = r3 ^ r5; // 3 xor
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r5 = r5 ^ ds[r7 & mask]; // 5 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl
|
||||
r1 = mul_hi(r1, r5); // 8 mulhi
|
||||
r6 = rotr_var(r6, r3); // 9 rotr
|
||||
r3 = r3 | r4; // 10 or
|
||||
r4 = r4 ^ ds[r6 & mask]; // 11 load
|
||||
r0 = mul_hi(r0, r4); // 12 mulhi
|
||||
r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add
|
||||
r0 = r0 ^ ds[r5 & mask]; // 14 load
|
||||
r2 = r2 - r4; // 15 sub
|
||||
r2 = r2 ^ ds[r7 & mask]; // 16 load
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl
|
||||
r5 = r5 * r0; // 19 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl
|
||||
r6 = mul_hi(r6, r2); // 22 mulhi
|
||||
r6 = r6 ^ ds[r3 & mask]; // 23 load
|
||||
r5 = r5 * r0; // 24 mul
|
||||
r5 = rotl_imm(r5, 19u); // 25 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl
|
||||
r0 = r0 ^ r5; // 27 xor
|
||||
r0 = r0 ^ r4; // 28 xor
|
||||
r3 = r3 - r0; // 29 sub
|
||||
r5 = r5 * r1; // 30 mul
|
||||
r7 = r7 ^ ds[r3 & mask]; // 31 load
|
||||
r1 = r1 ^ ds[r2 & mask]; // 32 load
|
||||
r5 = r5 ^ r6; // 33 xor
|
||||
r5 = r5 ^ ds[r0 & mask]; // 34 load
|
||||
r0 = mul_hi(r0, r5); // 35 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl
|
||||
r7 = r7 ^ ds[r6 & mask]; // 37 load
|
||||
r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl
|
||||
r2 = r2 ^ r5; // 40 xor
|
||||
r3 = r6 * r3 + r3; // 41 mad
|
||||
r6 = r6 - r7; // 42 sub
|
||||
r7 = r7 ^ r0; // 43 xor
|
||||
r1 = r1 ^ ds[r7 & mask]; // 44 load
|
||||
r2 = r2 * r3; // 45 mul
|
||||
r1 = mul_hi(r1, r5); // 46 mulhi
|
||||
r4 = r4 - r3; // 47 sub
|
||||
r2 = rotr_var(r2, r6); // 48 rotr
|
||||
r3 = r3 ^ ds[r4 & mask]; // 49 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add
|
||||
r0 = r0 * r2; // 51 mul
|
||||
r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add
|
||||
r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add
|
||||
r7 = rotl_imm(r7, 14u); // 54 rotl
|
||||
r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add
|
||||
r6 = r6 ^ ds[r7 & mask]; // 56 load
|
||||
r1 = rotr_var(r1, r5); // 57 rotr
|
||||
r5 = r5 ^ ds[r3 & mask]; // 58 load
|
||||
r6 = r6 ^ ds[r1 & mask]; // 59 load
|
||||
r3 = r5 * r0 + r3; // 60 mad
|
||||
r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add
|
||||
r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add
|
||||
r5 = rotl_imm(r5, 19u); // 63 rotl
|
||||
// latency-shadow block (Counter ASIC 3.0 item 8): 256 ALU instructions x 27 passes after instruction 63, no load
|
||||
for (uint sh = 0u; sh < 27u; ++sh) {
|
||||
r5 = r5 + r2 + ((((sel >> 18u) & 1u) != 0u) ? 0x8bc12da9u : 0x92199f99u); // s0 add
|
||||
r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x63079e5au : 0x8d72d3adu); // s1 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 2u); r6 = r6 ^ t_; } // s2 shfl
|
||||
r4 = r4 - r2; // s3 sub
|
||||
r7 = r7 + r0 + ((((sel >> 31u) & 1u) != 0u) ? 0x5d4c7a60u : 0xb21b4babu); // s4 add
|
||||
r0 = rotl_imm(r0, 11u); // s5 rotl
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // s6 add
|
||||
r7 = r7 ^ r1; // s7 xor
|
||||
r1 = r6 * r5 + r1; // s8 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 4u); r6 = r6 ^ t_; } // s9 shfl
|
||||
r1 = r2 * r2 + r1; // s10 mad
|
||||
r5 = r0 * r3 + r5; // s11 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 4u); r2 = r2 ^ t_; } // s12 shfl
|
||||
r1 = r1 + r5 + ((((sel >> 25u) & 1u) != 0u) ? 0xbc48c63eu : 0xbdf8f9a5u); // s13 add
|
||||
r1 = rotl_imm(r1, 29u); // s14 rotl
|
||||
r1 = r1 - r4; // s15 sub
|
||||
r7 = r7 | r1; // s16 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r2 = r2 ^ t_; } // s17 shfl
|
||||
r7 = r7 ^ r4; // s18 xor
|
||||
r6 = r6 * r1; // s19 mul
|
||||
r5 = r6 * r0 + r5; // s20 mad
|
||||
r3 = r3 - r1; // s21 sub
|
||||
r6 = r6 * r0; // s22 mul
|
||||
r2 = r2 + r0 + ((((sel >> 1u) & 1u) != 0u) ? 0x96e8f127u : 0x45c37cecu); // s23 add
|
||||
r6 = r6 - r4; // s24 sub
|
||||
r7 = r3 * r4 + r7; // s25 mad
|
||||
r3 = rotl_imm(r3, 9u); // s26 rotl
|
||||
r2 = r2 - r1; // s27 sub
|
||||
r6 = mul_hi(r6, r0); // s28 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r2 = r2 ^ t_; } // s29 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r1 = r1 ^ t_; } // s30 shfl
|
||||
r1 = r1 + r6 + ((((sel >> 1u) & 1u) != 0u) ? 0xb1cdb2abu : 0x37985632u); // s31 add
|
||||
r0 = rotr_var(r0, r1); // s32 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // s33 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r0 = r0 ^ t_; } // s34 shfl
|
||||
r7 = r7 * r0; // s35 mul
|
||||
r3 = r3 + r1 + ((((sel >> 0u) & 1u) != 0u) ? 0x856e0180u : 0x804e777eu); // s36 add
|
||||
r1 = r1 ^ r6; // s37 xor
|
||||
r2 = r2 * r7; // s38 mul
|
||||
r6 = r6 + r5 + ((((sel >> 2u) & 1u) != 0u) ? 0xe9bd0cc4u : 0x0b06c8a7u); // s39 add
|
||||
r5 = r5 * r7; // s40 mul
|
||||
r2 = mul_hi(r2, r3); // s41 mulhi
|
||||
r2 = r2 ^ r0; // s42 xor
|
||||
r0 = rotl_imm(r0, 4u); // s43 rotl
|
||||
r4 = mul_hi(r4, r3); // s44 mulhi
|
||||
r6 = r6 + r7 + ((((sel >> 12u) & 1u) != 0u) ? 0xcdcb63f6u : 0xee822e17u); // s45 add
|
||||
r3 = r2 * r0 + r3; // s46 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r4 = r4 ^ t_; } // s47 shfl
|
||||
r6 = mul_hi(r6, r5); // s48 mulhi
|
||||
r0 = r0 - r2; // s49 sub
|
||||
r3 = r3 - r5; // s50 sub
|
||||
r1 = rotr_var(r1, r4); // s51 rotr
|
||||
r6 = r7 * r7 + r6; // s52 mad
|
||||
r5 = r5 ^ r3; // s53 xor
|
||||
r1 = r1 - r0; // s54 sub
|
||||
r5 = r5 - r6; // s55 sub
|
||||
r3 = r3 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0xd4758987u : 0x3dfad1b6u); // s56 add
|
||||
r4 = r4 + r1 + ((((sel >> 0u) & 1u) != 0u) ? 0x0602d6beu : 0xfca75bc2u); // s57 add
|
||||
r5 = mul_hi(r5, r6); // s58 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 2u); r2 = r2 ^ t_; } // s59 shfl
|
||||
r2 = rotl_imm(r2, 9u); // s60 rotl
|
||||
r4 = r4 | r6; // s61 or
|
||||
r6 = rotr_var(r6, r4); // s62 rotr
|
||||
r2 = r2 * r4; // s63 mul
|
||||
r0 = r0 ^ r5; // s64 xor
|
||||
r2 = r2 ^ r7; // s65 xor
|
||||
r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x8596687au : 0xd71c02ffu); // s66 add
|
||||
r1 = rotl_imm(r1, 21u); // s67 rotl
|
||||
r3 = r3 * r2; // s68 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 2u); r7 = r7 ^ t_; } // s69 shfl
|
||||
r3 = r3 * r2; // s70 mul
|
||||
r0 = r2 * r5 + r0; // s71 mad
|
||||
r6 = r6 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0x08ac5733u : 0x3c6fe15du); // s72 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r3 = r3 ^ t_; } // s73 shfl
|
||||
r3 = r3 * r6; // s74 mul
|
||||
r6 = r6 ^ r7; // s75 xor
|
||||
r3 = r3 | r0; // s76 or
|
||||
r2 = r2 + r4 + ((((sel >> 1u) & 1u) != 0u) ? 0xc100b495u : 0x295d5faeu); // s77 add
|
||||
r3 = mul_hi(r3, r7); // s78 mulhi
|
||||
r4 = r4 | r1; // s79 or
|
||||
r4 = rotr_var(r4, r3); // s80 rotr
|
||||
r4 = r4 + r3 + ((((sel >> 15u) & 1u) != 0u) ? 0x5cc59530u : 0x274a9221u); // s81 add
|
||||
r7 = r7 + r3 + ((((sel >> 5u) & 1u) != 0u) ? 0x141479ecu : 0xc8651f8eu); // s82 add
|
||||
r3 = rotr_var(r3, r6); // s83 rotr
|
||||
r2 = r4 * r7 + r2; // s84 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r2 = r2 ^ t_; } // s85 shfl
|
||||
r3 = r3 ^ r2; // s86 xor
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 2u); r5 = r5 ^ t_; } // s87 shfl
|
||||
r0 = r0 ^ r4; // s88 xor
|
||||
r3 = r3 + r2 + ((((sel >> 18u) & 1u) != 0u) ? 0x883c0c92u : 0x53f915c2u); // s89 add
|
||||
r7 = rotr_var(r7, r5); // s90 rotr
|
||||
r3 = r3 + r1 + ((((sel >> 31u) & 1u) != 0u) ? 0x3cc5bd20u : 0xe2be00a5u); // s91 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 1u); r7 = r7 ^ t_; } // s92 shfl
|
||||
r6 = rotr_var(r6, r2); // s93 rotr
|
||||
r7 = rotl_imm(r7, 14u); // s94 rotl
|
||||
r4 = r5 * r5 + r4; // s95 mad
|
||||
r2 = r4 * r5 + r2; // s96 mad
|
||||
r1 = r1 ^ r0; // s97 xor
|
||||
r5 = r5 * r4; // s98 mul
|
||||
r2 = r2 - r0; // s99 sub
|
||||
r7 = rotl_imm(r7, 30u); // s100 rotl
|
||||
r5 = r5 + r6 + ((((sel >> 17u) & 1u) != 0u) ? 0xbfd646cbu : 0x423fd9c9u); // s101 add
|
||||
r7 = r7 + r6 + ((((sel >> 11u) & 1u) != 0u) ? 0x19b74a43u : 0x8d3c011du); // s102 add
|
||||
r5 = rotl_imm(r5, 6u); // s103 rotl
|
||||
r0 = r0 * r4; // s104 mul
|
||||
r0 = r0 | r5; // s105 or
|
||||
r0 = r0 + r1 + ((((sel >> 15u) & 1u) != 0u) ? 0x7dcb7e18u : 0x8ce14721u); // s106 add
|
||||
r0 = rotl_imm(r0, 15u); // s107 rotl
|
||||
r4 = r4 - r2; // s108 sub
|
||||
r2 = mul_hi(r2, r0); // s109 mulhi
|
||||
r1 = mul_hi(r1, r0); // s110 mulhi
|
||||
r0 = r0 + r2 + ((((sel >> 28u) & 1u) != 0u) ? 0x3c179ad8u : 0x2d9fc6b9u); // s111 add
|
||||
r2 = mul_hi(r2, r0); // s112 mulhi
|
||||
r6 = r6 - r3; // s113 sub
|
||||
r7 = r6 * r3 + r7; // s114 mad
|
||||
r5 = rotr_var(r5, r1); // s115 rotr
|
||||
r6 = rotr_var(r6, r4); // s116 rotr
|
||||
r2 = r2 + r1 + ((((sel >> 22u) & 1u) != 0u) ? 0x47359729u : 0x502138e2u); // s117 add
|
||||
r3 = r2 * r0 + r3; // s118 mad
|
||||
r1 = r1 * r3; // s119 mul
|
||||
r2 = r0 * r4 + r2; // s120 mad
|
||||
r0 = r0 * r3; // s121 mul
|
||||
r2 = mul_hi(r2, r0); // s122 mulhi
|
||||
r5 = r5 * r6; // s123 mul
|
||||
r4 = r2 * r4 + r4; // s124 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 2u); r4 = r4 ^ t_; } // s125 shfl
|
||||
r2 = r2 ^ r0; // s126 xor
|
||||
r1 = rotl_imm(r1, 7u); // s127 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r7 = r7 ^ t_; } // s128 shfl
|
||||
r6 = rotl_imm(r6, 17u); // s129 rotl
|
||||
r2 = r2 + r5 + ((((sel >> 15u) & 1u) != 0u) ? 0x6c57e4e7u : 0x33dff776u); // s130 add
|
||||
r0 = r0 | r3; // s131 or
|
||||
r5 = rotr_var(r5, r0); // s132 rotr
|
||||
r2 = r2 + r3 + ((((sel >> 30u) & 1u) != 0u) ? 0xe8212c0cu : 0x5db25b34u); // s133 add
|
||||
r4 = r4 ^ r3; // s134 xor
|
||||
r6 = r6 ^ r1; // s135 xor
|
||||
r1 = r1 - r7; // s136 sub
|
||||
r2 = r6 * r2 + r2; // s137 mad
|
||||
r0 = mul_hi(r0, r5); // s138 mulhi
|
||||
r2 = r2 + r7 + ((((sel >> 10u) & 1u) != 0u) ? 0xd44ff710u : 0x218d4090u); // s139 add
|
||||
r1 = rotr_var(r1, r4); // s140 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r3 = r3 ^ t_; } // s141 shfl
|
||||
r7 = r6 * r2 + r7; // s142 mad
|
||||
r2 = r2 * r3; // s143 mul
|
||||
r7 = r7 + r4 + ((((sel >> 29u) & 1u) != 0u) ? 0xb98942fau : 0xf4b1a8deu); // s144 add
|
||||
r7 = rotl_imm(r7, 15u); // s145 rotl
|
||||
r7 = r7 ^ r5; // s146 xor
|
||||
r4 = r7 * r4 + r4; // s147 mad
|
||||
r6 = rotr_var(r6, r5); // s148 rotr
|
||||
r1 = r2 * r4 + r1; // s149 mad
|
||||
r1 = r1 + r7 + ((((sel >> 17u) & 1u) != 0u) ? 0x0d48ba42u : 0x2bef10f2u); // s150 add
|
||||
r5 = r5 ^ r4; // s151 xor
|
||||
r7 = r7 + r0 + ((((sel >> 30u) & 1u) != 0u) ? 0xc98dea9cu : 0xdc5cc080u); // s152 add
|
||||
r4 = r4 * r6; // s153 mul
|
||||
r0 = r0 * r2; // s154 mul
|
||||
r6 = r6 ^ r5; // s155 xor
|
||||
r4 = rotr_var(r4, r2); // s156 rotr
|
||||
r1 = rotl_imm(r1, 11u); // s157 rotl
|
||||
r5 = r5 + r0 + ((((sel >> 11u) & 1u) != 0u) ? 0x6c752dcbu : 0x18197438u); // s158 add
|
||||
r4 = r1 * r5 + r4; // s159 mad
|
||||
r4 = rotl_imm(r4, 26u); // s160 rotl
|
||||
r3 = r0 * r7 + r3; // s161 mad
|
||||
r3 = rotr_var(r3, r1); // s162 rotr
|
||||
r4 = r4 + r0 + ((((sel >> 3u) & 1u) != 0u) ? 0x8e481727u : 0xf1c46574u); // s163 add
|
||||
r0 = mul_hi(r0, r4); // s164 mulhi
|
||||
r2 = r2 + r6 + ((((sel >> 20u) & 1u) != 0u) ? 0x550ab406u : 0xac578137u); // s165 add
|
||||
r0 = rotl_imm(r0, 13u); // s166 rotl
|
||||
r3 = r3 + r1 + ((((sel >> 14u) & 1u) != 0u) ? 0xaebb5966u : 0xa33e6706u); // s167 add
|
||||
r3 = r3 | r5; // s168 or
|
||||
r6 = rotr_var(r6, r2); // s169 rotr
|
||||
r4 = r4 ^ r6; // s170 xor
|
||||
r6 = r6 - r1; // s171 sub
|
||||
r7 = rotl_imm(r7, 22u); // s172 rotl
|
||||
r5 = rotl_imm(r5, 15u); // s173 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 8u); r7 = r7 ^ t_; } // s174 shfl
|
||||
r0 = r0 ^ r5; // s175 xor
|
||||
r7 = rotl_imm(r7, 6u); // s176 rotl
|
||||
r7 = r7 - r0; // s177 sub
|
||||
r3 = rotl_imm(r3, 30u); // s178 rotl
|
||||
r7 = r6 * r1 + r7; // s179 mad
|
||||
r6 = rotl_imm(r6, 9u); // s180 rotl
|
||||
r2 = r2 ^ r4; // s181 xor
|
||||
r2 = r2 ^ r7; // s182 xor
|
||||
r7 = r7 ^ r2; // s183 xor
|
||||
r1 = r1 + r2 + ((((sel >> 21u) & 1u) != 0u) ? 0x5bb7550fu : 0xd94d55acu); // s184 add
|
||||
r3 = r3 | r5; // s185 or
|
||||
r6 = mul_hi(r6, r3); // s186 mulhi
|
||||
r4 = r4 | r0; // s187 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 2u); r7 = r7 ^ t_; } // s188 shfl
|
||||
r6 = r6 + r5 + ((((sel >> 6u) & 1u) != 0u) ? 0x2a354e2du : 0x89e747feu); // s189 add
|
||||
r1 = r1 ^ r4; // s190 xor
|
||||
r7 = mul_hi(r7, r4); // s191 mulhi
|
||||
r2 = rotl_imm(r2, 26u); // s192 rotl
|
||||
r5 = rotl_imm(r5, 8u); // s193 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r4 = r4 ^ t_; } // s194 shfl
|
||||
r4 = r4 ^ r5; // s195 xor
|
||||
r1 = r1 * r3; // s196 mul
|
||||
r5 = r5 + r2 + ((((sel >> 7u) & 1u) != 0u) ? 0x10cfdc71u : 0xea03e8e7u); // s197 add
|
||||
r7 = r7 + r5 + ((((sel >> 15u) & 1u) != 0u) ? 0x66e148d3u : 0xa7ee0102u); // s198 add
|
||||
r5 = r5 - r2; // s199 sub
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 2u); r5 = r5 ^ t_; } // s200 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 8u); r7 = r7 ^ t_; } // s201 shfl
|
||||
r4 = rotr_var(r4, r5); // s202 rotr
|
||||
r7 = r7 ^ r5; // s203 xor
|
||||
r7 = r7 ^ r0; // s204 xor
|
||||
r6 = r6 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0x8558b619u : 0x8379a4deu); // s205 add
|
||||
r4 = rotl_imm(r4, 13u); // s206 rotl
|
||||
r1 = mul_hi(r1, r3); // s207 mulhi
|
||||
r1 = r4 * r4 + r1; // s208 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 4u); r2 = r2 ^ t_; } // s209 shfl
|
||||
r7 = r7 + r0 + ((((sel >> 11u) & 1u) != 0u) ? 0xf4049c4cu : 0xaa8cb14eu); // s210 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 16u); r7 = r7 ^ t_; } // s211 shfl
|
||||
r1 = r1 ^ r0; // s212 xor
|
||||
r7 = rotl_imm(r7, 3u); // s213 rotl
|
||||
r4 = rotr_var(r4, r7); // s214 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 16u); r3 = r3 ^ t_; } // s215 shfl
|
||||
r5 = r5 | r1; // s216 or
|
||||
r1 = r1 - r2; // s217 sub
|
||||
r6 = r6 - r5; // s218 sub
|
||||
r6 = rotl_imm(r6, 4u); // s219 rotl
|
||||
r2 = mul_hi(r2, r0); // s220 mulhi
|
||||
r2 = r2 | r0; // s221 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r5 = r5 ^ t_; } // s222 shfl
|
||||
r5 = mul_hi(r5, r6); // s223 mulhi
|
||||
r0 = r0 - r6; // s224 sub
|
||||
r7 = rotl_imm(r7, 23u); // s225 rotl
|
||||
r4 = r4 | r2; // s226 or
|
||||
r2 = r2 * r4; // s227 mul
|
||||
r3 = rotl_imm(r3, 12u); // s228 rotl
|
||||
r0 = rotr_var(r0, r4); // s229 rotr
|
||||
r0 = r0 + r6 + ((((sel >> 16u) & 1u) != 0u) ? 0x2a7fecb2u : 0x1d176220u); // s230 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r1 = r1 ^ t_; } // s231 shfl
|
||||
r2 = r2 * r1; // s232 mul
|
||||
r7 = r0 * r1 + r7; // s233 mad
|
||||
r5 = rotl_imm(r5, 22u); // s234 rotl
|
||||
r4 = rotr_var(r4, r6); // s235 rotr
|
||||
r0 = r5 * r1 + r0; // s236 mad
|
||||
r6 = r6 ^ r4; // s237 xor
|
||||
r4 = r4 ^ r6; // s238 xor
|
||||
r6 = rotl_imm(r6, 18u); // s239 rotl
|
||||
r4 = r4 + r7 + ((((sel >> 25u) & 1u) != 0u) ? 0xadce39f3u : 0x17dafb4du); // s240 add
|
||||
r0 = r0 - r7; // s241 sub
|
||||
r1 = rotr_var(r1, r6); // s242 rotr
|
||||
r3 = r3 + r0 + ((((sel >> 29u) & 1u) != 0u) ? 0x0e7033b6u : 0xf2f77d26u); // s243 add
|
||||
r2 = r7 * r4 + r2; // s244 mad
|
||||
r4 = r0 * r6 + r4; // s245 mad
|
||||
r5 = r5 * r7; // s246 mul
|
||||
r3 = r3 + r5 + ((((sel >> 31u) & 1u) != 0u) ? 0x7b2ff6b7u : 0xfed2da4eu); // s247 add
|
||||
r0 = r0 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0xb5fad7b1u : 0x03b2891cu); // s248 add
|
||||
r5 = mul_hi(r5, r4); // s249 mulhi
|
||||
r5 = r5 + r2 + ((((sel >> 11u) & 1u) != 0u) ? 0xa7e71c2fu : 0x042cd6e3u); // s250 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 1u); r6 = r6 ^ t_; } // s251 shfl
|
||||
r6 = r6 ^ r1; // s252 xor
|
||||
r2 = r2 * r5; // s253 mul
|
||||
r0 = r0 + r6 + ((((sel >> 16u) & 1u) != 0u) ? 0xb83d78deu : 0x784a302bu); // s254 add
|
||||
r2 = r2 - r4; // s255 sub
|
||||
}
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
|
||||
#if IGNEUM_EXCHANGE != 0
|
||||
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
|
||||
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
|
||||
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
|
||||
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
|
||||
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
|
||||
}
|
||||
#endif
|
||||
423
proto-cuda/packs-ca3-v5/v4-genesis/kernel.cu
Normal file
423
proto-cuda/packs-ca3-v5/v4-genesis/kernel.cu
Normal file
|
|
@ -0,0 +1,423 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal).
|
||||
// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC.
|
||||
#include <cuda_runtime.h>
|
||||
#include <cstdint>
|
||||
#include "program.h"
|
||||
#include "memhard.h"
|
||||
|
||||
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31.
|
||||
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
|
||||
// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x.
|
||||
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) {
|
||||
uint32_t x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item.
|
||||
// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host.
|
||||
__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) {
|
||||
uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
|
||||
uint32_t t = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (t < nItems) {
|
||||
uint32_t s[16];
|
||||
mh_item(cache, t, s);
|
||||
uint32_t* d = ds + (size_t)t * 16u;
|
||||
for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
|
||||
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
|
||||
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
|
||||
__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask) {
|
||||
uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
uint32_t nonce = baseNonce + gid;
|
||||
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint32_t x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint32_t x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint32_t x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint32_t x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint32_t x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint32_t x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint32_t x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint32_t x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint32_t it = 0u; it < 8u; ++it) {
|
||||
uint32_t sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r2 = r1 * r1 + r2; // 1 mad
|
||||
r2 = r3 * r2 + r2; // 2 mad
|
||||
r3 = r3 ^ r5; // 3 xor
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r5 = r5 ^ ds[r7 & mask]; // 5 load
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 6 shfl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 8); // 7 shfl
|
||||
r1 = __umulhi(r1, r5); // 8 mulhi
|
||||
r6 = rotr_var(r6, r3); // 9 rotr
|
||||
r3 = r3 | r4; // 10 or
|
||||
r4 = r4 ^ ds[r6 & mask]; // 11 load
|
||||
r0 = __umulhi(r0, r4); // 12 mulhi
|
||||
r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add
|
||||
r0 = r0 ^ ds[r5 & mask]; // 14 load
|
||||
r2 = r2 - r4; // 15 sub
|
||||
r2 = r2 ^ ds[r7 & mask]; // 16 load
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 18 shfl
|
||||
r5 = r5 * r0; // 19 mul
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 20 shfl
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 16); // 21 shfl
|
||||
r6 = __umulhi(r6, r2); // 22 mulhi
|
||||
r6 = r6 ^ ds[r3 & mask]; // 23 load
|
||||
r5 = r5 * r0; // 24 mul
|
||||
r5 = rotl_imm(r5, 19u); // 25 rotl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r6, 2); // 26 shfl
|
||||
r0 = r0 ^ r5; // 27 xor
|
||||
r0 = r0 ^ r4; // 28 xor
|
||||
r3 = r3 - r0; // 29 sub
|
||||
r5 = r5 * r1; // 30 mul
|
||||
r7 = r7 ^ ds[r3 & mask]; // 31 load
|
||||
r1 = r1 ^ ds[r2 & mask]; // 32 load
|
||||
r5 = r5 ^ r6; // 33 xor
|
||||
r5 = r5 ^ ds[r0 & mask]; // 34 load
|
||||
r0 = __umulhi(r0, r5); // 35 mulhi
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // 36 shfl
|
||||
r7 = r7 ^ ds[r6 & mask]; // 37 load
|
||||
r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 4); // 39 shfl
|
||||
r2 = r2 ^ r5; // 40 xor
|
||||
r3 = r6 * r3 + r3; // 41 mad
|
||||
r6 = r6 - r7; // 42 sub
|
||||
r7 = r7 ^ r0; // 43 xor
|
||||
r1 = r1 ^ ds[r7 & mask]; // 44 load
|
||||
r2 = r2 * r3; // 45 mul
|
||||
r1 = __umulhi(r1, r5); // 46 mulhi
|
||||
r4 = r4 - r3; // 47 sub
|
||||
r2 = rotr_var(r2, r6); // 48 rotr
|
||||
r3 = r3 ^ ds[r4 & mask]; // 49 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add
|
||||
r0 = r0 * r2; // 51 mul
|
||||
r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add
|
||||
r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add
|
||||
r7 = rotl_imm(r7, 14u); // 54 rotl
|
||||
r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add
|
||||
r6 = r6 ^ ds[r7 & mask]; // 56 load
|
||||
r1 = rotr_var(r1, r5); // 57 rotr
|
||||
r5 = r5 ^ ds[r3 & mask]; // 58 load
|
||||
r6 = r6 ^ ds[r1 & mask]; // 59 load
|
||||
r3 = r5 * r0 + r3; // 60 mad
|
||||
r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add
|
||||
r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add
|
||||
r5 = rotl_imm(r5, 19u); // 63 rotl
|
||||
// latency-shadow block (Counter ASIC 3.0 item 8): 256 ALU instructions x 27 passes after instruction 63, no load
|
||||
for (uint32_t sh = 0u; sh < 27u; ++sh) {
|
||||
r5 = r5 + r2 + ((((sel >> 18u) & 1u) != 0u) ? 0x8bc12da9u : 0x92199f99u); // s0 add
|
||||
r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x63079e5au : 0x8d72d3adu); // s1 add
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 2); // s2 shfl
|
||||
r4 = r4 - r2; // s3 sub
|
||||
r7 = r7 + r0 + ((((sel >> 31u) & 1u) != 0u) ? 0x5d4c7a60u : 0xb21b4babu); // s4 add
|
||||
r0 = rotl_imm(r0, 11u); // s5 rotl
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // s6 add
|
||||
r7 = r7 ^ r1; // s7 xor
|
||||
r1 = r6 * r5 + r1; // s8 mad
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r1, 4); // s9 shfl
|
||||
r1 = r2 * r2 + r1; // s10 mad
|
||||
r5 = r0 * r3 + r5; // s11 mad
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r6, 4); // s12 shfl
|
||||
r1 = r1 + r5 + ((((sel >> 25u) & 1u) != 0u) ? 0xbc48c63eu : 0xbdf8f9a5u); // s13 add
|
||||
r1 = rotl_imm(r1, 29u); // s14 rotl
|
||||
r1 = r1 - r4; // s15 sub
|
||||
r7 = r7 | r1; // s16 or
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // s17 shfl
|
||||
r7 = r7 ^ r4; // s18 xor
|
||||
r6 = r6 * r1; // s19 mul
|
||||
r5 = r6 * r0 + r5; // s20 mad
|
||||
r3 = r3 - r1; // s21 sub
|
||||
r6 = r6 * r0; // s22 mul
|
||||
r2 = r2 + r0 + ((((sel >> 1u) & 1u) != 0u) ? 0x96e8f127u : 0x45c37cecu); // s23 add
|
||||
r6 = r6 - r4; // s24 sub
|
||||
r7 = r3 * r4 + r7; // s25 mad
|
||||
r3 = rotl_imm(r3, 9u); // s26 rotl
|
||||
r2 = r2 - r1; // s27 sub
|
||||
r6 = __umulhi(r6, r0); // s28 mulhi
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // s29 shfl
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r6, 1); // s30 shfl
|
||||
r1 = r1 + r6 + ((((sel >> 1u) & 1u) != 0u) ? 0xb1cdb2abu : 0x37985632u); // s31 add
|
||||
r0 = rotr_var(r0, r1); // s32 rotr
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // s33 shfl
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // s34 shfl
|
||||
r7 = r7 * r0; // s35 mul
|
||||
r3 = r3 + r1 + ((((sel >> 0u) & 1u) != 0u) ? 0x856e0180u : 0x804e777eu); // s36 add
|
||||
r1 = r1 ^ r6; // s37 xor
|
||||
r2 = r2 * r7; // s38 mul
|
||||
r6 = r6 + r5 + ((((sel >> 2u) & 1u) != 0u) ? 0xe9bd0cc4u : 0x0b06c8a7u); // s39 add
|
||||
r5 = r5 * r7; // s40 mul
|
||||
r2 = __umulhi(r2, r3); // s41 mulhi
|
||||
r2 = r2 ^ r0; // s42 xor
|
||||
r0 = rotl_imm(r0, 4u); // s43 rotl
|
||||
r4 = __umulhi(r4, r3); // s44 mulhi
|
||||
r6 = r6 + r7 + ((((sel >> 12u) & 1u) != 0u) ? 0xcdcb63f6u : 0xee822e17u); // s45 add
|
||||
r3 = r2 * r0 + r3; // s46 mad
|
||||
r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r3, 8); // s47 shfl
|
||||
r6 = __umulhi(r6, r5); // s48 mulhi
|
||||
r0 = r0 - r2; // s49 sub
|
||||
r3 = r3 - r5; // s50 sub
|
||||
r1 = rotr_var(r1, r4); // s51 rotr
|
||||
r6 = r7 * r7 + r6; // s52 mad
|
||||
r5 = r5 ^ r3; // s53 xor
|
||||
r1 = r1 - r0; // s54 sub
|
||||
r5 = r5 - r6; // s55 sub
|
||||
r3 = r3 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0xd4758987u : 0x3dfad1b6u); // s56 add
|
||||
r4 = r4 + r1 + ((((sel >> 0u) & 1u) != 0u) ? 0x0602d6beu : 0xfca75bc2u); // s57 add
|
||||
r5 = __umulhi(r5, r6); // s58 mulhi
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r1, 2); // s59 shfl
|
||||
r2 = rotl_imm(r2, 9u); // s60 rotl
|
||||
r4 = r4 | r6; // s61 or
|
||||
r6 = rotr_var(r6, r4); // s62 rotr
|
||||
r2 = r2 * r4; // s63 mul
|
||||
r0 = r0 ^ r5; // s64 xor
|
||||
r2 = r2 ^ r7; // s65 xor
|
||||
r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x8596687au : 0xd71c02ffu); // s66 add
|
||||
r1 = rotl_imm(r1, 21u); // s67 rotl
|
||||
r3 = r3 * r2; // s68 mul
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r5, 2); // s69 shfl
|
||||
r3 = r3 * r2; // s70 mul
|
||||
r0 = r2 * r5 + r0; // s71 mad
|
||||
r6 = r6 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0x08ac5733u : 0x3c6fe15du); // s72 add
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r2, 8); // s73 shfl
|
||||
r3 = r3 * r6; // s74 mul
|
||||
r6 = r6 ^ r7; // s75 xor
|
||||
r3 = r3 | r0; // s76 or
|
||||
r2 = r2 + r4 + ((((sel >> 1u) & 1u) != 0u) ? 0xc100b495u : 0x295d5faeu); // s77 add
|
||||
r3 = __umulhi(r3, r7); // s78 mulhi
|
||||
r4 = r4 | r1; // s79 or
|
||||
r4 = rotr_var(r4, r3); // s80 rotr
|
||||
r4 = r4 + r3 + ((((sel >> 15u) & 1u) != 0u) ? 0x5cc59530u : 0x274a9221u); // s81 add
|
||||
r7 = r7 + r3 + ((((sel >> 5u) & 1u) != 0u) ? 0x141479ecu : 0xc8651f8eu); // s82 add
|
||||
r3 = rotr_var(r3, r6); // s83 rotr
|
||||
r2 = r4 * r7 + r2; // s84 mad
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r5, 16); // s85 shfl
|
||||
r3 = r3 ^ r2; // s86 xor
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r7, 2); // s87 shfl
|
||||
r0 = r0 ^ r4; // s88 xor
|
||||
r3 = r3 + r2 + ((((sel >> 18u) & 1u) != 0u) ? 0x883c0c92u : 0x53f915c2u); // s89 add
|
||||
r7 = rotr_var(r7, r5); // s90 rotr
|
||||
r3 = r3 + r1 + ((((sel >> 31u) & 1u) != 0u) ? 0x3cc5bd20u : 0xe2be00a5u); // s91 add
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r2, 1); // s92 shfl
|
||||
r6 = rotr_var(r6, r2); // s93 rotr
|
||||
r7 = rotl_imm(r7, 14u); // s94 rotl
|
||||
r4 = r5 * r5 + r4; // s95 mad
|
||||
r2 = r4 * r5 + r2; // s96 mad
|
||||
r1 = r1 ^ r0; // s97 xor
|
||||
r5 = r5 * r4; // s98 mul
|
||||
r2 = r2 - r0; // s99 sub
|
||||
r7 = rotl_imm(r7, 30u); // s100 rotl
|
||||
r5 = r5 + r6 + ((((sel >> 17u) & 1u) != 0u) ? 0xbfd646cbu : 0x423fd9c9u); // s101 add
|
||||
r7 = r7 + r6 + ((((sel >> 11u) & 1u) != 0u) ? 0x19b74a43u : 0x8d3c011du); // s102 add
|
||||
r5 = rotl_imm(r5, 6u); // s103 rotl
|
||||
r0 = r0 * r4; // s104 mul
|
||||
r0 = r0 | r5; // s105 or
|
||||
r0 = r0 + r1 + ((((sel >> 15u) & 1u) != 0u) ? 0x7dcb7e18u : 0x8ce14721u); // s106 add
|
||||
r0 = rotl_imm(r0, 15u); // s107 rotl
|
||||
r4 = r4 - r2; // s108 sub
|
||||
r2 = __umulhi(r2, r0); // s109 mulhi
|
||||
r1 = __umulhi(r1, r0); // s110 mulhi
|
||||
r0 = r0 + r2 + ((((sel >> 28u) & 1u) != 0u) ? 0x3c179ad8u : 0x2d9fc6b9u); // s111 add
|
||||
r2 = __umulhi(r2, r0); // s112 mulhi
|
||||
r6 = r6 - r3; // s113 sub
|
||||
r7 = r6 * r3 + r7; // s114 mad
|
||||
r5 = rotr_var(r5, r1); // s115 rotr
|
||||
r6 = rotr_var(r6, r4); // s116 rotr
|
||||
r2 = r2 + r1 + ((((sel >> 22u) & 1u) != 0u) ? 0x47359729u : 0x502138e2u); // s117 add
|
||||
r3 = r2 * r0 + r3; // s118 mad
|
||||
r1 = r1 * r3; // s119 mul
|
||||
r2 = r0 * r4 + r2; // s120 mad
|
||||
r0 = r0 * r3; // s121 mul
|
||||
r2 = __umulhi(r2, r0); // s122 mulhi
|
||||
r5 = r5 * r6; // s123 mul
|
||||
r4 = r2 * r4 + r4; // s124 mad
|
||||
r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r2, 2); // s125 shfl
|
||||
r2 = r2 ^ r0; // s126 xor
|
||||
r1 = rotl_imm(r1, 7u); // s127 rotl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // s128 shfl
|
||||
r6 = rotl_imm(r6, 17u); // s129 rotl
|
||||
r2 = r2 + r5 + ((((sel >> 15u) & 1u) != 0u) ? 0x6c57e4e7u : 0x33dff776u); // s130 add
|
||||
r0 = r0 | r3; // s131 or
|
||||
r5 = rotr_var(r5, r0); // s132 rotr
|
||||
r2 = r2 + r3 + ((((sel >> 30u) & 1u) != 0u) ? 0xe8212c0cu : 0x5db25b34u); // s133 add
|
||||
r4 = r4 ^ r3; // s134 xor
|
||||
r6 = r6 ^ r1; // s135 xor
|
||||
r1 = r1 - r7; // s136 sub
|
||||
r2 = r6 * r2 + r2; // s137 mad
|
||||
r0 = __umulhi(r0, r5); // s138 mulhi
|
||||
r2 = r2 + r7 + ((((sel >> 10u) & 1u) != 0u) ? 0xd44ff710u : 0x218d4090u); // s139 add
|
||||
r1 = rotr_var(r1, r4); // s140 rotr
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r6, 1); // s141 shfl
|
||||
r7 = r6 * r2 + r7; // s142 mad
|
||||
r2 = r2 * r3; // s143 mul
|
||||
r7 = r7 + r4 + ((((sel >> 29u) & 1u) != 0u) ? 0xb98942fau : 0xf4b1a8deu); // s144 add
|
||||
r7 = rotl_imm(r7, 15u); // s145 rotl
|
||||
r7 = r7 ^ r5; // s146 xor
|
||||
r4 = r7 * r4 + r4; // s147 mad
|
||||
r6 = rotr_var(r6, r5); // s148 rotr
|
||||
r1 = r2 * r4 + r1; // s149 mad
|
||||
r1 = r1 + r7 + ((((sel >> 17u) & 1u) != 0u) ? 0x0d48ba42u : 0x2bef10f2u); // s150 add
|
||||
r5 = r5 ^ r4; // s151 xor
|
||||
r7 = r7 + r0 + ((((sel >> 30u) & 1u) != 0u) ? 0xc98dea9cu : 0xdc5cc080u); // s152 add
|
||||
r4 = r4 * r6; // s153 mul
|
||||
r0 = r0 * r2; // s154 mul
|
||||
r6 = r6 ^ r5; // s155 xor
|
||||
r4 = rotr_var(r4, r2); // s156 rotr
|
||||
r1 = rotl_imm(r1, 11u); // s157 rotl
|
||||
r5 = r5 + r0 + ((((sel >> 11u) & 1u) != 0u) ? 0x6c752dcbu : 0x18197438u); // s158 add
|
||||
r4 = r1 * r5 + r4; // s159 mad
|
||||
r4 = rotl_imm(r4, 26u); // s160 rotl
|
||||
r3 = r0 * r7 + r3; // s161 mad
|
||||
r3 = rotr_var(r3, r1); // s162 rotr
|
||||
r4 = r4 + r0 + ((((sel >> 3u) & 1u) != 0u) ? 0x8e481727u : 0xf1c46574u); // s163 add
|
||||
r0 = __umulhi(r0, r4); // s164 mulhi
|
||||
r2 = r2 + r6 + ((((sel >> 20u) & 1u) != 0u) ? 0x550ab406u : 0xac578137u); // s165 add
|
||||
r0 = rotl_imm(r0, 13u); // s166 rotl
|
||||
r3 = r3 + r1 + ((((sel >> 14u) & 1u) != 0u) ? 0xaebb5966u : 0xa33e6706u); // s167 add
|
||||
r3 = r3 | r5; // s168 or
|
||||
r6 = rotr_var(r6, r2); // s169 rotr
|
||||
r4 = r4 ^ r6; // s170 xor
|
||||
r6 = r6 - r1; // s171 sub
|
||||
r7 = rotl_imm(r7, 22u); // s172 rotl
|
||||
r5 = rotl_imm(r5, 15u); // s173 rotl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r0, 8); // s174 shfl
|
||||
r0 = r0 ^ r5; // s175 xor
|
||||
r7 = rotl_imm(r7, 6u); // s176 rotl
|
||||
r7 = r7 - r0; // s177 sub
|
||||
r3 = rotl_imm(r3, 30u); // s178 rotl
|
||||
r7 = r6 * r1 + r7; // s179 mad
|
||||
r6 = rotl_imm(r6, 9u); // s180 rotl
|
||||
r2 = r2 ^ r4; // s181 xor
|
||||
r2 = r2 ^ r7; // s182 xor
|
||||
r7 = r7 ^ r2; // s183 xor
|
||||
r1 = r1 + r2 + ((((sel >> 21u) & 1u) != 0u) ? 0x5bb7550fu : 0xd94d55acu); // s184 add
|
||||
r3 = r3 | r5; // s185 or
|
||||
r6 = __umulhi(r6, r3); // s186 mulhi
|
||||
r4 = r4 | r0; // s187 or
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r1, 2); // s188 shfl
|
||||
r6 = r6 + r5 + ((((sel >> 6u) & 1u) != 0u) ? 0x2a354e2du : 0x89e747feu); // s189 add
|
||||
r1 = r1 ^ r4; // s190 xor
|
||||
r7 = __umulhi(r7, r4); // s191 mulhi
|
||||
r2 = rotl_imm(r2, 26u); // s192 rotl
|
||||
r5 = rotl_imm(r5, 8u); // s193 rotl
|
||||
r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r5, 16); // s194 shfl
|
||||
r4 = r4 ^ r5; // s195 xor
|
||||
r1 = r1 * r3; // s196 mul
|
||||
r5 = r5 + r2 + ((((sel >> 7u) & 1u) != 0u) ? 0x10cfdc71u : 0xea03e8e7u); // s197 add
|
||||
r7 = r7 + r5 + ((((sel >> 15u) & 1u) != 0u) ? 0x66e148d3u : 0xa7ee0102u); // s198 add
|
||||
r5 = r5 - r2; // s199 sub
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r1, 2); // s200 shfl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r1, 8); // s201 shfl
|
||||
r4 = rotr_var(r4, r5); // s202 rotr
|
||||
r7 = r7 ^ r5; // s203 xor
|
||||
r7 = r7 ^ r0; // s204 xor
|
||||
r6 = r6 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0x8558b619u : 0x8379a4deu); // s205 add
|
||||
r4 = rotl_imm(r4, 13u); // s206 rotl
|
||||
r1 = __umulhi(r1, r3); // s207 mulhi
|
||||
r1 = r4 * r4 + r1; // s208 mad
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r0, 4); // s209 shfl
|
||||
r7 = r7 + r0 + ((((sel >> 11u) & 1u) != 0u) ? 0xf4049c4cu : 0xaa8cb14eu); // s210 add
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r1, 16); // s211 shfl
|
||||
r1 = r1 ^ r0; // s212 xor
|
||||
r7 = rotl_imm(r7, 3u); // s213 rotl
|
||||
r4 = rotr_var(r4, r7); // s214 rotr
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r2, 16); // s215 shfl
|
||||
r5 = r5 | r1; // s216 or
|
||||
r1 = r1 - r2; // s217 sub
|
||||
r6 = r6 - r5; // s218 sub
|
||||
r6 = rotl_imm(r6, 4u); // s219 rotl
|
||||
r2 = __umulhi(r2, r0); // s220 mulhi
|
||||
r2 = r2 | r0; // s221 or
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r2, 8); // s222 shfl
|
||||
r5 = __umulhi(r5, r6); // s223 mulhi
|
||||
r0 = r0 - r6; // s224 sub
|
||||
r7 = rotl_imm(r7, 23u); // s225 rotl
|
||||
r4 = r4 | r2; // s226 or
|
||||
r2 = r2 * r4; // s227 mul
|
||||
r3 = rotl_imm(r3, 12u); // s228 rotl
|
||||
r0 = rotr_var(r0, r4); // s229 rotr
|
||||
r0 = r0 + r6 + ((((sel >> 16u) & 1u) != 0u) ? 0x2a7fecb2u : 0x1d176220u); // s230 add
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // s231 shfl
|
||||
r2 = r2 * r1; // s232 mul
|
||||
r7 = r0 * r1 + r7; // s233 mad
|
||||
r5 = rotl_imm(r5, 22u); // s234 rotl
|
||||
r4 = rotr_var(r4, r6); // s235 rotr
|
||||
r0 = r5 * r1 + r0; // s236 mad
|
||||
r6 = r6 ^ r4; // s237 xor
|
||||
r4 = r4 ^ r6; // s238 xor
|
||||
r6 = rotl_imm(r6, 18u); // s239 rotl
|
||||
r4 = r4 + r7 + ((((sel >> 25u) & 1u) != 0u) ? 0xadce39f3u : 0x17dafb4du); // s240 add
|
||||
r0 = r0 - r7; // s241 sub
|
||||
r1 = rotr_var(r1, r6); // s242 rotr
|
||||
r3 = r3 + r0 + ((((sel >> 29u) & 1u) != 0u) ? 0x0e7033b6u : 0xf2f77d26u); // s243 add
|
||||
r2 = r7 * r4 + r2; // s244 mad
|
||||
r4 = r0 * r6 + r4; // s245 mad
|
||||
r5 = r5 * r7; // s246 mul
|
||||
r3 = r3 + r5 + ((((sel >> 31u) & 1u) != 0u) ? 0x7b2ff6b7u : 0xfed2da4eu); // s247 add
|
||||
r0 = r0 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0xb5fad7b1u : 0x03b2891cu); // s248 add
|
||||
r5 = __umulhi(r5, r4); // s249 mulhi
|
||||
r5 = r5 + r2 + ((((sel >> 11u) & 1u) != 0u) ? 0xa7e71c2fu : 0x042cd6e3u); // s250 add
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r1, 1); // s251 shfl
|
||||
r6 = r6 ^ r1; // s252 xor
|
||||
r2 = r2 * r5; // s253 mul
|
||||
r0 = r0 + r6 + ((((sel >> 16u) & 1u) != 0u) ? 0xb83d78deu : 0x784a302bu); // s254 add
|
||||
r2 = r2 - r4; // s255 sub
|
||||
}
|
||||
}
|
||||
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
|
||||
}
|
||||
|
||||
// Host-side launch wrappers. Declared in program.h, called from host.cu.
|
||||
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) {
|
||||
if (nSegments == 0u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 256u;
|
||||
uint32_t grid = (nSegments + block - 1u) / block;
|
||||
igneum_cache_fill<<<grid, block>>>(cache, nSegments);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
|
||||
if (nItems == 0u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 256u;
|
||||
uint32_t grid = (nItems + block - 1u) / block;
|
||||
igneum_build<<<grid, block>>>(ds, cache, nItems);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
uint32_t nonces, uint32_t blockWarps) {
|
||||
if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 32u * blockWarps;
|
||||
if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;
|
||||
igneum_hash<<<nonces / block, block>>>(ds, out, baseNonce, mask);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
|
||||
cudaFuncAttributes attr;
|
||||
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash);
|
||||
if (e != cudaSuccess) return e;
|
||||
*numRegs = attr.numRegs;
|
||||
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0);
|
||||
}
|
||||
891
proto-cuda/packs-ca3-v5/v4-genesis/kernel_bound.cl
Normal file
891
proto-cuda/packs-ca3-v5/v4-genesis/kernel_bound.cl
Normal file
|
|
@ -0,0 +1,891 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
|
||||
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
|
||||
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
|
||||
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
|
||||
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
|
||||
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
|
||||
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
|
||||
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
|
||||
#ifndef IGNEUM_GROUP
|
||||
#define IGNEUM_GROUP 32
|
||||
#endif
|
||||
#ifndef IGNEUM_EXCHANGE
|
||||
#define IGNEUM_EXCHANGE 0
|
||||
#endif
|
||||
#ifdef __OPENCL_VERSION__
|
||||
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
|
||||
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#ifdef cl_khr_subgroups
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
|
||||
#endif
|
||||
#ifdef cl_khr_subgroup_shuffle
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
|
||||
#endif
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
|
||||
#endif
|
||||
#else
|
||||
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
|
||||
#include "emu_opencl.h"
|
||||
#endif
|
||||
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#else
|
||||
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
|
||||
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
|
||||
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
|
||||
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
|
||||
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
|
||||
#endif
|
||||
|
||||
static inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
|
||||
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
|
||||
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
|
||||
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
|
||||
static inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of 8 x seed-parameterised mixer + one 64-byte cache read, then 8 x final mixer (class v3, mixer multiplier 8,
|
||||
// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
static inline void mh_chacha_block(const uint* x, uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
static inline void mh_cache_segment(__global uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
static inline void mh_mixer(uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of 8 x mixer + cache line s[0] & mask; 8 x final mixer.
|
||||
static inline void mh_item(__global const uint* cache, uint t, uint* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
for (uint j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 8u + j + 1u));
|
||||
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
for (uint j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (64u + j + 1u));
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
|
||||
// The same constants as memhard.h in this pack (one emitter, three dialects).
|
||||
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
|
||||
uint seg = (uint)get_global_id(0);
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
|
||||
uint t = (uint)get_global_id(0);
|
||||
if (t < nItems) {
|
||||
uint s[16];
|
||||
mh_item(cache, t, s);
|
||||
__global uint* d = ds + ((ulong)t * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask) {
|
||||
uint gid = (uint)get_global_id(0);
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r2 = r1 * r1 + r2; // 1 mad
|
||||
r2 = r3 * r2 + r2; // 2 mad
|
||||
r3 = r3 ^ r5; // 3 xor
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r5 = r5 ^ ds[r7 & mask]; // 5 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl
|
||||
r1 = mul_hi(r1, r5); // 8 mulhi
|
||||
r6 = rotr_var(r6, r3); // 9 rotr
|
||||
r3 = r3 | r4; // 10 or
|
||||
r4 = r4 ^ ds[r6 & mask]; // 11 load
|
||||
r0 = mul_hi(r0, r4); // 12 mulhi
|
||||
r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add
|
||||
r0 = r0 ^ ds[r5 & mask]; // 14 load
|
||||
r2 = r2 - r4; // 15 sub
|
||||
r2 = r2 ^ ds[r7 & mask]; // 16 load
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl
|
||||
r5 = r5 * r0; // 19 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl
|
||||
r6 = mul_hi(r6, r2); // 22 mulhi
|
||||
r6 = r6 ^ ds[r3 & mask]; // 23 load
|
||||
r5 = r5 * r0; // 24 mul
|
||||
r5 = rotl_imm(r5, 19u); // 25 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl
|
||||
r0 = r0 ^ r5; // 27 xor
|
||||
r0 = r0 ^ r4; // 28 xor
|
||||
r3 = r3 - r0; // 29 sub
|
||||
r5 = r5 * r1; // 30 mul
|
||||
r7 = r7 ^ ds[r3 & mask]; // 31 load
|
||||
r1 = r1 ^ ds[r2 & mask]; // 32 load
|
||||
r5 = r5 ^ r6; // 33 xor
|
||||
r5 = r5 ^ ds[r0 & mask]; // 34 load
|
||||
r0 = mul_hi(r0, r5); // 35 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl
|
||||
r7 = r7 ^ ds[r6 & mask]; // 37 load
|
||||
r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl
|
||||
r2 = r2 ^ r5; // 40 xor
|
||||
r3 = r6 * r3 + r3; // 41 mad
|
||||
r6 = r6 - r7; // 42 sub
|
||||
r7 = r7 ^ r0; // 43 xor
|
||||
r1 = r1 ^ ds[r7 & mask]; // 44 load
|
||||
r2 = r2 * r3; // 45 mul
|
||||
r1 = mul_hi(r1, r5); // 46 mulhi
|
||||
r4 = r4 - r3; // 47 sub
|
||||
r2 = rotr_var(r2, r6); // 48 rotr
|
||||
r3 = r3 ^ ds[r4 & mask]; // 49 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add
|
||||
r0 = r0 * r2; // 51 mul
|
||||
r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add
|
||||
r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add
|
||||
r7 = rotl_imm(r7, 14u); // 54 rotl
|
||||
r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add
|
||||
r6 = r6 ^ ds[r7 & mask]; // 56 load
|
||||
r1 = rotr_var(r1, r5); // 57 rotr
|
||||
r5 = r5 ^ ds[r3 & mask]; // 58 load
|
||||
r6 = r6 ^ ds[r1 & mask]; // 59 load
|
||||
r3 = r5 * r0 + r3; // 60 mad
|
||||
r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add
|
||||
r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add
|
||||
r5 = rotl_imm(r5, 19u); // 63 rotl
|
||||
// latency-shadow block (Counter ASIC 3.0 item 8): 256 ALU instructions x 27 passes after instruction 63, no load
|
||||
for (uint sh = 0u; sh < 27u; ++sh) {
|
||||
r5 = r5 + r2 + ((((sel >> 18u) & 1u) != 0u) ? 0x8bc12da9u : 0x92199f99u); // s0 add
|
||||
r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x63079e5au : 0x8d72d3adu); // s1 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 2u); r6 = r6 ^ t_; } // s2 shfl
|
||||
r4 = r4 - r2; // s3 sub
|
||||
r7 = r7 + r0 + ((((sel >> 31u) & 1u) != 0u) ? 0x5d4c7a60u : 0xb21b4babu); // s4 add
|
||||
r0 = rotl_imm(r0, 11u); // s5 rotl
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // s6 add
|
||||
r7 = r7 ^ r1; // s7 xor
|
||||
r1 = r6 * r5 + r1; // s8 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 4u); r6 = r6 ^ t_; } // s9 shfl
|
||||
r1 = r2 * r2 + r1; // s10 mad
|
||||
r5 = r0 * r3 + r5; // s11 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 4u); r2 = r2 ^ t_; } // s12 shfl
|
||||
r1 = r1 + r5 + ((((sel >> 25u) & 1u) != 0u) ? 0xbc48c63eu : 0xbdf8f9a5u); // s13 add
|
||||
r1 = rotl_imm(r1, 29u); // s14 rotl
|
||||
r1 = r1 - r4; // s15 sub
|
||||
r7 = r7 | r1; // s16 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r2 = r2 ^ t_; } // s17 shfl
|
||||
r7 = r7 ^ r4; // s18 xor
|
||||
r6 = r6 * r1; // s19 mul
|
||||
r5 = r6 * r0 + r5; // s20 mad
|
||||
r3 = r3 - r1; // s21 sub
|
||||
r6 = r6 * r0; // s22 mul
|
||||
r2 = r2 + r0 + ((((sel >> 1u) & 1u) != 0u) ? 0x96e8f127u : 0x45c37cecu); // s23 add
|
||||
r6 = r6 - r4; // s24 sub
|
||||
r7 = r3 * r4 + r7; // s25 mad
|
||||
r3 = rotl_imm(r3, 9u); // s26 rotl
|
||||
r2 = r2 - r1; // s27 sub
|
||||
r6 = mul_hi(r6, r0); // s28 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r2 = r2 ^ t_; } // s29 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r1 = r1 ^ t_; } // s30 shfl
|
||||
r1 = r1 + r6 + ((((sel >> 1u) & 1u) != 0u) ? 0xb1cdb2abu : 0x37985632u); // s31 add
|
||||
r0 = rotr_var(r0, r1); // s32 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // s33 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r0 = r0 ^ t_; } // s34 shfl
|
||||
r7 = r7 * r0; // s35 mul
|
||||
r3 = r3 + r1 + ((((sel >> 0u) & 1u) != 0u) ? 0x856e0180u : 0x804e777eu); // s36 add
|
||||
r1 = r1 ^ r6; // s37 xor
|
||||
r2 = r2 * r7; // s38 mul
|
||||
r6 = r6 + r5 + ((((sel >> 2u) & 1u) != 0u) ? 0xe9bd0cc4u : 0x0b06c8a7u); // s39 add
|
||||
r5 = r5 * r7; // s40 mul
|
||||
r2 = mul_hi(r2, r3); // s41 mulhi
|
||||
r2 = r2 ^ r0; // s42 xor
|
||||
r0 = rotl_imm(r0, 4u); // s43 rotl
|
||||
r4 = mul_hi(r4, r3); // s44 mulhi
|
||||
r6 = r6 + r7 + ((((sel >> 12u) & 1u) != 0u) ? 0xcdcb63f6u : 0xee822e17u); // s45 add
|
||||
r3 = r2 * r0 + r3; // s46 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r4 = r4 ^ t_; } // s47 shfl
|
||||
r6 = mul_hi(r6, r5); // s48 mulhi
|
||||
r0 = r0 - r2; // s49 sub
|
||||
r3 = r3 - r5; // s50 sub
|
||||
r1 = rotr_var(r1, r4); // s51 rotr
|
||||
r6 = r7 * r7 + r6; // s52 mad
|
||||
r5 = r5 ^ r3; // s53 xor
|
||||
r1 = r1 - r0; // s54 sub
|
||||
r5 = r5 - r6; // s55 sub
|
||||
r3 = r3 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0xd4758987u : 0x3dfad1b6u); // s56 add
|
||||
r4 = r4 + r1 + ((((sel >> 0u) & 1u) != 0u) ? 0x0602d6beu : 0xfca75bc2u); // s57 add
|
||||
r5 = mul_hi(r5, r6); // s58 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 2u); r2 = r2 ^ t_; } // s59 shfl
|
||||
r2 = rotl_imm(r2, 9u); // s60 rotl
|
||||
r4 = r4 | r6; // s61 or
|
||||
r6 = rotr_var(r6, r4); // s62 rotr
|
||||
r2 = r2 * r4; // s63 mul
|
||||
r0 = r0 ^ r5; // s64 xor
|
||||
r2 = r2 ^ r7; // s65 xor
|
||||
r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x8596687au : 0xd71c02ffu); // s66 add
|
||||
r1 = rotl_imm(r1, 21u); // s67 rotl
|
||||
r3 = r3 * r2; // s68 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 2u); r7 = r7 ^ t_; } // s69 shfl
|
||||
r3 = r3 * r2; // s70 mul
|
||||
r0 = r2 * r5 + r0; // s71 mad
|
||||
r6 = r6 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0x08ac5733u : 0x3c6fe15du); // s72 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r3 = r3 ^ t_; } // s73 shfl
|
||||
r3 = r3 * r6; // s74 mul
|
||||
r6 = r6 ^ r7; // s75 xor
|
||||
r3 = r3 | r0; // s76 or
|
||||
r2 = r2 + r4 + ((((sel >> 1u) & 1u) != 0u) ? 0xc100b495u : 0x295d5faeu); // s77 add
|
||||
r3 = mul_hi(r3, r7); // s78 mulhi
|
||||
r4 = r4 | r1; // s79 or
|
||||
r4 = rotr_var(r4, r3); // s80 rotr
|
||||
r4 = r4 + r3 + ((((sel >> 15u) & 1u) != 0u) ? 0x5cc59530u : 0x274a9221u); // s81 add
|
||||
r7 = r7 + r3 + ((((sel >> 5u) & 1u) != 0u) ? 0x141479ecu : 0xc8651f8eu); // s82 add
|
||||
r3 = rotr_var(r3, r6); // s83 rotr
|
||||
r2 = r4 * r7 + r2; // s84 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r2 = r2 ^ t_; } // s85 shfl
|
||||
r3 = r3 ^ r2; // s86 xor
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 2u); r5 = r5 ^ t_; } // s87 shfl
|
||||
r0 = r0 ^ r4; // s88 xor
|
||||
r3 = r3 + r2 + ((((sel >> 18u) & 1u) != 0u) ? 0x883c0c92u : 0x53f915c2u); // s89 add
|
||||
r7 = rotr_var(r7, r5); // s90 rotr
|
||||
r3 = r3 + r1 + ((((sel >> 31u) & 1u) != 0u) ? 0x3cc5bd20u : 0xe2be00a5u); // s91 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 1u); r7 = r7 ^ t_; } // s92 shfl
|
||||
r6 = rotr_var(r6, r2); // s93 rotr
|
||||
r7 = rotl_imm(r7, 14u); // s94 rotl
|
||||
r4 = r5 * r5 + r4; // s95 mad
|
||||
r2 = r4 * r5 + r2; // s96 mad
|
||||
r1 = r1 ^ r0; // s97 xor
|
||||
r5 = r5 * r4; // s98 mul
|
||||
r2 = r2 - r0; // s99 sub
|
||||
r7 = rotl_imm(r7, 30u); // s100 rotl
|
||||
r5 = r5 + r6 + ((((sel >> 17u) & 1u) != 0u) ? 0xbfd646cbu : 0x423fd9c9u); // s101 add
|
||||
r7 = r7 + r6 + ((((sel >> 11u) & 1u) != 0u) ? 0x19b74a43u : 0x8d3c011du); // s102 add
|
||||
r5 = rotl_imm(r5, 6u); // s103 rotl
|
||||
r0 = r0 * r4; // s104 mul
|
||||
r0 = r0 | r5; // s105 or
|
||||
r0 = r0 + r1 + ((((sel >> 15u) & 1u) != 0u) ? 0x7dcb7e18u : 0x8ce14721u); // s106 add
|
||||
r0 = rotl_imm(r0, 15u); // s107 rotl
|
||||
r4 = r4 - r2; // s108 sub
|
||||
r2 = mul_hi(r2, r0); // s109 mulhi
|
||||
r1 = mul_hi(r1, r0); // s110 mulhi
|
||||
r0 = r0 + r2 + ((((sel >> 28u) & 1u) != 0u) ? 0x3c179ad8u : 0x2d9fc6b9u); // s111 add
|
||||
r2 = mul_hi(r2, r0); // s112 mulhi
|
||||
r6 = r6 - r3; // s113 sub
|
||||
r7 = r6 * r3 + r7; // s114 mad
|
||||
r5 = rotr_var(r5, r1); // s115 rotr
|
||||
r6 = rotr_var(r6, r4); // s116 rotr
|
||||
r2 = r2 + r1 + ((((sel >> 22u) & 1u) != 0u) ? 0x47359729u : 0x502138e2u); // s117 add
|
||||
r3 = r2 * r0 + r3; // s118 mad
|
||||
r1 = r1 * r3; // s119 mul
|
||||
r2 = r0 * r4 + r2; // s120 mad
|
||||
r0 = r0 * r3; // s121 mul
|
||||
r2 = mul_hi(r2, r0); // s122 mulhi
|
||||
r5 = r5 * r6; // s123 mul
|
||||
r4 = r2 * r4 + r4; // s124 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 2u); r4 = r4 ^ t_; } // s125 shfl
|
||||
r2 = r2 ^ r0; // s126 xor
|
||||
r1 = rotl_imm(r1, 7u); // s127 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r7 = r7 ^ t_; } // s128 shfl
|
||||
r6 = rotl_imm(r6, 17u); // s129 rotl
|
||||
r2 = r2 + r5 + ((((sel >> 15u) & 1u) != 0u) ? 0x6c57e4e7u : 0x33dff776u); // s130 add
|
||||
r0 = r0 | r3; // s131 or
|
||||
r5 = rotr_var(r5, r0); // s132 rotr
|
||||
r2 = r2 + r3 + ((((sel >> 30u) & 1u) != 0u) ? 0xe8212c0cu : 0x5db25b34u); // s133 add
|
||||
r4 = r4 ^ r3; // s134 xor
|
||||
r6 = r6 ^ r1; // s135 xor
|
||||
r1 = r1 - r7; // s136 sub
|
||||
r2 = r6 * r2 + r2; // s137 mad
|
||||
r0 = mul_hi(r0, r5); // s138 mulhi
|
||||
r2 = r2 + r7 + ((((sel >> 10u) & 1u) != 0u) ? 0xd44ff710u : 0x218d4090u); // s139 add
|
||||
r1 = rotr_var(r1, r4); // s140 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r3 = r3 ^ t_; } // s141 shfl
|
||||
r7 = r6 * r2 + r7; // s142 mad
|
||||
r2 = r2 * r3; // s143 mul
|
||||
r7 = r7 + r4 + ((((sel >> 29u) & 1u) != 0u) ? 0xb98942fau : 0xf4b1a8deu); // s144 add
|
||||
r7 = rotl_imm(r7, 15u); // s145 rotl
|
||||
r7 = r7 ^ r5; // s146 xor
|
||||
r4 = r7 * r4 + r4; // s147 mad
|
||||
r6 = rotr_var(r6, r5); // s148 rotr
|
||||
r1 = r2 * r4 + r1; // s149 mad
|
||||
r1 = r1 + r7 + ((((sel >> 17u) & 1u) != 0u) ? 0x0d48ba42u : 0x2bef10f2u); // s150 add
|
||||
r5 = r5 ^ r4; // s151 xor
|
||||
r7 = r7 + r0 + ((((sel >> 30u) & 1u) != 0u) ? 0xc98dea9cu : 0xdc5cc080u); // s152 add
|
||||
r4 = r4 * r6; // s153 mul
|
||||
r0 = r0 * r2; // s154 mul
|
||||
r6 = r6 ^ r5; // s155 xor
|
||||
r4 = rotr_var(r4, r2); // s156 rotr
|
||||
r1 = rotl_imm(r1, 11u); // s157 rotl
|
||||
r5 = r5 + r0 + ((((sel >> 11u) & 1u) != 0u) ? 0x6c752dcbu : 0x18197438u); // s158 add
|
||||
r4 = r1 * r5 + r4; // s159 mad
|
||||
r4 = rotl_imm(r4, 26u); // s160 rotl
|
||||
r3 = r0 * r7 + r3; // s161 mad
|
||||
r3 = rotr_var(r3, r1); // s162 rotr
|
||||
r4 = r4 + r0 + ((((sel >> 3u) & 1u) != 0u) ? 0x8e481727u : 0xf1c46574u); // s163 add
|
||||
r0 = mul_hi(r0, r4); // s164 mulhi
|
||||
r2 = r2 + r6 + ((((sel >> 20u) & 1u) != 0u) ? 0x550ab406u : 0xac578137u); // s165 add
|
||||
r0 = rotl_imm(r0, 13u); // s166 rotl
|
||||
r3 = r3 + r1 + ((((sel >> 14u) & 1u) != 0u) ? 0xaebb5966u : 0xa33e6706u); // s167 add
|
||||
r3 = r3 | r5; // s168 or
|
||||
r6 = rotr_var(r6, r2); // s169 rotr
|
||||
r4 = r4 ^ r6; // s170 xor
|
||||
r6 = r6 - r1; // s171 sub
|
||||
r7 = rotl_imm(r7, 22u); // s172 rotl
|
||||
r5 = rotl_imm(r5, 15u); // s173 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 8u); r7 = r7 ^ t_; } // s174 shfl
|
||||
r0 = r0 ^ r5; // s175 xor
|
||||
r7 = rotl_imm(r7, 6u); // s176 rotl
|
||||
r7 = r7 - r0; // s177 sub
|
||||
r3 = rotl_imm(r3, 30u); // s178 rotl
|
||||
r7 = r6 * r1 + r7; // s179 mad
|
||||
r6 = rotl_imm(r6, 9u); // s180 rotl
|
||||
r2 = r2 ^ r4; // s181 xor
|
||||
r2 = r2 ^ r7; // s182 xor
|
||||
r7 = r7 ^ r2; // s183 xor
|
||||
r1 = r1 + r2 + ((((sel >> 21u) & 1u) != 0u) ? 0x5bb7550fu : 0xd94d55acu); // s184 add
|
||||
r3 = r3 | r5; // s185 or
|
||||
r6 = mul_hi(r6, r3); // s186 mulhi
|
||||
r4 = r4 | r0; // s187 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 2u); r7 = r7 ^ t_; } // s188 shfl
|
||||
r6 = r6 + r5 + ((((sel >> 6u) & 1u) != 0u) ? 0x2a354e2du : 0x89e747feu); // s189 add
|
||||
r1 = r1 ^ r4; // s190 xor
|
||||
r7 = mul_hi(r7, r4); // s191 mulhi
|
||||
r2 = rotl_imm(r2, 26u); // s192 rotl
|
||||
r5 = rotl_imm(r5, 8u); // s193 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r4 = r4 ^ t_; } // s194 shfl
|
||||
r4 = r4 ^ r5; // s195 xor
|
||||
r1 = r1 * r3; // s196 mul
|
||||
r5 = r5 + r2 + ((((sel >> 7u) & 1u) != 0u) ? 0x10cfdc71u : 0xea03e8e7u); // s197 add
|
||||
r7 = r7 + r5 + ((((sel >> 15u) & 1u) != 0u) ? 0x66e148d3u : 0xa7ee0102u); // s198 add
|
||||
r5 = r5 - r2; // s199 sub
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 2u); r5 = r5 ^ t_; } // s200 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 8u); r7 = r7 ^ t_; } // s201 shfl
|
||||
r4 = rotr_var(r4, r5); // s202 rotr
|
||||
r7 = r7 ^ r5; // s203 xor
|
||||
r7 = r7 ^ r0; // s204 xor
|
||||
r6 = r6 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0x8558b619u : 0x8379a4deu); // s205 add
|
||||
r4 = rotl_imm(r4, 13u); // s206 rotl
|
||||
r1 = mul_hi(r1, r3); // s207 mulhi
|
||||
r1 = r4 * r4 + r1; // s208 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 4u); r2 = r2 ^ t_; } // s209 shfl
|
||||
r7 = r7 + r0 + ((((sel >> 11u) & 1u) != 0u) ? 0xf4049c4cu : 0xaa8cb14eu); // s210 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 16u); r7 = r7 ^ t_; } // s211 shfl
|
||||
r1 = r1 ^ r0; // s212 xor
|
||||
r7 = rotl_imm(r7, 3u); // s213 rotl
|
||||
r4 = rotr_var(r4, r7); // s214 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 16u); r3 = r3 ^ t_; } // s215 shfl
|
||||
r5 = r5 | r1; // s216 or
|
||||
r1 = r1 - r2; // s217 sub
|
||||
r6 = r6 - r5; // s218 sub
|
||||
r6 = rotl_imm(r6, 4u); // s219 rotl
|
||||
r2 = mul_hi(r2, r0); // s220 mulhi
|
||||
r2 = r2 | r0; // s221 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r5 = r5 ^ t_; } // s222 shfl
|
||||
r5 = mul_hi(r5, r6); // s223 mulhi
|
||||
r0 = r0 - r6; // s224 sub
|
||||
r7 = rotl_imm(r7, 23u); // s225 rotl
|
||||
r4 = r4 | r2; // s226 or
|
||||
r2 = r2 * r4; // s227 mul
|
||||
r3 = rotl_imm(r3, 12u); // s228 rotl
|
||||
r0 = rotr_var(r0, r4); // s229 rotr
|
||||
r0 = r0 + r6 + ((((sel >> 16u) & 1u) != 0u) ? 0x2a7fecb2u : 0x1d176220u); // s230 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r1 = r1 ^ t_; } // s231 shfl
|
||||
r2 = r2 * r1; // s232 mul
|
||||
r7 = r0 * r1 + r7; // s233 mad
|
||||
r5 = rotl_imm(r5, 22u); // s234 rotl
|
||||
r4 = rotr_var(r4, r6); // s235 rotr
|
||||
r0 = r5 * r1 + r0; // s236 mad
|
||||
r6 = r6 ^ r4; // s237 xor
|
||||
r4 = r4 ^ r6; // s238 xor
|
||||
r6 = rotl_imm(r6, 18u); // s239 rotl
|
||||
r4 = r4 + r7 + ((((sel >> 25u) & 1u) != 0u) ? 0xadce39f3u : 0x17dafb4du); // s240 add
|
||||
r0 = r0 - r7; // s241 sub
|
||||
r1 = rotr_var(r1, r6); // s242 rotr
|
||||
r3 = r3 + r0 + ((((sel >> 29u) & 1u) != 0u) ? 0x0e7033b6u : 0xf2f77d26u); // s243 add
|
||||
r2 = r7 * r4 + r2; // s244 mad
|
||||
r4 = r0 * r6 + r4; // s245 mad
|
||||
r5 = r5 * r7; // s246 mul
|
||||
r3 = r3 + r5 + ((((sel >> 31u) & 1u) != 0u) ? 0x7b2ff6b7u : 0xfed2da4eu); // s247 add
|
||||
r0 = r0 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0xb5fad7b1u : 0x03b2891cu); // s248 add
|
||||
r5 = mul_hi(r5, r4); // s249 mulhi
|
||||
r5 = r5 + r2 + ((((sel >> 11u) & 1u) != 0u) ? 0xa7e71c2fu : 0x042cd6e3u); // s250 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 1u); r6 = r6 ^ t_; } // s251 shfl
|
||||
r6 = r6 ^ r1; // s252 xor
|
||||
r2 = r2 * r5; // s253 mul
|
||||
r0 = r0 + r6 + ((((sel >> 16u) & 1u) != 0u) ? 0xb83d78deu : 0x784a302bu); // s254 add
|
||||
r2 = r2 - r4; // s255 sub
|
||||
}
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
|
||||
#if IGNEUM_EXCHANGE != 0
|
||||
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
|
||||
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
|
||||
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
|
||||
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
|
||||
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
|
||||
}
|
||||
#endif
|
||||
|
||||
// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash.
|
||||
IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw) {
|
||||
uint gid = (uint)get_global_id(0);
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; }
|
||||
{ uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; }
|
||||
{ uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; }
|
||||
{ uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; }
|
||||
{ uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; }
|
||||
{ uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; }
|
||||
{ uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; }
|
||||
{ uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r2 = r1 * r1 + r2; // 1 mad
|
||||
r2 = r3 * r2 + r2; // 2 mad
|
||||
r3 = r3 ^ r5; // 3 xor
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r5 = r5 ^ ds[r7 & mask]; // 5 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl
|
||||
r1 = mul_hi(r1, r5); // 8 mulhi
|
||||
r6 = rotr_var(r6, r3); // 9 rotr
|
||||
r3 = r3 | r4; // 10 or
|
||||
r4 = r4 ^ ds[r6 & mask]; // 11 load
|
||||
r0 = mul_hi(r0, r4); // 12 mulhi
|
||||
r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add
|
||||
r0 = r0 ^ ds[r5 & mask]; // 14 load
|
||||
r2 = r2 - r4; // 15 sub
|
||||
r2 = r2 ^ ds[r7 & mask]; // 16 load
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl
|
||||
r5 = r5 * r0; // 19 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl
|
||||
r6 = mul_hi(r6, r2); // 22 mulhi
|
||||
r6 = r6 ^ ds[r3 & mask]; // 23 load
|
||||
r5 = r5 * r0; // 24 mul
|
||||
r5 = rotl_imm(r5, 19u); // 25 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl
|
||||
r0 = r0 ^ r5; // 27 xor
|
||||
r0 = r0 ^ r4; // 28 xor
|
||||
r3 = r3 - r0; // 29 sub
|
||||
r5 = r5 * r1; // 30 mul
|
||||
r7 = r7 ^ ds[r3 & mask]; // 31 load
|
||||
r1 = r1 ^ ds[r2 & mask]; // 32 load
|
||||
r5 = r5 ^ r6; // 33 xor
|
||||
r5 = r5 ^ ds[r0 & mask]; // 34 load
|
||||
r0 = mul_hi(r0, r5); // 35 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl
|
||||
r7 = r7 ^ ds[r6 & mask]; // 37 load
|
||||
r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl
|
||||
r2 = r2 ^ r5; // 40 xor
|
||||
r3 = r6 * r3 + r3; // 41 mad
|
||||
r6 = r6 - r7; // 42 sub
|
||||
r7 = r7 ^ r0; // 43 xor
|
||||
r1 = r1 ^ ds[r7 & mask]; // 44 load
|
||||
r2 = r2 * r3; // 45 mul
|
||||
r1 = mul_hi(r1, r5); // 46 mulhi
|
||||
r4 = r4 - r3; // 47 sub
|
||||
r2 = rotr_var(r2, r6); // 48 rotr
|
||||
r3 = r3 ^ ds[r4 & mask]; // 49 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add
|
||||
r0 = r0 * r2; // 51 mul
|
||||
r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add
|
||||
r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add
|
||||
r7 = rotl_imm(r7, 14u); // 54 rotl
|
||||
r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add
|
||||
r6 = r6 ^ ds[r7 & mask]; // 56 load
|
||||
r1 = rotr_var(r1, r5); // 57 rotr
|
||||
r5 = r5 ^ ds[r3 & mask]; // 58 load
|
||||
r6 = r6 ^ ds[r1 & mask]; // 59 load
|
||||
r3 = r5 * r0 + r3; // 60 mad
|
||||
r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add
|
||||
r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add
|
||||
r5 = rotl_imm(r5, 19u); // 63 rotl
|
||||
// latency-shadow block (Counter ASIC 3.0 item 8): 256 ALU instructions x 27 passes after instruction 63, no load
|
||||
for (uint sh = 0u; sh < 27u; ++sh) {
|
||||
r5 = r5 + r2 + ((((sel >> 18u) & 1u) != 0u) ? 0x8bc12da9u : 0x92199f99u); // s0 add
|
||||
r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x63079e5au : 0x8d72d3adu); // s1 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 2u); r6 = r6 ^ t_; } // s2 shfl
|
||||
r4 = r4 - r2; // s3 sub
|
||||
r7 = r7 + r0 + ((((sel >> 31u) & 1u) != 0u) ? 0x5d4c7a60u : 0xb21b4babu); // s4 add
|
||||
r0 = rotl_imm(r0, 11u); // s5 rotl
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // s6 add
|
||||
r7 = r7 ^ r1; // s7 xor
|
||||
r1 = r6 * r5 + r1; // s8 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 4u); r6 = r6 ^ t_; } // s9 shfl
|
||||
r1 = r2 * r2 + r1; // s10 mad
|
||||
r5 = r0 * r3 + r5; // s11 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 4u); r2 = r2 ^ t_; } // s12 shfl
|
||||
r1 = r1 + r5 + ((((sel >> 25u) & 1u) != 0u) ? 0xbc48c63eu : 0xbdf8f9a5u); // s13 add
|
||||
r1 = rotl_imm(r1, 29u); // s14 rotl
|
||||
r1 = r1 - r4; // s15 sub
|
||||
r7 = r7 | r1; // s16 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r2 = r2 ^ t_; } // s17 shfl
|
||||
r7 = r7 ^ r4; // s18 xor
|
||||
r6 = r6 * r1; // s19 mul
|
||||
r5 = r6 * r0 + r5; // s20 mad
|
||||
r3 = r3 - r1; // s21 sub
|
||||
r6 = r6 * r0; // s22 mul
|
||||
r2 = r2 + r0 + ((((sel >> 1u) & 1u) != 0u) ? 0x96e8f127u : 0x45c37cecu); // s23 add
|
||||
r6 = r6 - r4; // s24 sub
|
||||
r7 = r3 * r4 + r7; // s25 mad
|
||||
r3 = rotl_imm(r3, 9u); // s26 rotl
|
||||
r2 = r2 - r1; // s27 sub
|
||||
r6 = mul_hi(r6, r0); // s28 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r2 = r2 ^ t_; } // s29 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r1 = r1 ^ t_; } // s30 shfl
|
||||
r1 = r1 + r6 + ((((sel >> 1u) & 1u) != 0u) ? 0xb1cdb2abu : 0x37985632u); // s31 add
|
||||
r0 = rotr_var(r0, r1); // s32 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // s33 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r0 = r0 ^ t_; } // s34 shfl
|
||||
r7 = r7 * r0; // s35 mul
|
||||
r3 = r3 + r1 + ((((sel >> 0u) & 1u) != 0u) ? 0x856e0180u : 0x804e777eu); // s36 add
|
||||
r1 = r1 ^ r6; // s37 xor
|
||||
r2 = r2 * r7; // s38 mul
|
||||
r6 = r6 + r5 + ((((sel >> 2u) & 1u) != 0u) ? 0xe9bd0cc4u : 0x0b06c8a7u); // s39 add
|
||||
r5 = r5 * r7; // s40 mul
|
||||
r2 = mul_hi(r2, r3); // s41 mulhi
|
||||
r2 = r2 ^ r0; // s42 xor
|
||||
r0 = rotl_imm(r0, 4u); // s43 rotl
|
||||
r4 = mul_hi(r4, r3); // s44 mulhi
|
||||
r6 = r6 + r7 + ((((sel >> 12u) & 1u) != 0u) ? 0xcdcb63f6u : 0xee822e17u); // s45 add
|
||||
r3 = r2 * r0 + r3; // s46 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r4 = r4 ^ t_; } // s47 shfl
|
||||
r6 = mul_hi(r6, r5); // s48 mulhi
|
||||
r0 = r0 - r2; // s49 sub
|
||||
r3 = r3 - r5; // s50 sub
|
||||
r1 = rotr_var(r1, r4); // s51 rotr
|
||||
r6 = r7 * r7 + r6; // s52 mad
|
||||
r5 = r5 ^ r3; // s53 xor
|
||||
r1 = r1 - r0; // s54 sub
|
||||
r5 = r5 - r6; // s55 sub
|
||||
r3 = r3 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0xd4758987u : 0x3dfad1b6u); // s56 add
|
||||
r4 = r4 + r1 + ((((sel >> 0u) & 1u) != 0u) ? 0x0602d6beu : 0xfca75bc2u); // s57 add
|
||||
r5 = mul_hi(r5, r6); // s58 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 2u); r2 = r2 ^ t_; } // s59 shfl
|
||||
r2 = rotl_imm(r2, 9u); // s60 rotl
|
||||
r4 = r4 | r6; // s61 or
|
||||
r6 = rotr_var(r6, r4); // s62 rotr
|
||||
r2 = r2 * r4; // s63 mul
|
||||
r0 = r0 ^ r5; // s64 xor
|
||||
r2 = r2 ^ r7; // s65 xor
|
||||
r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x8596687au : 0xd71c02ffu); // s66 add
|
||||
r1 = rotl_imm(r1, 21u); // s67 rotl
|
||||
r3 = r3 * r2; // s68 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 2u); r7 = r7 ^ t_; } // s69 shfl
|
||||
r3 = r3 * r2; // s70 mul
|
||||
r0 = r2 * r5 + r0; // s71 mad
|
||||
r6 = r6 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0x08ac5733u : 0x3c6fe15du); // s72 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r3 = r3 ^ t_; } // s73 shfl
|
||||
r3 = r3 * r6; // s74 mul
|
||||
r6 = r6 ^ r7; // s75 xor
|
||||
r3 = r3 | r0; // s76 or
|
||||
r2 = r2 + r4 + ((((sel >> 1u) & 1u) != 0u) ? 0xc100b495u : 0x295d5faeu); // s77 add
|
||||
r3 = mul_hi(r3, r7); // s78 mulhi
|
||||
r4 = r4 | r1; // s79 or
|
||||
r4 = rotr_var(r4, r3); // s80 rotr
|
||||
r4 = r4 + r3 + ((((sel >> 15u) & 1u) != 0u) ? 0x5cc59530u : 0x274a9221u); // s81 add
|
||||
r7 = r7 + r3 + ((((sel >> 5u) & 1u) != 0u) ? 0x141479ecu : 0xc8651f8eu); // s82 add
|
||||
r3 = rotr_var(r3, r6); // s83 rotr
|
||||
r2 = r4 * r7 + r2; // s84 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r2 = r2 ^ t_; } // s85 shfl
|
||||
r3 = r3 ^ r2; // s86 xor
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 2u); r5 = r5 ^ t_; } // s87 shfl
|
||||
r0 = r0 ^ r4; // s88 xor
|
||||
r3 = r3 + r2 + ((((sel >> 18u) & 1u) != 0u) ? 0x883c0c92u : 0x53f915c2u); // s89 add
|
||||
r7 = rotr_var(r7, r5); // s90 rotr
|
||||
r3 = r3 + r1 + ((((sel >> 31u) & 1u) != 0u) ? 0x3cc5bd20u : 0xe2be00a5u); // s91 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 1u); r7 = r7 ^ t_; } // s92 shfl
|
||||
r6 = rotr_var(r6, r2); // s93 rotr
|
||||
r7 = rotl_imm(r7, 14u); // s94 rotl
|
||||
r4 = r5 * r5 + r4; // s95 mad
|
||||
r2 = r4 * r5 + r2; // s96 mad
|
||||
r1 = r1 ^ r0; // s97 xor
|
||||
r5 = r5 * r4; // s98 mul
|
||||
r2 = r2 - r0; // s99 sub
|
||||
r7 = rotl_imm(r7, 30u); // s100 rotl
|
||||
r5 = r5 + r6 + ((((sel >> 17u) & 1u) != 0u) ? 0xbfd646cbu : 0x423fd9c9u); // s101 add
|
||||
r7 = r7 + r6 + ((((sel >> 11u) & 1u) != 0u) ? 0x19b74a43u : 0x8d3c011du); // s102 add
|
||||
r5 = rotl_imm(r5, 6u); // s103 rotl
|
||||
r0 = r0 * r4; // s104 mul
|
||||
r0 = r0 | r5; // s105 or
|
||||
r0 = r0 + r1 + ((((sel >> 15u) & 1u) != 0u) ? 0x7dcb7e18u : 0x8ce14721u); // s106 add
|
||||
r0 = rotl_imm(r0, 15u); // s107 rotl
|
||||
r4 = r4 - r2; // s108 sub
|
||||
r2 = mul_hi(r2, r0); // s109 mulhi
|
||||
r1 = mul_hi(r1, r0); // s110 mulhi
|
||||
r0 = r0 + r2 + ((((sel >> 28u) & 1u) != 0u) ? 0x3c179ad8u : 0x2d9fc6b9u); // s111 add
|
||||
r2 = mul_hi(r2, r0); // s112 mulhi
|
||||
r6 = r6 - r3; // s113 sub
|
||||
r7 = r6 * r3 + r7; // s114 mad
|
||||
r5 = rotr_var(r5, r1); // s115 rotr
|
||||
r6 = rotr_var(r6, r4); // s116 rotr
|
||||
r2 = r2 + r1 + ((((sel >> 22u) & 1u) != 0u) ? 0x47359729u : 0x502138e2u); // s117 add
|
||||
r3 = r2 * r0 + r3; // s118 mad
|
||||
r1 = r1 * r3; // s119 mul
|
||||
r2 = r0 * r4 + r2; // s120 mad
|
||||
r0 = r0 * r3; // s121 mul
|
||||
r2 = mul_hi(r2, r0); // s122 mulhi
|
||||
r5 = r5 * r6; // s123 mul
|
||||
r4 = r2 * r4 + r4; // s124 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 2u); r4 = r4 ^ t_; } // s125 shfl
|
||||
r2 = r2 ^ r0; // s126 xor
|
||||
r1 = rotl_imm(r1, 7u); // s127 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r7 = r7 ^ t_; } // s128 shfl
|
||||
r6 = rotl_imm(r6, 17u); // s129 rotl
|
||||
r2 = r2 + r5 + ((((sel >> 15u) & 1u) != 0u) ? 0x6c57e4e7u : 0x33dff776u); // s130 add
|
||||
r0 = r0 | r3; // s131 or
|
||||
r5 = rotr_var(r5, r0); // s132 rotr
|
||||
r2 = r2 + r3 + ((((sel >> 30u) & 1u) != 0u) ? 0xe8212c0cu : 0x5db25b34u); // s133 add
|
||||
r4 = r4 ^ r3; // s134 xor
|
||||
r6 = r6 ^ r1; // s135 xor
|
||||
r1 = r1 - r7; // s136 sub
|
||||
r2 = r6 * r2 + r2; // s137 mad
|
||||
r0 = mul_hi(r0, r5); // s138 mulhi
|
||||
r2 = r2 + r7 + ((((sel >> 10u) & 1u) != 0u) ? 0xd44ff710u : 0x218d4090u); // s139 add
|
||||
r1 = rotr_var(r1, r4); // s140 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r3 = r3 ^ t_; } // s141 shfl
|
||||
r7 = r6 * r2 + r7; // s142 mad
|
||||
r2 = r2 * r3; // s143 mul
|
||||
r7 = r7 + r4 + ((((sel >> 29u) & 1u) != 0u) ? 0xb98942fau : 0xf4b1a8deu); // s144 add
|
||||
r7 = rotl_imm(r7, 15u); // s145 rotl
|
||||
r7 = r7 ^ r5; // s146 xor
|
||||
r4 = r7 * r4 + r4; // s147 mad
|
||||
r6 = rotr_var(r6, r5); // s148 rotr
|
||||
r1 = r2 * r4 + r1; // s149 mad
|
||||
r1 = r1 + r7 + ((((sel >> 17u) & 1u) != 0u) ? 0x0d48ba42u : 0x2bef10f2u); // s150 add
|
||||
r5 = r5 ^ r4; // s151 xor
|
||||
r7 = r7 + r0 + ((((sel >> 30u) & 1u) != 0u) ? 0xc98dea9cu : 0xdc5cc080u); // s152 add
|
||||
r4 = r4 * r6; // s153 mul
|
||||
r0 = r0 * r2; // s154 mul
|
||||
r6 = r6 ^ r5; // s155 xor
|
||||
r4 = rotr_var(r4, r2); // s156 rotr
|
||||
r1 = rotl_imm(r1, 11u); // s157 rotl
|
||||
r5 = r5 + r0 + ((((sel >> 11u) & 1u) != 0u) ? 0x6c752dcbu : 0x18197438u); // s158 add
|
||||
r4 = r1 * r5 + r4; // s159 mad
|
||||
r4 = rotl_imm(r4, 26u); // s160 rotl
|
||||
r3 = r0 * r7 + r3; // s161 mad
|
||||
r3 = rotr_var(r3, r1); // s162 rotr
|
||||
r4 = r4 + r0 + ((((sel >> 3u) & 1u) != 0u) ? 0x8e481727u : 0xf1c46574u); // s163 add
|
||||
r0 = mul_hi(r0, r4); // s164 mulhi
|
||||
r2 = r2 + r6 + ((((sel >> 20u) & 1u) != 0u) ? 0x550ab406u : 0xac578137u); // s165 add
|
||||
r0 = rotl_imm(r0, 13u); // s166 rotl
|
||||
r3 = r3 + r1 + ((((sel >> 14u) & 1u) != 0u) ? 0xaebb5966u : 0xa33e6706u); // s167 add
|
||||
r3 = r3 | r5; // s168 or
|
||||
r6 = rotr_var(r6, r2); // s169 rotr
|
||||
r4 = r4 ^ r6; // s170 xor
|
||||
r6 = r6 - r1; // s171 sub
|
||||
r7 = rotl_imm(r7, 22u); // s172 rotl
|
||||
r5 = rotl_imm(r5, 15u); // s173 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 8u); r7 = r7 ^ t_; } // s174 shfl
|
||||
r0 = r0 ^ r5; // s175 xor
|
||||
r7 = rotl_imm(r7, 6u); // s176 rotl
|
||||
r7 = r7 - r0; // s177 sub
|
||||
r3 = rotl_imm(r3, 30u); // s178 rotl
|
||||
r7 = r6 * r1 + r7; // s179 mad
|
||||
r6 = rotl_imm(r6, 9u); // s180 rotl
|
||||
r2 = r2 ^ r4; // s181 xor
|
||||
r2 = r2 ^ r7; // s182 xor
|
||||
r7 = r7 ^ r2; // s183 xor
|
||||
r1 = r1 + r2 + ((((sel >> 21u) & 1u) != 0u) ? 0x5bb7550fu : 0xd94d55acu); // s184 add
|
||||
r3 = r3 | r5; // s185 or
|
||||
r6 = mul_hi(r6, r3); // s186 mulhi
|
||||
r4 = r4 | r0; // s187 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 2u); r7 = r7 ^ t_; } // s188 shfl
|
||||
r6 = r6 + r5 + ((((sel >> 6u) & 1u) != 0u) ? 0x2a354e2du : 0x89e747feu); // s189 add
|
||||
r1 = r1 ^ r4; // s190 xor
|
||||
r7 = mul_hi(r7, r4); // s191 mulhi
|
||||
r2 = rotl_imm(r2, 26u); // s192 rotl
|
||||
r5 = rotl_imm(r5, 8u); // s193 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r4 = r4 ^ t_; } // s194 shfl
|
||||
r4 = r4 ^ r5; // s195 xor
|
||||
r1 = r1 * r3; // s196 mul
|
||||
r5 = r5 + r2 + ((((sel >> 7u) & 1u) != 0u) ? 0x10cfdc71u : 0xea03e8e7u); // s197 add
|
||||
r7 = r7 + r5 + ((((sel >> 15u) & 1u) != 0u) ? 0x66e148d3u : 0xa7ee0102u); // s198 add
|
||||
r5 = r5 - r2; // s199 sub
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 2u); r5 = r5 ^ t_; } // s200 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 8u); r7 = r7 ^ t_; } // s201 shfl
|
||||
r4 = rotr_var(r4, r5); // s202 rotr
|
||||
r7 = r7 ^ r5; // s203 xor
|
||||
r7 = r7 ^ r0; // s204 xor
|
||||
r6 = r6 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0x8558b619u : 0x8379a4deu); // s205 add
|
||||
r4 = rotl_imm(r4, 13u); // s206 rotl
|
||||
r1 = mul_hi(r1, r3); // s207 mulhi
|
||||
r1 = r4 * r4 + r1; // s208 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 4u); r2 = r2 ^ t_; } // s209 shfl
|
||||
r7 = r7 + r0 + ((((sel >> 11u) & 1u) != 0u) ? 0xf4049c4cu : 0xaa8cb14eu); // s210 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 16u); r7 = r7 ^ t_; } // s211 shfl
|
||||
r1 = r1 ^ r0; // s212 xor
|
||||
r7 = rotl_imm(r7, 3u); // s213 rotl
|
||||
r4 = rotr_var(r4, r7); // s214 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 16u); r3 = r3 ^ t_; } // s215 shfl
|
||||
r5 = r5 | r1; // s216 or
|
||||
r1 = r1 - r2; // s217 sub
|
||||
r6 = r6 - r5; // s218 sub
|
||||
r6 = rotl_imm(r6, 4u); // s219 rotl
|
||||
r2 = mul_hi(r2, r0); // s220 mulhi
|
||||
r2 = r2 | r0; // s221 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r5 = r5 ^ t_; } // s222 shfl
|
||||
r5 = mul_hi(r5, r6); // s223 mulhi
|
||||
r0 = r0 - r6; // s224 sub
|
||||
r7 = rotl_imm(r7, 23u); // s225 rotl
|
||||
r4 = r4 | r2; // s226 or
|
||||
r2 = r2 * r4; // s227 mul
|
||||
r3 = rotl_imm(r3, 12u); // s228 rotl
|
||||
r0 = rotr_var(r0, r4); // s229 rotr
|
||||
r0 = r0 + r6 + ((((sel >> 16u) & 1u) != 0u) ? 0x2a7fecb2u : 0x1d176220u); // s230 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r1 = r1 ^ t_; } // s231 shfl
|
||||
r2 = r2 * r1; // s232 mul
|
||||
r7 = r0 * r1 + r7; // s233 mad
|
||||
r5 = rotl_imm(r5, 22u); // s234 rotl
|
||||
r4 = rotr_var(r4, r6); // s235 rotr
|
||||
r0 = r5 * r1 + r0; // s236 mad
|
||||
r6 = r6 ^ r4; // s237 xor
|
||||
r4 = r4 ^ r6; // s238 xor
|
||||
r6 = rotl_imm(r6, 18u); // s239 rotl
|
||||
r4 = r4 + r7 + ((((sel >> 25u) & 1u) != 0u) ? 0xadce39f3u : 0x17dafb4du); // s240 add
|
||||
r0 = r0 - r7; // s241 sub
|
||||
r1 = rotr_var(r1, r6); // s242 rotr
|
||||
r3 = r3 + r0 + ((((sel >> 29u) & 1u) != 0u) ? 0x0e7033b6u : 0xf2f77d26u); // s243 add
|
||||
r2 = r7 * r4 + r2; // s244 mad
|
||||
r4 = r0 * r6 + r4; // s245 mad
|
||||
r5 = r5 * r7; // s246 mul
|
||||
r3 = r3 + r5 + ((((sel >> 31u) & 1u) != 0u) ? 0x7b2ff6b7u : 0xfed2da4eu); // s247 add
|
||||
r0 = r0 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0xb5fad7b1u : 0x03b2891cu); // s248 add
|
||||
r5 = mul_hi(r5, r4); // s249 mulhi
|
||||
r5 = r5 + r2 + ((((sel >> 11u) & 1u) != 0u) ? 0xa7e71c2fu : 0x042cd6e3u); // s250 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 1u); r6 = r6 ^ t_; } // s251 shfl
|
||||
r6 = r6 ^ r1; // s252 xor
|
||||
r2 = r2 * r5; // s253 mul
|
||||
r0 = r0 + r6 + ((((sel >> 16u) & 1u) != 0u) ? 0xb83d78deu : 0x784a302bu); // s254 add
|
||||
r2 = r2 - r4; // s255 sub
|
||||
}
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
382
proto-cuda/packs-ca3-v5/v4-genesis/kernel_bound.cu
Normal file
382
proto-cuda/packs-ca3-v5/v4-genesis/kernel_bound.cu
Normal file
|
|
@ -0,0 +1,382 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW.
|
||||
// Host declarations (also in program_bound.h if present):
|
||||
// struct IgneumInitWords { uint32_t w[8]; };
|
||||
// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps);
|
||||
// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
|
||||
#include <cuda_runtime.h>
|
||||
#include <cstdint>
|
||||
#include "program.h"
|
||||
|
||||
struct IgneumInitWords { uint32_t w[8]; };
|
||||
|
||||
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
|
||||
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
|
||||
__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw) {
|
||||
uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
uint32_t nonce = baseNonce + gid;
|
||||
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; }
|
||||
{ uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; }
|
||||
{ uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; }
|
||||
{ uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; }
|
||||
{ uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; }
|
||||
{ uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; }
|
||||
{ uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; }
|
||||
{ uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; }
|
||||
|
||||
for (uint32_t it = 0u; it < 8u; ++it) {
|
||||
uint32_t sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r2 = r1 * r1 + r2; // 1 mad
|
||||
r2 = r3 * r2 + r2; // 2 mad
|
||||
r3 = r3 ^ r5; // 3 xor
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r5 = r5 ^ ds[r7 & mask]; // 5 load
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 6 shfl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 8); // 7 shfl
|
||||
r1 = __umulhi(r1, r5); // 8 mulhi
|
||||
r6 = rotr_var(r6, r3); // 9 rotr
|
||||
r3 = r3 | r4; // 10 or
|
||||
r4 = r4 ^ ds[r6 & mask]; // 11 load
|
||||
r0 = __umulhi(r0, r4); // 12 mulhi
|
||||
r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add
|
||||
r0 = r0 ^ ds[r5 & mask]; // 14 load
|
||||
r2 = r2 - r4; // 15 sub
|
||||
r2 = r2 ^ ds[r7 & mask]; // 16 load
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 18 shfl
|
||||
r5 = r5 * r0; // 19 mul
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 20 shfl
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 16); // 21 shfl
|
||||
r6 = __umulhi(r6, r2); // 22 mulhi
|
||||
r6 = r6 ^ ds[r3 & mask]; // 23 load
|
||||
r5 = r5 * r0; // 24 mul
|
||||
r5 = rotl_imm(r5, 19u); // 25 rotl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r6, 2); // 26 shfl
|
||||
r0 = r0 ^ r5; // 27 xor
|
||||
r0 = r0 ^ r4; // 28 xor
|
||||
r3 = r3 - r0; // 29 sub
|
||||
r5 = r5 * r1; // 30 mul
|
||||
r7 = r7 ^ ds[r3 & mask]; // 31 load
|
||||
r1 = r1 ^ ds[r2 & mask]; // 32 load
|
||||
r5 = r5 ^ r6; // 33 xor
|
||||
r5 = r5 ^ ds[r0 & mask]; // 34 load
|
||||
r0 = __umulhi(r0, r5); // 35 mulhi
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // 36 shfl
|
||||
r7 = r7 ^ ds[r6 & mask]; // 37 load
|
||||
r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 4); // 39 shfl
|
||||
r2 = r2 ^ r5; // 40 xor
|
||||
r3 = r6 * r3 + r3; // 41 mad
|
||||
r6 = r6 - r7; // 42 sub
|
||||
r7 = r7 ^ r0; // 43 xor
|
||||
r1 = r1 ^ ds[r7 & mask]; // 44 load
|
||||
r2 = r2 * r3; // 45 mul
|
||||
r1 = __umulhi(r1, r5); // 46 mulhi
|
||||
r4 = r4 - r3; // 47 sub
|
||||
r2 = rotr_var(r2, r6); // 48 rotr
|
||||
r3 = r3 ^ ds[r4 & mask]; // 49 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add
|
||||
r0 = r0 * r2; // 51 mul
|
||||
r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add
|
||||
r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add
|
||||
r7 = rotl_imm(r7, 14u); // 54 rotl
|
||||
r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add
|
||||
r6 = r6 ^ ds[r7 & mask]; // 56 load
|
||||
r1 = rotr_var(r1, r5); // 57 rotr
|
||||
r5 = r5 ^ ds[r3 & mask]; // 58 load
|
||||
r6 = r6 ^ ds[r1 & mask]; // 59 load
|
||||
r3 = r5 * r0 + r3; // 60 mad
|
||||
r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add
|
||||
r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add
|
||||
r5 = rotl_imm(r5, 19u); // 63 rotl
|
||||
// latency-shadow block (Counter ASIC 3.0 item 8): 256 ALU instructions x 27 passes after instruction 63, no load
|
||||
for (uint32_t sh = 0u; sh < 27u; ++sh) {
|
||||
r5 = r5 + r2 + ((((sel >> 18u) & 1u) != 0u) ? 0x8bc12da9u : 0x92199f99u); // s0 add
|
||||
r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x63079e5au : 0x8d72d3adu); // s1 add
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 2); // s2 shfl
|
||||
r4 = r4 - r2; // s3 sub
|
||||
r7 = r7 + r0 + ((((sel >> 31u) & 1u) != 0u) ? 0x5d4c7a60u : 0xb21b4babu); // s4 add
|
||||
r0 = rotl_imm(r0, 11u); // s5 rotl
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // s6 add
|
||||
r7 = r7 ^ r1; // s7 xor
|
||||
r1 = r6 * r5 + r1; // s8 mad
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r1, 4); // s9 shfl
|
||||
r1 = r2 * r2 + r1; // s10 mad
|
||||
r5 = r0 * r3 + r5; // s11 mad
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r6, 4); // s12 shfl
|
||||
r1 = r1 + r5 + ((((sel >> 25u) & 1u) != 0u) ? 0xbc48c63eu : 0xbdf8f9a5u); // s13 add
|
||||
r1 = rotl_imm(r1, 29u); // s14 rotl
|
||||
r1 = r1 - r4; // s15 sub
|
||||
r7 = r7 | r1; // s16 or
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // s17 shfl
|
||||
r7 = r7 ^ r4; // s18 xor
|
||||
r6 = r6 * r1; // s19 mul
|
||||
r5 = r6 * r0 + r5; // s20 mad
|
||||
r3 = r3 - r1; // s21 sub
|
||||
r6 = r6 * r0; // s22 mul
|
||||
r2 = r2 + r0 + ((((sel >> 1u) & 1u) != 0u) ? 0x96e8f127u : 0x45c37cecu); // s23 add
|
||||
r6 = r6 - r4; // s24 sub
|
||||
r7 = r3 * r4 + r7; // s25 mad
|
||||
r3 = rotl_imm(r3, 9u); // s26 rotl
|
||||
r2 = r2 - r1; // s27 sub
|
||||
r6 = __umulhi(r6, r0); // s28 mulhi
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // s29 shfl
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r6, 1); // s30 shfl
|
||||
r1 = r1 + r6 + ((((sel >> 1u) & 1u) != 0u) ? 0xb1cdb2abu : 0x37985632u); // s31 add
|
||||
r0 = rotr_var(r0, r1); // s32 rotr
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // s33 shfl
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // s34 shfl
|
||||
r7 = r7 * r0; // s35 mul
|
||||
r3 = r3 + r1 + ((((sel >> 0u) & 1u) != 0u) ? 0x856e0180u : 0x804e777eu); // s36 add
|
||||
r1 = r1 ^ r6; // s37 xor
|
||||
r2 = r2 * r7; // s38 mul
|
||||
r6 = r6 + r5 + ((((sel >> 2u) & 1u) != 0u) ? 0xe9bd0cc4u : 0x0b06c8a7u); // s39 add
|
||||
r5 = r5 * r7; // s40 mul
|
||||
r2 = __umulhi(r2, r3); // s41 mulhi
|
||||
r2 = r2 ^ r0; // s42 xor
|
||||
r0 = rotl_imm(r0, 4u); // s43 rotl
|
||||
r4 = __umulhi(r4, r3); // s44 mulhi
|
||||
r6 = r6 + r7 + ((((sel >> 12u) & 1u) != 0u) ? 0xcdcb63f6u : 0xee822e17u); // s45 add
|
||||
r3 = r2 * r0 + r3; // s46 mad
|
||||
r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r3, 8); // s47 shfl
|
||||
r6 = __umulhi(r6, r5); // s48 mulhi
|
||||
r0 = r0 - r2; // s49 sub
|
||||
r3 = r3 - r5; // s50 sub
|
||||
r1 = rotr_var(r1, r4); // s51 rotr
|
||||
r6 = r7 * r7 + r6; // s52 mad
|
||||
r5 = r5 ^ r3; // s53 xor
|
||||
r1 = r1 - r0; // s54 sub
|
||||
r5 = r5 - r6; // s55 sub
|
||||
r3 = r3 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0xd4758987u : 0x3dfad1b6u); // s56 add
|
||||
r4 = r4 + r1 + ((((sel >> 0u) & 1u) != 0u) ? 0x0602d6beu : 0xfca75bc2u); // s57 add
|
||||
r5 = __umulhi(r5, r6); // s58 mulhi
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r1, 2); // s59 shfl
|
||||
r2 = rotl_imm(r2, 9u); // s60 rotl
|
||||
r4 = r4 | r6; // s61 or
|
||||
r6 = rotr_var(r6, r4); // s62 rotr
|
||||
r2 = r2 * r4; // s63 mul
|
||||
r0 = r0 ^ r5; // s64 xor
|
||||
r2 = r2 ^ r7; // s65 xor
|
||||
r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x8596687au : 0xd71c02ffu); // s66 add
|
||||
r1 = rotl_imm(r1, 21u); // s67 rotl
|
||||
r3 = r3 * r2; // s68 mul
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r5, 2); // s69 shfl
|
||||
r3 = r3 * r2; // s70 mul
|
||||
r0 = r2 * r5 + r0; // s71 mad
|
||||
r6 = r6 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0x08ac5733u : 0x3c6fe15du); // s72 add
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r2, 8); // s73 shfl
|
||||
r3 = r3 * r6; // s74 mul
|
||||
r6 = r6 ^ r7; // s75 xor
|
||||
r3 = r3 | r0; // s76 or
|
||||
r2 = r2 + r4 + ((((sel >> 1u) & 1u) != 0u) ? 0xc100b495u : 0x295d5faeu); // s77 add
|
||||
r3 = __umulhi(r3, r7); // s78 mulhi
|
||||
r4 = r4 | r1; // s79 or
|
||||
r4 = rotr_var(r4, r3); // s80 rotr
|
||||
r4 = r4 + r3 + ((((sel >> 15u) & 1u) != 0u) ? 0x5cc59530u : 0x274a9221u); // s81 add
|
||||
r7 = r7 + r3 + ((((sel >> 5u) & 1u) != 0u) ? 0x141479ecu : 0xc8651f8eu); // s82 add
|
||||
r3 = rotr_var(r3, r6); // s83 rotr
|
||||
r2 = r4 * r7 + r2; // s84 mad
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r5, 16); // s85 shfl
|
||||
r3 = r3 ^ r2; // s86 xor
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r7, 2); // s87 shfl
|
||||
r0 = r0 ^ r4; // s88 xor
|
||||
r3 = r3 + r2 + ((((sel >> 18u) & 1u) != 0u) ? 0x883c0c92u : 0x53f915c2u); // s89 add
|
||||
r7 = rotr_var(r7, r5); // s90 rotr
|
||||
r3 = r3 + r1 + ((((sel >> 31u) & 1u) != 0u) ? 0x3cc5bd20u : 0xe2be00a5u); // s91 add
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r2, 1); // s92 shfl
|
||||
r6 = rotr_var(r6, r2); // s93 rotr
|
||||
r7 = rotl_imm(r7, 14u); // s94 rotl
|
||||
r4 = r5 * r5 + r4; // s95 mad
|
||||
r2 = r4 * r5 + r2; // s96 mad
|
||||
r1 = r1 ^ r0; // s97 xor
|
||||
r5 = r5 * r4; // s98 mul
|
||||
r2 = r2 - r0; // s99 sub
|
||||
r7 = rotl_imm(r7, 30u); // s100 rotl
|
||||
r5 = r5 + r6 + ((((sel >> 17u) & 1u) != 0u) ? 0xbfd646cbu : 0x423fd9c9u); // s101 add
|
||||
r7 = r7 + r6 + ((((sel >> 11u) & 1u) != 0u) ? 0x19b74a43u : 0x8d3c011du); // s102 add
|
||||
r5 = rotl_imm(r5, 6u); // s103 rotl
|
||||
r0 = r0 * r4; // s104 mul
|
||||
r0 = r0 | r5; // s105 or
|
||||
r0 = r0 + r1 + ((((sel >> 15u) & 1u) != 0u) ? 0x7dcb7e18u : 0x8ce14721u); // s106 add
|
||||
r0 = rotl_imm(r0, 15u); // s107 rotl
|
||||
r4 = r4 - r2; // s108 sub
|
||||
r2 = __umulhi(r2, r0); // s109 mulhi
|
||||
r1 = __umulhi(r1, r0); // s110 mulhi
|
||||
r0 = r0 + r2 + ((((sel >> 28u) & 1u) != 0u) ? 0x3c179ad8u : 0x2d9fc6b9u); // s111 add
|
||||
r2 = __umulhi(r2, r0); // s112 mulhi
|
||||
r6 = r6 - r3; // s113 sub
|
||||
r7 = r6 * r3 + r7; // s114 mad
|
||||
r5 = rotr_var(r5, r1); // s115 rotr
|
||||
r6 = rotr_var(r6, r4); // s116 rotr
|
||||
r2 = r2 + r1 + ((((sel >> 22u) & 1u) != 0u) ? 0x47359729u : 0x502138e2u); // s117 add
|
||||
r3 = r2 * r0 + r3; // s118 mad
|
||||
r1 = r1 * r3; // s119 mul
|
||||
r2 = r0 * r4 + r2; // s120 mad
|
||||
r0 = r0 * r3; // s121 mul
|
||||
r2 = __umulhi(r2, r0); // s122 mulhi
|
||||
r5 = r5 * r6; // s123 mul
|
||||
r4 = r2 * r4 + r4; // s124 mad
|
||||
r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r2, 2); // s125 shfl
|
||||
r2 = r2 ^ r0; // s126 xor
|
||||
r1 = rotl_imm(r1, 7u); // s127 rotl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // s128 shfl
|
||||
r6 = rotl_imm(r6, 17u); // s129 rotl
|
||||
r2 = r2 + r5 + ((((sel >> 15u) & 1u) != 0u) ? 0x6c57e4e7u : 0x33dff776u); // s130 add
|
||||
r0 = r0 | r3; // s131 or
|
||||
r5 = rotr_var(r5, r0); // s132 rotr
|
||||
r2 = r2 + r3 + ((((sel >> 30u) & 1u) != 0u) ? 0xe8212c0cu : 0x5db25b34u); // s133 add
|
||||
r4 = r4 ^ r3; // s134 xor
|
||||
r6 = r6 ^ r1; // s135 xor
|
||||
r1 = r1 - r7; // s136 sub
|
||||
r2 = r6 * r2 + r2; // s137 mad
|
||||
r0 = __umulhi(r0, r5); // s138 mulhi
|
||||
r2 = r2 + r7 + ((((sel >> 10u) & 1u) != 0u) ? 0xd44ff710u : 0x218d4090u); // s139 add
|
||||
r1 = rotr_var(r1, r4); // s140 rotr
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r6, 1); // s141 shfl
|
||||
r7 = r6 * r2 + r7; // s142 mad
|
||||
r2 = r2 * r3; // s143 mul
|
||||
r7 = r7 + r4 + ((((sel >> 29u) & 1u) != 0u) ? 0xb98942fau : 0xf4b1a8deu); // s144 add
|
||||
r7 = rotl_imm(r7, 15u); // s145 rotl
|
||||
r7 = r7 ^ r5; // s146 xor
|
||||
r4 = r7 * r4 + r4; // s147 mad
|
||||
r6 = rotr_var(r6, r5); // s148 rotr
|
||||
r1 = r2 * r4 + r1; // s149 mad
|
||||
r1 = r1 + r7 + ((((sel >> 17u) & 1u) != 0u) ? 0x0d48ba42u : 0x2bef10f2u); // s150 add
|
||||
r5 = r5 ^ r4; // s151 xor
|
||||
r7 = r7 + r0 + ((((sel >> 30u) & 1u) != 0u) ? 0xc98dea9cu : 0xdc5cc080u); // s152 add
|
||||
r4 = r4 * r6; // s153 mul
|
||||
r0 = r0 * r2; // s154 mul
|
||||
r6 = r6 ^ r5; // s155 xor
|
||||
r4 = rotr_var(r4, r2); // s156 rotr
|
||||
r1 = rotl_imm(r1, 11u); // s157 rotl
|
||||
r5 = r5 + r0 + ((((sel >> 11u) & 1u) != 0u) ? 0x6c752dcbu : 0x18197438u); // s158 add
|
||||
r4 = r1 * r5 + r4; // s159 mad
|
||||
r4 = rotl_imm(r4, 26u); // s160 rotl
|
||||
r3 = r0 * r7 + r3; // s161 mad
|
||||
r3 = rotr_var(r3, r1); // s162 rotr
|
||||
r4 = r4 + r0 + ((((sel >> 3u) & 1u) != 0u) ? 0x8e481727u : 0xf1c46574u); // s163 add
|
||||
r0 = __umulhi(r0, r4); // s164 mulhi
|
||||
r2 = r2 + r6 + ((((sel >> 20u) & 1u) != 0u) ? 0x550ab406u : 0xac578137u); // s165 add
|
||||
r0 = rotl_imm(r0, 13u); // s166 rotl
|
||||
r3 = r3 + r1 + ((((sel >> 14u) & 1u) != 0u) ? 0xaebb5966u : 0xa33e6706u); // s167 add
|
||||
r3 = r3 | r5; // s168 or
|
||||
r6 = rotr_var(r6, r2); // s169 rotr
|
||||
r4 = r4 ^ r6; // s170 xor
|
||||
r6 = r6 - r1; // s171 sub
|
||||
r7 = rotl_imm(r7, 22u); // s172 rotl
|
||||
r5 = rotl_imm(r5, 15u); // s173 rotl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r0, 8); // s174 shfl
|
||||
r0 = r0 ^ r5; // s175 xor
|
||||
r7 = rotl_imm(r7, 6u); // s176 rotl
|
||||
r7 = r7 - r0; // s177 sub
|
||||
r3 = rotl_imm(r3, 30u); // s178 rotl
|
||||
r7 = r6 * r1 + r7; // s179 mad
|
||||
r6 = rotl_imm(r6, 9u); // s180 rotl
|
||||
r2 = r2 ^ r4; // s181 xor
|
||||
r2 = r2 ^ r7; // s182 xor
|
||||
r7 = r7 ^ r2; // s183 xor
|
||||
r1 = r1 + r2 + ((((sel >> 21u) & 1u) != 0u) ? 0x5bb7550fu : 0xd94d55acu); // s184 add
|
||||
r3 = r3 | r5; // s185 or
|
||||
r6 = __umulhi(r6, r3); // s186 mulhi
|
||||
r4 = r4 | r0; // s187 or
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r1, 2); // s188 shfl
|
||||
r6 = r6 + r5 + ((((sel >> 6u) & 1u) != 0u) ? 0x2a354e2du : 0x89e747feu); // s189 add
|
||||
r1 = r1 ^ r4; // s190 xor
|
||||
r7 = __umulhi(r7, r4); // s191 mulhi
|
||||
r2 = rotl_imm(r2, 26u); // s192 rotl
|
||||
r5 = rotl_imm(r5, 8u); // s193 rotl
|
||||
r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r5, 16); // s194 shfl
|
||||
r4 = r4 ^ r5; // s195 xor
|
||||
r1 = r1 * r3; // s196 mul
|
||||
r5 = r5 + r2 + ((((sel >> 7u) & 1u) != 0u) ? 0x10cfdc71u : 0xea03e8e7u); // s197 add
|
||||
r7 = r7 + r5 + ((((sel >> 15u) & 1u) != 0u) ? 0x66e148d3u : 0xa7ee0102u); // s198 add
|
||||
r5 = r5 - r2; // s199 sub
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r1, 2); // s200 shfl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r1, 8); // s201 shfl
|
||||
r4 = rotr_var(r4, r5); // s202 rotr
|
||||
r7 = r7 ^ r5; // s203 xor
|
||||
r7 = r7 ^ r0; // s204 xor
|
||||
r6 = r6 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0x8558b619u : 0x8379a4deu); // s205 add
|
||||
r4 = rotl_imm(r4, 13u); // s206 rotl
|
||||
r1 = __umulhi(r1, r3); // s207 mulhi
|
||||
r1 = r4 * r4 + r1; // s208 mad
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r0, 4); // s209 shfl
|
||||
r7 = r7 + r0 + ((((sel >> 11u) & 1u) != 0u) ? 0xf4049c4cu : 0xaa8cb14eu); // s210 add
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r1, 16); // s211 shfl
|
||||
r1 = r1 ^ r0; // s212 xor
|
||||
r7 = rotl_imm(r7, 3u); // s213 rotl
|
||||
r4 = rotr_var(r4, r7); // s214 rotr
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r2, 16); // s215 shfl
|
||||
r5 = r5 | r1; // s216 or
|
||||
r1 = r1 - r2; // s217 sub
|
||||
r6 = r6 - r5; // s218 sub
|
||||
r6 = rotl_imm(r6, 4u); // s219 rotl
|
||||
r2 = __umulhi(r2, r0); // s220 mulhi
|
||||
r2 = r2 | r0; // s221 or
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r2, 8); // s222 shfl
|
||||
r5 = __umulhi(r5, r6); // s223 mulhi
|
||||
r0 = r0 - r6; // s224 sub
|
||||
r7 = rotl_imm(r7, 23u); // s225 rotl
|
||||
r4 = r4 | r2; // s226 or
|
||||
r2 = r2 * r4; // s227 mul
|
||||
r3 = rotl_imm(r3, 12u); // s228 rotl
|
||||
r0 = rotr_var(r0, r4); // s229 rotr
|
||||
r0 = r0 + r6 + ((((sel >> 16u) & 1u) != 0u) ? 0x2a7fecb2u : 0x1d176220u); // s230 add
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // s231 shfl
|
||||
r2 = r2 * r1; // s232 mul
|
||||
r7 = r0 * r1 + r7; // s233 mad
|
||||
r5 = rotl_imm(r5, 22u); // s234 rotl
|
||||
r4 = rotr_var(r4, r6); // s235 rotr
|
||||
r0 = r5 * r1 + r0; // s236 mad
|
||||
r6 = r6 ^ r4; // s237 xor
|
||||
r4 = r4 ^ r6; // s238 xor
|
||||
r6 = rotl_imm(r6, 18u); // s239 rotl
|
||||
r4 = r4 + r7 + ((((sel >> 25u) & 1u) != 0u) ? 0xadce39f3u : 0x17dafb4du); // s240 add
|
||||
r0 = r0 - r7; // s241 sub
|
||||
r1 = rotr_var(r1, r6); // s242 rotr
|
||||
r3 = r3 + r0 + ((((sel >> 29u) & 1u) != 0u) ? 0x0e7033b6u : 0xf2f77d26u); // s243 add
|
||||
r2 = r7 * r4 + r2; // s244 mad
|
||||
r4 = r0 * r6 + r4; // s245 mad
|
||||
r5 = r5 * r7; // s246 mul
|
||||
r3 = r3 + r5 + ((((sel >> 31u) & 1u) != 0u) ? 0x7b2ff6b7u : 0xfed2da4eu); // s247 add
|
||||
r0 = r0 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0xb5fad7b1u : 0x03b2891cu); // s248 add
|
||||
r5 = __umulhi(r5, r4); // s249 mulhi
|
||||
r5 = r5 + r2 + ((((sel >> 11u) & 1u) != 0u) ? 0xa7e71c2fu : 0x042cd6e3u); // s250 add
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r1, 1); // s251 shfl
|
||||
r6 = r6 ^ r1; // s252 xor
|
||||
r2 = r2 * r5; // s253 mul
|
||||
r0 = r0 + r6 + ((((sel >> 16u) & 1u) != 0u) ? 0xb83d78deu : 0x784a302bu); // s254 add
|
||||
r2 = r2 - r4; // s255 sub
|
||||
}
|
||||
}
|
||||
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
|
||||
}
|
||||
|
||||
cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps) {
|
||||
if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 32u * blockWarps;
|
||||
if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;
|
||||
igneum_hash_bound<<<nonces / block, block>>>(ds, out, baseNonce, mask, iw);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
|
||||
cudaFuncAttributes attr;
|
||||
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound);
|
||||
if (e != cudaSuccess) return e;
|
||||
*numRegs = attr.numRegs;
|
||||
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0);
|
||||
}
|
||||
109
proto-cuda/packs-ca3-v5/v4-genesis/memhard.h
Normal file
109
proto-cuda/packs-ca3-v5/v4-genesis/memhard.h
Normal file
|
|
@ -0,0 +1,109 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against.
|
||||
// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference).
|
||||
// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C.
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
#if defined(__CUDACC__)
|
||||
#define IGNEUM_HD __host__ __device__ __forceinline__
|
||||
#elif defined(_MSC_VER) && !defined(__cplusplus)
|
||||
#define IGNEUM_HD static __inline
|
||||
#else
|
||||
#define IGNEUM_HD static inline
|
||||
#endif
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of 8 x seed-parameterised mixer + one 64-byte cache read, then 8 x final mixer (class v3, mixer multiplier 8,
|
||||
// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) {
|
||||
for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint32_t r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) {
|
||||
uint32_t prev[16]; uint32_t x[16]; uint32_t y[16];
|
||||
for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of 8 x mixer + cache line s[0] & mask; 8 x final mixer.
|
||||
IGNEUM_HD void mh_item(const uint32_t* cache, uint32_t t, uint32_t* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint32_t r = 0u; r < 8u; ++r) {
|
||||
for (uint32_t j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 8u + j + 1u));
|
||||
const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
for (uint32_t j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (64u + j + 1u));
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
IGNEUM_HD uint32_t mh_word(const uint32_t* cache, uint32_t w) { uint32_t s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
107
proto-cuda/packs-ca3-v5/v4-genesis/memhard.metal
Normal file
107
proto-cuda/packs-ca3-v5/v4-genesis/memhard.metal
Normal file
|
|
@ -0,0 +1,107 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of 8 x seed-parameterised mixer + one 64-byte cache read, then 8 x final mixer (class v3, mixer multiplier 8,
|
||||
// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
inline void mh_chacha_block(const thread uint* x, thread uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
inline void mh_cache_segment(device uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
inline void mh_mixer(thread uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of 8 x mixer + cache line s[0] & mask; 8 x final mixer.
|
||||
inline void mh_item(device const uint* cache, uint t, thread uint* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
for (uint j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 8u + j + 1u));
|
||||
device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
for (uint j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (64u + j + 1u));
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
inline uint mh_word(device const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
|
||||
// One thread per segment (2^16 threads).
|
||||
kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) {
|
||||
mh_cache_segment(cache, gid);
|
||||
}
|
||||
// One thread per 64-byte item (dataset words / 16 threads).
|
||||
kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],
|
||||
uint gid [[thread_position_in_grid]]) {
|
||||
uint s[16];
|
||||
mh_item(cache, gid, s);
|
||||
device uint* d = dataset + gid * 16u;
|
||||
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
75
proto-cuda/packs-ca3-v5/v4-genesis/program.h
Normal file
75
proto-cuda/packs-ca3-v5/v4-genesis/program.h
Normal file
|
|
@ -0,0 +1,75 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.
|
||||
// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros.
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
#ifndef IGNEUM_NO_CUDA
|
||||
#include <cuda_runtime.h>
|
||||
#endif
|
||||
|
||||
#define IGNEUM_SEED_STRING "igneum-genesis"
|
||||
#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973"
|
||||
#define IGNEUM_GENERATOR 4
|
||||
#define IGNEUM_PROGRAM_ATTEMPT 0
|
||||
#define IGNEUM_PROGRAM_ID 0xa217c7f698880830ull
|
||||
#define IGNEUM_DAY_STRING "2026-10-03"
|
||||
#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033"
|
||||
#define IGNEUM_DAY0 0x3067619fu
|
||||
#define IGNEUM_DAY1 0x3c269176u
|
||||
#define IGNEUM_DATASET_LOG2 28
|
||||
#define IGNEUM_MASK 0x0fffffffu
|
||||
#define IGNEUM_LANES 32
|
||||
#define IGNEUM_ITERATIONS 8
|
||||
#define IGNEUM_INSTR_COUNT 64
|
||||
#define IGNEUM_LOADS_PER_HASH 128
|
||||
#define IGNEUM_WIDE_LOADS_PER_HASH 0
|
||||
#define IGNEUM_OP_MIX "load=16 add=8 shfl=8 xor=6 mad=5 mul=5 mulhi=5 sub=4 rotl=3 rotr=3 or=1"
|
||||
// Program class v4 (Counter ASIC 3.0, docs/plans/counter-asic-3-node.md): generator version 4, class v3 plus the
|
||||
// latency-shadow block (IGNEUM_SHADOW_INSTRS x IGNEUM_SHADOW_REPS per iteration); a worker that runs another class
|
||||
// refuses this pack, and a job line names the class it wants (class=v4 era=<hex>).
|
||||
#define IGNEUM_PROGRAM_CLASS "v4"
|
||||
#define IGNEUM_PROGRAM_SUBVERSION 3
|
||||
// Class v3 construction (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): version 2 loads; the dataset item
|
||||
// derivation applies the mixer IGNEUM_MIXER_MULT times per round (memhard.h), and the cache follows the growth rule.
|
||||
#define IGNEUM_LOAD_CLASS "mx8+sh256x27"
|
||||
#define IGNEUM_CLASS_MIXER_MULT 8
|
||||
#define IGNEUM_CACHE_GROWTH 1 // 1: cache words = 2^(26 + doublings(day)), doublings = floor(log2(1 + day / 1460))
|
||||
#define IGNEUM_LOAD_SLOTS 16
|
||||
#define IGNEUM_LOAD_MIX { 100, 0, 0 }
|
||||
#define IGNEUM_LOAD_WIDTH_COUNTS { 16, 0, 0 } // loads of 4, 16, 64 bytes per program
|
||||
#define IGNEUM_BYTES_PER_HASH 512
|
||||
#define IGNEUM_FOLD_ROT 11
|
||||
#define IGNEUM_FOLD_MUL 0x9e3779b1u
|
||||
// Latency-shadow block (Counter ASIC 3.0 item 8, docs/analysis/latency-shadow-2026-10-06.md): NOT the lottery hash. A block of
|
||||
// IGNEUM_SHADOW_INSTRS ALU instructions (the ten non-load families) runs IGNEUM_SHADOW_REPS times at the end of every
|
||||
// iteration; the 16 loads, the acceptance rule and the base program are the class's without the shadow.
|
||||
#define IGNEUM_SHADOW_INSTRS 256
|
||||
#define IGNEUM_SHADOW_REPS 27
|
||||
#define IGNEUM_SHADOW_INSTRS_PER_HASH 55296
|
||||
#define IGNEUM_SHADOW_OP_MIX "add=47 rotl=30 xor=30 shfl=29 mad=27 mul=22 sub=21 rotr=20 mulhi=18 or=12"
|
||||
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
|
||||
#define IGNEUM_DATASET_MODE 1
|
||||
|
||||
#define IGNEUM_SEEDW_INIT { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }
|
||||
#define IGNEUM_KEY_INIT { 0x3067619fu, 0x3c269176u, 0x84a03b03u, 0xf8c63294u, 0xff977c5bu, 0xe60def3eu, 0x63630141u, 0xb8fbcb58u }
|
||||
#define IGNEUM_CACHE_LOG2_WORDS 26
|
||||
#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6
|
||||
#define IGNEUM_CACHE_SEGMENTS 65536u
|
||||
#define IGNEUM_ITEM_ROUNDS 8
|
||||
#define IGNEUM_MIXER_MULT 8 // mixer applications per round and after the last read (class v3, docs/plans/mixer-x4.md)
|
||||
#define IGNEUM_MIX_ROT_INIT { 20u, 20u, 19u, 4u, 26u, 3u, 3u, 27u }
|
||||
#define IGNEUM_MIX_MUL_INIT { 0x42146205u, 0x52cbe0fbu, 0x7ecf4a03u, 0x6728907fu, 0xd81d9751u, 0x132952c3u, 0xf60de277u, 0x05358035u, 0xbaf6499du, 0xe4db9667u, 0x3e98f45du, 0xd0004eddu, 0x2691630du, 0x9beb3bcfu, 0xab310379u, 0x99cfb423u }
|
||||
#define IGNEUM_MIX_RC_INIT { 0xbab68293u, 0xcc162340u, 0x6ce151ccu, 0xe62b8997u, 0xc9c80297u, 0xf74a1654u, 0x3d704af5u, 0x3cf522b7u, 0x2b9cac04u, 0xa880ac10u, 0x13e5dd1du, 0x6fc3e233u, 0x2d83eeacu, 0x9006e8bfu, 0x2c4b5362u, 0x31b49ee2u }
|
||||
|
||||
#ifndef IGNEUM_NO_CUDA
|
||||
// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().
|
||||
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);
|
||||
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);
|
||||
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
uint32_t nonces, uint32_t blockWarps);
|
||||
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
|
||||
#endif
|
||||
391
proto-cuda/packs-ca3-v5/v4-genesis/program.json
Normal file
391
proto-cuda/packs-ca3-v5/v4-genesis/program.json
Normal file
|
|
@ -0,0 +1,391 @@
|
|||
{
|
||||
"format": "igneum-program-pack-3",
|
||||
"generator": 4,
|
||||
"attempt": 0,
|
||||
"program_id": "0xa217c7f698880830",
|
||||
"program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32",
|
||||
"dataset_mode": "memory-hard",
|
||||
"seed": "igneum-genesis",
|
||||
"seed_bytes": "69676e65756d2d67656e65736973",
|
||||
"seed_words": ["0x67a9a7be", "0x1a155b25", "0xfddfb732", "0x4b5af2e8", "0xc55caf33", "0xa27c13b7", "0x06628a48", "0x03852469"],
|
||||
"seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32",
|
||||
"generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried",
|
||||
"lanes": 32,
|
||||
"registers": 8,
|
||||
"iterations": 8,
|
||||
"instruction_count": 64,
|
||||
"loads_per_hash": 128,
|
||||
"program_class": "v4",
|
||||
"sub_version": 3,
|
||||
"load_class": "mx8+sh256x27",
|
||||
"mixer_mult": 8,
|
||||
"cache_growth": true,
|
||||
"mixer": "class v3 (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): every mixer application of the item derivation is 8 applications with round keys (r * 8 + j + 1) * 0x9E3779B9, the 8 dependent cache reads per item unchanged; cache growth rule option C: cache words = 2^(26 + doublings(day)), dataset words = 2^(genesis_log2 + doublings(day)), doublings(day) = floor(log2(1 + day / 1460)) for day = days since genesis",
|
||||
"load_slots": 16,
|
||||
"load_mix_percent_4_16_64": [100, 0, 0],
|
||||
"load_width_counts_4_16_64": [16, 0, 0],
|
||||
"bytes_per_hash": 512,
|
||||
"wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots",
|
||||
"op_mix": {"load": 16, "add": 8, "shfl": 8, "xor": 6, "mad": 5, "mul": 5, "mulhi": 5, "sub": 4, "rotl": 3, "rotr": 3, "or": 1},
|
||||
"register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]",
|
||||
"splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16",
|
||||
"iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order",
|
||||
"output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo",
|
||||
"op_semantics": {
|
||||
"add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)",
|
||||
"sub": "dst = dst - src",
|
||||
"mul": "dst = dst * src (low 32)",
|
||||
"mulhi": "dst = high 32 bits of dst * src",
|
||||
"xor": "dst = dst ^ src",
|
||||
"or": "dst = dst | src",
|
||||
"rotl": "dst = rotl(dst, rot), rot in 1..31",
|
||||
"rotr": "dst = rotr(dst, src & 31)",
|
||||
"mad": "dst = src * src2 + dst",
|
||||
"shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp",
|
||||
"load": "dst = dst ^ dataset[src & dataset.mask]",
|
||||
"wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)"
|
||||
},
|
||||
"dataset": {
|
||||
"log2_words": 28,
|
||||
"bytes": 1073741824,
|
||||
"mask": "0x0fffffff",
|
||||
"day": "2026-10-03",
|
||||
"day_bytes": "6461792f323032362d31302d3033",
|
||||
"day_words_from": "seed_words_from_bytes(day_bytes)",
|
||||
"d0": "0x3067619f",
|
||||
"d1": "0x3c269176",
|
||||
"mode": "memory-hard",
|
||||
"spec": "proto-metal/MEMHARD.md",
|
||||
"key": ["0x3067619f", "0x3c269176", "0x84a03b03", "0xf8c63294", "0xff977c5b", "0xe60def3e", "0x63630141", "0xb8fbcb58"],
|
||||
"key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]",
|
||||
"cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"},
|
||||
"mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [20, 20, 19, 4, 26, 3, 3, 27], "mul": ["0x42146205", "0x52cbe0fb", "0x7ecf4a03", "0x6728907f", "0xd81d9751", "0x132952c3", "0xf60de277", "0x05358035", "0xbaf6499d", "0xe4db9667", "0x3e98f45d", "0xd0004edd", "0x2691630d", "0x9beb3bcf", "0xab310379", "0x99cfb423"], "rc": ["0xbab68293", "0xcc162340", "0x6ce151cc", "0xe62b8997", "0xc9c80297", "0xf74a1654", "0x3d704af5", "0x3cf522b7", "0x2b9cac04", "0xa880ac10", "0x13e5dd1d", "0x6fc3e233", "0x2d83eeac", "0x9006e8bf", "0x2c4b5362", "0x31b49ee2"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"},
|
||||
"mixer_mult": 8,
|
||||
"item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: for j in 0..7: s = M(s, rk = (r * 8 + j + 1) * 0x9E3779B9); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then for j in 0..7: s = M(s, rk = (64 + j + 1) * 0x9E3779B9); item(t) = s",
|
||||
"word": "dataset[w] = item(w >> 4)[w & 15]"
|
||||
},
|
||||
"shadow": {"instrs": 256, "reps": 27, "instrs_per_hash": 55296, "op_mix": {"add": 47, "rotl": 30, "xor": 30, "shfl": 29, "mad": 27, "mul": 22, "sub": 21, "rotr": 20, "mulhi": 18, "or": 12}, "rule": "Counter ASIC 3.0 item 8 (docs/analysis/latency-shadow-2026-10-06.md): after the 64 base instructions the program stream draws instrs more ALU instructions (the non-load table, nine draws each, the source as on an ALU slot); the block runs reps times at the end of every iteration with the iteration's sel; the acceptance rule interprets the base program only", "program_id_suffix": "'shadow/' || instrs_le16 || reps_le16", "instructions": [
|
||||
{"i": 0, "op": "add", "dst": 5, "src": 2, "src2": 1, "imm": "0x92199f99", "imm2": "0x8bc12da9", "rot": 9, "bit": 18, "mask": 4},
|
||||
{"i": 1, "op": "add", "dst": 0, "src": 7, "src2": 1, "imm": "0x8d72d3ad", "imm2": "0x63079e5a", "rot": 20, "bit": 26, "mask": 4},
|
||||
{"i": 2, "op": "shfl", "dst": 6, "src": 3, "src2": 6, "imm": "0x9beaeddf", "imm2": "0x744ecb00", "rot": 10, "bit": 4, "mask": 2},
|
||||
{"i": 3, "op": "sub", "dst": 4, "src": 2, "src2": 0, "imm": "0x2a3ddc67", "imm2": "0x72c80241", "rot": 30, "bit": 1, "mask": 16},
|
||||
{"i": 4, "op": "add", "dst": 7, "src": 0, "src2": 5, "imm": "0xb21b4bab", "imm2": "0x5d4c7a60", "rot": 17, "bit": 31, "mask": 4},
|
||||
{"i": 5, "op": "rotl", "dst": 0, "src": 6, "src2": 3, "imm": "0xe69d7919", "imm2": "0xa048c61e", "rot": 11, "bit": 1, "mask": 8},
|
||||
{"i": 6, "op": "add", "dst": 0, "src": 1, "src2": 3, "imm": "0x2fe0e98b", "imm2": "0xc88e2942", "rot": 5, "bit": 16, "mask": 16},
|
||||
{"i": 7, "op": "xor", "dst": 7, "src": 1, "src2": 0, "imm": "0xc16efe98", "imm2": "0x01a638c8", "rot": 8, "bit": 17, "mask": 1},
|
||||
{"i": 8, "op": "mad", "dst": 1, "src": 6, "src2": 5, "imm": "0x76ec7b8b", "imm2": "0x25663feb", "rot": 29, "bit": 30, "mask": 2},
|
||||
{"i": 9, "op": "shfl", "dst": 6, "src": 1, "src2": 0, "imm": "0x6c8ee3cb", "imm2": "0xea93237e", "rot": 27, "bit": 1, "mask": 4},
|
||||
{"i": 10, "op": "mad", "dst": 1, "src": 2, "src2": 2, "imm": "0x6dc4ea18", "imm2": "0x6efde6f5", "rot": 3, "bit": 20, "mask": 2},
|
||||
{"i": 11, "op": "mad", "dst": 5, "src": 0, "src2": 3, "imm": "0x023613fc", "imm2": "0x18c51939", "rot": 19, "bit": 31, "mask": 1},
|
||||
{"i": 12, "op": "shfl", "dst": 2, "src": 6, "src2": 1, "imm": "0x74aec8d2", "imm2": "0x7f7ad29c", "rot": 7, "bit": 8, "mask": 4},
|
||||
{"i": 13, "op": "add", "dst": 1, "src": 5, "src2": 6, "imm": "0xbdf8f9a5", "imm2": "0xbc48c63e", "rot": 6, "bit": 25, "mask": 2},
|
||||
{"i": 14, "op": "rotl", "dst": 1, "src": 2, "src2": 6, "imm": "0x7ad8ca8b", "imm2": "0x14c712ad", "rot": 29, "bit": 25, "mask": 8},
|
||||
{"i": 15, "op": "sub", "dst": 1, "src": 4, "src2": 5, "imm": "0xd3349f69", "imm2": "0x70aac45a", "rot": 29, "bit": 9, "mask": 16},
|
||||
{"i": 16, "op": "or", "dst": 7, "src": 1, "src2": 2, "imm": "0x08168ed1", "imm2": "0x28964037", "rot": 26, "bit": 0, "mask": 2},
|
||||
{"i": 17, "op": "shfl", "dst": 2, "src": 4, "src2": 6, "imm": "0x98d85f72", "imm2": "0x3dc87042", "rot": 29, "bit": 22, "mask": 8},
|
||||
{"i": 18, "op": "xor", "dst": 7, "src": 4, "src2": 0, "imm": "0x651d4a0e", "imm2": "0x93c198bd", "rot": 26, "bit": 12, "mask": 8},
|
||||
{"i": 19, "op": "mul", "dst": 6, "src": 1, "src2": 6, "imm": "0xd38ce89d", "imm2": "0x3dfad388", "rot": 5, "bit": 28, "mask": 8},
|
||||
{"i": 20, "op": "mad", "dst": 5, "src": 6, "src2": 0, "imm": "0x59d78b36", "imm2": "0xc4db274e", "rot": 25, "bit": 24, "mask": 1},
|
||||
{"i": 21, "op": "sub", "dst": 3, "src": 1, "src2": 7, "imm": "0xf6c6a6c7", "imm2": "0x8536f4e6", "rot": 6, "bit": 12, "mask": 4},
|
||||
{"i": 22, "op": "mul", "dst": 6, "src": 0, "src2": 7, "imm": "0x599445b4", "imm2": "0x632f8c32", "rot": 19, "bit": 4, "mask": 8},
|
||||
{"i": 23, "op": "add", "dst": 2, "src": 0, "src2": 7, "imm": "0x45c37cec", "imm2": "0x96e8f127", "rot": 15, "bit": 1, "mask": 4},
|
||||
{"i": 24, "op": "sub", "dst": 6, "src": 4, "src2": 3, "imm": "0xc1c44491", "imm2": "0x92e3ce57", "rot": 20, "bit": 25, "mask": 4},
|
||||
{"i": 25, "op": "mad", "dst": 7, "src": 3, "src2": 4, "imm": "0x89bfb8d3", "imm2": "0x19b5455e", "rot": 22, "bit": 2, "mask": 16},
|
||||
{"i": 26, "op": "rotl", "dst": 3, "src": 5, "src2": 7, "imm": "0x2f47ce8d", "imm2": "0x8b458ec5", "rot": 9, "bit": 0, "mask": 4},
|
||||
{"i": 27, "op": "sub", "dst": 2, "src": 1, "src2": 2, "imm": "0x9f0dce23", "imm2": "0x3cdca814", "rot": 25, "bit": 1, "mask": 2},
|
||||
{"i": 28, "op": "mulhi", "dst": 6, "src": 0, "src2": 3, "imm": "0x61fc9eb8", "imm2": "0x23202e9d", "rot": 30, "bit": 16, "mask": 16},
|
||||
{"i": 29, "op": "shfl", "dst": 2, "src": 4, "src2": 3, "imm": "0x339dbd65", "imm2": "0xc175f639", "rot": 15, "bit": 22, "mask": 2},
|
||||
{"i": 30, "op": "shfl", "dst": 1, "src": 6, "src2": 1, "imm": "0x5bd14589", "imm2": "0xa68a2bed", "rot": 31, "bit": 31, "mask": 1},
|
||||
{"i": 31, "op": "add", "dst": 1, "src": 6, "src2": 1, "imm": "0x37985632", "imm2": "0xb1cdb2ab", "rot": 29, "bit": 1, "mask": 8},
|
||||
{"i": 32, "op": "rotr", "dst": 0, "src": 1, "src2": 6, "imm": "0x4b2058f4", "imm2": "0xf06d8ac9", "rot": 9, "bit": 15, "mask": 16},
|
||||
{"i": 33, "op": "shfl", "dst": 3, "src": 4, "src2": 4, "imm": "0x2081626c", "imm2": "0x08d1bb87", "rot": 24, "bit": 29, "mask": 2},
|
||||
{"i": 34, "op": "shfl", "dst": 0, "src": 3, "src2": 6, "imm": "0xe4fcfe03", "imm2": "0x78a46b13", "rot": 15, "bit": 29, "mask": 4},
|
||||
{"i": 35, "op": "mul", "dst": 7, "src": 0, "src2": 4, "imm": "0x33148d30", "imm2": "0x5780a3d6", "rot": 6, "bit": 11, "mask": 2},
|
||||
{"i": 36, "op": "add", "dst": 3, "src": 1, "src2": 1, "imm": "0x804e777e", "imm2": "0x856e0180", "rot": 22, "bit": 0, "mask": 16},
|
||||
{"i": 37, "op": "xor", "dst": 1, "src": 6, "src2": 0, "imm": "0xc69aa2f6", "imm2": "0xafebe3e8", "rot": 15, "bit": 9, "mask": 16},
|
||||
{"i": 38, "op": "mul", "dst": 2, "src": 7, "src2": 4, "imm": "0xeb4c4082", "imm2": "0x3f1e0da7", "rot": 25, "bit": 20, "mask": 16},
|
||||
{"i": 39, "op": "add", "dst": 6, "src": 5, "src2": 5, "imm": "0x0b06c8a7", "imm2": "0xe9bd0cc4", "rot": 6, "bit": 2, "mask": 4},
|
||||
{"i": 40, "op": "mul", "dst": 5, "src": 7, "src2": 7, "imm": "0x070af1b6", "imm2": "0x6b648e75", "rot": 10, "bit": 6, "mask": 16},
|
||||
{"i": 41, "op": "mulhi", "dst": 2, "src": 3, "src2": 2, "imm": "0x389753b2", "imm2": "0x9e657305", "rot": 30, "bit": 2, "mask": 4},
|
||||
{"i": 42, "op": "xor", "dst": 2, "src": 0, "src2": 4, "imm": "0x0960e5b9", "imm2": "0xa925a406", "rot": 4, "bit": 6, "mask": 16},
|
||||
{"i": 43, "op": "rotl", "dst": 0, "src": 1, "src2": 1, "imm": "0x8eabe09f", "imm2": "0x76ef71e2", "rot": 4, "bit": 19, "mask": 8},
|
||||
{"i": 44, "op": "mulhi", "dst": 4, "src": 3, "src2": 7, "imm": "0x6a34768a", "imm2": "0x2e427c64", "rot": 21, "bit": 5, "mask": 4},
|
||||
{"i": 45, "op": "add", "dst": 6, "src": 7, "src2": 5, "imm": "0xee822e17", "imm2": "0xcdcb63f6", "rot": 27, "bit": 12, "mask": 16},
|
||||
{"i": 46, "op": "mad", "dst": 3, "src": 2, "src2": 0, "imm": "0xc89ead43", "imm2": "0x4d3108b2", "rot": 26, "bit": 17, "mask": 1},
|
||||
{"i": 47, "op": "shfl", "dst": 4, "src": 3, "src2": 0, "imm": "0xdd62be9e", "imm2": "0xf243f6f2", "rot": 8, "bit": 4, "mask": 8},
|
||||
{"i": 48, "op": "mulhi", "dst": 6, "src": 5, "src2": 0, "imm": "0xd3f7fa72", "imm2": "0x0debc83f", "rot": 25, "bit": 5, "mask": 2},
|
||||
{"i": 49, "op": "sub", "dst": 0, "src": 2, "src2": 6, "imm": "0x7afadd15", "imm2": "0xc07fc39c", "rot": 30, "bit": 29, "mask": 8},
|
||||
{"i": 50, "op": "sub", "dst": 3, "src": 5, "src2": 3, "imm": "0x179b78ec", "imm2": "0xaeccde37", "rot": 30, "bit": 17, "mask": 1},
|
||||
{"i": 51, "op": "rotr", "dst": 1, "src": 4, "src2": 1, "imm": "0xa4bfcea6", "imm2": "0xbf63bb2f", "rot": 2, "bit": 21, "mask": 2},
|
||||
{"i": 52, "op": "mad", "dst": 6, "src": 7, "src2": 7, "imm": "0xabf5ed10", "imm2": "0x8a285f51", "rot": 3, "bit": 23, "mask": 2},
|
||||
{"i": 53, "op": "xor", "dst": 5, "src": 3, "src2": 5, "imm": "0x88b416eb", "imm2": "0x36271d87", "rot": 19, "bit": 10, "mask": 2},
|
||||
{"i": 54, "op": "sub", "dst": 1, "src": 0, "src2": 1, "imm": "0xa3417dd3", "imm2": "0xcd98f620", "rot": 28, "bit": 2, "mask": 1},
|
||||
{"i": 55, "op": "sub", "dst": 5, "src": 6, "src2": 2, "imm": "0x45996c6f", "imm2": "0xe3d41087", "rot": 4, "bit": 31, "mask": 1},
|
||||
{"i": 56, "op": "add", "dst": 3, "src": 2, "src2": 0, "imm": "0x3dfad1b6", "imm2": "0xd4758987", "rot": 27, "bit": 10, "mask": 2},
|
||||
{"i": 57, "op": "add", "dst": 4, "src": 1, "src2": 3, "imm": "0xfca75bc2", "imm2": "0x0602d6be", "rot": 19, "bit": 0, "mask": 16},
|
||||
{"i": 58, "op": "mulhi", "dst": 5, "src": 6, "src2": 5, "imm": "0x83250a7b", "imm2": "0x2f93d53b", "rot": 25, "bit": 7, "mask": 2},
|
||||
{"i": 59, "op": "shfl", "dst": 2, "src": 1, "src2": 0, "imm": "0x13f4a089", "imm2": "0x145ea125", "rot": 12, "bit": 3, "mask": 2},
|
||||
{"i": 60, "op": "rotl", "dst": 2, "src": 5, "src2": 6, "imm": "0xad3170e3", "imm2": "0x15db04d1", "rot": 9, "bit": 13, "mask": 2},
|
||||
{"i": 61, "op": "or", "dst": 4, "src": 6, "src2": 2, "imm": "0x4b6305b2", "imm2": "0x6e7b2e9c", "rot": 27, "bit": 16, "mask": 4},
|
||||
{"i": 62, "op": "rotr", "dst": 6, "src": 4, "src2": 6, "imm": "0xfcd4b1c9", "imm2": "0xaf5c733b", "rot": 6, "bit": 21, "mask": 16},
|
||||
{"i": 63, "op": "mul", "dst": 2, "src": 4, "src2": 6, "imm": "0x95cebb3e", "imm2": "0xbba9cdfc", "rot": 19, "bit": 23, "mask": 1},
|
||||
{"i": 64, "op": "xor", "dst": 0, "src": 5, "src2": 7, "imm": "0x5f7579ff", "imm2": "0xb104e01a", "rot": 25, "bit": 12, "mask": 8},
|
||||
{"i": 65, "op": "xor", "dst": 2, "src": 7, "src2": 3, "imm": "0x749c7611", "imm2": "0xf17df3bb", "rot": 27, "bit": 26, "mask": 2},
|
||||
{"i": 66, "op": "add", "dst": 2, "src": 1, "src2": 6, "imm": "0xd71c02ff", "imm2": "0x8596687a", "rot": 4, "bit": 6, "mask": 2},
|
||||
{"i": 67, "op": "rotl", "dst": 1, "src": 3, "src2": 2, "imm": "0xd2864071", "imm2": "0xaa644ef3", "rot": 21, "bit": 5, "mask": 1},
|
||||
{"i": 68, "op": "mul", "dst": 3, "src": 2, "src2": 1, "imm": "0xd0689a5f", "imm2": "0xa3a1f063", "rot": 13, "bit": 28, "mask": 2},
|
||||
{"i": 69, "op": "shfl", "dst": 7, "src": 5, "src2": 6, "imm": "0xd01476eb", "imm2": "0x76abb5c2", "rot": 7, "bit": 30, "mask": 2},
|
||||
{"i": 70, "op": "mul", "dst": 3, "src": 2, "src2": 5, "imm": "0x59a75c10", "imm2": "0x1e1fbb72", "rot": 20, "bit": 16, "mask": 1},
|
||||
{"i": 71, "op": "mad", "dst": 0, "src": 2, "src2": 5, "imm": "0x39cca1df", "imm2": "0x22c057ac", "rot": 5, "bit": 23, "mask": 16},
|
||||
{"i": 72, "op": "add", "dst": 6, "src": 7, "src2": 3, "imm": "0x3c6fe15d", "imm2": "0x08ac5733", "rot": 14, "bit": 0, "mask": 4},
|
||||
{"i": 73, "op": "shfl", "dst": 3, "src": 2, "src2": 6, "imm": "0x2098627d", "imm2": "0xf4fecc81", "rot": 20, "bit": 30, "mask": 8},
|
||||
{"i": 74, "op": "mul", "dst": 3, "src": 6, "src2": 5, "imm": "0xf82e9e23", "imm2": "0x3ad62132", "rot": 10, "bit": 14, "mask": 16},
|
||||
{"i": 75, "op": "xor", "dst": 6, "src": 7, "src2": 4, "imm": "0x70548a91", "imm2": "0xa9715d2e", "rot": 6, "bit": 24, "mask": 1},
|
||||
{"i": 76, "op": "or", "dst": 3, "src": 0, "src2": 7, "imm": "0xa164325f", "imm2": "0xf300838b", "rot": 23, "bit": 23, "mask": 16},
|
||||
{"i": 77, "op": "add", "dst": 2, "src": 4, "src2": 6, "imm": "0x295d5fae", "imm2": "0xc100b495", "rot": 7, "bit": 1, "mask": 2},
|
||||
{"i": 78, "op": "mulhi", "dst": 3, "src": 7, "src2": 4, "imm": "0xb62cca87", "imm2": "0x2ebde415", "rot": 14, "bit": 7, "mask": 2},
|
||||
{"i": 79, "op": "or", "dst": 4, "src": 1, "src2": 7, "imm": "0xfcbc482d", "imm2": "0x8876e6cd", "rot": 29, "bit": 5, "mask": 2},
|
||||
{"i": 80, "op": "rotr", "dst": 4, "src": 3, "src2": 0, "imm": "0x6d64013b", "imm2": "0x675f4a8d", "rot": 26, "bit": 18, "mask": 8},
|
||||
{"i": 81, "op": "add", "dst": 4, "src": 3, "src2": 3, "imm": "0x274a9221", "imm2": "0x5cc59530", "rot": 15, "bit": 15, "mask": 2},
|
||||
{"i": 82, "op": "add", "dst": 7, "src": 3, "src2": 0, "imm": "0xc8651f8e", "imm2": "0x141479ec", "rot": 18, "bit": 5, "mask": 1},
|
||||
{"i": 83, "op": "rotr", "dst": 3, "src": 6, "src2": 0, "imm": "0xcc8a7766", "imm2": "0xc2eb5161", "rot": 4, "bit": 29, "mask": 2},
|
||||
{"i": 84, "op": "mad", "dst": 2, "src": 4, "src2": 7, "imm": "0xbc507d51", "imm2": "0x0b1196fd", "rot": 9, "bit": 7, "mask": 8},
|
||||
{"i": 85, "op": "shfl", "dst": 2, "src": 5, "src2": 5, "imm": "0x5a156c90", "imm2": "0xa6b3fbfa", "rot": 11, "bit": 2, "mask": 16},
|
||||
{"i": 86, "op": "xor", "dst": 3, "src": 2, "src2": 1, "imm": "0x8b042658", "imm2": "0xacf37a8f", "rot": 12, "bit": 23, "mask": 16},
|
||||
{"i": 87, "op": "shfl", "dst": 5, "src": 7, "src2": 2, "imm": "0x1a214238", "imm2": "0x017fdf5d", "rot": 29, "bit": 14, "mask": 2},
|
||||
{"i": 88, "op": "xor", "dst": 0, "src": 4, "src2": 2, "imm": "0xd523e612", "imm2": "0x2158c2ed", "rot": 30, "bit": 14, "mask": 4},
|
||||
{"i": 89, "op": "add", "dst": 3, "src": 2, "src2": 7, "imm": "0x53f915c2", "imm2": "0x883c0c92", "rot": 9, "bit": 18, "mask": 8},
|
||||
{"i": 90, "op": "rotr", "dst": 7, "src": 5, "src2": 5, "imm": "0xbd633b21", "imm2": "0xcf8c356c", "rot": 25, "bit": 31, "mask": 4},
|
||||
{"i": 91, "op": "add", "dst": 3, "src": 1, "src2": 1, "imm": "0xe2be00a5", "imm2": "0x3cc5bd20", "rot": 10, "bit": 31, "mask": 1},
|
||||
{"i": 92, "op": "shfl", "dst": 7, "src": 2, "src2": 4, "imm": "0x3d3da600", "imm2": "0x1f22df89", "rot": 4, "bit": 24, "mask": 1},
|
||||
{"i": 93, "op": "rotr", "dst": 6, "src": 2, "src2": 7, "imm": "0xb0f57471", "imm2": "0x8f2a7eca", "rot": 16, "bit": 25, "mask": 1},
|
||||
{"i": 94, "op": "rotl", "dst": 7, "src": 0, "src2": 4, "imm": "0x00a21815", "imm2": "0xeb4d7218", "rot": 14, "bit": 18, "mask": 16},
|
||||
{"i": 95, "op": "mad", "dst": 4, "src": 5, "src2": 5, "imm": "0xc4c3828e", "imm2": "0xfb7bba17", "rot": 16, "bit": 10, "mask": 16},
|
||||
{"i": 96, "op": "mad", "dst": 2, "src": 4, "src2": 5, "imm": "0xfc5bbc75", "imm2": "0xfc43468f", "rot": 31, "bit": 20, "mask": 16},
|
||||
{"i": 97, "op": "xor", "dst": 1, "src": 0, "src2": 2, "imm": "0x1fe9c249", "imm2": "0x164bb16b", "rot": 15, "bit": 9, "mask": 4},
|
||||
{"i": 98, "op": "mul", "dst": 5, "src": 4, "src2": 1, "imm": "0xeda725fa", "imm2": "0x66f7e9a2", "rot": 21, "bit": 6, "mask": 2},
|
||||
{"i": 99, "op": "sub", "dst": 2, "src": 0, "src2": 7, "imm": "0x8fb29f7d", "imm2": "0xf530eda1", "rot": 27, "bit": 30, "mask": 4},
|
||||
{"i": 100, "op": "rotl", "dst": 7, "src": 2, "src2": 0, "imm": "0x9f4de742", "imm2": "0x9b5ff871", "rot": 30, "bit": 21, "mask": 16},
|
||||
{"i": 101, "op": "add", "dst": 5, "src": 6, "src2": 7, "imm": "0x423fd9c9", "imm2": "0xbfd646cb", "rot": 17, "bit": 17, "mask": 2},
|
||||
{"i": 102, "op": "add", "dst": 7, "src": 6, "src2": 3, "imm": "0x8d3c011d", "imm2": "0x19b74a43", "rot": 12, "bit": 11, "mask": 4},
|
||||
{"i": 103, "op": "rotl", "dst": 5, "src": 6, "src2": 0, "imm": "0xcfc70303", "imm2": "0xf2b3e8ef", "rot": 6, "bit": 24, "mask": 1},
|
||||
{"i": 104, "op": "mul", "dst": 0, "src": 4, "src2": 5, "imm": "0x650475eb", "imm2": "0x11dcbd94", "rot": 15, "bit": 10, "mask": 1},
|
||||
{"i": 105, "op": "or", "dst": 0, "src": 5, "src2": 3, "imm": "0xaacaa145", "imm2": "0x9139d3fe", "rot": 4, "bit": 18, "mask": 2},
|
||||
{"i": 106, "op": "add", "dst": 0, "src": 1, "src2": 7, "imm": "0x8ce14721", "imm2": "0x7dcb7e18", "rot": 24, "bit": 15, "mask": 8},
|
||||
{"i": 107, "op": "rotl", "dst": 0, "src": 3, "src2": 3, "imm": "0xb36d6d98", "imm2": "0x2c3390c8", "rot": 15, "bit": 10, "mask": 16},
|
||||
{"i": 108, "op": "sub", "dst": 4, "src": 2, "src2": 5, "imm": "0xf6bbdaef", "imm2": "0x9db6f65e", "rot": 25, "bit": 10, "mask": 1},
|
||||
{"i": 109, "op": "mulhi", "dst": 2, "src": 0, "src2": 0, "imm": "0xb1721fd6", "imm2": "0xd96d52c9", "rot": 1, "bit": 27, "mask": 8},
|
||||
{"i": 110, "op": "mulhi", "dst": 1, "src": 0, "src2": 5, "imm": "0xfb4ca37f", "imm2": "0xb6ec7dbe", "rot": 27, "bit": 12, "mask": 8},
|
||||
{"i": 111, "op": "add", "dst": 0, "src": 2, "src2": 0, "imm": "0x2d9fc6b9", "imm2": "0x3c179ad8", "rot": 9, "bit": 28, "mask": 16},
|
||||
{"i": 112, "op": "mulhi", "dst": 2, "src": 0, "src2": 6, "imm": "0x71a9f6bd", "imm2": "0xd3417bf5", "rot": 2, "bit": 14, "mask": 8},
|
||||
{"i": 113, "op": "sub", "dst": 6, "src": 3, "src2": 5, "imm": "0xa3d19400", "imm2": "0x15df7330", "rot": 5, "bit": 2, "mask": 8},
|
||||
{"i": 114, "op": "mad", "dst": 7, "src": 6, "src2": 3, "imm": "0x1400d92e", "imm2": "0xfa4a158e", "rot": 16, "bit": 9, "mask": 1},
|
||||
{"i": 115, "op": "rotr", "dst": 5, "src": 1, "src2": 3, "imm": "0xd01684c3", "imm2": "0x6367febb", "rot": 23, "bit": 19, "mask": 2},
|
||||
{"i": 116, "op": "rotr", "dst": 6, "src": 4, "src2": 2, "imm": "0x8cd9eb96", "imm2": "0x98f33b24", "rot": 25, "bit": 1, "mask": 2},
|
||||
{"i": 117, "op": "add", "dst": 2, "src": 1, "src2": 7, "imm": "0x502138e2", "imm2": "0x47359729", "rot": 5, "bit": 22, "mask": 4},
|
||||
{"i": 118, "op": "mad", "dst": 3, "src": 2, "src2": 0, "imm": "0xd94a35c7", "imm2": "0xb83416c3", "rot": 11, "bit": 3, "mask": 4},
|
||||
{"i": 119, "op": "mul", "dst": 1, "src": 3, "src2": 6, "imm": "0x09b30cf7", "imm2": "0xef3b98e7", "rot": 17, "bit": 23, "mask": 8},
|
||||
{"i": 120, "op": "mad", "dst": 2, "src": 0, "src2": 4, "imm": "0x56416fe0", "imm2": "0x5e1dc8f3", "rot": 17, "bit": 14, "mask": 2},
|
||||
{"i": 121, "op": "mul", "dst": 0, "src": 3, "src2": 2, "imm": "0x2892697c", "imm2": "0x9cb3b14e", "rot": 29, "bit": 25, "mask": 2},
|
||||
{"i": 122, "op": "mulhi", "dst": 2, "src": 0, "src2": 3, "imm": "0xc33089a1", "imm2": "0xd6c5b530", "rot": 27, "bit": 13, "mask": 8},
|
||||
{"i": 123, "op": "mul", "dst": 5, "src": 6, "src2": 4, "imm": "0xdfe04e4a", "imm2": "0x3cc40160", "rot": 3, "bit": 14, "mask": 2},
|
||||
{"i": 124, "op": "mad", "dst": 4, "src": 2, "src2": 4, "imm": "0xb33dc1b7", "imm2": "0xab97743a", "rot": 7, "bit": 21, "mask": 4},
|
||||
{"i": 125, "op": "shfl", "dst": 4, "src": 2, "src2": 0, "imm": "0xfd469909", "imm2": "0xfc85dc55", "rot": 28, "bit": 24, "mask": 2},
|
||||
{"i": 126, "op": "xor", "dst": 2, "src": 0, "src2": 5, "imm": "0xa0094578", "imm2": "0xf1bee474", "rot": 31, "bit": 7, "mask": 2},
|
||||
{"i": 127, "op": "rotl", "dst": 1, "src": 7, "src2": 2, "imm": "0xb7ae00a0", "imm2": "0x86cce297", "rot": 7, "bit": 31, "mask": 8},
|
||||
{"i": 128, "op": "shfl", "dst": 7, "src": 2, "src2": 7, "imm": "0x019d1940", "imm2": "0x3fe9e7dd", "rot": 14, "bit": 4, "mask": 4},
|
||||
{"i": 129, "op": "rotl", "dst": 6, "src": 7, "src2": 7, "imm": "0x40a74cc4", "imm2": "0x09eb4adf", "rot": 17, "bit": 30, "mask": 1},
|
||||
{"i": 130, "op": "add", "dst": 2, "src": 5, "src2": 5, "imm": "0x33dff776", "imm2": "0x6c57e4e7", "rot": 27, "bit": 15, "mask": 4},
|
||||
{"i": 131, "op": "or", "dst": 0, "src": 3, "src2": 1, "imm": "0xe0c53bb9", "imm2": "0x124404f6", "rot": 4, "bit": 21, "mask": 16},
|
||||
{"i": 132, "op": "rotr", "dst": 5, "src": 0, "src2": 1, "imm": "0xe4353fce", "imm2": "0x559d0118", "rot": 2, "bit": 29, "mask": 4},
|
||||
{"i": 133, "op": "add", "dst": 2, "src": 3, "src2": 6, "imm": "0x5db25b34", "imm2": "0xe8212c0c", "rot": 21, "bit": 30, "mask": 1},
|
||||
{"i": 134, "op": "xor", "dst": 4, "src": 3, "src2": 6, "imm": "0x7366e50e", "imm2": "0x7cc1ffbd", "rot": 19, "bit": 18, "mask": 16},
|
||||
{"i": 135, "op": "xor", "dst": 6, "src": 1, "src2": 2, "imm": "0xa36decb7", "imm2": "0x79291734", "rot": 26, "bit": 0, "mask": 8},
|
||||
{"i": 136, "op": "sub", "dst": 1, "src": 7, "src2": 0, "imm": "0x225b03e4", "imm2": "0x7183e193", "rot": 16, "bit": 6, "mask": 2},
|
||||
{"i": 137, "op": "mad", "dst": 2, "src": 6, "src2": 2, "imm": "0x6f620d51", "imm2": "0x4e814e19", "rot": 6, "bit": 6, "mask": 2},
|
||||
{"i": 138, "op": "mulhi", "dst": 0, "src": 5, "src2": 6, "imm": "0x919d6bf3", "imm2": "0xc6240638", "rot": 17, "bit": 11, "mask": 16},
|
||||
{"i": 139, "op": "add", "dst": 2, "src": 7, "src2": 7, "imm": "0x218d4090", "imm2": "0xd44ff710", "rot": 1, "bit": 10, "mask": 16},
|
||||
{"i": 140, "op": "rotr", "dst": 1, "src": 4, "src2": 4, "imm": "0x4cbfa722", "imm2": "0x114e9564", "rot": 28, "bit": 9, "mask": 16},
|
||||
{"i": 141, "op": "shfl", "dst": 3, "src": 6, "src2": 2, "imm": "0xf702c6a1", "imm2": "0xf8f3c5d9", "rot": 7, "bit": 24, "mask": 1},
|
||||
{"i": 142, "op": "mad", "dst": 7, "src": 6, "src2": 2, "imm": "0xa2d285be", "imm2": "0x9cc94532", "rot": 15, "bit": 20, "mask": 8},
|
||||
{"i": 143, "op": "mul", "dst": 2, "src": 3, "src2": 7, "imm": "0x08b7ed80", "imm2": "0xa8abced4", "rot": 18, "bit": 10, "mask": 4},
|
||||
{"i": 144, "op": "add", "dst": 7, "src": 4, "src2": 2, "imm": "0xf4b1a8de", "imm2": "0xb98942fa", "rot": 18, "bit": 29, "mask": 8},
|
||||
{"i": 145, "op": "rotl", "dst": 7, "src": 6, "src2": 1, "imm": "0x64b6ba2d", "imm2": "0xf9e86793", "rot": 15, "bit": 24, "mask": 8},
|
||||
{"i": 146, "op": "xor", "dst": 7, "src": 5, "src2": 3, "imm": "0x41c42b5f", "imm2": "0x7c7e0e39", "rot": 8, "bit": 23, "mask": 2},
|
||||
{"i": 147, "op": "mad", "dst": 4, "src": 7, "src2": 4, "imm": "0x3caa807a", "imm2": "0x553a0cec", "rot": 11, "bit": 22, "mask": 8},
|
||||
{"i": 148, "op": "rotr", "dst": 6, "src": 5, "src2": 3, "imm": "0x3cb1289a", "imm2": "0x6b36f78a", "rot": 1, "bit": 9, "mask": 16},
|
||||
{"i": 149, "op": "mad", "dst": 1, "src": 2, "src2": 4, "imm": "0x95310ea8", "imm2": "0xc533aa6a", "rot": 16, "bit": 17, "mask": 1},
|
||||
{"i": 150, "op": "add", "dst": 1, "src": 7, "src2": 7, "imm": "0x2bef10f2", "imm2": "0x0d48ba42", "rot": 8, "bit": 17, "mask": 4},
|
||||
{"i": 151, "op": "xor", "dst": 5, "src": 4, "src2": 7, "imm": "0xda997ab2", "imm2": "0xae31d69e", "rot": 11, "bit": 31, "mask": 2},
|
||||
{"i": 152, "op": "add", "dst": 7, "src": 0, "src2": 6, "imm": "0xdc5cc080", "imm2": "0xc98dea9c", "rot": 16, "bit": 30, "mask": 2},
|
||||
{"i": 153, "op": "mul", "dst": 4, "src": 6, "src2": 7, "imm": "0xbfbf5f6c", "imm2": "0x61f611bb", "rot": 14, "bit": 22, "mask": 2},
|
||||
{"i": 154, "op": "mul", "dst": 0, "src": 2, "src2": 3, "imm": "0xa78008c3", "imm2": "0xbad5eeb1", "rot": 10, "bit": 28, "mask": 8},
|
||||
{"i": 155, "op": "xor", "dst": 6, "src": 5, "src2": 4, "imm": "0x1a73b866", "imm2": "0x1f5a62c9", "rot": 9, "bit": 27, "mask": 8},
|
||||
{"i": 156, "op": "rotr", "dst": 4, "src": 2, "src2": 0, "imm": "0x9c88500c", "imm2": "0xe25dccf9", "rot": 11, "bit": 2, "mask": 16},
|
||||
{"i": 157, "op": "rotl", "dst": 1, "src": 4, "src2": 4, "imm": "0xb05e7669", "imm2": "0x9db704b1", "rot": 11, "bit": 28, "mask": 4},
|
||||
{"i": 158, "op": "add", "dst": 5, "src": 0, "src2": 6, "imm": "0x18197438", "imm2": "0x6c752dcb", "rot": 11, "bit": 11, "mask": 2},
|
||||
{"i": 159, "op": "mad", "dst": 4, "src": 1, "src2": 5, "imm": "0x4796a65e", "imm2": "0x00e08c7a", "rot": 8, "bit": 19, "mask": 4},
|
||||
{"i": 160, "op": "rotl", "dst": 4, "src": 7, "src2": 5, "imm": "0x08a03052", "imm2": "0x0204f0ba", "rot": 26, "bit": 5, "mask": 2},
|
||||
{"i": 161, "op": "mad", "dst": 3, "src": 0, "src2": 7, "imm": "0xa0603b0e", "imm2": "0x7eee83d5", "rot": 28, "bit": 22, "mask": 16},
|
||||
{"i": 162, "op": "rotr", "dst": 3, "src": 1, "src2": 6, "imm": "0x78a3c69d", "imm2": "0x684693a0", "rot": 21, "bit": 23, "mask": 4},
|
||||
{"i": 163, "op": "add", "dst": 4, "src": 0, "src2": 7, "imm": "0xf1c46574", "imm2": "0x8e481727", "rot": 9, "bit": 3, "mask": 4},
|
||||
{"i": 164, "op": "mulhi", "dst": 0, "src": 4, "src2": 3, "imm": "0x04644afa", "imm2": "0x64bda2b5", "rot": 26, "bit": 16, "mask": 16},
|
||||
{"i": 165, "op": "add", "dst": 2, "src": 6, "src2": 2, "imm": "0xac578137", "imm2": "0x550ab406", "rot": 13, "bit": 20, "mask": 16},
|
||||
{"i": 166, "op": "rotl", "dst": 0, "src": 4, "src2": 5, "imm": "0x7f564760", "imm2": "0xb9a8b4f8", "rot": 13, "bit": 29, "mask": 1},
|
||||
{"i": 167, "op": "add", "dst": 3, "src": 1, "src2": 0, "imm": "0xa33e6706", "imm2": "0xaebb5966", "rot": 12, "bit": 14, "mask": 16},
|
||||
{"i": 168, "op": "or", "dst": 3, "src": 5, "src2": 3, "imm": "0x65b2f3eb", "imm2": "0xb1d00d20", "rot": 12, "bit": 2, "mask": 8},
|
||||
{"i": 169, "op": "rotr", "dst": 6, "src": 2, "src2": 3, "imm": "0x7f21faf5", "imm2": "0xbbf0d3f9", "rot": 17, "bit": 13, "mask": 8},
|
||||
{"i": 170, "op": "xor", "dst": 4, "src": 6, "src2": 2, "imm": "0x162c7140", "imm2": "0x90d404ad", "rot": 26, "bit": 1, "mask": 4},
|
||||
{"i": 171, "op": "sub", "dst": 6, "src": 1, "src2": 7, "imm": "0x6ef9c76e", "imm2": "0xfb7ba272", "rot": 31, "bit": 25, "mask": 1},
|
||||
{"i": 172, "op": "rotl", "dst": 7, "src": 5, "src2": 4, "imm": "0xde04eb3b", "imm2": "0xd56caa00", "rot": 22, "bit": 21, "mask": 4},
|
||||
{"i": 173, "op": "rotl", "dst": 5, "src": 3, "src2": 5, "imm": "0xa9eac934", "imm2": "0x2c338e51", "rot": 15, "bit": 22, "mask": 2},
|
||||
{"i": 174, "op": "shfl", "dst": 7, "src": 0, "src2": 3, "imm": "0x700be4e3", "imm2": "0x4bcfc732", "rot": 19, "bit": 6, "mask": 8},
|
||||
{"i": 175, "op": "xor", "dst": 0, "src": 5, "src2": 1, "imm": "0x02ccdba9", "imm2": "0xd0915be0", "rot": 15, "bit": 2, "mask": 16},
|
||||
{"i": 176, "op": "rotl", "dst": 7, "src": 4, "src2": 1, "imm": "0x489c8165", "imm2": "0xf24b5a4f", "rot": 6, "bit": 22, "mask": 1},
|
||||
{"i": 177, "op": "sub", "dst": 7, "src": 0, "src2": 6, "imm": "0x23ad9693", "imm2": "0x9a8c2f7b", "rot": 24, "bit": 2, "mask": 2},
|
||||
{"i": 178, "op": "rotl", "dst": 3, "src": 0, "src2": 6, "imm": "0xafa72a42", "imm2": "0x371d74ee", "rot": 30, "bit": 16, "mask": 1},
|
||||
{"i": 179, "op": "mad", "dst": 7, "src": 6, "src2": 1, "imm": "0x18a2a3f3", "imm2": "0xb811b951", "rot": 2, "bit": 9, "mask": 2},
|
||||
{"i": 180, "op": "rotl", "dst": 6, "src": 4, "src2": 1, "imm": "0xe6c69c0e", "imm2": "0xfc46a951", "rot": 9, "bit": 31, "mask": 4},
|
||||
{"i": 181, "op": "xor", "dst": 2, "src": 4, "src2": 7, "imm": "0xbd1b89e4", "imm2": "0xdf4bce5c", "rot": 15, "bit": 19, "mask": 4},
|
||||
{"i": 182, "op": "xor", "dst": 2, "src": 7, "src2": 5, "imm": "0x3dd12aed", "imm2": "0xd0756a69", "rot": 13, "bit": 16, "mask": 4},
|
||||
{"i": 183, "op": "xor", "dst": 7, "src": 2, "src2": 3, "imm": "0x77ce69d6", "imm2": "0x2b39bdf2", "rot": 8, "bit": 22, "mask": 16},
|
||||
{"i": 184, "op": "add", "dst": 1, "src": 2, "src2": 4, "imm": "0xd94d55ac", "imm2": "0x5bb7550f", "rot": 31, "bit": 21, "mask": 1},
|
||||
{"i": 185, "op": "or", "dst": 3, "src": 5, "src2": 6, "imm": "0x7b1ce846", "imm2": "0xcc3b8509", "rot": 28, "bit": 9, "mask": 4},
|
||||
{"i": 186, "op": "mulhi", "dst": 6, "src": 3, "src2": 0, "imm": "0xa5e24690", "imm2": "0x2200ba81", "rot": 26, "bit": 10, "mask": 16},
|
||||
{"i": 187, "op": "or", "dst": 4, "src": 0, "src2": 3, "imm": "0xf6efe759", "imm2": "0xae1f7118", "rot": 19, "bit": 20, "mask": 4},
|
||||
{"i": 188, "op": "shfl", "dst": 7, "src": 1, "src2": 5, "imm": "0xdb318b45", "imm2": "0xcec459c9", "rot": 20, "bit": 11, "mask": 2},
|
||||
{"i": 189, "op": "add", "dst": 6, "src": 5, "src2": 1, "imm": "0x89e747fe", "imm2": "0x2a354e2d", "rot": 22, "bit": 6, "mask": 1},
|
||||
{"i": 190, "op": "xor", "dst": 1, "src": 4, "src2": 2, "imm": "0xc379e617", "imm2": "0x75e9d63b", "rot": 23, "bit": 3, "mask": 2},
|
||||
{"i": 191, "op": "mulhi", "dst": 7, "src": 4, "src2": 3, "imm": "0x5a710287", "imm2": "0x7fe4ead6", "rot": 10, "bit": 25, "mask": 4},
|
||||
{"i": 192, "op": "rotl", "dst": 2, "src": 1, "src2": 2, "imm": "0x4d5e59a3", "imm2": "0x1e4fef28", "rot": 26, "bit": 0, "mask": 1},
|
||||
{"i": 193, "op": "rotl", "dst": 5, "src": 3, "src2": 7, "imm": "0x7444c47d", "imm2": "0xdad5f8be", "rot": 8, "bit": 30, "mask": 2},
|
||||
{"i": 194, "op": "shfl", "dst": 4, "src": 5, "src2": 7, "imm": "0x6a225bbb", "imm2": "0xd6532cd7", "rot": 13, "bit": 17, "mask": 16},
|
||||
{"i": 195, "op": "xor", "dst": 4, "src": 5, "src2": 3, "imm": "0x68344b9a", "imm2": "0xcb46a38b", "rot": 1, "bit": 27, "mask": 16},
|
||||
{"i": 196, "op": "mul", "dst": 1, "src": 3, "src2": 4, "imm": "0xbebf7359", "imm2": "0x3f0890ba", "rot": 18, "bit": 12, "mask": 4},
|
||||
{"i": 197, "op": "add", "dst": 5, "src": 2, "src2": 3, "imm": "0xea03e8e7", "imm2": "0x10cfdc71", "rot": 31, "bit": 7, "mask": 8},
|
||||
{"i": 198, "op": "add", "dst": 7, "src": 5, "src2": 1, "imm": "0xa7ee0102", "imm2": "0x66e148d3", "rot": 12, "bit": 15, "mask": 1},
|
||||
{"i": 199, "op": "sub", "dst": 5, "src": 2, "src2": 4, "imm": "0xc215584e", "imm2": "0x55fce30f", "rot": 30, "bit": 9, "mask": 2},
|
||||
{"i": 200, "op": "shfl", "dst": 5, "src": 1, "src2": 1, "imm": "0x308f8358", "imm2": "0x489f1f93", "rot": 13, "bit": 13, "mask": 2},
|
||||
{"i": 201, "op": "shfl", "dst": 7, "src": 1, "src2": 5, "imm": "0x713f1957", "imm2": "0x2239f757", "rot": 15, "bit": 7, "mask": 8},
|
||||
{"i": 202, "op": "rotr", "dst": 4, "src": 5, "src2": 1, "imm": "0xb037b6e7", "imm2": "0x49a8b528", "rot": 27, "bit": 13, "mask": 16},
|
||||
{"i": 203, "op": "xor", "dst": 7, "src": 5, "src2": 1, "imm": "0x56110241", "imm2": "0xefc8386d", "rot": 22, "bit": 20, "mask": 1},
|
||||
{"i": 204, "op": "xor", "dst": 7, "src": 0, "src2": 0, "imm": "0x0827fcc7", "imm2": "0xf4f5dd07", "rot": 13, "bit": 7, "mask": 2},
|
||||
{"i": 205, "op": "add", "dst": 6, "src": 2, "src2": 0, "imm": "0x8379a4de", "imm2": "0x8558b619", "rot": 26, "bit": 10, "mask": 1},
|
||||
{"i": 206, "op": "rotl", "dst": 4, "src": 3, "src2": 3, "imm": "0x091ceef0", "imm2": "0x69b0f72f", "rot": 13, "bit": 7, "mask": 1},
|
||||
{"i": 207, "op": "mulhi", "dst": 1, "src": 3, "src2": 4, "imm": "0x758edac1", "imm2": "0x98dd2f21", "rot": 15, "bit": 9, "mask": 1},
|
||||
{"i": 208, "op": "mad", "dst": 1, "src": 4, "src2": 4, "imm": "0x78871a1a", "imm2": "0x5e14e1c8", "rot": 19, "bit": 6, "mask": 2},
|
||||
{"i": 209, "op": "shfl", "dst": 2, "src": 0, "src2": 2, "imm": "0xf7e0b9aa", "imm2": "0xaecfd347", "rot": 21, "bit": 9, "mask": 4},
|
||||
{"i": 210, "op": "add", "dst": 7, "src": 0, "src2": 7, "imm": "0xaa8cb14e", "imm2": "0xf4049c4c", "rot": 24, "bit": 11, "mask": 8},
|
||||
{"i": 211, "op": "shfl", "dst": 7, "src": 1, "src2": 6, "imm": "0xc230d919", "imm2": "0xf76d08fb", "rot": 21, "bit": 13, "mask": 16},
|
||||
{"i": 212, "op": "xor", "dst": 1, "src": 0, "src2": 2, "imm": "0x31a5af90", "imm2": "0x7eef58ee", "rot": 9, "bit": 12, "mask": 4},
|
||||
{"i": 213, "op": "rotl", "dst": 7, "src": 1, "src2": 6, "imm": "0x0db26138", "imm2": "0x8e3c31f9", "rot": 3, "bit": 26, "mask": 2},
|
||||
{"i": 214, "op": "rotr", "dst": 4, "src": 7, "src2": 3, "imm": "0x29558100", "imm2": "0xe4b13ad6", "rot": 29, "bit": 24, "mask": 8},
|
||||
{"i": 215, "op": "shfl", "dst": 3, "src": 2, "src2": 6, "imm": "0x1b48c3d0", "imm2": "0x5c674ff6", "rot": 5, "bit": 9, "mask": 16},
|
||||
{"i": 216, "op": "or", "dst": 5, "src": 1, "src2": 0, "imm": "0xef768632", "imm2": "0x6de9d10d", "rot": 10, "bit": 4, "mask": 4},
|
||||
{"i": 217, "op": "sub", "dst": 1, "src": 2, "src2": 5, "imm": "0x88ca7f5a", "imm2": "0x24718a36", "rot": 18, "bit": 31, "mask": 1},
|
||||
{"i": 218, "op": "sub", "dst": 6, "src": 5, "src2": 0, "imm": "0xc4a06728", "imm2": "0xdc2a4fd8", "rot": 9, "bit": 9, "mask": 2},
|
||||
{"i": 219, "op": "rotl", "dst": 6, "src": 5, "src2": 7, "imm": "0xb7489e47", "imm2": "0xf13795c5", "rot": 4, "bit": 3, "mask": 8},
|
||||
{"i": 220, "op": "mulhi", "dst": 2, "src": 0, "src2": 2, "imm": "0x873cd31b", "imm2": "0x3dfbc55d", "rot": 12, "bit": 23, "mask": 8},
|
||||
{"i": 221, "op": "or", "dst": 2, "src": 0, "src2": 3, "imm": "0xdc21f099", "imm2": "0xee06f01e", "rot": 2, "bit": 17, "mask": 8},
|
||||
{"i": 222, "op": "shfl", "dst": 5, "src": 2, "src2": 6, "imm": "0x36def499", "imm2": "0xa2849d59", "rot": 23, "bit": 4, "mask": 8},
|
||||
{"i": 223, "op": "mulhi", "dst": 5, "src": 6, "src2": 7, "imm": "0xfb95fbca", "imm2": "0xc1aac427", "rot": 14, "bit": 11, "mask": 2},
|
||||
{"i": 224, "op": "sub", "dst": 0, "src": 6, "src2": 7, "imm": "0x3d9d29c4", "imm2": "0x34d0dcc0", "rot": 17, "bit": 6, "mask": 4},
|
||||
{"i": 225, "op": "rotl", "dst": 7, "src": 0, "src2": 5, "imm": "0x93b01b8e", "imm2": "0xfe1d75ac", "rot": 23, "bit": 15, "mask": 1},
|
||||
{"i": 226, "op": "or", "dst": 4, "src": 2, "src2": 4, "imm": "0xfed76e8e", "imm2": "0x1c24ecd8", "rot": 2, "bit": 6, "mask": 16},
|
||||
{"i": 227, "op": "mul", "dst": 2, "src": 4, "src2": 1, "imm": "0x5795f5b0", "imm2": "0x0566ea2a", "rot": 6, "bit": 28, "mask": 4},
|
||||
{"i": 228, "op": "rotl", "dst": 3, "src": 0, "src2": 2, "imm": "0x8c8486de", "imm2": "0xd066aa8f", "rot": 12, "bit": 20, "mask": 1},
|
||||
{"i": 229, "op": "rotr", "dst": 0, "src": 4, "src2": 0, "imm": "0x23a3e882", "imm2": "0xaca23902", "rot": 5, "bit": 22, "mask": 16},
|
||||
{"i": 230, "op": "add", "dst": 0, "src": 6, "src2": 4, "imm": "0x1d176220", "imm2": "0x2a7fecb2", "rot": 20, "bit": 16, "mask": 2},
|
||||
{"i": 231, "op": "shfl", "dst": 1, "src": 2, "src2": 0, "imm": "0x91172787", "imm2": "0xc5d7af28", "rot": 3, "bit": 24, "mask": 4},
|
||||
{"i": 232, "op": "mul", "dst": 2, "src": 1, "src2": 2, "imm": "0x0be68835", "imm2": "0xde692bdb", "rot": 30, "bit": 17, "mask": 1},
|
||||
{"i": 233, "op": "mad", "dst": 7, "src": 0, "src2": 1, "imm": "0xf90d2db5", "imm2": "0x96c4c175", "rot": 28, "bit": 7, "mask": 4},
|
||||
{"i": 234, "op": "rotl", "dst": 5, "src": 2, "src2": 1, "imm": "0x16379736", "imm2": "0x6973b905", "rot": 22, "bit": 17, "mask": 4},
|
||||
{"i": 235, "op": "rotr", "dst": 4, "src": 6, "src2": 2, "imm": "0x84292a13", "imm2": "0x0f897740", "rot": 12, "bit": 6, "mask": 8},
|
||||
{"i": 236, "op": "mad", "dst": 0, "src": 5, "src2": 1, "imm": "0xeb7de837", "imm2": "0x64f0c302", "rot": 4, "bit": 19, "mask": 1},
|
||||
{"i": 237, "op": "xor", "dst": 6, "src": 4, "src2": 4, "imm": "0x6ab4b683", "imm2": "0x20f17adb", "rot": 1, "bit": 0, "mask": 16},
|
||||
{"i": 238, "op": "xor", "dst": 4, "src": 6, "src2": 1, "imm": "0x5836b35c", "imm2": "0x3293cc4a", "rot": 16, "bit": 22, "mask": 16},
|
||||
{"i": 239, "op": "rotl", "dst": 6, "src": 1, "src2": 6, "imm": "0x769d8bc0", "imm2": "0xdc86c9cc", "rot": 18, "bit": 27, "mask": 16},
|
||||
{"i": 240, "op": "add", "dst": 4, "src": 7, "src2": 1, "imm": "0x17dafb4d", "imm2": "0xadce39f3", "rot": 12, "bit": 25, "mask": 8},
|
||||
{"i": 241, "op": "sub", "dst": 0, "src": 7, "src2": 4, "imm": "0xd4690bda", "imm2": "0xdab9b27b", "rot": 30, "bit": 21, "mask": 1},
|
||||
{"i": 242, "op": "rotr", "dst": 1, "src": 6, "src2": 2, "imm": "0x670a2d0a", "imm2": "0x0612e33c", "rot": 31, "bit": 25, "mask": 2},
|
||||
{"i": 243, "op": "add", "dst": 3, "src": 0, "src2": 2, "imm": "0xf2f77d26", "imm2": "0x0e7033b6", "rot": 27, "bit": 29, "mask": 1},
|
||||
{"i": 244, "op": "mad", "dst": 2, "src": 7, "src2": 4, "imm": "0xa9eefc9d", "imm2": "0x16166c85", "rot": 18, "bit": 23, "mask": 16},
|
||||
{"i": 245, "op": "mad", "dst": 4, "src": 0, "src2": 6, "imm": "0xb26f6c0f", "imm2": "0x0630e821", "rot": 23, "bit": 30, "mask": 4},
|
||||
{"i": 246, "op": "mul", "dst": 5, "src": 7, "src2": 2, "imm": "0x269ce4bf", "imm2": "0x1ce28d32", "rot": 30, "bit": 15, "mask": 16},
|
||||
{"i": 247, "op": "add", "dst": 3, "src": 5, "src2": 0, "imm": "0xfed2da4e", "imm2": "0x7b2ff6b7", "rot": 25, "bit": 31, "mask": 2},
|
||||
{"i": 248, "op": "add", "dst": 0, "src": 7, "src2": 5, "imm": "0x03b2891c", "imm2": "0xb5fad7b1", "rot": 2, "bit": 0, "mask": 4},
|
||||
{"i": 249, "op": "mulhi", "dst": 5, "src": 4, "src2": 6, "imm": "0xa69a1e71", "imm2": "0x15f0c0ea", "rot": 4, "bit": 1, "mask": 4},
|
||||
{"i": 250, "op": "add", "dst": 5, "src": 2, "src2": 3, "imm": "0x042cd6e3", "imm2": "0xa7e71c2f", "rot": 24, "bit": 11, "mask": 8},
|
||||
{"i": 251, "op": "shfl", "dst": 6, "src": 1, "src2": 2, "imm": "0x89c6b683", "imm2": "0x10bbf661", "rot": 24, "bit": 5, "mask": 1},
|
||||
{"i": 252, "op": "xor", "dst": 6, "src": 1, "src2": 3, "imm": "0x265c66d6", "imm2": "0xd4a689ed", "rot": 6, "bit": 14, "mask": 8},
|
||||
{"i": 253, "op": "mul", "dst": 2, "src": 5, "src2": 5, "imm": "0x890b8201", "imm2": "0x97c36bf3", "rot": 17, "bit": 22, "mask": 4},
|
||||
{"i": 254, "op": "add", "dst": 0, "src": 6, "src2": 0, "imm": "0x784a302b", "imm2": "0xb83d78de", "rot": 27, "bit": 16, "mask": 4},
|
||||
{"i": 255, "op": "sub", "dst": 2, "src": 4, "src2": 0, "imm": "0x817adb38", "imm2": "0xf3a3534b", "rot": 7, "bit": 22, "mask": 4}
|
||||
]},
|
||||
"instructions": [
|
||||
{"i": 0, "op": "mad", "dst": 2, "src": 3, "src2": 4, "imm": "0xbf7b174d", "imm2": "0x337b762e", "rot": 17, "bit": 2, "mask": 2, "width": 1},
|
||||
{"i": 1, "op": "mad", "dst": 2, "src": 1, "src2": 1, "imm": "0xdd04a5da", "imm2": "0x42da7657", "rot": 15, "bit": 30, "mask": 16, "width": 1},
|
||||
{"i": 2, "op": "mad", "dst": 2, "src": 3, "src2": 2, "imm": "0x734003fa", "imm2": "0x5bb67700", "rot": 3, "bit": 20, "mask": 1, "width": 1},
|
||||
{"i": 3, "op": "xor", "dst": 3, "src": 5, "src2": 5, "imm": "0xc55a1b1c", "imm2": "0xa19720f3", "rot": 7, "bit": 8, "mask": 1, "width": 1},
|
||||
{"i": 4, "op": "load", "dst": 7, "src": 2, "src2": 5, "imm": "0xad572dd7", "imm2": "0x9ceb3ea7", "rot": 18, "bit": 30, "mask": 2, "width": 1},
|
||||
{"i": 5, "op": "load", "dst": 5, "src": 7, "src2": 3, "imm": "0x769a53be", "imm2": "0x80f9067e", "rot": 12, "bit": 22, "mask": 1, "width": 1},
|
||||
{"i": 6, "op": "shfl", "dst": 1, "src": 4, "src2": 0, "imm": "0xd3613d88", "imm2": "0x262fb219", "rot": 10, "bit": 30, "mask": 8, "width": 1},
|
||||
{"i": 7, "op": "shfl", "dst": 7, "src": 3, "src2": 4, "imm": "0xce38e42f", "imm2": "0xb868b818", "rot": 11, "bit": 8, "mask": 8, "width": 1},
|
||||
{"i": 8, "op": "mulhi", "dst": 1, "src": 5, "src2": 2, "imm": "0xa5eebca5", "imm2": "0x5703a72b", "rot": 13, "bit": 13, "mask": 16, "width": 1},
|
||||
{"i": 9, "op": "rotr", "dst": 6, "src": 3, "src2": 4, "imm": "0x17a5a9c7", "imm2": "0xdcfb93a1", "rot": 20, "bit": 27, "mask": 2, "width": 1},
|
||||
{"i": 10, "op": "or", "dst": 3, "src": 4, "src2": 1, "imm": "0xccb7d785", "imm2": "0xc335364c", "rot": 14, "bit": 12, "mask": 4, "width": 1},
|
||||
{"i": 11, "op": "load", "dst": 4, "src": 6, "src2": 2, "imm": "0x88cb9af3", "imm2": "0x4e7dc10d", "rot": 24, "bit": 17, "mask": 4, "width": 1},
|
||||
{"i": 12, "op": "mulhi", "dst": 0, "src": 4, "src2": 4, "imm": "0x45374321", "imm2": "0x3cd91989", "rot": 11, "bit": 4, "mask": 2, "width": 1},
|
||||
{"i": 13, "op": "add", "dst": 5, "src": 1, "src2": 2, "imm": "0xc7934706", "imm2": "0xd3177981", "rot": 16, "bit": 30, "mask": 2, "width": 1},
|
||||
{"i": 14, "op": "load", "dst": 0, "src": 5, "src2": 3, "imm": "0xfd7f56bb", "imm2": "0x65e14f52", "rot": 13, "bit": 22, "mask": 2, "width": 1},
|
||||
{"i": 15, "op": "sub", "dst": 2, "src": 4, "src2": 4, "imm": "0x35a80b49", "imm2": "0x060f2d13", "rot": 16, "bit": 20, "mask": 16, "width": 1},
|
||||
{"i": 16, "op": "load", "dst": 2, "src": 7, "src2": 2, "imm": "0xae0a32c2", "imm2": "0x4c2a4cfe", "rot": 8, "bit": 31, "mask": 16, "width": 1},
|
||||
{"i": 17, "op": "load", "dst": 7, "src": 4, "src2": 6, "imm": "0x82a84cc3", "imm2": "0x21a38d68", "rot": 15, "bit": 21, "mask": 2, "width": 1},
|
||||
{"i": 18, "op": "shfl", "dst": 7, "src": 3, "src2": 3, "imm": "0xa3818806", "imm2": "0x8f66b5c8", "rot": 14, "bit": 6, "mask": 4, "width": 1},
|
||||
{"i": 19, "op": "mul", "dst": 5, "src": 0, "src2": 1, "imm": "0xa00de107", "imm2": "0x77bfcaa5", "rot": 3, "bit": 10, "mask": 2, "width": 1},
|
||||
{"i": 20, "op": "shfl", "dst": 3, "src": 4, "src2": 7, "imm": "0x1d2b8cab", "imm2": "0x80b4f9a2", "rot": 14, "bit": 25, "mask": 2, "width": 1},
|
||||
{"i": 21, "op": "shfl", "dst": 2, "src": 4, "src2": 5, "imm": "0x3ac915d2", "imm2": "0x5fba7bc2", "rot": 16, "bit": 1, "mask": 16, "width": 1},
|
||||
{"i": 22, "op": "mulhi", "dst": 6, "src": 2, "src2": 0, "imm": "0xdc3ec8fd", "imm2": "0x599e2fa3", "rot": 22, "bit": 3, "mask": 2, "width": 1},
|
||||
{"i": 23, "op": "load", "dst": 6, "src": 3, "src2": 5, "imm": "0x2a6b16d5", "imm2": "0xd73e396f", "rot": 28, "bit": 29, "mask": 2, "width": 1},
|
||||
{"i": 24, "op": "mul", "dst": 5, "src": 0, "src2": 7, "imm": "0x376d0223", "imm2": "0xe1c2169a", "rot": 4, "bit": 16, "mask": 16, "width": 1},
|
||||
{"i": 25, "op": "rotl", "dst": 5, "src": 7, "src2": 3, "imm": "0x78ad8c60", "imm2": "0x6f5b77d5", "rot": 19, "bit": 11, "mask": 16, "width": 1},
|
||||
{"i": 26, "op": "shfl", "dst": 7, "src": 6, "src2": 5, "imm": "0x93915b9f", "imm2": "0x1e61fb6b", "rot": 28, "bit": 23, "mask": 2, "width": 1},
|
||||
{"i": 27, "op": "xor", "dst": 0, "src": 5, "src2": 7, "imm": "0x6378fe15", "imm2": "0x66c78f42", "rot": 12, "bit": 31, "mask": 8, "width": 1},
|
||||
{"i": 28, "op": "xor", "dst": 0, "src": 4, "src2": 7, "imm": "0x20a57fda", "imm2": "0x088c848e", "rot": 16, "bit": 13, "mask": 4, "width": 1},
|
||||
{"i": 29, "op": "sub", "dst": 3, "src": 0, "src2": 2, "imm": "0x49d95fd5", "imm2": "0x1a5f946a", "rot": 6, "bit": 12, "mask": 1, "width": 1},
|
||||
{"i": 30, "op": "mul", "dst": 5, "src": 1, "src2": 7, "imm": "0x0a816217", "imm2": "0x405c4f73", "rot": 13, "bit": 27, "mask": 4, "width": 1},
|
||||
{"i": 31, "op": "load", "dst": 7, "src": 3, "src2": 2, "imm": "0x09ed045e", "imm2": "0xd69c4715", "rot": 5, "bit": 9, "mask": 2, "width": 1},
|
||||
{"i": 32, "op": "load", "dst": 1, "src": 2, "src2": 6, "imm": "0xeb79ea49", "imm2": "0xcc587f5a", "rot": 6, "bit": 8, "mask": 16, "width": 1},
|
||||
{"i": 33, "op": "xor", "dst": 5, "src": 6, "src2": 1, "imm": "0x3027401e", "imm2": "0x5f20c27e", "rot": 18, "bit": 9, "mask": 2, "width": 1},
|
||||
{"i": 34, "op": "load", "dst": 5, "src": 0, "src2": 3, "imm": "0x0e1cab07", "imm2": "0x09356c5b", "rot": 19, "bit": 31, "mask": 1, "width": 1},
|
||||
{"i": 35, "op": "mulhi", "dst": 0, "src": 5, "src2": 2, "imm": "0x90e31357", "imm2": "0xabd32484", "rot": 26, "bit": 5, "mask": 8, "width": 1},
|
||||
{"i": 36, "op": "shfl", "dst": 5, "src": 2, "src2": 5, "imm": "0xee9a955f", "imm2": "0x31b3faed", "rot": 8, "bit": 24, "mask": 4, "width": 1},
|
||||
{"i": 37, "op": "load", "dst": 7, "src": 6, "src2": 7, "imm": "0x3ba2f832", "imm2": "0x1160dcd3", "rot": 4, "bit": 29, "mask": 1, "width": 1},
|
||||
{"i": 38, "op": "add", "dst": 3, "src": 1, "src2": 7, "imm": "0x75ba2fad", "imm2": "0x230c005c", "rot": 4, "bit": 27, "mask": 1, "width": 1},
|
||||
{"i": 39, "op": "shfl", "dst": 1, "src": 5, "src2": 3, "imm": "0xdbf37e75", "imm2": "0xb5ac1969", "rot": 30, "bit": 13, "mask": 4, "width": 1},
|
||||
{"i": 40, "op": "xor", "dst": 2, "src": 5, "src2": 5, "imm": "0x47f136c5", "imm2": "0x06ce9153", "rot": 19, "bit": 10, "mask": 2, "width": 1},
|
||||
{"i": 41, "op": "mad", "dst": 3, "src": 6, "src2": 3, "imm": "0xce13eff8", "imm2": "0x04cc1d55", "rot": 3, "bit": 1, "mask": 4, "width": 1},
|
||||
{"i": 42, "op": "sub", "dst": 6, "src": 7, "src2": 1, "imm": "0x6a65ab71", "imm2": "0x8fbc1bcd", "rot": 4, "bit": 1, "mask": 8, "width": 1},
|
||||
{"i": 43, "op": "xor", "dst": 7, "src": 0, "src2": 7, "imm": "0xdaeb4928", "imm2": "0xc0423027", "rot": 24, "bit": 11, "mask": 8, "width": 1},
|
||||
{"i": 44, "op": "load", "dst": 1, "src": 7, "src2": 7, "imm": "0x778f01c9", "imm2": "0x28cedcea", "rot": 12, "bit": 4, "mask": 16, "width": 1},
|
||||
{"i": 45, "op": "mul", "dst": 2, "src": 3, "src2": 2, "imm": "0xf4264f1b", "imm2": "0x0f627d56", "rot": 5, "bit": 28, "mask": 8, "width": 1},
|
||||
{"i": 46, "op": "mulhi", "dst": 1, "src": 5, "src2": 1, "imm": "0xffb2147a", "imm2": "0xccde9b05", "rot": 13, "bit": 9, "mask": 2, "width": 1},
|
||||
{"i": 47, "op": "sub", "dst": 4, "src": 3, "src2": 5, "imm": "0x73b36234", "imm2": "0x3f5d5997", "rot": 7, "bit": 18, "mask": 2, "width": 1},
|
||||
{"i": 48, "op": "rotr", "dst": 2, "src": 6, "src2": 3, "imm": "0x3a4d9aa9", "imm2": "0x212bec7b", "rot": 4, "bit": 29, "mask": 16, "width": 1},
|
||||
{"i": 49, "op": "load", "dst": 3, "src": 4, "src2": 2, "imm": "0x626f11df", "imm2": "0x56cd5bfd", "rot": 7, "bit": 1, "mask": 1, "width": 1},
|
||||
{"i": 50, "op": "add", "dst": 1, "src": 5, "src2": 6, "imm": "0x81b8bc2c", "imm2": "0x1907970c", "rot": 28, "bit": 7, "mask": 4, "width": 1},
|
||||
{"i": 51, "op": "mul", "dst": 0, "src": 2, "src2": 2, "imm": "0xa8848b30", "imm2": "0xef6ac348", "rot": 9, "bit": 15, "mask": 8, "width": 1},
|
||||
{"i": 52, "op": "add", "dst": 0, "src": 2, "src2": 0, "imm": "0x4f92b968", "imm2": "0x699fd448", "rot": 22, "bit": 6, "mask": 4, "width": 1},
|
||||
{"i": 53, "op": "add", "dst": 1, "src": 0, "src2": 2, "imm": "0x2bb965af", "imm2": "0x77b1520d", "rot": 2, "bit": 12, "mask": 8, "width": 1},
|
||||
{"i": 54, "op": "rotl", "dst": 7, "src": 1, "src2": 0, "imm": "0x553e678b", "imm2": "0x3cc8eae0", "rot": 14, "bit": 20, "mask": 2, "width": 1},
|
||||
{"i": 55, "op": "add", "dst": 3, "src": 7, "src2": 2, "imm": "0x7b0fe07a", "imm2": "0xa54c55a0", "rot": 10, "bit": 1, "mask": 1, "width": 1},
|
||||
{"i": 56, "op": "load", "dst": 6, "src": 7, "src2": 4, "imm": "0x01eba9aa", "imm2": "0x2758c0f7", "rot": 14, "bit": 15, "mask": 4, "width": 1},
|
||||
{"i": 57, "op": "rotr", "dst": 1, "src": 5, "src2": 2, "imm": "0x1f5267b3", "imm2": "0x236f5a27", "rot": 2, "bit": 31, "mask": 16, "width": 1},
|
||||
{"i": 58, "op": "load", "dst": 5, "src": 3, "src2": 3, "imm": "0xa9954a9b", "imm2": "0x6a54d4e8", "rot": 11, "bit": 10, "mask": 16, "width": 1},
|
||||
{"i": 59, "op": "load", "dst": 6, "src": 1, "src2": 4, "imm": "0x9923ff88", "imm2": "0x9357254e", "rot": 16, "bit": 1, "mask": 16, "width": 1},
|
||||
{"i": 60, "op": "mad", "dst": 3, "src": 5, "src2": 0, "imm": "0xc0cc51a6", "imm2": "0x3fd7701b", "rot": 20, "bit": 1, "mask": 4, "width": 1},
|
||||
{"i": 61, "op": "add", "dst": 5, "src": 7, "src2": 7, "imm": "0xaf9dd72d", "imm2": "0xad7493e7", "rot": 7, "bit": 31, "mask": 16, "width": 1},
|
||||
{"i": 62, "op": "add", "dst": 4, "src": 6, "src2": 2, "imm": "0x89841d87", "imm2": "0x1e07c3d9", "rot": 6, "bit": 27, "mask": 1, "width": 1},
|
||||
{"i": 63, "op": "rotl", "dst": 5, "src": 4, "src2": 2, "imm": "0xae210f8d", "imm2": "0x8e499ba4", "rot": 19, "bit": 9, "mask": 1, "width": 1}
|
||||
]
|
||||
}
|
||||
368
proto-cuda/packs-ca3-v5/v4-genesis/program.metal
Normal file
368
proto-cuda/packs-ca3-v5/v4-genesis/program.metal
Normal file
|
|
@ -0,0 +1,368 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
|
||||
#define MASK 0x0fffffffu
|
||||
constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u };
|
||||
|
||||
inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
|
||||
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
|
||||
device ulong* out [[buffer(1)]],
|
||||
constant uint& baseNonce [[buffer(2)]],
|
||||
uint gid [[thread_position_in_grid]]) {
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint x = nonce ^ SEEDW[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ SEEDW[1]; }
|
||||
{ uint x = nonce ^ SEEDW[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ SEEDW[2]; }
|
||||
{ uint x = nonce ^ SEEDW[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ SEEDW[3]; }
|
||||
{ uint x = nonce ^ SEEDW[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ SEEDW[4]; }
|
||||
{ uint x = nonce ^ SEEDW[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ SEEDW[5]; }
|
||||
{ uint x = nonce ^ SEEDW[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ SEEDW[6]; }
|
||||
{ uint x = nonce ^ SEEDW[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ SEEDW[7]; }
|
||||
{ uint x = nonce ^ SEEDW[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ SEEDW[0]; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0
|
||||
r2 = r1 * r1 + r2; // 1
|
||||
r2 = r3 * r2 + r2; // 2
|
||||
r3 = r3 ^ r5; // 3
|
||||
r7 = r7 ^ dataset[r2 & MASK]; // 4
|
||||
r5 = r5 ^ dataset[r7 & MASK]; // 5
|
||||
r1 = r1 ^ simd_shuffle_xor(r4, (ushort)8); // 6
|
||||
r7 = r7 ^ simd_shuffle_xor(r3, (ushort)8); // 7
|
||||
r1 = mulhi(r1, r5); // 8
|
||||
r6 = rotr_var(r6, r3); // 9
|
||||
r3 = r3 | r4; // 10
|
||||
r4 = r4 ^ dataset[r6 & MASK]; // 11
|
||||
r0 = mulhi(r0, r4); // 12
|
||||
r5 = r5 + r1 + select(0xc7934706u, 0xd3177981u, ((sel >> 30u) & 1u) != 0u); // 13
|
||||
r0 = r0 ^ dataset[r5 & MASK]; // 14
|
||||
r2 = r2 - r4; // 15
|
||||
r2 = r2 ^ dataset[r7 & MASK]; // 16
|
||||
r7 = r7 ^ dataset[r4 & MASK]; // 17
|
||||
r7 = r7 ^ simd_shuffle_xor(r3, (ushort)4); // 18
|
||||
r5 = r5 * r0; // 19
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 20
|
||||
r2 = r2 ^ simd_shuffle_xor(r4, (ushort)16); // 21
|
||||
r6 = mulhi(r6, r2); // 22
|
||||
r6 = r6 ^ dataset[r3 & MASK]; // 23
|
||||
r5 = r5 * r0; // 24
|
||||
r5 = rotl_imm(r5, 19u); // 25
|
||||
r7 = r7 ^ simd_shuffle_xor(r6, (ushort)2); // 26
|
||||
r0 = r0 ^ r5; // 27
|
||||
r0 = r0 ^ r4; // 28
|
||||
r3 = r3 - r0; // 29
|
||||
r5 = r5 * r1; // 30
|
||||
r7 = r7 ^ dataset[r3 & MASK]; // 31
|
||||
r1 = r1 ^ dataset[r2 & MASK]; // 32
|
||||
r5 = r5 ^ r6; // 33
|
||||
r5 = r5 ^ dataset[r0 & MASK]; // 34
|
||||
r0 = mulhi(r0, r5); // 35
|
||||
r5 = r5 ^ simd_shuffle_xor(r2, (ushort)4); // 36
|
||||
r7 = r7 ^ dataset[r6 & MASK]; // 37
|
||||
r3 = r3 + r1 + select(0x75ba2fadu, 0x230c005cu, ((sel >> 27u) & 1u) != 0u); // 38
|
||||
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)4); // 39
|
||||
r2 = r2 ^ r5; // 40
|
||||
r3 = r6 * r3 + r3; // 41
|
||||
r6 = r6 - r7; // 42
|
||||
r7 = r7 ^ r0; // 43
|
||||
r1 = r1 ^ dataset[r7 & MASK]; // 44
|
||||
r2 = r2 * r3; // 45
|
||||
r1 = mulhi(r1, r5); // 46
|
||||
r4 = r4 - r3; // 47
|
||||
r2 = rotr_var(r2, r6); // 48
|
||||
r3 = r3 ^ dataset[r4 & MASK]; // 49
|
||||
r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 50
|
||||
r0 = r0 * r2; // 51
|
||||
r0 = r0 + r2 + select(0x4f92b968u, 0x699fd448u, ((sel >> 6u) & 1u) != 0u); // 52
|
||||
r1 = r1 + r0 + select(0x2bb965afu, 0x77b1520du, ((sel >> 12u) & 1u) != 0u); // 53
|
||||
r7 = rotl_imm(r7, 14u); // 54
|
||||
r3 = r3 + r7 + select(0x7b0fe07au, 0xa54c55a0u, ((sel >> 1u) & 1u) != 0u); // 55
|
||||
r6 = r6 ^ dataset[r7 & MASK]; // 56
|
||||
r1 = rotr_var(r1, r5); // 57
|
||||
r5 = r5 ^ dataset[r3 & MASK]; // 58
|
||||
r6 = r6 ^ dataset[r1 & MASK]; // 59
|
||||
r3 = r5 * r0 + r3; // 60
|
||||
r5 = r5 + r7 + select(0xaf9dd72du, 0xad7493e7u, ((sel >> 31u) & 1u) != 0u); // 61
|
||||
r4 = r4 + r6 + select(0x89841d87u, 0x1e07c3d9u, ((sel >> 27u) & 1u) != 0u); // 62
|
||||
r5 = rotl_imm(r5, 19u); // 63
|
||||
// latency-shadow block (Counter ASIC 3.0 item 8): 256 ALU instructions x 27 passes after instruction 63, no load
|
||||
for (uint sh = 0u; sh < 27u; ++sh) {
|
||||
r5 = r5 + r2 + select(0x92199f99u, 0x8bc12da9u, ((sel >> 18u) & 1u) != 0u); // s0 add
|
||||
r0 = r0 + r7 + select(0x8d72d3adu, 0x63079e5au, ((sel >> 26u) & 1u) != 0u); // s1 add
|
||||
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)2); // s2 shfl
|
||||
r4 = r4 - r2; // s3 sub
|
||||
r7 = r7 + r0 + select(0xb21b4babu, 0x5d4c7a60u, ((sel >> 31u) & 1u) != 0u); // s4 add
|
||||
r0 = rotl_imm(r0, 11u); // s5 rotl
|
||||
r0 = r0 + r1 + select(0x2fe0e98bu, 0xc88e2942u, ((sel >> 16u) & 1u) != 0u); // s6 add
|
||||
r7 = r7 ^ r1; // s7 xor
|
||||
r1 = r6 * r5 + r1; // s8 mad
|
||||
r6 = r6 ^ simd_shuffle_xor(r1, (ushort)4); // s9 shfl
|
||||
r1 = r2 * r2 + r1; // s10 mad
|
||||
r5 = r0 * r3 + r5; // s11 mad
|
||||
r2 = r2 ^ simd_shuffle_xor(r6, (ushort)4); // s12 shfl
|
||||
r1 = r1 + r5 + select(0xbdf8f9a5u, 0xbc48c63eu, ((sel >> 25u) & 1u) != 0u); // s13 add
|
||||
r1 = rotl_imm(r1, 29u); // s14 rotl
|
||||
r1 = r1 - r4; // s15 sub
|
||||
r7 = r7 | r1; // s16 or
|
||||
r2 = r2 ^ simd_shuffle_xor(r4, (ushort)8); // s17 shfl
|
||||
r7 = r7 ^ r4; // s18 xor
|
||||
r6 = r6 * r1; // s19 mul
|
||||
r5 = r6 * r0 + r5; // s20 mad
|
||||
r3 = r3 - r1; // s21 sub
|
||||
r6 = r6 * r0; // s22 mul
|
||||
r2 = r2 + r0 + select(0x45c37cecu, 0x96e8f127u, ((sel >> 1u) & 1u) != 0u); // s23 add
|
||||
r6 = r6 - r4; // s24 sub
|
||||
r7 = r3 * r4 + r7; // s25 mad
|
||||
r3 = rotl_imm(r3, 9u); // s26 rotl
|
||||
r2 = r2 - r1; // s27 sub
|
||||
r6 = mulhi(r6, r0); // s28 mulhi
|
||||
r2 = r2 ^ simd_shuffle_xor(r4, (ushort)2); // s29 shfl
|
||||
r1 = r1 ^ simd_shuffle_xor(r6, (ushort)1); // s30 shfl
|
||||
r1 = r1 + r6 + select(0x37985632u, 0xb1cdb2abu, ((sel >> 1u) & 1u) != 0u); // s31 add
|
||||
r0 = rotr_var(r0, r1); // s32 rotr
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // s33 shfl
|
||||
r0 = r0 ^ simd_shuffle_xor(r3, (ushort)4); // s34 shfl
|
||||
r7 = r7 * r0; // s35 mul
|
||||
r3 = r3 + r1 + select(0x804e777eu, 0x856e0180u, ((sel >> 0u) & 1u) != 0u); // s36 add
|
||||
r1 = r1 ^ r6; // s37 xor
|
||||
r2 = r2 * r7; // s38 mul
|
||||
r6 = r6 + r5 + select(0x0b06c8a7u, 0xe9bd0cc4u, ((sel >> 2u) & 1u) != 0u); // s39 add
|
||||
r5 = r5 * r7; // s40 mul
|
||||
r2 = mulhi(r2, r3); // s41 mulhi
|
||||
r2 = r2 ^ r0; // s42 xor
|
||||
r0 = rotl_imm(r0, 4u); // s43 rotl
|
||||
r4 = mulhi(r4, r3); // s44 mulhi
|
||||
r6 = r6 + r7 + select(0xee822e17u, 0xcdcb63f6u, ((sel >> 12u) & 1u) != 0u); // s45 add
|
||||
r3 = r2 * r0 + r3; // s46 mad
|
||||
r4 = r4 ^ simd_shuffle_xor(r3, (ushort)8); // s47 shfl
|
||||
r6 = mulhi(r6, r5); // s48 mulhi
|
||||
r0 = r0 - r2; // s49 sub
|
||||
r3 = r3 - r5; // s50 sub
|
||||
r1 = rotr_var(r1, r4); // s51 rotr
|
||||
r6 = r7 * r7 + r6; // s52 mad
|
||||
r5 = r5 ^ r3; // s53 xor
|
||||
r1 = r1 - r0; // s54 sub
|
||||
r5 = r5 - r6; // s55 sub
|
||||
r3 = r3 + r2 + select(0x3dfad1b6u, 0xd4758987u, ((sel >> 10u) & 1u) != 0u); // s56 add
|
||||
r4 = r4 + r1 + select(0xfca75bc2u, 0x0602d6beu, ((sel >> 0u) & 1u) != 0u); // s57 add
|
||||
r5 = mulhi(r5, r6); // s58 mulhi
|
||||
r2 = r2 ^ simd_shuffle_xor(r1, (ushort)2); // s59 shfl
|
||||
r2 = rotl_imm(r2, 9u); // s60 rotl
|
||||
r4 = r4 | r6; // s61 or
|
||||
r6 = rotr_var(r6, r4); // s62 rotr
|
||||
r2 = r2 * r4; // s63 mul
|
||||
r0 = r0 ^ r5; // s64 xor
|
||||
r2 = r2 ^ r7; // s65 xor
|
||||
r2 = r2 + r1 + select(0xd71c02ffu, 0x8596687au, ((sel >> 6u) & 1u) != 0u); // s66 add
|
||||
r1 = rotl_imm(r1, 21u); // s67 rotl
|
||||
r3 = r3 * r2; // s68 mul
|
||||
r7 = r7 ^ simd_shuffle_xor(r5, (ushort)2); // s69 shfl
|
||||
r3 = r3 * r2; // s70 mul
|
||||
r0 = r2 * r5 + r0; // s71 mad
|
||||
r6 = r6 + r7 + select(0x3c6fe15du, 0x08ac5733u, ((sel >> 0u) & 1u) != 0u); // s72 add
|
||||
r3 = r3 ^ simd_shuffle_xor(r2, (ushort)8); // s73 shfl
|
||||
r3 = r3 * r6; // s74 mul
|
||||
r6 = r6 ^ r7; // s75 xor
|
||||
r3 = r3 | r0; // s76 or
|
||||
r2 = r2 + r4 + select(0x295d5faeu, 0xc100b495u, ((sel >> 1u) & 1u) != 0u); // s77 add
|
||||
r3 = mulhi(r3, r7); // s78 mulhi
|
||||
r4 = r4 | r1; // s79 or
|
||||
r4 = rotr_var(r4, r3); // s80 rotr
|
||||
r4 = r4 + r3 + select(0x274a9221u, 0x5cc59530u, ((sel >> 15u) & 1u) != 0u); // s81 add
|
||||
r7 = r7 + r3 + select(0xc8651f8eu, 0x141479ecu, ((sel >> 5u) & 1u) != 0u); // s82 add
|
||||
r3 = rotr_var(r3, r6); // s83 rotr
|
||||
r2 = r4 * r7 + r2; // s84 mad
|
||||
r2 = r2 ^ simd_shuffle_xor(r5, (ushort)16); // s85 shfl
|
||||
r3 = r3 ^ r2; // s86 xor
|
||||
r5 = r5 ^ simd_shuffle_xor(r7, (ushort)2); // s87 shfl
|
||||
r0 = r0 ^ r4; // s88 xor
|
||||
r3 = r3 + r2 + select(0x53f915c2u, 0x883c0c92u, ((sel >> 18u) & 1u) != 0u); // s89 add
|
||||
r7 = rotr_var(r7, r5); // s90 rotr
|
||||
r3 = r3 + r1 + select(0xe2be00a5u, 0x3cc5bd20u, ((sel >> 31u) & 1u) != 0u); // s91 add
|
||||
r7 = r7 ^ simd_shuffle_xor(r2, (ushort)1); // s92 shfl
|
||||
r6 = rotr_var(r6, r2); // s93 rotr
|
||||
r7 = rotl_imm(r7, 14u); // s94 rotl
|
||||
r4 = r5 * r5 + r4; // s95 mad
|
||||
r2 = r4 * r5 + r2; // s96 mad
|
||||
r1 = r1 ^ r0; // s97 xor
|
||||
r5 = r5 * r4; // s98 mul
|
||||
r2 = r2 - r0; // s99 sub
|
||||
r7 = rotl_imm(r7, 30u); // s100 rotl
|
||||
r5 = r5 + r6 + select(0x423fd9c9u, 0xbfd646cbu, ((sel >> 17u) & 1u) != 0u); // s101 add
|
||||
r7 = r7 + r6 + select(0x8d3c011du, 0x19b74a43u, ((sel >> 11u) & 1u) != 0u); // s102 add
|
||||
r5 = rotl_imm(r5, 6u); // s103 rotl
|
||||
r0 = r0 * r4; // s104 mul
|
||||
r0 = r0 | r5; // s105 or
|
||||
r0 = r0 + r1 + select(0x8ce14721u, 0x7dcb7e18u, ((sel >> 15u) & 1u) != 0u); // s106 add
|
||||
r0 = rotl_imm(r0, 15u); // s107 rotl
|
||||
r4 = r4 - r2; // s108 sub
|
||||
r2 = mulhi(r2, r0); // s109 mulhi
|
||||
r1 = mulhi(r1, r0); // s110 mulhi
|
||||
r0 = r0 + r2 + select(0x2d9fc6b9u, 0x3c179ad8u, ((sel >> 28u) & 1u) != 0u); // s111 add
|
||||
r2 = mulhi(r2, r0); // s112 mulhi
|
||||
r6 = r6 - r3; // s113 sub
|
||||
r7 = r6 * r3 + r7; // s114 mad
|
||||
r5 = rotr_var(r5, r1); // s115 rotr
|
||||
r6 = rotr_var(r6, r4); // s116 rotr
|
||||
r2 = r2 + r1 + select(0x502138e2u, 0x47359729u, ((sel >> 22u) & 1u) != 0u); // s117 add
|
||||
r3 = r2 * r0 + r3; // s118 mad
|
||||
r1 = r1 * r3; // s119 mul
|
||||
r2 = r0 * r4 + r2; // s120 mad
|
||||
r0 = r0 * r3; // s121 mul
|
||||
r2 = mulhi(r2, r0); // s122 mulhi
|
||||
r5 = r5 * r6; // s123 mul
|
||||
r4 = r2 * r4 + r4; // s124 mad
|
||||
r4 = r4 ^ simd_shuffle_xor(r2, (ushort)2); // s125 shfl
|
||||
r2 = r2 ^ r0; // s126 xor
|
||||
r1 = rotl_imm(r1, 7u); // s127 rotl
|
||||
r7 = r7 ^ simd_shuffle_xor(r2, (ushort)4); // s128 shfl
|
||||
r6 = rotl_imm(r6, 17u); // s129 rotl
|
||||
r2 = r2 + r5 + select(0x33dff776u, 0x6c57e4e7u, ((sel >> 15u) & 1u) != 0u); // s130 add
|
||||
r0 = r0 | r3; // s131 or
|
||||
r5 = rotr_var(r5, r0); // s132 rotr
|
||||
r2 = r2 + r3 + select(0x5db25b34u, 0xe8212c0cu, ((sel >> 30u) & 1u) != 0u); // s133 add
|
||||
r4 = r4 ^ r3; // s134 xor
|
||||
r6 = r6 ^ r1; // s135 xor
|
||||
r1 = r1 - r7; // s136 sub
|
||||
r2 = r6 * r2 + r2; // s137 mad
|
||||
r0 = mulhi(r0, r5); // s138 mulhi
|
||||
r2 = r2 + r7 + select(0x218d4090u, 0xd44ff710u, ((sel >> 10u) & 1u) != 0u); // s139 add
|
||||
r1 = rotr_var(r1, r4); // s140 rotr
|
||||
r3 = r3 ^ simd_shuffle_xor(r6, (ushort)1); // s141 shfl
|
||||
r7 = r6 * r2 + r7; // s142 mad
|
||||
r2 = r2 * r3; // s143 mul
|
||||
r7 = r7 + r4 + select(0xf4b1a8deu, 0xb98942fau, ((sel >> 29u) & 1u) != 0u); // s144 add
|
||||
r7 = rotl_imm(r7, 15u); // s145 rotl
|
||||
r7 = r7 ^ r5; // s146 xor
|
||||
r4 = r7 * r4 + r4; // s147 mad
|
||||
r6 = rotr_var(r6, r5); // s148 rotr
|
||||
r1 = r2 * r4 + r1; // s149 mad
|
||||
r1 = r1 + r7 + select(0x2bef10f2u, 0x0d48ba42u, ((sel >> 17u) & 1u) != 0u); // s150 add
|
||||
r5 = r5 ^ r4; // s151 xor
|
||||
r7 = r7 + r0 + select(0xdc5cc080u, 0xc98dea9cu, ((sel >> 30u) & 1u) != 0u); // s152 add
|
||||
r4 = r4 * r6; // s153 mul
|
||||
r0 = r0 * r2; // s154 mul
|
||||
r6 = r6 ^ r5; // s155 xor
|
||||
r4 = rotr_var(r4, r2); // s156 rotr
|
||||
r1 = rotl_imm(r1, 11u); // s157 rotl
|
||||
r5 = r5 + r0 + select(0x18197438u, 0x6c752dcbu, ((sel >> 11u) & 1u) != 0u); // s158 add
|
||||
r4 = r1 * r5 + r4; // s159 mad
|
||||
r4 = rotl_imm(r4, 26u); // s160 rotl
|
||||
r3 = r0 * r7 + r3; // s161 mad
|
||||
r3 = rotr_var(r3, r1); // s162 rotr
|
||||
r4 = r4 + r0 + select(0xf1c46574u, 0x8e481727u, ((sel >> 3u) & 1u) != 0u); // s163 add
|
||||
r0 = mulhi(r0, r4); // s164 mulhi
|
||||
r2 = r2 + r6 + select(0xac578137u, 0x550ab406u, ((sel >> 20u) & 1u) != 0u); // s165 add
|
||||
r0 = rotl_imm(r0, 13u); // s166 rotl
|
||||
r3 = r3 + r1 + select(0xa33e6706u, 0xaebb5966u, ((sel >> 14u) & 1u) != 0u); // s167 add
|
||||
r3 = r3 | r5; // s168 or
|
||||
r6 = rotr_var(r6, r2); // s169 rotr
|
||||
r4 = r4 ^ r6; // s170 xor
|
||||
r6 = r6 - r1; // s171 sub
|
||||
r7 = rotl_imm(r7, 22u); // s172 rotl
|
||||
r5 = rotl_imm(r5, 15u); // s173 rotl
|
||||
r7 = r7 ^ simd_shuffle_xor(r0, (ushort)8); // s174 shfl
|
||||
r0 = r0 ^ r5; // s175 xor
|
||||
r7 = rotl_imm(r7, 6u); // s176 rotl
|
||||
r7 = r7 - r0; // s177 sub
|
||||
r3 = rotl_imm(r3, 30u); // s178 rotl
|
||||
r7 = r6 * r1 + r7; // s179 mad
|
||||
r6 = rotl_imm(r6, 9u); // s180 rotl
|
||||
r2 = r2 ^ r4; // s181 xor
|
||||
r2 = r2 ^ r7; // s182 xor
|
||||
r7 = r7 ^ r2; // s183 xor
|
||||
r1 = r1 + r2 + select(0xd94d55acu, 0x5bb7550fu, ((sel >> 21u) & 1u) != 0u); // s184 add
|
||||
r3 = r3 | r5; // s185 or
|
||||
r6 = mulhi(r6, r3); // s186 mulhi
|
||||
r4 = r4 | r0; // s187 or
|
||||
r7 = r7 ^ simd_shuffle_xor(r1, (ushort)2); // s188 shfl
|
||||
r6 = r6 + r5 + select(0x89e747feu, 0x2a354e2du, ((sel >> 6u) & 1u) != 0u); // s189 add
|
||||
r1 = r1 ^ r4; // s190 xor
|
||||
r7 = mulhi(r7, r4); // s191 mulhi
|
||||
r2 = rotl_imm(r2, 26u); // s192 rotl
|
||||
r5 = rotl_imm(r5, 8u); // s193 rotl
|
||||
r4 = r4 ^ simd_shuffle_xor(r5, (ushort)16); // s194 shfl
|
||||
r4 = r4 ^ r5; // s195 xor
|
||||
r1 = r1 * r3; // s196 mul
|
||||
r5 = r5 + r2 + select(0xea03e8e7u, 0x10cfdc71u, ((sel >> 7u) & 1u) != 0u); // s197 add
|
||||
r7 = r7 + r5 + select(0xa7ee0102u, 0x66e148d3u, ((sel >> 15u) & 1u) != 0u); // s198 add
|
||||
r5 = r5 - r2; // s199 sub
|
||||
r5 = r5 ^ simd_shuffle_xor(r1, (ushort)2); // s200 shfl
|
||||
r7 = r7 ^ simd_shuffle_xor(r1, (ushort)8); // s201 shfl
|
||||
r4 = rotr_var(r4, r5); // s202 rotr
|
||||
r7 = r7 ^ r5; // s203 xor
|
||||
r7 = r7 ^ r0; // s204 xor
|
||||
r6 = r6 + r2 + select(0x8379a4deu, 0x8558b619u, ((sel >> 10u) & 1u) != 0u); // s205 add
|
||||
r4 = rotl_imm(r4, 13u); // s206 rotl
|
||||
r1 = mulhi(r1, r3); // s207 mulhi
|
||||
r1 = r4 * r4 + r1; // s208 mad
|
||||
r2 = r2 ^ simd_shuffle_xor(r0, (ushort)4); // s209 shfl
|
||||
r7 = r7 + r0 + select(0xaa8cb14eu, 0xf4049c4cu, ((sel >> 11u) & 1u) != 0u); // s210 add
|
||||
r7 = r7 ^ simd_shuffle_xor(r1, (ushort)16); // s211 shfl
|
||||
r1 = r1 ^ r0; // s212 xor
|
||||
r7 = rotl_imm(r7, 3u); // s213 rotl
|
||||
r4 = rotr_var(r4, r7); // s214 rotr
|
||||
r3 = r3 ^ simd_shuffle_xor(r2, (ushort)16); // s215 shfl
|
||||
r5 = r5 | r1; // s216 or
|
||||
r1 = r1 - r2; // s217 sub
|
||||
r6 = r6 - r5; // s218 sub
|
||||
r6 = rotl_imm(r6, 4u); // s219 rotl
|
||||
r2 = mulhi(r2, r0); // s220 mulhi
|
||||
r2 = r2 | r0; // s221 or
|
||||
r5 = r5 ^ simd_shuffle_xor(r2, (ushort)8); // s222 shfl
|
||||
r5 = mulhi(r5, r6); // s223 mulhi
|
||||
r0 = r0 - r6; // s224 sub
|
||||
r7 = rotl_imm(r7, 23u); // s225 rotl
|
||||
r4 = r4 | r2; // s226 or
|
||||
r2 = r2 * r4; // s227 mul
|
||||
r3 = rotl_imm(r3, 12u); // s228 rotl
|
||||
r0 = rotr_var(r0, r4); // s229 rotr
|
||||
r0 = r0 + r6 + select(0x1d176220u, 0x2a7fecb2u, ((sel >> 16u) & 1u) != 0u); // s230 add
|
||||
r1 = r1 ^ simd_shuffle_xor(r2, (ushort)4); // s231 shfl
|
||||
r2 = r2 * r1; // s232 mul
|
||||
r7 = r0 * r1 + r7; // s233 mad
|
||||
r5 = rotl_imm(r5, 22u); // s234 rotl
|
||||
r4 = rotr_var(r4, r6); // s235 rotr
|
||||
r0 = r5 * r1 + r0; // s236 mad
|
||||
r6 = r6 ^ r4; // s237 xor
|
||||
r4 = r4 ^ r6; // s238 xor
|
||||
r6 = rotl_imm(r6, 18u); // s239 rotl
|
||||
r4 = r4 + r7 + select(0x17dafb4du, 0xadce39f3u, ((sel >> 25u) & 1u) != 0u); // s240 add
|
||||
r0 = r0 - r7; // s241 sub
|
||||
r1 = rotr_var(r1, r6); // s242 rotr
|
||||
r3 = r3 + r0 + select(0xf2f77d26u, 0x0e7033b6u, ((sel >> 29u) & 1u) != 0u); // s243 add
|
||||
r2 = r7 * r4 + r2; // s244 mad
|
||||
r4 = r0 * r6 + r4; // s245 mad
|
||||
r5 = r5 * r7; // s246 mul
|
||||
r3 = r3 + r5 + select(0xfed2da4eu, 0x7b2ff6b7u, ((sel >> 31u) & 1u) != 0u); // s247 add
|
||||
r0 = r0 + r7 + select(0x03b2891cu, 0xb5fad7b1u, ((sel >> 0u) & 1u) != 0u); // s248 add
|
||||
r5 = mulhi(r5, r4); // s249 mulhi
|
||||
r5 = r5 + r2 + select(0x042cd6e3u, 0xa7e71c2fu, ((sel >> 11u) & 1u) != 0u); // s250 add
|
||||
r6 = r6 ^ simd_shuffle_xor(r1, (ushort)1); // s251 shfl
|
||||
r6 = r6 ^ r1; // s252 xor
|
||||
r2 = r2 * r5; // s253 mul
|
||||
r0 = r0 + r6 + select(0x784a302bu, 0xb83d78deu, ((sel >> 16u) & 1u) != 0u); // s254 add
|
||||
r2 = r2 - r4; // s255 sub
|
||||
}
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
370
proto-cuda/packs-ca3-v5/v4-genesis/program_bound.metal
Normal file
370
proto-cuda/packs-ca3-v5/v4-genesis/program_bound.metal
Normal file
|
|
@ -0,0 +1,370 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
|
||||
#define MASK 0x0fffffffu
|
||||
constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u };
|
||||
|
||||
inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
|
||||
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW.
|
||||
kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
|
||||
device ulong* out [[buffer(1)]],
|
||||
constant uint& baseNonce [[buffer(2)]],
|
||||
constant uint* initw [[buffer(3)]],
|
||||
uint gid [[thread_position_in_grid]]) {
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint x = nonce ^ initw[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ initw[1]; }
|
||||
{ uint x = nonce ^ initw[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ initw[2]; }
|
||||
{ uint x = nonce ^ initw[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ initw[3]; }
|
||||
{ uint x = nonce ^ initw[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ initw[4]; }
|
||||
{ uint x = nonce ^ initw[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ initw[5]; }
|
||||
{ uint x = nonce ^ initw[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ initw[6]; }
|
||||
{ uint x = nonce ^ initw[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ initw[7]; }
|
||||
{ uint x = nonce ^ initw[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ initw[0]; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0
|
||||
r2 = r1 * r1 + r2; // 1
|
||||
r2 = r3 * r2 + r2; // 2
|
||||
r3 = r3 ^ r5; // 3
|
||||
r7 = r7 ^ dataset[r2 & MASK]; // 4
|
||||
r5 = r5 ^ dataset[r7 & MASK]; // 5
|
||||
r1 = r1 ^ simd_shuffle_xor(r4, (ushort)8); // 6
|
||||
r7 = r7 ^ simd_shuffle_xor(r3, (ushort)8); // 7
|
||||
r1 = mulhi(r1, r5); // 8
|
||||
r6 = rotr_var(r6, r3); // 9
|
||||
r3 = r3 | r4; // 10
|
||||
r4 = r4 ^ dataset[r6 & MASK]; // 11
|
||||
r0 = mulhi(r0, r4); // 12
|
||||
r5 = r5 + r1 + select(0xc7934706u, 0xd3177981u, ((sel >> 30u) & 1u) != 0u); // 13
|
||||
r0 = r0 ^ dataset[r5 & MASK]; // 14
|
||||
r2 = r2 - r4; // 15
|
||||
r2 = r2 ^ dataset[r7 & MASK]; // 16
|
||||
r7 = r7 ^ dataset[r4 & MASK]; // 17
|
||||
r7 = r7 ^ simd_shuffle_xor(r3, (ushort)4); // 18
|
||||
r5 = r5 * r0; // 19
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 20
|
||||
r2 = r2 ^ simd_shuffle_xor(r4, (ushort)16); // 21
|
||||
r6 = mulhi(r6, r2); // 22
|
||||
r6 = r6 ^ dataset[r3 & MASK]; // 23
|
||||
r5 = r5 * r0; // 24
|
||||
r5 = rotl_imm(r5, 19u); // 25
|
||||
r7 = r7 ^ simd_shuffle_xor(r6, (ushort)2); // 26
|
||||
r0 = r0 ^ r5; // 27
|
||||
r0 = r0 ^ r4; // 28
|
||||
r3 = r3 - r0; // 29
|
||||
r5 = r5 * r1; // 30
|
||||
r7 = r7 ^ dataset[r3 & MASK]; // 31
|
||||
r1 = r1 ^ dataset[r2 & MASK]; // 32
|
||||
r5 = r5 ^ r6; // 33
|
||||
r5 = r5 ^ dataset[r0 & MASK]; // 34
|
||||
r0 = mulhi(r0, r5); // 35
|
||||
r5 = r5 ^ simd_shuffle_xor(r2, (ushort)4); // 36
|
||||
r7 = r7 ^ dataset[r6 & MASK]; // 37
|
||||
r3 = r3 + r1 + select(0x75ba2fadu, 0x230c005cu, ((sel >> 27u) & 1u) != 0u); // 38
|
||||
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)4); // 39
|
||||
r2 = r2 ^ r5; // 40
|
||||
r3 = r6 * r3 + r3; // 41
|
||||
r6 = r6 - r7; // 42
|
||||
r7 = r7 ^ r0; // 43
|
||||
r1 = r1 ^ dataset[r7 & MASK]; // 44
|
||||
r2 = r2 * r3; // 45
|
||||
r1 = mulhi(r1, r5); // 46
|
||||
r4 = r4 - r3; // 47
|
||||
r2 = rotr_var(r2, r6); // 48
|
||||
r3 = r3 ^ dataset[r4 & MASK]; // 49
|
||||
r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 50
|
||||
r0 = r0 * r2; // 51
|
||||
r0 = r0 + r2 + select(0x4f92b968u, 0x699fd448u, ((sel >> 6u) & 1u) != 0u); // 52
|
||||
r1 = r1 + r0 + select(0x2bb965afu, 0x77b1520du, ((sel >> 12u) & 1u) != 0u); // 53
|
||||
r7 = rotl_imm(r7, 14u); // 54
|
||||
r3 = r3 + r7 + select(0x7b0fe07au, 0xa54c55a0u, ((sel >> 1u) & 1u) != 0u); // 55
|
||||
r6 = r6 ^ dataset[r7 & MASK]; // 56
|
||||
r1 = rotr_var(r1, r5); // 57
|
||||
r5 = r5 ^ dataset[r3 & MASK]; // 58
|
||||
r6 = r6 ^ dataset[r1 & MASK]; // 59
|
||||
r3 = r5 * r0 + r3; // 60
|
||||
r5 = r5 + r7 + select(0xaf9dd72du, 0xad7493e7u, ((sel >> 31u) & 1u) != 0u); // 61
|
||||
r4 = r4 + r6 + select(0x89841d87u, 0x1e07c3d9u, ((sel >> 27u) & 1u) != 0u); // 62
|
||||
r5 = rotl_imm(r5, 19u); // 63
|
||||
// latency-shadow block (Counter ASIC 3.0 item 8): 256 ALU instructions x 27 passes after instruction 63, no load
|
||||
for (uint sh = 0u; sh < 27u; ++sh) {
|
||||
r5 = r5 + r2 + select(0x92199f99u, 0x8bc12da9u, ((sel >> 18u) & 1u) != 0u); // s0 add
|
||||
r0 = r0 + r7 + select(0x8d72d3adu, 0x63079e5au, ((sel >> 26u) & 1u) != 0u); // s1 add
|
||||
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)2); // s2 shfl
|
||||
r4 = r4 - r2; // s3 sub
|
||||
r7 = r7 + r0 + select(0xb21b4babu, 0x5d4c7a60u, ((sel >> 31u) & 1u) != 0u); // s4 add
|
||||
r0 = rotl_imm(r0, 11u); // s5 rotl
|
||||
r0 = r0 + r1 + select(0x2fe0e98bu, 0xc88e2942u, ((sel >> 16u) & 1u) != 0u); // s6 add
|
||||
r7 = r7 ^ r1; // s7 xor
|
||||
r1 = r6 * r5 + r1; // s8 mad
|
||||
r6 = r6 ^ simd_shuffle_xor(r1, (ushort)4); // s9 shfl
|
||||
r1 = r2 * r2 + r1; // s10 mad
|
||||
r5 = r0 * r3 + r5; // s11 mad
|
||||
r2 = r2 ^ simd_shuffle_xor(r6, (ushort)4); // s12 shfl
|
||||
r1 = r1 + r5 + select(0xbdf8f9a5u, 0xbc48c63eu, ((sel >> 25u) & 1u) != 0u); // s13 add
|
||||
r1 = rotl_imm(r1, 29u); // s14 rotl
|
||||
r1 = r1 - r4; // s15 sub
|
||||
r7 = r7 | r1; // s16 or
|
||||
r2 = r2 ^ simd_shuffle_xor(r4, (ushort)8); // s17 shfl
|
||||
r7 = r7 ^ r4; // s18 xor
|
||||
r6 = r6 * r1; // s19 mul
|
||||
r5 = r6 * r0 + r5; // s20 mad
|
||||
r3 = r3 - r1; // s21 sub
|
||||
r6 = r6 * r0; // s22 mul
|
||||
r2 = r2 + r0 + select(0x45c37cecu, 0x96e8f127u, ((sel >> 1u) & 1u) != 0u); // s23 add
|
||||
r6 = r6 - r4; // s24 sub
|
||||
r7 = r3 * r4 + r7; // s25 mad
|
||||
r3 = rotl_imm(r3, 9u); // s26 rotl
|
||||
r2 = r2 - r1; // s27 sub
|
||||
r6 = mulhi(r6, r0); // s28 mulhi
|
||||
r2 = r2 ^ simd_shuffle_xor(r4, (ushort)2); // s29 shfl
|
||||
r1 = r1 ^ simd_shuffle_xor(r6, (ushort)1); // s30 shfl
|
||||
r1 = r1 + r6 + select(0x37985632u, 0xb1cdb2abu, ((sel >> 1u) & 1u) != 0u); // s31 add
|
||||
r0 = rotr_var(r0, r1); // s32 rotr
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // s33 shfl
|
||||
r0 = r0 ^ simd_shuffle_xor(r3, (ushort)4); // s34 shfl
|
||||
r7 = r7 * r0; // s35 mul
|
||||
r3 = r3 + r1 + select(0x804e777eu, 0x856e0180u, ((sel >> 0u) & 1u) != 0u); // s36 add
|
||||
r1 = r1 ^ r6; // s37 xor
|
||||
r2 = r2 * r7; // s38 mul
|
||||
r6 = r6 + r5 + select(0x0b06c8a7u, 0xe9bd0cc4u, ((sel >> 2u) & 1u) != 0u); // s39 add
|
||||
r5 = r5 * r7; // s40 mul
|
||||
r2 = mulhi(r2, r3); // s41 mulhi
|
||||
r2 = r2 ^ r0; // s42 xor
|
||||
r0 = rotl_imm(r0, 4u); // s43 rotl
|
||||
r4 = mulhi(r4, r3); // s44 mulhi
|
||||
r6 = r6 + r7 + select(0xee822e17u, 0xcdcb63f6u, ((sel >> 12u) & 1u) != 0u); // s45 add
|
||||
r3 = r2 * r0 + r3; // s46 mad
|
||||
r4 = r4 ^ simd_shuffle_xor(r3, (ushort)8); // s47 shfl
|
||||
r6 = mulhi(r6, r5); // s48 mulhi
|
||||
r0 = r0 - r2; // s49 sub
|
||||
r3 = r3 - r5; // s50 sub
|
||||
r1 = rotr_var(r1, r4); // s51 rotr
|
||||
r6 = r7 * r7 + r6; // s52 mad
|
||||
r5 = r5 ^ r3; // s53 xor
|
||||
r1 = r1 - r0; // s54 sub
|
||||
r5 = r5 - r6; // s55 sub
|
||||
r3 = r3 + r2 + select(0x3dfad1b6u, 0xd4758987u, ((sel >> 10u) & 1u) != 0u); // s56 add
|
||||
r4 = r4 + r1 + select(0xfca75bc2u, 0x0602d6beu, ((sel >> 0u) & 1u) != 0u); // s57 add
|
||||
r5 = mulhi(r5, r6); // s58 mulhi
|
||||
r2 = r2 ^ simd_shuffle_xor(r1, (ushort)2); // s59 shfl
|
||||
r2 = rotl_imm(r2, 9u); // s60 rotl
|
||||
r4 = r4 | r6; // s61 or
|
||||
r6 = rotr_var(r6, r4); // s62 rotr
|
||||
r2 = r2 * r4; // s63 mul
|
||||
r0 = r0 ^ r5; // s64 xor
|
||||
r2 = r2 ^ r7; // s65 xor
|
||||
r2 = r2 + r1 + select(0xd71c02ffu, 0x8596687au, ((sel >> 6u) & 1u) != 0u); // s66 add
|
||||
r1 = rotl_imm(r1, 21u); // s67 rotl
|
||||
r3 = r3 * r2; // s68 mul
|
||||
r7 = r7 ^ simd_shuffle_xor(r5, (ushort)2); // s69 shfl
|
||||
r3 = r3 * r2; // s70 mul
|
||||
r0 = r2 * r5 + r0; // s71 mad
|
||||
r6 = r6 + r7 + select(0x3c6fe15du, 0x08ac5733u, ((sel >> 0u) & 1u) != 0u); // s72 add
|
||||
r3 = r3 ^ simd_shuffle_xor(r2, (ushort)8); // s73 shfl
|
||||
r3 = r3 * r6; // s74 mul
|
||||
r6 = r6 ^ r7; // s75 xor
|
||||
r3 = r3 | r0; // s76 or
|
||||
r2 = r2 + r4 + select(0x295d5faeu, 0xc100b495u, ((sel >> 1u) & 1u) != 0u); // s77 add
|
||||
r3 = mulhi(r3, r7); // s78 mulhi
|
||||
r4 = r4 | r1; // s79 or
|
||||
r4 = rotr_var(r4, r3); // s80 rotr
|
||||
r4 = r4 + r3 + select(0x274a9221u, 0x5cc59530u, ((sel >> 15u) & 1u) != 0u); // s81 add
|
||||
r7 = r7 + r3 + select(0xc8651f8eu, 0x141479ecu, ((sel >> 5u) & 1u) != 0u); // s82 add
|
||||
r3 = rotr_var(r3, r6); // s83 rotr
|
||||
r2 = r4 * r7 + r2; // s84 mad
|
||||
r2 = r2 ^ simd_shuffle_xor(r5, (ushort)16); // s85 shfl
|
||||
r3 = r3 ^ r2; // s86 xor
|
||||
r5 = r5 ^ simd_shuffle_xor(r7, (ushort)2); // s87 shfl
|
||||
r0 = r0 ^ r4; // s88 xor
|
||||
r3 = r3 + r2 + select(0x53f915c2u, 0x883c0c92u, ((sel >> 18u) & 1u) != 0u); // s89 add
|
||||
r7 = rotr_var(r7, r5); // s90 rotr
|
||||
r3 = r3 + r1 + select(0xe2be00a5u, 0x3cc5bd20u, ((sel >> 31u) & 1u) != 0u); // s91 add
|
||||
r7 = r7 ^ simd_shuffle_xor(r2, (ushort)1); // s92 shfl
|
||||
r6 = rotr_var(r6, r2); // s93 rotr
|
||||
r7 = rotl_imm(r7, 14u); // s94 rotl
|
||||
r4 = r5 * r5 + r4; // s95 mad
|
||||
r2 = r4 * r5 + r2; // s96 mad
|
||||
r1 = r1 ^ r0; // s97 xor
|
||||
r5 = r5 * r4; // s98 mul
|
||||
r2 = r2 - r0; // s99 sub
|
||||
r7 = rotl_imm(r7, 30u); // s100 rotl
|
||||
r5 = r5 + r6 + select(0x423fd9c9u, 0xbfd646cbu, ((sel >> 17u) & 1u) != 0u); // s101 add
|
||||
r7 = r7 + r6 + select(0x8d3c011du, 0x19b74a43u, ((sel >> 11u) & 1u) != 0u); // s102 add
|
||||
r5 = rotl_imm(r5, 6u); // s103 rotl
|
||||
r0 = r0 * r4; // s104 mul
|
||||
r0 = r0 | r5; // s105 or
|
||||
r0 = r0 + r1 + select(0x8ce14721u, 0x7dcb7e18u, ((sel >> 15u) & 1u) != 0u); // s106 add
|
||||
r0 = rotl_imm(r0, 15u); // s107 rotl
|
||||
r4 = r4 - r2; // s108 sub
|
||||
r2 = mulhi(r2, r0); // s109 mulhi
|
||||
r1 = mulhi(r1, r0); // s110 mulhi
|
||||
r0 = r0 + r2 + select(0x2d9fc6b9u, 0x3c179ad8u, ((sel >> 28u) & 1u) != 0u); // s111 add
|
||||
r2 = mulhi(r2, r0); // s112 mulhi
|
||||
r6 = r6 - r3; // s113 sub
|
||||
r7 = r6 * r3 + r7; // s114 mad
|
||||
r5 = rotr_var(r5, r1); // s115 rotr
|
||||
r6 = rotr_var(r6, r4); // s116 rotr
|
||||
r2 = r2 + r1 + select(0x502138e2u, 0x47359729u, ((sel >> 22u) & 1u) != 0u); // s117 add
|
||||
r3 = r2 * r0 + r3; // s118 mad
|
||||
r1 = r1 * r3; // s119 mul
|
||||
r2 = r0 * r4 + r2; // s120 mad
|
||||
r0 = r0 * r3; // s121 mul
|
||||
r2 = mulhi(r2, r0); // s122 mulhi
|
||||
r5 = r5 * r6; // s123 mul
|
||||
r4 = r2 * r4 + r4; // s124 mad
|
||||
r4 = r4 ^ simd_shuffle_xor(r2, (ushort)2); // s125 shfl
|
||||
r2 = r2 ^ r0; // s126 xor
|
||||
r1 = rotl_imm(r1, 7u); // s127 rotl
|
||||
r7 = r7 ^ simd_shuffle_xor(r2, (ushort)4); // s128 shfl
|
||||
r6 = rotl_imm(r6, 17u); // s129 rotl
|
||||
r2 = r2 + r5 + select(0x33dff776u, 0x6c57e4e7u, ((sel >> 15u) & 1u) != 0u); // s130 add
|
||||
r0 = r0 | r3; // s131 or
|
||||
r5 = rotr_var(r5, r0); // s132 rotr
|
||||
r2 = r2 + r3 + select(0x5db25b34u, 0xe8212c0cu, ((sel >> 30u) & 1u) != 0u); // s133 add
|
||||
r4 = r4 ^ r3; // s134 xor
|
||||
r6 = r6 ^ r1; // s135 xor
|
||||
r1 = r1 - r7; // s136 sub
|
||||
r2 = r6 * r2 + r2; // s137 mad
|
||||
r0 = mulhi(r0, r5); // s138 mulhi
|
||||
r2 = r2 + r7 + select(0x218d4090u, 0xd44ff710u, ((sel >> 10u) & 1u) != 0u); // s139 add
|
||||
r1 = rotr_var(r1, r4); // s140 rotr
|
||||
r3 = r3 ^ simd_shuffle_xor(r6, (ushort)1); // s141 shfl
|
||||
r7 = r6 * r2 + r7; // s142 mad
|
||||
r2 = r2 * r3; // s143 mul
|
||||
r7 = r7 + r4 + select(0xf4b1a8deu, 0xb98942fau, ((sel >> 29u) & 1u) != 0u); // s144 add
|
||||
r7 = rotl_imm(r7, 15u); // s145 rotl
|
||||
r7 = r7 ^ r5; // s146 xor
|
||||
r4 = r7 * r4 + r4; // s147 mad
|
||||
r6 = rotr_var(r6, r5); // s148 rotr
|
||||
r1 = r2 * r4 + r1; // s149 mad
|
||||
r1 = r1 + r7 + select(0x2bef10f2u, 0x0d48ba42u, ((sel >> 17u) & 1u) != 0u); // s150 add
|
||||
r5 = r5 ^ r4; // s151 xor
|
||||
r7 = r7 + r0 + select(0xdc5cc080u, 0xc98dea9cu, ((sel >> 30u) & 1u) != 0u); // s152 add
|
||||
r4 = r4 * r6; // s153 mul
|
||||
r0 = r0 * r2; // s154 mul
|
||||
r6 = r6 ^ r5; // s155 xor
|
||||
r4 = rotr_var(r4, r2); // s156 rotr
|
||||
r1 = rotl_imm(r1, 11u); // s157 rotl
|
||||
r5 = r5 + r0 + select(0x18197438u, 0x6c752dcbu, ((sel >> 11u) & 1u) != 0u); // s158 add
|
||||
r4 = r1 * r5 + r4; // s159 mad
|
||||
r4 = rotl_imm(r4, 26u); // s160 rotl
|
||||
r3 = r0 * r7 + r3; // s161 mad
|
||||
r3 = rotr_var(r3, r1); // s162 rotr
|
||||
r4 = r4 + r0 + select(0xf1c46574u, 0x8e481727u, ((sel >> 3u) & 1u) != 0u); // s163 add
|
||||
r0 = mulhi(r0, r4); // s164 mulhi
|
||||
r2 = r2 + r6 + select(0xac578137u, 0x550ab406u, ((sel >> 20u) & 1u) != 0u); // s165 add
|
||||
r0 = rotl_imm(r0, 13u); // s166 rotl
|
||||
r3 = r3 + r1 + select(0xa33e6706u, 0xaebb5966u, ((sel >> 14u) & 1u) != 0u); // s167 add
|
||||
r3 = r3 | r5; // s168 or
|
||||
r6 = rotr_var(r6, r2); // s169 rotr
|
||||
r4 = r4 ^ r6; // s170 xor
|
||||
r6 = r6 - r1; // s171 sub
|
||||
r7 = rotl_imm(r7, 22u); // s172 rotl
|
||||
r5 = rotl_imm(r5, 15u); // s173 rotl
|
||||
r7 = r7 ^ simd_shuffle_xor(r0, (ushort)8); // s174 shfl
|
||||
r0 = r0 ^ r5; // s175 xor
|
||||
r7 = rotl_imm(r7, 6u); // s176 rotl
|
||||
r7 = r7 - r0; // s177 sub
|
||||
r3 = rotl_imm(r3, 30u); // s178 rotl
|
||||
r7 = r6 * r1 + r7; // s179 mad
|
||||
r6 = rotl_imm(r6, 9u); // s180 rotl
|
||||
r2 = r2 ^ r4; // s181 xor
|
||||
r2 = r2 ^ r7; // s182 xor
|
||||
r7 = r7 ^ r2; // s183 xor
|
||||
r1 = r1 + r2 + select(0xd94d55acu, 0x5bb7550fu, ((sel >> 21u) & 1u) != 0u); // s184 add
|
||||
r3 = r3 | r5; // s185 or
|
||||
r6 = mulhi(r6, r3); // s186 mulhi
|
||||
r4 = r4 | r0; // s187 or
|
||||
r7 = r7 ^ simd_shuffle_xor(r1, (ushort)2); // s188 shfl
|
||||
r6 = r6 + r5 + select(0x89e747feu, 0x2a354e2du, ((sel >> 6u) & 1u) != 0u); // s189 add
|
||||
r1 = r1 ^ r4; // s190 xor
|
||||
r7 = mulhi(r7, r4); // s191 mulhi
|
||||
r2 = rotl_imm(r2, 26u); // s192 rotl
|
||||
r5 = rotl_imm(r5, 8u); // s193 rotl
|
||||
r4 = r4 ^ simd_shuffle_xor(r5, (ushort)16); // s194 shfl
|
||||
r4 = r4 ^ r5; // s195 xor
|
||||
r1 = r1 * r3; // s196 mul
|
||||
r5 = r5 + r2 + select(0xea03e8e7u, 0x10cfdc71u, ((sel >> 7u) & 1u) != 0u); // s197 add
|
||||
r7 = r7 + r5 + select(0xa7ee0102u, 0x66e148d3u, ((sel >> 15u) & 1u) != 0u); // s198 add
|
||||
r5 = r5 - r2; // s199 sub
|
||||
r5 = r5 ^ simd_shuffle_xor(r1, (ushort)2); // s200 shfl
|
||||
r7 = r7 ^ simd_shuffle_xor(r1, (ushort)8); // s201 shfl
|
||||
r4 = rotr_var(r4, r5); // s202 rotr
|
||||
r7 = r7 ^ r5; // s203 xor
|
||||
r7 = r7 ^ r0; // s204 xor
|
||||
r6 = r6 + r2 + select(0x8379a4deu, 0x8558b619u, ((sel >> 10u) & 1u) != 0u); // s205 add
|
||||
r4 = rotl_imm(r4, 13u); // s206 rotl
|
||||
r1 = mulhi(r1, r3); // s207 mulhi
|
||||
r1 = r4 * r4 + r1; // s208 mad
|
||||
r2 = r2 ^ simd_shuffle_xor(r0, (ushort)4); // s209 shfl
|
||||
r7 = r7 + r0 + select(0xaa8cb14eu, 0xf4049c4cu, ((sel >> 11u) & 1u) != 0u); // s210 add
|
||||
r7 = r7 ^ simd_shuffle_xor(r1, (ushort)16); // s211 shfl
|
||||
r1 = r1 ^ r0; // s212 xor
|
||||
r7 = rotl_imm(r7, 3u); // s213 rotl
|
||||
r4 = rotr_var(r4, r7); // s214 rotr
|
||||
r3 = r3 ^ simd_shuffle_xor(r2, (ushort)16); // s215 shfl
|
||||
r5 = r5 | r1; // s216 or
|
||||
r1 = r1 - r2; // s217 sub
|
||||
r6 = r6 - r5; // s218 sub
|
||||
r6 = rotl_imm(r6, 4u); // s219 rotl
|
||||
r2 = mulhi(r2, r0); // s220 mulhi
|
||||
r2 = r2 | r0; // s221 or
|
||||
r5 = r5 ^ simd_shuffle_xor(r2, (ushort)8); // s222 shfl
|
||||
r5 = mulhi(r5, r6); // s223 mulhi
|
||||
r0 = r0 - r6; // s224 sub
|
||||
r7 = rotl_imm(r7, 23u); // s225 rotl
|
||||
r4 = r4 | r2; // s226 or
|
||||
r2 = r2 * r4; // s227 mul
|
||||
r3 = rotl_imm(r3, 12u); // s228 rotl
|
||||
r0 = rotr_var(r0, r4); // s229 rotr
|
||||
r0 = r0 + r6 + select(0x1d176220u, 0x2a7fecb2u, ((sel >> 16u) & 1u) != 0u); // s230 add
|
||||
r1 = r1 ^ simd_shuffle_xor(r2, (ushort)4); // s231 shfl
|
||||
r2 = r2 * r1; // s232 mul
|
||||
r7 = r0 * r1 + r7; // s233 mad
|
||||
r5 = rotl_imm(r5, 22u); // s234 rotl
|
||||
r4 = rotr_var(r4, r6); // s235 rotr
|
||||
r0 = r5 * r1 + r0; // s236 mad
|
||||
r6 = r6 ^ r4; // s237 xor
|
||||
r4 = r4 ^ r6; // s238 xor
|
||||
r6 = rotl_imm(r6, 18u); // s239 rotl
|
||||
r4 = r4 + r7 + select(0x17dafb4du, 0xadce39f3u, ((sel >> 25u) & 1u) != 0u); // s240 add
|
||||
r0 = r0 - r7; // s241 sub
|
||||
r1 = rotr_var(r1, r6); // s242 rotr
|
||||
r3 = r3 + r0 + select(0xf2f77d26u, 0x0e7033b6u, ((sel >> 29u) & 1u) != 0u); // s243 add
|
||||
r2 = r7 * r4 + r2; // s244 mad
|
||||
r4 = r0 * r6 + r4; // s245 mad
|
||||
r5 = r5 * r7; // s246 mul
|
||||
r3 = r3 + r5 + select(0xfed2da4eu, 0x7b2ff6b7u, ((sel >> 31u) & 1u) != 0u); // s247 add
|
||||
r0 = r0 + r7 + select(0x03b2891cu, 0xb5fad7b1u, ((sel >> 0u) & 1u) != 0u); // s248 add
|
||||
r5 = mulhi(r5, r4); // s249 mulhi
|
||||
r5 = r5 + r2 + select(0x042cd6e3u, 0xa7e71c2fu, ((sel >> 11u) & 1u) != 0u); // s250 add
|
||||
r6 = r6 ^ simd_shuffle_xor(r1, (ushort)1); // s251 shfl
|
||||
r6 = r6 ^ r1; // s252 xor
|
||||
r2 = r2 * r5; // s253 mul
|
||||
r0 = r0 + r6 + select(0x784a302bu, 0xb83d78deu, ((sel >> 16u) & 1u) != 0u); // s254 add
|
||||
r2 = r2 - r4; // s255 sub
|
||||
}
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
57
proto-cuda/packs-ca3-v5/v4-genesis/vectors.h
Normal file
57
proto-cuda/packs-ca3-v5/v4-genesis/vectors.h
Normal file
|
|
@ -0,0 +1,57 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Expected outputs: igneum-pow (Rust) CPU interpreter, generator v4, memory-hard dataset
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
|
||||
#define IGNEUM_VEC_WARPS 3
|
||||
static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u };
|
||||
static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {
|
||||
{ // base nonce 0
|
||||
0xdfbc8db1c06dacd8ull, 0xd56061f5411cc2cfull, 0x66c9911a020ba1e5ull, 0xc8b93c36c4c7a2a6ull, 0xac045e8343ea2521ull, 0x5cdddf6fc438de8bull, 0xad88eea3ea678e28ull, 0x6fbc5b7160399767ull,
|
||||
0x8073ccabd0764f0cull, 0x0e32bd34d9f4cd5full, 0x7d23625bccd7703bull, 0x481c64f93034050aull, 0x4d2493acc84787cfull, 0x853490ed669c5c6aull, 0x4e3b6fc3a3e9f661ull, 0xc87a8caad36cb1daull,
|
||||
0xd592f24a236b35d8ull, 0xbf8fceb08aa32461ull, 0xd174d9f2943500f2ull, 0xe5127777dd2876b0ull, 0xd709d6bdcc633059ull, 0xaa30a9b82f58568full, 0x1b05070b769f1f6cull, 0xfffc24d3fb2a74eeull,
|
||||
0x3c1187b38d01ba95ull, 0x90ef34ea51d4eb79ull, 0xfe124596a63bf602ull, 0x9ad0ff64e27608aeull, 0x7d4f699998541fb9ull, 0xc142eb65cadaaabaull, 0x7a73a36d7faedaa9ull, 0xb92715f32b721587ull
|
||||
},
|
||||
{ // base nonce 4096
|
||||
0x433afad2164a8ddaull, 0x1fc487979b4dffb2ull, 0xbaf10a7e22b67b7dull, 0xe2f4c61b908c014aull, 0x8d588a69173db01full, 0x334c34ae7d8da448ull, 0xabf638cea27cf935ull, 0x06e3d4a1d6e0fbeeull,
|
||||
0xec3d53a3ecf99e86ull, 0x78987c6911255367ull, 0xbbc9c743d9576be5ull, 0x43d622ee7ed30a35ull, 0x339860b0560c988cull, 0x347fba26cb169315ull, 0xd510af804aa26d21ull, 0x5c3ca3e10e783ff8ull,
|
||||
0xc806f8afa12a937aull, 0xb6213f67b09846a2ull, 0xd875236923df5af6ull, 0x760e944b06c76d50ull, 0x87c2740caf1a4c63ull, 0xb80f96fe2edc6e9bull, 0xad6b33f2e4158ebeull, 0xac291295afd15709ull,
|
||||
0x33236c6ca9918da3ull, 0xc7365c50ce60c222ull, 0x53f84641982af8c6ull, 0xd6ada09dbb253a7bull, 0xab39a0bb6f49d28bull, 0xdf0d63f31cb955ecull, 0xe18f65eb0b5ccb6bull, 0x207c50f288cbb677ull
|
||||
},
|
||||
{ // base nonce 1000000
|
||||
0x4ffd090a02db43b4ull, 0x3df015d557eb7dffull, 0xb11e898d6f219179ull, 0x0ffff9c4715ea693ull, 0xc51e5ab1fff3e42dull, 0x3cbe173c74768e58ull, 0x0bc9c3d99d9082dcull, 0x190b0881a508b8b4ull,
|
||||
0x38e7b4a21c3c7d18ull, 0x34aa4afba0dfc716ull, 0x4a8f8a8027f08cd6ull, 0xe7ab045630e34a7eull, 0xa2bdea296316ac03ull, 0xdc85eb2818449894ull, 0xecf5cc78cb02cb05ull, 0xab138a76db952f3eull,
|
||||
0xf625a5704aab7146ull, 0x927d175dd5de47fcull, 0x249aedcf78d89c84ull, 0x66da14f5abee33faull, 0x8584faba12d0e357ull, 0xc1906204df361c55ull, 0x7eb470a084fc181dull, 0x74c114faf8ece5cfull,
|
||||
0xd15cb62a0e1eda86ull, 0x8ba7f1a7ab236d41ull, 0xfda39d1b301db5f5ull, 0x4a008ea52f3c438bull, 0x7b5486d62e2bbb45ull, 0x05440eaf2d1168a2ull, 0x0299a1d3e6c9e4e7ull, 0x60308a0e058e1e29ull
|
||||
}
|
||||
};
|
||||
|
||||
// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (268435455).
|
||||
static const uint32_t IGNEUM_DS_HEAD[16] = {
|
||||
0xfdad4319u, 0x1a7b68e1u, 0xde6db608u, 0x13d73892u, 0xd17f447au, 0xb2221ccfu, 0x9db004bdu, 0x57d7d367u,
|
||||
0xdbc4cf34u, 0x697c009au, 0xc43af1d4u, 0x97f12b2eu, 0x74c37cd0u, 0xc651ea15u, 0x665a6d29u, 0x22330a2du
|
||||
};
|
||||
static const uint32_t IGNEUM_DS_LAST_INDEX = 268435455u;
|
||||
static const uint32_t IGNEUM_DS_LAST = 0xa83e7aa6u;
|
||||
// 64 sampled dataset words (index, value) computed on the Mac.
|
||||
#define IGNEUM_DS_SAMPLES 64
|
||||
static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = {
|
||||
59471966u, 217795994u, 208353206u, 42483309u, 172547758u, 148076330u, 183853158u, 214389424u, 267488061u, 169781097u, 184093494u, 153880993u, 84977930u, 46426879u, 3093825u, 225364072u, 44593546u, 260713159u, 168250303u, 52384140u, 223401610u, 45554030u, 95410555u, 175039924u, 79171087u, 267580473u, 24168642u, 37981670u, 171551130u, 195559979u, 204611762u, 140997658u, 138925853u, 86637313u, 20736778u, 219665210u, 160430336u, 264654675u, 8013395u, 228945585u, 213884386u, 104419827u, 44185464u, 142737231u, 99284897u, 132475900u, 61861762u, 132056166u, 262388043u, 91878046u, 117353561u, 124768597u, 71352993u, 190698941u, 46055428u, 55281366u, 165145231u, 106810753u, 171985651u, 232085256u, 159510492u, 40072060u, 209107596u, 39023794u
|
||||
};
|
||||
static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = {
|
||||
0x3230bc7bu, 0x7fbfe2c9u, 0xb2690991u, 0x1745c7c5u, 0x0ab0ccafu, 0x1bf87d6bu, 0x160139fdu, 0x719817acu, 0x0155df4bu, 0xbe1e86c3u, 0x680bcd6cu, 0x79c3dc6cu, 0x181e7e5fu, 0x0713a109u, 0xc705dd9fu, 0x3933b7a8u, 0xdd1c0431u, 0x50522b30u, 0xa0020b38u, 0xbff39e96u, 0x21b67e18u, 0x740f8db3u, 0x2baba568u, 0x2c9bef83u, 0x0ad9b671u, 0xc4327869u, 0x7b4fd7d0u, 0x2c29965fu, 0xec56f15fu, 0x61111746u, 0x303a1d6eu, 0xbddcfd1au, 0xf829a355u, 0x6d5df2a9u, 0x01ab8e44u, 0x06d13507u, 0xda8dcfc6u, 0x01a703e1u, 0xafe7d2c1u, 0xc091c3a2u, 0xac1814feu, 0x6e6ff62au, 0x8fdf01bau, 0xdd3f7159u, 0xdfa0d75cu, 0x26684c35u, 0x7f441e63u, 0x88df2570u, 0x8aa4d5ebu, 0xcc816c05u, 0x434df890u, 0xcd392ad6u, 0x1ab4cb63u, 0x595926fau, 0x7cd76b41u, 0x20cb95c4u, 0x13cf823fu, 0xf9daf901u, 0xff9af40au, 0x2c7dfa51u, 0x871206dbu, 0x938c116cu, 0xb64bf199u, 0x5751f874u
|
||||
};
|
||||
// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^26 words.
|
||||
static const uint32_t IGNEUM_CACHE_HEAD[16] = {
|
||||
0x355a86d2u, 0x7957db1cu, 0xd21772afu, 0x6fc1e09bu, 0xd55ce61du, 0x6e6a278bu, 0xd3f543ceu, 0x223d8e82u,
|
||||
0x143ab337u, 0x2e9f05bdu, 0x2eb389bfu, 0x0c6e449eu, 0x5cfa4222u, 0xba6560feu, 0x8e3e1aa4u, 0xdbcc1d53u
|
||||
};
|
||||
static const uint32_t IGNEUM_CACHE_LAST[16] = {
|
||||
0x41190d91u, 0xbd277957u, 0x22ddbb49u, 0x6986f207u, 0xdf69a4d6u, 0x26401a3au, 0x818230fbu, 0xc417122du,
|
||||
0x3597b211u, 0xb553ce55u, 0xcf39cc0du, 0x3b7fc43au, 0x3fd43b00u, 0x67e1c80eu, 0xffa7ea7du, 0xca2960abu
|
||||
};
|
||||
static const uint64_t IGNEUM_CACHE_FNV64 = 0x48c4f5bf24166b2eull;
|
||||
36
proto-cuda/packs-ca3-v5/v4-genesis/vectors.json
Normal file
36
proto-cuda/packs-ca3-v5/v4-genesis/vectors.json
Normal file
|
|
@ -0,0 +1,36 @@
|
|||
{
|
||||
"seed": "igneum-genesis",
|
||||
"day": "2026-10-03",
|
||||
"dataset_mode": "memory-hard",
|
||||
"dataset_log2_words": 28,
|
||||
"mask": "0x0fffffff",
|
||||
"lanes": 32,
|
||||
"source": "igneum-pow (Rust) CPU interpreter, generator v4, memory-hard dataset",
|
||||
"warps": [
|
||||
{"base_nonce": 0, "expected": [
|
||||
"0xdfbc8db1c06dacd8", "0xd56061f5411cc2cf", "0x66c9911a020ba1e5", "0xc8b93c36c4c7a2a6", "0xac045e8343ea2521", "0x5cdddf6fc438de8b", "0xad88eea3ea678e28", "0x6fbc5b7160399767",
|
||||
"0x8073ccabd0764f0c", "0x0e32bd34d9f4cd5f", "0x7d23625bccd7703b", "0x481c64f93034050a", "0x4d2493acc84787cf", "0x853490ed669c5c6a", "0x4e3b6fc3a3e9f661", "0xc87a8caad36cb1da",
|
||||
"0xd592f24a236b35d8", "0xbf8fceb08aa32461", "0xd174d9f2943500f2", "0xe5127777dd2876b0", "0xd709d6bdcc633059", "0xaa30a9b82f58568f", "0x1b05070b769f1f6c", "0xfffc24d3fb2a74ee",
|
||||
"0x3c1187b38d01ba95", "0x90ef34ea51d4eb79", "0xfe124596a63bf602", "0x9ad0ff64e27608ae", "0x7d4f699998541fb9", "0xc142eb65cadaaaba", "0x7a73a36d7faedaa9", "0xb92715f32b721587"
|
||||
]},
|
||||
{"base_nonce": 4096, "expected": [
|
||||
"0x433afad2164a8dda", "0x1fc487979b4dffb2", "0xbaf10a7e22b67b7d", "0xe2f4c61b908c014a", "0x8d588a69173db01f", "0x334c34ae7d8da448", "0xabf638cea27cf935", "0x06e3d4a1d6e0fbee",
|
||||
"0xec3d53a3ecf99e86", "0x78987c6911255367", "0xbbc9c743d9576be5", "0x43d622ee7ed30a35", "0x339860b0560c988c", "0x347fba26cb169315", "0xd510af804aa26d21", "0x5c3ca3e10e783ff8",
|
||||
"0xc806f8afa12a937a", "0xb6213f67b09846a2", "0xd875236923df5af6", "0x760e944b06c76d50", "0x87c2740caf1a4c63", "0xb80f96fe2edc6e9b", "0xad6b33f2e4158ebe", "0xac291295afd15709",
|
||||
"0x33236c6ca9918da3", "0xc7365c50ce60c222", "0x53f84641982af8c6", "0xd6ada09dbb253a7b", "0xab39a0bb6f49d28b", "0xdf0d63f31cb955ec", "0xe18f65eb0b5ccb6b", "0x207c50f288cbb677"
|
||||
]},
|
||||
{"base_nonce": 1000000, "expected": [
|
||||
"0x4ffd090a02db43b4", "0x3df015d557eb7dff", "0xb11e898d6f219179", "0x0ffff9c4715ea693", "0xc51e5ab1fff3e42d", "0x3cbe173c74768e58", "0x0bc9c3d99d9082dc", "0x190b0881a508b8b4",
|
||||
"0x38e7b4a21c3c7d18", "0x34aa4afba0dfc716", "0x4a8f8a8027f08cd6", "0xe7ab045630e34a7e", "0xa2bdea296316ac03", "0xdc85eb2818449894", "0xecf5cc78cb02cb05", "0xab138a76db952f3e",
|
||||
"0xf625a5704aab7146", "0x927d175dd5de47fc", "0x249aedcf78d89c84", "0x66da14f5abee33fa", "0x8584faba12d0e357", "0xc1906204df361c55", "0x7eb470a084fc181d", "0x74c114faf8ece5cf",
|
||||
"0xd15cb62a0e1eda86", "0x8ba7f1a7ab236d41", "0xfda39d1b301db5f5", "0x4a008ea52f3c438b", "0x7b5486d62e2bbb45", "0x05440eaf2d1168a2", "0x0299a1d3e6c9e4e7", "0x60308a0e058e1e29"
|
||||
]}
|
||||
],
|
||||
"dataset_head": ["0xfdad4319", "0x1a7b68e1", "0xde6db608", "0x13d73892", "0xd17f447a", "0xb2221ccf", "0x9db004bd", "0x57d7d367", "0xdbc4cf34", "0x697c009a", "0xc43af1d4", "0x97f12b2e", "0x74c37cd0", "0xc651ea15", "0x665a6d29", "0x22330a2d"],
|
||||
"dataset_last_index": 268435455,
|
||||
"dataset_last": "0xa83e7aa6",
|
||||
"dataset_samples": [{"index": 59471966, "value": "0x3230bc7b"}, {"index": 217795994, "value": "0x7fbfe2c9"}, {"index": 208353206, "value": "0xb2690991"}, {"index": 42483309, "value": "0x1745c7c5"}, {"index": 172547758, "value": "0x0ab0ccaf"}, {"index": 148076330, "value": "0x1bf87d6b"}, {"index": 183853158, "value": "0x160139fd"}, {"index": 214389424, "value": "0x719817ac"}, {"index": 267488061, "value": "0x0155df4b"}, {"index": 169781097, "value": "0xbe1e86c3"}, {"index": 184093494, "value": "0x680bcd6c"}, {"index": 153880993, "value": "0x79c3dc6c"}, {"index": 84977930, "value": "0x181e7e5f"}, {"index": 46426879, "value": "0x0713a109"}, {"index": 3093825, "value": "0xc705dd9f"}, {"index": 225364072, "value": "0x3933b7a8"}, {"index": 44593546, "value": "0xdd1c0431"}, {"index": 260713159, "value": "0x50522b30"}, {"index": 168250303, "value": "0xa0020b38"}, {"index": 52384140, "value": "0xbff39e96"}, {"index": 223401610, "value": "0x21b67e18"}, {"index": 45554030, "value": "0x740f8db3"}, {"index": 95410555, "value": "0x2baba568"}, {"index": 175039924, "value": "0x2c9bef83"}, {"index": 79171087, "value": "0x0ad9b671"}, {"index": 267580473, "value": "0xc4327869"}, {"index": 24168642, "value": "0x7b4fd7d0"}, {"index": 37981670, "value": "0x2c29965f"}, {"index": 171551130, "value": "0xec56f15f"}, {"index": 195559979, "value": "0x61111746"}, {"index": 204611762, "value": "0x303a1d6e"}, {"index": 140997658, "value": "0xbddcfd1a"}, {"index": 138925853, "value": "0xf829a355"}, {"index": 86637313, "value": "0x6d5df2a9"}, {"index": 20736778, "value": "0x01ab8e44"}, {"index": 219665210, "value": "0x06d13507"}, {"index": 160430336, "value": "0xda8dcfc6"}, {"index": 264654675, "value": "0x01a703e1"}, {"index": 8013395, "value": "0xafe7d2c1"}, {"index": 228945585, "value": "0xc091c3a2"}, {"index": 213884386, "value": "0xac1814fe"}, {"index": 104419827, "value": "0x6e6ff62a"}, {"index": 44185464, "value": "0x8fdf01ba"}, {"index": 142737231, "value": "0xdd3f7159"}, {"index": 99284897, "value": "0xdfa0d75c"}, {"index": 132475900, "value": "0x26684c35"}, {"index": 61861762, "value": "0x7f441e63"}, {"index": 132056166, "value": "0x88df2570"}, {"index": 262388043, "value": "0x8aa4d5eb"}, {"index": 91878046, "value": "0xcc816c05"}, {"index": 117353561, "value": "0x434df890"}, {"index": 124768597, "value": "0xcd392ad6"}, {"index": 71352993, "value": "0x1ab4cb63"}, {"index": 190698941, "value": "0x595926fa"}, {"index": 46055428, "value": "0x7cd76b41"}, {"index": 55281366, "value": "0x20cb95c4"}, {"index": 165145231, "value": "0x13cf823f"}, {"index": 106810753, "value": "0xf9daf901"}, {"index": 171985651, "value": "0xff9af40a"}, {"index": 232085256, "value": "0x2c7dfa51"}, {"index": 159510492, "value": "0x871206db"}, {"index": 40072060, "value": "0x938c116c"}, {"index": 209107596, "value": "0xb64bf199"}, {"index": 39023794, "value": "0x5751f874"}],
|
||||
"cache_head": ["0x355a86d2", "0x7957db1c", "0xd21772af", "0x6fc1e09b", "0xd55ce61d", "0x6e6a278b", "0xd3f543ce", "0x223d8e82", "0x143ab337", "0x2e9f05bd", "0x2eb389bf", "0x0c6e449e", "0x5cfa4222", "0xba6560fe", "0x8e3e1aa4", "0xdbcc1d53"],
|
||||
"cache_last_line": ["0x41190d91", "0xbd277957", "0x22ddbb49", "0x6986f207", "0xdf69a4d6", "0x26401a3a", "0x818230fb", "0xc417122d", "0x3597b211", "0xb553ce55", "0xcf39cc0d", "0x3b7fc43a", "0x3fd43b00", "0x67e1c80e", "0xffa7ea7d", "0xca2960ab"],
|
||||
"cache_fnv1a64": "0x48c4f5bf24166b2e"
|
||||
}
|
||||
545
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/kernel.cl
Normal file
545
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/kernel.cl
Normal file
|
|
@ -0,0 +1,545 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925/day/69676e65756d2d6461792ffd50000000000000". Do not edit by hand.
|
||||
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
|
||||
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
|
||||
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
|
||||
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
|
||||
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
|
||||
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
|
||||
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
|
||||
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
|
||||
#ifndef IGNEUM_GROUP
|
||||
#define IGNEUM_GROUP 32
|
||||
#endif
|
||||
#ifndef IGNEUM_EXCHANGE
|
||||
#define IGNEUM_EXCHANGE 0
|
||||
#endif
|
||||
#ifdef __OPENCL_VERSION__
|
||||
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
|
||||
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#ifdef cl_khr_subgroups
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
|
||||
#endif
|
||||
#ifdef cl_khr_subgroup_shuffle
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
|
||||
#endif
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
|
||||
#endif
|
||||
#else
|
||||
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
|
||||
#include "emu_opencl.h"
|
||||
#endif
|
||||
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#else
|
||||
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
|
||||
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
|
||||
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
|
||||
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
|
||||
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
|
||||
#endif
|
||||
|
||||
static inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
|
||||
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
|
||||
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
|
||||
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
|
||||
static inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of 8 x seed-parameterised mixer + one 64-byte cache read, then 8 x final mixer (class v3, mixer multiplier 8,
|
||||
// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
static inline void mh_chacha_block(const uint* x, uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
static inline void mh_cache_segment(__global uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0xe23f9008u ^ prev[4];
|
||||
x[5] = 0xde33e763u ^ prev[5];
|
||||
x[6] = 0xc5ba415du ^ prev[6];
|
||||
x[7] = 0x8ddf6786u ^ prev[7];
|
||||
x[8] = 0x59f4a4beu ^ prev[8];
|
||||
x[9] = 0x8a3bc680u ^ prev[9];
|
||||
x[10] = 0x701b8e40u ^ prev[10];
|
||||
x[11] = 0x3b59025au ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
static inline void mh_mixer(uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xe5bef5a3u + rk)) * 0xd1d79e7fu;
|
||||
s[1] = (s[1] ^ (0xdb2a7d90u + rk)) * 0xbcb61f8bu;
|
||||
s[2] = (s[2] ^ (0xcd1fa7e1u + rk)) * 0x970e91abu;
|
||||
s[3] = (s[3] ^ (0x30289419u + rk)) * 0xd749d96du;
|
||||
s[4] = (s[4] ^ (0x0a730d58u + rk)) * 0x143d7339u;
|
||||
s[5] = (s[5] ^ (0x432f8579u + rk)) * 0x2cde0d69u;
|
||||
s[6] = (s[6] ^ (0x6ab978a5u + rk)) * 0x12d0a8c1u;
|
||||
s[7] = (s[7] ^ (0x8c984f49u + rk)) * 0x2d335be9u;
|
||||
s[8] = (s[8] ^ (0x788c3c9eu + rk)) * 0x80a8aae9u;
|
||||
s[9] = (s[9] ^ (0x051eef02u + rk)) * 0x7ac896c7u;
|
||||
s[10] = (s[10] ^ (0x05e77db9u + rk)) * 0x9de23db7u;
|
||||
s[11] = (s[11] ^ (0x8b3cd4f3u + rk)) * 0xc362827du;
|
||||
s[12] = (s[12] ^ (0x6948d6cfu + rk)) * 0x5f4cdb5bu;
|
||||
s[13] = (s[13] ^ (0x0d8677f6u + rk)) * 0xfc6c5097u;
|
||||
s[14] = (s[14] ^ (0xf9505c8cu + rk)) * 0x6f547f83u;
|
||||
s[15] = (s[15] ^ (0x4513c163u + rk)) * 0x1ac31b47u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 28u, 11u, 18u, 18u) MH_QR(s[1], s[5], s[9], s[13], 28u, 11u, 18u, 18u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 28u, 11u, 18u, 18u) MH_QR(s[3], s[7], s[11], s[15], 28u, 11u, 18u, 18u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 13u, 28u, 5u, 13u) MH_QR(s[1], s[6], s[11], s[12], 13u, 28u, 5u, 13u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 13u, 28u, 5u, 13u) MH_QR(s[3], s[4], s[9], s[14], 13u, 28u, 5u, 13u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of 8 x mixer + cache line s[0] & mask; 8 x final mixer.
|
||||
static inline void mh_item(__global const uint* cache, __global const uint* leaf, uint t, uint* s) {
|
||||
s[0] = 0xe23f9008u;
|
||||
s[1] = 0xde33e763u;
|
||||
s[2] = 0xc5ba415du;
|
||||
s[3] = 0x8ddf6786u;
|
||||
s[4] = 0x59f4a4beu;
|
||||
s[5] = 0x8a3bc680u;
|
||||
s[6] = 0x701b8e40u;
|
||||
s[7] = 0x3b59025au;
|
||||
s[8] = t * 0xd1d79e7fu + 0xe5bef5a3u;
|
||||
s[9] = t * 0xbcb61f8bu + 0xdb2a7d90u;
|
||||
s[10] = t * 0x970e91abu + 0xcd1fa7e1u;
|
||||
s[11] = t * 0xd749d96du + 0x30289419u;
|
||||
s[12] = t * 0x143d7339u + 0x0a730d58u;
|
||||
s[13] = t * 0x2cde0d69u + 0x432f8579u;
|
||||
s[14] = t * 0x12d0a8c1u + 0x6ab978a5u;
|
||||
s[15] = t * 0x2d335be9u + 0x8c984f49u;
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= leaf[i];
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
for (uint j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 8u + j + 1u));
|
||||
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
for (uint j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (64u + j + 1u));
|
||||
}
|
||||
// Class v5 (docs/design/class-v5-stored-state.md): leaf(t) = leaves[t mod nLeaves], 16 words per leaf (leaves.bin).
|
||||
static inline __global const uint* mh_leaf(__global const uint* leaves, uint nLeaves, uint t) { return leaves + ((t % nLeaves) * 16u); }
|
||||
// Era layout (docs/plans/era-layout.md 1.2): dataset word w holds word mh_j(w) of item mh_t(w); j's bits sit at positions 1 9 13 14 of w.
|
||||
static inline uint mh_j(uint w) { return ((w >> 1u) & 1u) | (((w >> 9u) & 1u) << 1) | (((w >> 13u) & 1u) << 2) | (((w >> 14u) & 1u) << 3); }
|
||||
static inline uint mh_t(uint w) { w = (w & 0x00003fffu) | ((w >> 15u) << 14u); w = (w & 0x00001fffu) | ((w >> 14u) << 13u); w = (w & 0x000001ffu) | ((w >> 10u) << 9u); w = (w & 0x00000001u) | ((w >> 2u) << 1u); return w; }
|
||||
static inline uint mh_addr(uint t, uint j) { uint w = t; w = ((w >> 1u) << 2u) | (w & 0x00000001u) | (((j >> 0u) & 1u) << 1u); w = ((w >> 9u) << 10u) | (w & 0x000001ffu) | (((j >> 1u) & 1u) << 9u); w = ((w >> 13u) << 14u) | (w & 0x00001fffu) | (((j >> 2u) & 1u) << 13u); w = ((w >> 14u) << 15u) | (w & 0x00003fffu) | (((j >> 3u) & 1u) << 14u); return w; }
|
||||
// dataset[w] without the dataset: derive item mh_t(w) and take word mh_j(w).
|
||||
static inline uint mh_word(__global const uint* cache, __global const uint* leaves, uint nLeaves, uint w) { uint s[16]; mh_item(cache, mh_leaf(leaves, nLeaves, mh_t(w)), mh_t(w), s); return s[mh_j(w)]; }
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
|
||||
// The same constants as memhard.h in this pack (one emitter, three dialects).
|
||||
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
|
||||
uint seg = (uint)get_global_id(0);
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
// Class v5: the window's leaves (leaves.bin, IGNEUM_STATE_LEAVES x 16 words) and their count.
|
||||
__kernel void igneum_build(__global uint* ds, __global const uint* cache, __global const uint* leaves, uint nLeaves, uint nItems) {
|
||||
uint t = (uint)get_global_id(0);
|
||||
if (t < nItems) {
|
||||
uint s[16];
|
||||
mh_item(cache, mh_leaf(leaves, nLeaves, t), t, s);
|
||||
for (uint i = 0u; i < 16u; ++i) ds[(ulong)mh_addr(t, i)] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask) {
|
||||
uint gid = (uint)get_global_id(0);
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ 0xbe8c5a0cu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x7646e626u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint x = nonce ^ 0x7646e626u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0x508e29d0u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint x = nonce ^ 0x508e29d0u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0xfb8a5b4au; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint x = nonce ^ 0xfb8a5b4au; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0x53c50955u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint x = nonce ^ 0x53c50955u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0x685d62fcu; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint x = nonce ^ 0x685d62fcu; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x6065c013u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint x = nonce ^ 0x6065c013u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x881096ebu; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint x = nonce ^ 0x881096ebu; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0xbe8c5a0cu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r6 = r6 - r3; // 0 sub
|
||||
r2 = r2 | r6; // 1 or
|
||||
r4 = r4 + r0 + ((((sel >> 4u) & 1u) != 0u) ? 0x7731324bu : 0x1a579b38u); // 2 add
|
||||
r3 = r3 ^ ds[((rotl_imm(r6 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 3 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r0 = r0 ^ t_; } // 4 shfl
|
||||
r4 = r4 ^ r1; // 5 xor
|
||||
r3 = r5 * r3 + r3; // 6 mad
|
||||
r1 = r1 * r6; // 7 mul
|
||||
r2 = r2 ^ ds[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x03ffffffu) | 0x08000000u) & mask]; // 8 load
|
||||
r0 = r0 ^ r4; // 9 xor
|
||||
r0 = r0 + r7 + ((((sel >> 30u) & 1u) != 0u) ? 0xb10fcef8u : 0xee083919u); // 10 add
|
||||
r0 = mul_hi(r0, r2); // 11 mulhi
|
||||
r5 = r5 ^ r4; // 12 xor
|
||||
r7 = r3 * r0 + r7; // 13 mad
|
||||
r3 = r3 ^ ds[((rotl_imm(r4 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x00000000u) & mask]; // 14 load
|
||||
r2 = r2 ^ ds[((rotl_imm(r5 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 15 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r7 = r7 ^ t_; } // 16 shfl
|
||||
r7 = r7 * r6; // 17 mul
|
||||
r2 = r3 * r1 + r2; // 18 mad
|
||||
r5 = rotr_var(r5, r2); // 19 rotr
|
||||
r5 = r5 ^ ds[((rotl_imm(r3 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 20 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r6 = r6 ^ t_; } // 21 shfl
|
||||
r1 = mul_hi(r1, r2); // 22 mulhi
|
||||
r6 = r6 ^ r5; // 23 xor
|
||||
r5 = r2 * r0 + r5; // 24 mad
|
||||
r6 = r3 * r7 + r6; // 25 mad
|
||||
r6 = r6 ^ ds[((rotl_imm(r5 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 26 load
|
||||
r1 = r5 * r5 + r1; // 27 mad
|
||||
r0 = r0 ^ ds[((rotl_imm(r6 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 28 load
|
||||
r7 = r7 + r4 + ((((sel >> 7u) & 1u) != 0u) ? 0x0b1f02c7u : 0xd5e6c37cu); // 29 add
|
||||
r7 = r7 + r0 + ((((sel >> 0u) & 1u) != 0u) ? 0x2d6e8bd1u : 0x4e45ba51u); // 30 add
|
||||
r0 = r0 ^ r2; // 31 xor
|
||||
r6 = r6 + r2 + ((((sel >> 13u) & 1u) != 0u) ? 0x351422d2u : 0x1c91b2a1u); // 32 add
|
||||
r4 = mul_hi(r4, r5); // 33 mulhi
|
||||
r3 = rotr_var(r3, r5); // 34 rotr
|
||||
r3 = r3 ^ ds[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 35 load
|
||||
r4 = rotl_imm(r4, 13u); // 36 rotl
|
||||
r6 = r6 + r1 + ((((sel >> 7u) & 1u) != 0u) ? 0x64a28251u : 0x3f2970f7u); // 37 add
|
||||
r2 = r2 * r0; // 38 mul
|
||||
r3 = r3 + r4 + ((((sel >> 15u) & 1u) != 0u) ? 0x7378c955u : 0x3b7b3317u); // 39 add
|
||||
r0 = r0 ^ ds[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x03ffffffu) | 0x0c000000u) & mask]; // 40 load
|
||||
r6 = r6 ^ r5; // 41 xor
|
||||
r3 = r3 + r0 + ((((sel >> 2u) & 1u) != 0u) ? 0xb31f6a64u : 0x6678b059u); // 42 add
|
||||
r7 = r7 ^ ds[((rotl_imm(r3 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x00000000u) & mask]; // 43 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 4u); r3 = r3 ^ t_; } // 44 shfl
|
||||
r1 = mul_hi(r1, r5); // 45 mulhi
|
||||
r0 = rotl_imm(r0, 6u); // 46 rotl
|
||||
r1 = r1 ^ ds[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 47 load
|
||||
r4 = r4 + r0 + ((((sel >> 17u) & 1u) != 0u) ? 0x5feccee6u : 0x43095946u); // 48 add
|
||||
r2 = r2 ^ ds[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 49 load
|
||||
r2 = r2 - r3; // 50 sub
|
||||
r7 = r7 + r2 + ((((sel >> 24u) & 1u) != 0u) ? 0x26e2b582u : 0xf45ecdf8u); // 51 add
|
||||
r5 = r5 ^ ds[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x03ffffffu) | 0x0c000000u) & mask]; // 52 load
|
||||
r6 = r6 ^ ds[((rotl_imm(r3 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 53 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r7 = r7 ^ t_; } // 54 shfl
|
||||
r1 = rotr_var(r1, r2); // 55 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r0 = r0 ^ t_; } // 56 shfl
|
||||
r5 = rotl_imm(r5, 12u); // 57 rotl
|
||||
r5 = r5 + r3 + ((((sel >> 4u) & 1u) != 0u) ? 0x473b1718u : 0x6b4f35e8u); // 58 add
|
||||
r4 = rotr_var(r4, r1); // 59 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 1u); r4 = r4 ^ t_; } // 60 shfl
|
||||
r7 = r7 ^ ds[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 61 load
|
||||
r5 = r5 ^ ds[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 62 load
|
||||
r4 = r4 * r1; // 63 mul
|
||||
// latency-shadow block (Counter ASIC 3.0 item 8): 256 ALU instructions x 27 passes after instruction 63, no load
|
||||
for (uint sh = 0u; sh < 27u; ++sh) {
|
||||
r6 = r6 | r0; // s0 or
|
||||
r7 = r7 * r3; // s1 mul
|
||||
r3 = r5 * r5 + r3; // s2 mad
|
||||
r7 = r7 + r4 + ((((sel >> 6u) & 1u) != 0u) ? 0x38e58a06u : 0x97d3d105u); // s3 add
|
||||
r6 = r6 + r0 + ((((sel >> 20u) & 1u) != 0u) ? 0x9629673du : 0x29b87b9au); // s4 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 1u); r6 = r6 ^ t_; } // s5 shfl
|
||||
r5 = r3 * r5 + r5; // s6 mad
|
||||
r3 = rotl_imm(r3, 5u); // s7 rotl
|
||||
r0 = r0 + r7 + ((((sel >> 12u) & 1u) != 0u) ? 0x4a5f55e1u : 0xa7c1fb0fu); // s8 add
|
||||
r5 = r5 * r7; // s9 mul
|
||||
r4 = r4 ^ r5; // s10 xor
|
||||
r6 = mul_hi(r6, r4); // s11 mulhi
|
||||
r5 = rotl_imm(r5, 13u); // s12 rotl
|
||||
r0 = r2 * r1 + r0; // s13 mad
|
||||
r2 = mul_hi(r2, r4); // s14 mulhi
|
||||
r3 = r3 | r1; // s15 or
|
||||
r0 = r0 - r5; // s16 sub
|
||||
r0 = r0 * r6; // s17 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 4u); r5 = r5 ^ t_; } // s18 shfl
|
||||
r3 = r3 | r4; // s19 or
|
||||
r6 = r3 * r2 + r6; // s20 mad
|
||||
r0 = rotl_imm(r0, 11u); // s21 rotl
|
||||
r3 = rotl_imm(r3, 25u); // s22 rotl
|
||||
r6 = r6 ^ r4; // s23 xor
|
||||
r4 = r4 | r2; // s24 or
|
||||
r1 = r1 + r6 + ((((sel >> 7u) & 1u) != 0u) ? 0xfc2f6d17u : 0x53e74799u); // s25 add
|
||||
r7 = rotr_var(r7, r6); // s26 rotr
|
||||
r7 = r7 + r5 + ((((sel >> 27u) & 1u) != 0u) ? 0x787048dcu : 0xb8fae90eu); // s27 add
|
||||
r4 = r4 ^ r1; // s28 xor
|
||||
r6 = r6 ^ r1; // s29 xor
|
||||
r2 = mul_hi(r2, r0); // s30 mulhi
|
||||
r1 = rotr_var(r1, r4); // s31 rotr
|
||||
r7 = r7 ^ r4; // s32 xor
|
||||
r3 = r2 * r6 + r3; // s33 mad
|
||||
r3 = r3 + r1 + ((((sel >> 12u) & 1u) != 0u) ? 0x9ffd510bu : 0xd2dc42ebu); // s34 add
|
||||
r4 = r4 + r7 + ((((sel >> 7u) & 1u) != 0u) ? 0xf871ba37u : 0xad8eda6fu); // s35 add
|
||||
r7 = r7 + r4 + ((((sel >> 3u) & 1u) != 0u) ? 0xd02d30cau : 0xb0cef8f4u); // s36 add
|
||||
r4 = r3 * r2 + r4; // s37 mad
|
||||
r5 = r7 * r7 + r5; // s38 mad
|
||||
r3 = r3 + r6 + ((((sel >> 26u) & 1u) != 0u) ? 0x82e98a22u : 0x2bd506e2u); // s39 add
|
||||
r0 = rotl_imm(r0, 21u); // s40 rotl
|
||||
r1 = r1 | r0; // s41 or
|
||||
r7 = r7 + r5 + ((((sel >> 28u) & 1u) != 0u) ? 0x6435f6d8u : 0xc0386e49u); // s42 add
|
||||
r1 = mul_hi(r1, r0); // s43 mulhi
|
||||
r5 = r5 + r6 + ((((sel >> 12u) & 1u) != 0u) ? 0x1f682c68u : 0x3a8921a4u); // s44 add
|
||||
r4 = r4 + r5 + ((((sel >> 1u) & 1u) != 0u) ? 0xf246e180u : 0xeeca4334u); // s45 add
|
||||
r2 = r2 ^ r3; // s46 xor
|
||||
r2 = r2 ^ r5; // s47 xor
|
||||
r7 = r4 * r1 + r7; // s48 mad
|
||||
r2 = r6 * r1 + r2; // s49 mad
|
||||
r7 = r0 * r7 + r7; // s50 mad
|
||||
r5 = r5 + r6 + ((((sel >> 26u) & 1u) != 0u) ? 0x41bd68ccu : 0x745776d8u); // s51 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 1u); r3 = r3 ^ t_; } // s52 shfl
|
||||
r4 = r4 | r0; // s53 or
|
||||
r0 = r2 * r4 + r0; // s54 mad
|
||||
r5 = rotl_imm(r5, 9u); // s55 rotl
|
||||
r5 = r5 * r7; // s56 mul
|
||||
r7 = rotl_imm(r7, 27u); // s57 rotl
|
||||
r7 = r7 - r6; // s58 sub
|
||||
r2 = rotl_imm(r2, 22u); // s59 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 2u); r7 = r7 ^ t_; } // s60 shfl
|
||||
r4 = r4 + r0 + ((((sel >> 7u) & 1u) != 0u) ? 0xcba6a161u : 0x60d0ff55u); // s61 add
|
||||
r1 = r1 * r5; // s62 mul
|
||||
r7 = mul_hi(r7, r6); // s63 mulhi
|
||||
r5 = r5 * r0; // s64 mul
|
||||
r2 = rotr_var(r2, r5); // s65 rotr
|
||||
r0 = mul_hi(r0, r3); // s66 mulhi
|
||||
r0 = r0 | r7; // s67 or
|
||||
r0 = rotl_imm(r0, 19u); // s68 rotl
|
||||
r0 = rotl_imm(r0, 1u); // s69 rotl
|
||||
r2 = r2 | r3; // s70 or
|
||||
r3 = r7 * r5 + r3; // s71 mad
|
||||
r7 = r0 * r4 + r7; // s72 mad
|
||||
r4 = r4 | r3; // s73 or
|
||||
r3 = r3 ^ r1; // s74 xor
|
||||
r1 = r1 + r3 + ((((sel >> 15u) & 1u) != 0u) ? 0x8e35924eu : 0x3f4ffea2u); // s75 add
|
||||
r1 = r1 * r3; // s76 mul
|
||||
r0 = r3 * r2 + r0; // s77 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 16u); r2 = r2 ^ t_; } // s78 shfl
|
||||
r1 = mul_hi(r1, r0); // s79 mulhi
|
||||
r0 = mul_hi(r0, r4); // s80 mulhi
|
||||
r4 = r4 * r2; // s81 mul
|
||||
r5 = r5 - r7; // s82 sub
|
||||
r0 = r2 * r3 + r0; // s83 mad
|
||||
r7 = r7 - r2; // s84 sub
|
||||
r1 = r1 ^ r2; // s85 xor
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 16u); r6 = r6 ^ t_; } // s86 shfl
|
||||
r3 = r0 * r0 + r3; // s87 mad
|
||||
r0 = rotl_imm(r0, 20u); // s88 rotl
|
||||
r1 = r1 - r5; // s89 sub
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r6 = r6 ^ t_; } // s90 shfl
|
||||
r5 = r5 ^ r1; // s91 xor
|
||||
r5 = rotl_imm(r5, 12u); // s92 rotl
|
||||
r6 = r1 * r2 + r6; // s93 mad
|
||||
r2 = rotr_var(r2, r5); // s94 rotr
|
||||
r0 = r0 | r4; // s95 or
|
||||
r7 = r7 * r3; // s96 mul
|
||||
r5 = r5 | r0; // s97 or
|
||||
r3 = r3 + r1 + ((((sel >> 13u) & 1u) != 0u) ? 0x386bb642u : 0xbcfdd8a7u); // s98 add
|
||||
r0 = r0 + r6 + ((((sel >> 9u) & 1u) != 0u) ? 0xa3962d03u : 0xc921a021u); // s99 add
|
||||
r1 = rotr_var(r1, r0); // s100 rotr
|
||||
r2 = r2 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0xc5990c5cu : 0x70da4067u); // s101 add
|
||||
r2 = rotl_imm(r2, 3u); // s102 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 4u); r1 = r1 ^ t_; } // s103 shfl
|
||||
r0 = r0 ^ r2; // s104 xor
|
||||
r4 = r4 + r3 + ((((sel >> 28u) & 1u) != 0u) ? 0xb15aec75u : 0x1a5880cbu); // s105 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 2u); r4 = r4 ^ t_; } // s106 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 4u); r7 = r7 ^ t_; } // s107 shfl
|
||||
r1 = r7 * r3 + r1; // s108 mad
|
||||
r0 = r2 * r3 + r0; // s109 mad
|
||||
r2 = r2 | r4; // s110 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 1u); r6 = r6 ^ t_; } // s111 shfl
|
||||
r2 = r5 * r3 + r2; // s112 mad
|
||||
r4 = mul_hi(r4, r0); // s113 mulhi
|
||||
r7 = rotl_imm(r7, 29u); // s114 rotl
|
||||
r3 = r3 - r7; // s115 sub
|
||||
r0 = mul_hi(r0, r1); // s116 mulhi
|
||||
r4 = rotl_imm(r4, 12u); // s117 rotl
|
||||
r1 = r1 + r4 + ((((sel >> 7u) & 1u) != 0u) ? 0xc5e98ee5u : 0x24494ad3u); // s118 add
|
||||
r0 = r0 + r4 + ((((sel >> 4u) & 1u) != 0u) ? 0xa97bc949u : 0x1b2d1c83u); // s119 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 4u); r5 = r5 ^ t_; } // s120 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r3 = r3 ^ t_; } // s121 shfl
|
||||
r7 = rotl_imm(r7, 3u); // s122 rotl
|
||||
r4 = rotr_var(r4, r6); // s123 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 1u); r6 = r6 ^ t_; } // s124 shfl
|
||||
r5 = rotl_imm(r5, 21u); // s125 rotl
|
||||
r0 = r3 * r1 + r0; // s126 mad
|
||||
r0 = r0 + r3 + ((((sel >> 24u) & 1u) != 0u) ? 0xa7f6bb54u : 0xff14c08du); // s127 add
|
||||
r7 = r7 ^ r4; // s128 xor
|
||||
r3 = mul_hi(r3, r5); // s129 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 8u); r5 = r5 ^ t_; } // s130 shfl
|
||||
r6 = mul_hi(r6, r4); // s131 mulhi
|
||||
r5 = r5 ^ r3; // s132 xor
|
||||
r0 = rotl_imm(r0, 19u); // s133 rotl
|
||||
r4 = rotr_var(r4, r3); // s134 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 16u); r1 = r1 ^ t_; } // s135 shfl
|
||||
r5 = r5 - r2; // s136 sub
|
||||
r3 = r3 ^ r5; // s137 xor
|
||||
r0 = r0 - r2; // s138 sub
|
||||
r3 = r3 * r7; // s139 mul
|
||||
r5 = r5 + r3 + ((((sel >> 24u) & 1u) != 0u) ? 0x2d465ca8u : 0x82ab98f4u); // s140 add
|
||||
r4 = r4 + r1 + ((((sel >> 11u) & 1u) != 0u) ? 0x237a9d8eu : 0x80594390u); // s141 add
|
||||
r3 = r7 * r0 + r3; // s142 mad
|
||||
r0 = r0 | r3; // s143 or
|
||||
r0 = mul_hi(r0, r1); // s144 mulhi
|
||||
r6 = r6 ^ r2; // s145 xor
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 16u); r7 = r7 ^ t_; } // s146 shfl
|
||||
r0 = r0 + r1 + ((((sel >> 18u) & 1u) != 0u) ? 0xe68cf57eu : 0x07f15148u); // s147 add
|
||||
r7 = r7 | r3; // s148 or
|
||||
r4 = r4 ^ r2; // s149 xor
|
||||
r7 = r7 ^ r1; // s150 xor
|
||||
r5 = rotr_var(r5, r2); // s151 rotr
|
||||
r0 = r0 * r7; // s152 mul
|
||||
r4 = rotl_imm(r4, 16u); // s153 rotl
|
||||
r6 = rotr_var(r6, r4); // s154 rotr
|
||||
r6 = r6 ^ r1; // s155 xor
|
||||
r6 = r6 | r3; // s156 or
|
||||
r1 = r1 + r6 + ((((sel >> 19u) & 1u) != 0u) ? 0x84c2a09du : 0x14878c5au); // s157 add
|
||||
r4 = r4 - r1; // s158 sub
|
||||
r4 = r6 * r7 + r4; // s159 mad
|
||||
r1 = r1 * r5; // s160 mul
|
||||
r4 = r4 | r0; // s161 or
|
||||
r7 = rotl_imm(r7, 21u); // s162 rotl
|
||||
r0 = r0 - r1; // s163 sub
|
||||
r1 = r1 * r0; // s164 mul
|
||||
r2 = r2 - r1; // s165 sub
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r0 = r0 ^ t_; } // s166 shfl
|
||||
r0 = r0 - r4; // s167 sub
|
||||
r1 = rotr_var(r1, r6); // s168 rotr
|
||||
r7 = rotr_var(r7, r1); // s169 rotr
|
||||
r3 = r1 * r3 + r3; // s170 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 4u); r2 = r2 ^ t_; } // s171 shfl
|
||||
r2 = r2 - r1; // s172 sub
|
||||
r7 = r7 ^ r6; // s173 xor
|
||||
r4 = rotr_var(r4, r1); // s174 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 2u); r4 = r4 ^ t_; } // s175 shfl
|
||||
r6 = r6 * r2; // s176 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r0 = r0 ^ t_; } // s177 shfl
|
||||
r6 = r6 + r4 + ((((sel >> 21u) & 1u) != 0u) ? 0x509871e6u : 0x4fa43965u); // s178 add
|
||||
r0 = r0 ^ r6; // s179 xor
|
||||
r5 = r5 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0xb95b8a53u : 0xfb3c4bd8u); // s180 add
|
||||
r6 = r6 * r3; // s181 mul
|
||||
r7 = rotl_imm(r7, 1u); // s182 rotl
|
||||
r0 = r0 + r7 + ((((sel >> 11u) & 1u) != 0u) ? 0x7047b7cfu : 0xc448a197u); // s183 add
|
||||
r3 = r3 * r5; // s184 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 1u); r5 = r5 ^ t_; } // s185 shfl
|
||||
r0 = r0 * r2; // s186 mul
|
||||
r4 = mul_hi(r4, r7); // s187 mulhi
|
||||
r2 = r7 * r2 + r2; // s188 mad
|
||||
r7 = r7 - r2; // s189 sub
|
||||
r2 = r2 + r5 + ((((sel >> 19u) & 1u) != 0u) ? 0x5caccc5du : 0x2fceaf49u); // s190 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 1u); r1 = r1 ^ t_; } // s191 shfl
|
||||
r7 = r7 + r1 + ((((sel >> 20u) & 1u) != 0u) ? 0x857bb5feu : 0xba0cee62u); // s192 add
|
||||
r6 = rotl_imm(r6, 22u); // s193 rotl
|
||||
r3 = r3 ^ r7; // s194 xor
|
||||
r7 = r7 * r0; // s195 mul
|
||||
r3 = r3 ^ r5; // s196 xor
|
||||
r5 = r5 + r1 + ((((sel >> 29u) & 1u) != 0u) ? 0xeae84577u : 0x8519428cu); // s197 add
|
||||
r7 = r4 * r4 + r7; // s198 mad
|
||||
r3 = r3 + r6 + ((((sel >> 8u) & 1u) != 0u) ? 0x0bfdbfa1u : 0x468639d3u); // s199 add
|
||||
r5 = r3 * r4 + r5; // s200 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 4u); r2 = r2 ^ t_; } // s201 shfl
|
||||
r7 = r7 + r4 + ((((sel >> 23u) & 1u) != 0u) ? 0x53ee8e50u : 0x18ec9a69u); // s202 add
|
||||
r6 = r6 ^ r0; // s203 xor
|
||||
r4 = r4 + r5 + ((((sel >> 4u) & 1u) != 0u) ? 0xac63376eu : 0x3e81485bu); // s204 add
|
||||
r7 = r1 * r4 + r7; // s205 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r1 = r1 ^ t_; } // s206 shfl
|
||||
r4 = r4 * r0; // s207 mul
|
||||
r1 = r1 | r6; // s208 or
|
||||
r6 = r6 + r4 + ((((sel >> 28u) & 1u) != 0u) ? 0x5b745519u : 0x66ccb75du); // s209 add
|
||||
r1 = rotr_var(r1, r3); // s210 rotr
|
||||
r5 = r5 - r4; // s211 sub
|
||||
r4 = r4 ^ r7; // s212 xor
|
||||
r1 = r1 + r7 + ((((sel >> 8u) & 1u) != 0u) ? 0x1c3dccf7u : 0x0ce0553du); // s213 add
|
||||
r0 = mul_hi(r0, r1); // s214 mulhi
|
||||
r3 = r3 + r6 + ((((sel >> 9u) & 1u) != 0u) ? 0xfadc9205u : 0x96bfec89u); // s215 add
|
||||
r0 = rotl_imm(r0, 4u); // s216 rotl
|
||||
r6 = r6 ^ r4; // s217 xor
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 2u); r6 = r6 ^ t_; } // s218 shfl
|
||||
r7 = r0 * r1 + r7; // s219 mad
|
||||
r4 = r4 * r0; // s220 mul
|
||||
r2 = r6 * r0 + r2; // s221 mad
|
||||
r5 = r5 ^ r4; // s222 xor
|
||||
r0 = r0 * r5; // s223 mul
|
||||
r2 = r2 + r0 + ((((sel >> 9u) & 1u) != 0u) ? 0x66b15c04u : 0x61495681u); // s224 add
|
||||
r3 = rotr_var(r3, r2); // s225 rotr
|
||||
r2 = r3 * r3 + r2; // s226 mad
|
||||
r6 = r6 ^ r0; // s227 xor
|
||||
r4 = rotl_imm(r4, 30u); // s228 rotl
|
||||
r2 = r2 + r3 + ((((sel >> 29u) & 1u) != 0u) ? 0x25bc17c2u : 0x70d05c34u); // s229 add
|
||||
r1 = rotr_var(r1, r5); // s230 rotr
|
||||
r1 = r1 + r4 + ((((sel >> 8u) & 1u) != 0u) ? 0x2d8728f3u : 0x32a28384u); // s231 add
|
||||
r1 = rotl_imm(r1, 6u); // s232 rotl
|
||||
r2 = r2 + r0 + ((((sel >> 21u) & 1u) != 0u) ? 0x61e81fe6u : 0xcf0949f1u); // s233 add
|
||||
r4 = rotl_imm(r4, 2u); // s234 rotl
|
||||
r2 = r2 - r6; // s235 sub
|
||||
r6 = r6 | r3; // s236 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r7 = r7 ^ t_; } // s237 shfl
|
||||
r1 = r1 ^ r3; // s238 xor
|
||||
r2 = rotl_imm(r2, 1u); // s239 rotl
|
||||
r5 = rotr_var(r5, r2); // s240 rotr
|
||||
r4 = rotl_imm(r4, 10u); // s241 rotl
|
||||
r6 = r6 + r2 + ((((sel >> 16u) & 1u) != 0u) ? 0x0a67565du : 0xe204fb50u); // s242 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 2u); r1 = r1 ^ t_; } // s243 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 16u); r0 = r0 ^ t_; } // s244 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 2u); r5 = r5 ^ t_; } // s245 shfl
|
||||
r4 = r4 ^ r5; // s246 xor
|
||||
r7 = r7 ^ r2; // s247 xor
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 16u); r0 = r0 ^ t_; } // s248 shfl
|
||||
r1 = mul_hi(r1, r3); // s249 mulhi
|
||||
r7 = r7 + r5 + ((((sel >> 31u) & 1u) != 0u) ? 0x5a79fd6fu : 0x3d6daffdu); // s250 add
|
||||
r6 = r6 + r0 + ((((sel >> 28u) & 1u) != 0u) ? 0x6beb28c2u : 0x84334aeau); // s251 add
|
||||
r7 = r7 - r1; // s252 sub
|
||||
r5 = rotr_var(r5, r2); // s253 rotr
|
||||
r2 = mul_hi(r2, r5); // s254 mulhi
|
||||
r6 = r6 * r3; // s255 mul
|
||||
}
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
|
||||
#if IGNEUM_EXCHANGE != 0
|
||||
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
|
||||
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
|
||||
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
|
||||
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
|
||||
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
|
||||
}
|
||||
#endif
|
||||
423
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/kernel.cu
Normal file
423
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/kernel.cu
Normal file
|
|
@ -0,0 +1,423 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925/day/69676e65756d2d6461792ffd50000000000000". Do not edit by hand.
|
||||
// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal).
|
||||
// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC.
|
||||
#include <cuda_runtime.h>
|
||||
#include <cstdint>
|
||||
#include "program.h"
|
||||
#include "memhard.h"
|
||||
|
||||
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31.
|
||||
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
|
||||
// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x.
|
||||
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) {
|
||||
uint32_t x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item.
|
||||
// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host.
|
||||
__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) {
|
||||
uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
// Class v5: the window's leaves (leaves.bin, IGNEUM_STATE_LEAVES x 16 words) and their count.
|
||||
__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems) {
|
||||
uint32_t t = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (t < nItems) {
|
||||
uint32_t s[16];
|
||||
mh_item(cache, mh_leaf(leaves, nLeaves, t), t, s);
|
||||
for (uint32_t i = 0u; i < 16u; ++i) ds[(size_t)mh_addr(t, i)] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
|
||||
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
|
||||
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
|
||||
__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask) {
|
||||
uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
uint32_t nonce = baseNonce + gid;
|
||||
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint32_t x = nonce ^ 0xbe8c5a0cu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x7646e626u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint32_t x = nonce ^ 0x7646e626u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0x508e29d0u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint32_t x = nonce ^ 0x508e29d0u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0xfb8a5b4au; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint32_t x = nonce ^ 0xfb8a5b4au; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0x53c50955u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint32_t x = nonce ^ 0x53c50955u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0x685d62fcu; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint32_t x = nonce ^ 0x685d62fcu; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x6065c013u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint32_t x = nonce ^ 0x6065c013u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x881096ebu; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint32_t x = nonce ^ 0x881096ebu; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0xbe8c5a0cu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint32_t it = 0u; it < 8u; ++it) {
|
||||
uint32_t sel = r0;
|
||||
r6 = r6 - r3; // 0 sub
|
||||
r2 = r2 | r6; // 1 or
|
||||
r4 = r4 + r0 + ((((sel >> 4u) & 1u) != 0u) ? 0x7731324bu : 0x1a579b38u); // 2 add
|
||||
r3 = r3 ^ ds[((rotl_imm(r6 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 3 load
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 1); // 4 shfl
|
||||
r4 = r4 ^ r1; // 5 xor
|
||||
r3 = r5 * r3 + r3; // 6 mad
|
||||
r1 = r1 * r6; // 7 mul
|
||||
r2 = r2 ^ ds[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x03ffffffu) | 0x08000000u) & mask]; // 8 load
|
||||
r0 = r0 ^ r4; // 9 xor
|
||||
r0 = r0 + r7 + ((((sel >> 30u) & 1u) != 0u) ? 0xb10fcef8u : 0xee083919u); // 10 add
|
||||
r0 = __umulhi(r0, r2); // 11 mulhi
|
||||
r5 = r5 ^ r4; // 12 xor
|
||||
r7 = r3 * r0 + r7; // 13 mad
|
||||
r3 = r3 ^ ds[((rotl_imm(r4 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x00000000u) & mask]; // 14 load
|
||||
r2 = r2 ^ ds[((rotl_imm(r5 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 15 load
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r5, 16); // 16 shfl
|
||||
r7 = r7 * r6; // 17 mul
|
||||
r2 = r3 * r1 + r2; // 18 mad
|
||||
r5 = rotr_var(r5, r2); // 19 rotr
|
||||
r5 = r5 ^ ds[((rotl_imm(r3 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 20 load
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r4, 16); // 21 shfl
|
||||
r1 = __umulhi(r1, r2); // 22 mulhi
|
||||
r6 = r6 ^ r5; // 23 xor
|
||||
r5 = r2 * r0 + r5; // 24 mad
|
||||
r6 = r3 * r7 + r6; // 25 mad
|
||||
r6 = r6 ^ ds[((rotl_imm(r5 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 26 load
|
||||
r1 = r5 * r5 + r1; // 27 mad
|
||||
r0 = r0 ^ ds[((rotl_imm(r6 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 28 load
|
||||
r7 = r7 + r4 + ((((sel >> 7u) & 1u) != 0u) ? 0x0b1f02c7u : 0xd5e6c37cu); // 29 add
|
||||
r7 = r7 + r0 + ((((sel >> 0u) & 1u) != 0u) ? 0x2d6e8bd1u : 0x4e45ba51u); // 30 add
|
||||
r0 = r0 ^ r2; // 31 xor
|
||||
r6 = r6 + r2 + ((((sel >> 13u) & 1u) != 0u) ? 0x351422d2u : 0x1c91b2a1u); // 32 add
|
||||
r4 = __umulhi(r4, r5); // 33 mulhi
|
||||
r3 = rotr_var(r3, r5); // 34 rotr
|
||||
r3 = r3 ^ ds[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 35 load
|
||||
r4 = rotl_imm(r4, 13u); // 36 rotl
|
||||
r6 = r6 + r1 + ((((sel >> 7u) & 1u) != 0u) ? 0x64a28251u : 0x3f2970f7u); // 37 add
|
||||
r2 = r2 * r0; // 38 mul
|
||||
r3 = r3 + r4 + ((((sel >> 15u) & 1u) != 0u) ? 0x7378c955u : 0x3b7b3317u); // 39 add
|
||||
r0 = r0 ^ ds[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x03ffffffu) | 0x0c000000u) & mask]; // 40 load
|
||||
r6 = r6 ^ r5; // 41 xor
|
||||
r3 = r3 + r0 + ((((sel >> 2u) & 1u) != 0u) ? 0xb31f6a64u : 0x6678b059u); // 42 add
|
||||
r7 = r7 ^ ds[((rotl_imm(r3 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x00000000u) & mask]; // 43 load
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r7, 4); // 44 shfl
|
||||
r1 = __umulhi(r1, r5); // 45 mulhi
|
||||
r0 = rotl_imm(r0, 6u); // 46 rotl
|
||||
r1 = r1 ^ ds[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 47 load
|
||||
r4 = r4 + r0 + ((((sel >> 17u) & 1u) != 0u) ? 0x5feccee6u : 0x43095946u); // 48 add
|
||||
r2 = r2 ^ ds[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 49 load
|
||||
r2 = r2 - r3; // 50 sub
|
||||
r7 = r7 + r2 + ((((sel >> 24u) & 1u) != 0u) ? 0x26e2b582u : 0xf45ecdf8u); // 51 add
|
||||
r5 = r5 ^ ds[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x03ffffffu) | 0x0c000000u) & mask]; // 52 load
|
||||
r6 = r6 ^ ds[((rotl_imm(r3 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 53 load
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r5, 4); // 54 shfl
|
||||
r1 = rotr_var(r1, r2); // 55 rotr
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r5, 16); // 56 shfl
|
||||
r5 = rotl_imm(r5, 12u); // 57 rotl
|
||||
r5 = r5 + r3 + ((((sel >> 4u) & 1u) != 0u) ? 0x473b1718u : 0x6b4f35e8u); // 58 add
|
||||
r4 = rotr_var(r4, r1); // 59 rotr
|
||||
r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r5, 1); // 60 shfl
|
||||
r7 = r7 ^ ds[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 61 load
|
||||
r5 = r5 ^ ds[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 62 load
|
||||
r4 = r4 * r1; // 63 mul
|
||||
// latency-shadow block (Counter ASIC 3.0 item 8): 256 ALU instructions x 27 passes after instruction 63, no load
|
||||
for (uint32_t sh = 0u; sh < 27u; ++sh) {
|
||||
r6 = r6 | r0; // s0 or
|
||||
r7 = r7 * r3; // s1 mul
|
||||
r3 = r5 * r5 + r3; // s2 mad
|
||||
r7 = r7 + r4 + ((((sel >> 6u) & 1u) != 0u) ? 0x38e58a06u : 0x97d3d105u); // s3 add
|
||||
r6 = r6 + r0 + ((((sel >> 20u) & 1u) != 0u) ? 0x9629673du : 0x29b87b9au); // s4 add
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r2, 1); // s5 shfl
|
||||
r5 = r3 * r5 + r5; // s6 mad
|
||||
r3 = rotl_imm(r3, 5u); // s7 rotl
|
||||
r0 = r0 + r7 + ((((sel >> 12u) & 1u) != 0u) ? 0x4a5f55e1u : 0xa7c1fb0fu); // s8 add
|
||||
r5 = r5 * r7; // s9 mul
|
||||
r4 = r4 ^ r5; // s10 xor
|
||||
r6 = __umulhi(r6, r4); // s11 mulhi
|
||||
r5 = rotl_imm(r5, 13u); // s12 rotl
|
||||
r0 = r2 * r1 + r0; // s13 mad
|
||||
r2 = __umulhi(r2, r4); // s14 mulhi
|
||||
r3 = r3 | r1; // s15 or
|
||||
r0 = r0 - r5; // s16 sub
|
||||
r0 = r0 * r6; // s17 mul
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r1, 4); // s18 shfl
|
||||
r3 = r3 | r4; // s19 or
|
||||
r6 = r3 * r2 + r6; // s20 mad
|
||||
r0 = rotl_imm(r0, 11u); // s21 rotl
|
||||
r3 = rotl_imm(r3, 25u); // s22 rotl
|
||||
r6 = r6 ^ r4; // s23 xor
|
||||
r4 = r4 | r2; // s24 or
|
||||
r1 = r1 + r6 + ((((sel >> 7u) & 1u) != 0u) ? 0xfc2f6d17u : 0x53e74799u); // s25 add
|
||||
r7 = rotr_var(r7, r6); // s26 rotr
|
||||
r7 = r7 + r5 + ((((sel >> 27u) & 1u) != 0u) ? 0x787048dcu : 0xb8fae90eu); // s27 add
|
||||
r4 = r4 ^ r1; // s28 xor
|
||||
r6 = r6 ^ r1; // s29 xor
|
||||
r2 = __umulhi(r2, r0); // s30 mulhi
|
||||
r1 = rotr_var(r1, r4); // s31 rotr
|
||||
r7 = r7 ^ r4; // s32 xor
|
||||
r3 = r2 * r6 + r3; // s33 mad
|
||||
r3 = r3 + r1 + ((((sel >> 12u) & 1u) != 0u) ? 0x9ffd510bu : 0xd2dc42ebu); // s34 add
|
||||
r4 = r4 + r7 + ((((sel >> 7u) & 1u) != 0u) ? 0xf871ba37u : 0xad8eda6fu); // s35 add
|
||||
r7 = r7 + r4 + ((((sel >> 3u) & 1u) != 0u) ? 0xd02d30cau : 0xb0cef8f4u); // s36 add
|
||||
r4 = r3 * r2 + r4; // s37 mad
|
||||
r5 = r7 * r7 + r5; // s38 mad
|
||||
r3 = r3 + r6 + ((((sel >> 26u) & 1u) != 0u) ? 0x82e98a22u : 0x2bd506e2u); // s39 add
|
||||
r0 = rotl_imm(r0, 21u); // s40 rotl
|
||||
r1 = r1 | r0; // s41 or
|
||||
r7 = r7 + r5 + ((((sel >> 28u) & 1u) != 0u) ? 0x6435f6d8u : 0xc0386e49u); // s42 add
|
||||
r1 = __umulhi(r1, r0); // s43 mulhi
|
||||
r5 = r5 + r6 + ((((sel >> 12u) & 1u) != 0u) ? 0x1f682c68u : 0x3a8921a4u); // s44 add
|
||||
r4 = r4 + r5 + ((((sel >> 1u) & 1u) != 0u) ? 0xf246e180u : 0xeeca4334u); // s45 add
|
||||
r2 = r2 ^ r3; // s46 xor
|
||||
r2 = r2 ^ r5; // s47 xor
|
||||
r7 = r4 * r1 + r7; // s48 mad
|
||||
r2 = r6 * r1 + r2; // s49 mad
|
||||
r7 = r0 * r7 + r7; // s50 mad
|
||||
r5 = r5 + r6 + ((((sel >> 26u) & 1u) != 0u) ? 0x41bd68ccu : 0x745776d8u); // s51 add
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r0, 1); // s52 shfl
|
||||
r4 = r4 | r0; // s53 or
|
||||
r0 = r2 * r4 + r0; // s54 mad
|
||||
r5 = rotl_imm(r5, 9u); // s55 rotl
|
||||
r5 = r5 * r7; // s56 mul
|
||||
r7 = rotl_imm(r7, 27u); // s57 rotl
|
||||
r7 = r7 - r6; // s58 sub
|
||||
r2 = rotl_imm(r2, 22u); // s59 rotl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r1, 2); // s60 shfl
|
||||
r4 = r4 + r0 + ((((sel >> 7u) & 1u) != 0u) ? 0xcba6a161u : 0x60d0ff55u); // s61 add
|
||||
r1 = r1 * r5; // s62 mul
|
||||
r7 = __umulhi(r7, r6); // s63 mulhi
|
||||
r5 = r5 * r0; // s64 mul
|
||||
r2 = rotr_var(r2, r5); // s65 rotr
|
||||
r0 = __umulhi(r0, r3); // s66 mulhi
|
||||
r0 = r0 | r7; // s67 or
|
||||
r0 = rotl_imm(r0, 19u); // s68 rotl
|
||||
r0 = rotl_imm(r0, 1u); // s69 rotl
|
||||
r2 = r2 | r3; // s70 or
|
||||
r3 = r7 * r5 + r3; // s71 mad
|
||||
r7 = r0 * r4 + r7; // s72 mad
|
||||
r4 = r4 | r3; // s73 or
|
||||
r3 = r3 ^ r1; // s74 xor
|
||||
r1 = r1 + r3 + ((((sel >> 15u) & 1u) != 0u) ? 0x8e35924eu : 0x3f4ffea2u); // s75 add
|
||||
r1 = r1 * r3; // s76 mul
|
||||
r0 = r3 * r2 + r0; // s77 mad
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r0, 16); // s78 shfl
|
||||
r1 = __umulhi(r1, r0); // s79 mulhi
|
||||
r0 = __umulhi(r0, r4); // s80 mulhi
|
||||
r4 = r4 * r2; // s81 mul
|
||||
r5 = r5 - r7; // s82 sub
|
||||
r0 = r2 * r3 + r0; // s83 mad
|
||||
r7 = r7 - r2; // s84 sub
|
||||
r1 = r1 ^ r2; // s85 xor
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 16); // s86 shfl
|
||||
r3 = r0 * r0 + r3; // s87 mad
|
||||
r0 = rotl_imm(r0, 20u); // s88 rotl
|
||||
r1 = r1 - r5; // s89 sub
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r4, 16); // s90 shfl
|
||||
r5 = r5 ^ r1; // s91 xor
|
||||
r5 = rotl_imm(r5, 12u); // s92 rotl
|
||||
r6 = r1 * r2 + r6; // s93 mad
|
||||
r2 = rotr_var(r2, r5); // s94 rotr
|
||||
r0 = r0 | r4; // s95 or
|
||||
r7 = r7 * r3; // s96 mul
|
||||
r5 = r5 | r0; // s97 or
|
||||
r3 = r3 + r1 + ((((sel >> 13u) & 1u) != 0u) ? 0x386bb642u : 0xbcfdd8a7u); // s98 add
|
||||
r0 = r0 + r6 + ((((sel >> 9u) & 1u) != 0u) ? 0xa3962d03u : 0xc921a021u); // s99 add
|
||||
r1 = rotr_var(r1, r0); // s100 rotr
|
||||
r2 = r2 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0xc5990c5cu : 0x70da4067u); // s101 add
|
||||
r2 = rotl_imm(r2, 3u); // s102 rotl
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r4, 4); // s103 shfl
|
||||
r0 = r0 ^ r2; // s104 xor
|
||||
r4 = r4 + r3 + ((((sel >> 28u) & 1u) != 0u) ? 0xb15aec75u : 0x1a5880cbu); // s105 add
|
||||
r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r0, 2); // s106 shfl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r0, 4); // s107 shfl
|
||||
r1 = r7 * r3 + r1; // s108 mad
|
||||
r0 = r2 * r3 + r0; // s109 mad
|
||||
r2 = r2 | r4; // s110 or
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r1, 1); // s111 shfl
|
||||
r2 = r5 * r3 + r2; // s112 mad
|
||||
r4 = __umulhi(r4, r0); // s113 mulhi
|
||||
r7 = rotl_imm(r7, 29u); // s114 rotl
|
||||
r3 = r3 - r7; // s115 sub
|
||||
r0 = __umulhi(r0, r1); // s116 mulhi
|
||||
r4 = rotl_imm(r4, 12u); // s117 rotl
|
||||
r1 = r1 + r4 + ((((sel >> 7u) & 1u) != 0u) ? 0xc5e98ee5u : 0x24494ad3u); // s118 add
|
||||
r0 = r0 + r4 + ((((sel >> 4u) & 1u) != 0u) ? 0xa97bc949u : 0x1b2d1c83u); // s119 add
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r4, 4); // s120 shfl
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r2, 8); // s121 shfl
|
||||
r7 = rotl_imm(r7, 3u); // s122 rotl
|
||||
r4 = rotr_var(r4, r6); // s123 rotr
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r5, 1); // s124 shfl
|
||||
r5 = rotl_imm(r5, 21u); // s125 rotl
|
||||
r0 = r3 * r1 + r0; // s126 mad
|
||||
r0 = r0 + r3 + ((((sel >> 24u) & 1u) != 0u) ? 0xa7f6bb54u : 0xff14c08du); // s127 add
|
||||
r7 = r7 ^ r4; // s128 xor
|
||||
r3 = __umulhi(r3, r5); // s129 mulhi
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r7, 8); // s130 shfl
|
||||
r6 = __umulhi(r6, r4); // s131 mulhi
|
||||
r5 = r5 ^ r3; // s132 xor
|
||||
r0 = rotl_imm(r0, 19u); // s133 rotl
|
||||
r4 = rotr_var(r4, r3); // s134 rotr
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r2, 16); // s135 shfl
|
||||
r5 = r5 - r2; // s136 sub
|
||||
r3 = r3 ^ r5; // s137 xor
|
||||
r0 = r0 - r2; // s138 sub
|
||||
r3 = r3 * r7; // s139 mul
|
||||
r5 = r5 + r3 + ((((sel >> 24u) & 1u) != 0u) ? 0x2d465ca8u : 0x82ab98f4u); // s140 add
|
||||
r4 = r4 + r1 + ((((sel >> 11u) & 1u) != 0u) ? 0x237a9d8eu : 0x80594390u); // s141 add
|
||||
r3 = r7 * r0 + r3; // s142 mad
|
||||
r0 = r0 | r3; // s143 or
|
||||
r0 = __umulhi(r0, r1); // s144 mulhi
|
||||
r6 = r6 ^ r2; // s145 xor
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r1, 16); // s146 shfl
|
||||
r0 = r0 + r1 + ((((sel >> 18u) & 1u) != 0u) ? 0xe68cf57eu : 0x07f15148u); // s147 add
|
||||
r7 = r7 | r3; // s148 or
|
||||
r4 = r4 ^ r2; // s149 xor
|
||||
r7 = r7 ^ r1; // s150 xor
|
||||
r5 = rotr_var(r5, r2); // s151 rotr
|
||||
r0 = r0 * r7; // s152 mul
|
||||
r4 = rotl_imm(r4, 16u); // s153 rotl
|
||||
r6 = rotr_var(r6, r4); // s154 rotr
|
||||
r6 = r6 ^ r1; // s155 xor
|
||||
r6 = r6 | r3; // s156 or
|
||||
r1 = r1 + r6 + ((((sel >> 19u) & 1u) != 0u) ? 0x84c2a09du : 0x14878c5au); // s157 add
|
||||
r4 = r4 - r1; // s158 sub
|
||||
r4 = r6 * r7 + r4; // s159 mad
|
||||
r1 = r1 * r5; // s160 mul
|
||||
r4 = r4 | r0; // s161 or
|
||||
r7 = rotl_imm(r7, 21u); // s162 rotl
|
||||
r0 = r0 - r1; // s163 sub
|
||||
r1 = r1 * r0; // s164 mul
|
||||
r2 = r2 - r1; // s165 sub
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 2); // s166 shfl
|
||||
r0 = r0 - r4; // s167 sub
|
||||
r1 = rotr_var(r1, r6); // s168 rotr
|
||||
r7 = rotr_var(r7, r1); // s169 rotr
|
||||
r3 = r1 * r3 + r3; // s170 mad
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r0, 4); // s171 shfl
|
||||
r2 = r2 - r1; // s172 sub
|
||||
r7 = r7 ^ r6; // s173 xor
|
||||
r4 = rotr_var(r4, r1); // s174 rotr
|
||||
r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r7, 2); // s175 shfl
|
||||
r6 = r6 * r2; // s176 mul
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // s177 shfl
|
||||
r6 = r6 + r4 + ((((sel >> 21u) & 1u) != 0u) ? 0x509871e6u : 0x4fa43965u); // s178 add
|
||||
r0 = r0 ^ r6; // s179 xor
|
||||
r5 = r5 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0xb95b8a53u : 0xfb3c4bd8u); // s180 add
|
||||
r6 = r6 * r3; // s181 mul
|
||||
r7 = rotl_imm(r7, 1u); // s182 rotl
|
||||
r0 = r0 + r7 + ((((sel >> 11u) & 1u) != 0u) ? 0x7047b7cfu : 0xc448a197u); // s183 add
|
||||
r3 = r3 * r5; // s184 mul
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r7, 1); // s185 shfl
|
||||
r0 = r0 * r2; // s186 mul
|
||||
r4 = __umulhi(r4, r7); // s187 mulhi
|
||||
r2 = r7 * r2 + r2; // s188 mad
|
||||
r7 = r7 - r2; // s189 sub
|
||||
r2 = r2 + r5 + ((((sel >> 19u) & 1u) != 0u) ? 0x5caccc5du : 0x2fceaf49u); // s190 add
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r2, 1); // s191 shfl
|
||||
r7 = r7 + r1 + ((((sel >> 20u) & 1u) != 0u) ? 0x857bb5feu : 0xba0cee62u); // s192 add
|
||||
r6 = rotl_imm(r6, 22u); // s193 rotl
|
||||
r3 = r3 ^ r7; // s194 xor
|
||||
r7 = r7 * r0; // s195 mul
|
||||
r3 = r3 ^ r5; // s196 xor
|
||||
r5 = r5 + r1 + ((((sel >> 29u) & 1u) != 0u) ? 0xeae84577u : 0x8519428cu); // s197 add
|
||||
r7 = r4 * r4 + r7; // s198 mad
|
||||
r3 = r3 + r6 + ((((sel >> 8u) & 1u) != 0u) ? 0x0bfdbfa1u : 0x468639d3u); // s199 add
|
||||
r5 = r3 * r4 + r5; // s200 mad
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r7, 4); // s201 shfl
|
||||
r7 = r7 + r4 + ((((sel >> 23u) & 1u) != 0u) ? 0x53ee8e50u : 0x18ec9a69u); // s202 add
|
||||
r6 = r6 ^ r0; // s203 xor
|
||||
r4 = r4 + r5 + ((((sel >> 4u) & 1u) != 0u) ? 0xac63376eu : 0x3e81485bu); // s204 add
|
||||
r7 = r1 * r4 + r7; // s205 mad
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 16); // s206 shfl
|
||||
r4 = r4 * r0; // s207 mul
|
||||
r1 = r1 | r6; // s208 or
|
||||
r6 = r6 + r4 + ((((sel >> 28u) & 1u) != 0u) ? 0x5b745519u : 0x66ccb75du); // s209 add
|
||||
r1 = rotr_var(r1, r3); // s210 rotr
|
||||
r5 = r5 - r4; // s211 sub
|
||||
r4 = r4 ^ r7; // s212 xor
|
||||
r1 = r1 + r7 + ((((sel >> 8u) & 1u) != 0u) ? 0x1c3dccf7u : 0x0ce0553du); // s213 add
|
||||
r0 = __umulhi(r0, r1); // s214 mulhi
|
||||
r3 = r3 + r6 + ((((sel >> 9u) & 1u) != 0u) ? 0xfadc9205u : 0x96bfec89u); // s215 add
|
||||
r0 = rotl_imm(r0, 4u); // s216 rotl
|
||||
r6 = r6 ^ r4; // s217 xor
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r7, 2); // s218 shfl
|
||||
r7 = r0 * r1 + r7; // s219 mad
|
||||
r4 = r4 * r0; // s220 mul
|
||||
r2 = r6 * r0 + r2; // s221 mad
|
||||
r5 = r5 ^ r4; // s222 xor
|
||||
r0 = r0 * r5; // s223 mul
|
||||
r2 = r2 + r0 + ((((sel >> 9u) & 1u) != 0u) ? 0x66b15c04u : 0x61495681u); // s224 add
|
||||
r3 = rotr_var(r3, r2); // s225 rotr
|
||||
r2 = r3 * r3 + r2; // s226 mad
|
||||
r6 = r6 ^ r0; // s227 xor
|
||||
r4 = rotl_imm(r4, 30u); // s228 rotl
|
||||
r2 = r2 + r3 + ((((sel >> 29u) & 1u) != 0u) ? 0x25bc17c2u : 0x70d05c34u); // s229 add
|
||||
r1 = rotr_var(r1, r5); // s230 rotr
|
||||
r1 = r1 + r4 + ((((sel >> 8u) & 1u) != 0u) ? 0x2d8728f3u : 0x32a28384u); // s231 add
|
||||
r1 = rotl_imm(r1, 6u); // s232 rotl
|
||||
r2 = r2 + r0 + ((((sel >> 21u) & 1u) != 0u) ? 0x61e81fe6u : 0xcf0949f1u); // s233 add
|
||||
r4 = rotl_imm(r4, 2u); // s234 rotl
|
||||
r2 = r2 - r6; // s235 sub
|
||||
r6 = r6 | r3; // s236 or
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r5, 4); // s237 shfl
|
||||
r1 = r1 ^ r3; // s238 xor
|
||||
r2 = rotl_imm(r2, 1u); // s239 rotl
|
||||
r5 = rotr_var(r5, r2); // s240 rotr
|
||||
r4 = rotl_imm(r4, 10u); // s241 rotl
|
||||
r6 = r6 + r2 + ((((sel >> 16u) & 1u) != 0u) ? 0x0a67565du : 0xe204fb50u); // s242 add
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r3, 2); // s243 shfl
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r1, 16); // s244 shfl
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r7, 2); // s245 shfl
|
||||
r4 = r4 ^ r5; // s246 xor
|
||||
r7 = r7 ^ r2; // s247 xor
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 16); // s248 shfl
|
||||
r1 = __umulhi(r1, r3); // s249 mulhi
|
||||
r7 = r7 + r5 + ((((sel >> 31u) & 1u) != 0u) ? 0x5a79fd6fu : 0x3d6daffdu); // s250 add
|
||||
r6 = r6 + r0 + ((((sel >> 28u) & 1u) != 0u) ? 0x6beb28c2u : 0x84334aeau); // s251 add
|
||||
r7 = r7 - r1; // s252 sub
|
||||
r5 = rotr_var(r5, r2); // s253 rotr
|
||||
r2 = __umulhi(r2, r5); // s254 mulhi
|
||||
r6 = r6 * r3; // s255 mul
|
||||
}
|
||||
}
|
||||
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
|
||||
}
|
||||
|
||||
// Host-side launch wrappers. Declared in program.h, called from host.cu.
|
||||
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) {
|
||||
if (nSegments == 0u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 256u;
|
||||
uint32_t grid = (nSegments + block - 1u) / block;
|
||||
igneum_cache_fill<<<grid, block>>>(cache, nSegments);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems) {
|
||||
if (nItems == 0u || nLeaves == 0u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 256u;
|
||||
uint32_t grid = (nItems + block - 1u) / block;
|
||||
igneum_build<<<grid, block>>>(ds, cache, leaves, nLeaves, nItems);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
uint32_t nonces, uint32_t blockWarps) {
|
||||
if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 32u * blockWarps;
|
||||
if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;
|
||||
igneum_hash<<<nonces / block, block>>>(ds, out, baseNonce, mask);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
|
||||
cudaFuncAttributes attr;
|
||||
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash);
|
||||
if (e != cudaSuccess) return e;
|
||||
*numRegs = attr.numRegs;
|
||||
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0);
|
||||
}
|
||||
898
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/kernel_bound.cl
Normal file
898
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/kernel_bound.cl
Normal file
|
|
@ -0,0 +1,898 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925/day/69676e65756d2d6461792ffd50000000000000". Do not edit by hand.
|
||||
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
|
||||
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
|
||||
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
|
||||
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
|
||||
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
|
||||
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
|
||||
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
|
||||
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
|
||||
#ifndef IGNEUM_GROUP
|
||||
#define IGNEUM_GROUP 32
|
||||
#endif
|
||||
#ifndef IGNEUM_EXCHANGE
|
||||
#define IGNEUM_EXCHANGE 0
|
||||
#endif
|
||||
#ifdef __OPENCL_VERSION__
|
||||
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
|
||||
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#ifdef cl_khr_subgroups
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
|
||||
#endif
|
||||
#ifdef cl_khr_subgroup_shuffle
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
|
||||
#endif
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
|
||||
#endif
|
||||
#else
|
||||
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
|
||||
#include "emu_opencl.h"
|
||||
#endif
|
||||
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#else
|
||||
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
|
||||
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
|
||||
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
|
||||
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
|
||||
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
|
||||
#endif
|
||||
|
||||
static inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
|
||||
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
|
||||
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
|
||||
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
|
||||
static inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of 8 x seed-parameterised mixer + one 64-byte cache read, then 8 x final mixer (class v3, mixer multiplier 8,
|
||||
// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
static inline void mh_chacha_block(const uint* x, uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
static inline void mh_cache_segment(__global uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0xe23f9008u ^ prev[4];
|
||||
x[5] = 0xde33e763u ^ prev[5];
|
||||
x[6] = 0xc5ba415du ^ prev[6];
|
||||
x[7] = 0x8ddf6786u ^ prev[7];
|
||||
x[8] = 0x59f4a4beu ^ prev[8];
|
||||
x[9] = 0x8a3bc680u ^ prev[9];
|
||||
x[10] = 0x701b8e40u ^ prev[10];
|
||||
x[11] = 0x3b59025au ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
static inline void mh_mixer(uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xe5bef5a3u + rk)) * 0xd1d79e7fu;
|
||||
s[1] = (s[1] ^ (0xdb2a7d90u + rk)) * 0xbcb61f8bu;
|
||||
s[2] = (s[2] ^ (0xcd1fa7e1u + rk)) * 0x970e91abu;
|
||||
s[3] = (s[3] ^ (0x30289419u + rk)) * 0xd749d96du;
|
||||
s[4] = (s[4] ^ (0x0a730d58u + rk)) * 0x143d7339u;
|
||||
s[5] = (s[5] ^ (0x432f8579u + rk)) * 0x2cde0d69u;
|
||||
s[6] = (s[6] ^ (0x6ab978a5u + rk)) * 0x12d0a8c1u;
|
||||
s[7] = (s[7] ^ (0x8c984f49u + rk)) * 0x2d335be9u;
|
||||
s[8] = (s[8] ^ (0x788c3c9eu + rk)) * 0x80a8aae9u;
|
||||
s[9] = (s[9] ^ (0x051eef02u + rk)) * 0x7ac896c7u;
|
||||
s[10] = (s[10] ^ (0x05e77db9u + rk)) * 0x9de23db7u;
|
||||
s[11] = (s[11] ^ (0x8b3cd4f3u + rk)) * 0xc362827du;
|
||||
s[12] = (s[12] ^ (0x6948d6cfu + rk)) * 0x5f4cdb5bu;
|
||||
s[13] = (s[13] ^ (0x0d8677f6u + rk)) * 0xfc6c5097u;
|
||||
s[14] = (s[14] ^ (0xf9505c8cu + rk)) * 0x6f547f83u;
|
||||
s[15] = (s[15] ^ (0x4513c163u + rk)) * 0x1ac31b47u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 28u, 11u, 18u, 18u) MH_QR(s[1], s[5], s[9], s[13], 28u, 11u, 18u, 18u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 28u, 11u, 18u, 18u) MH_QR(s[3], s[7], s[11], s[15], 28u, 11u, 18u, 18u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 13u, 28u, 5u, 13u) MH_QR(s[1], s[6], s[11], s[12], 13u, 28u, 5u, 13u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 13u, 28u, 5u, 13u) MH_QR(s[3], s[4], s[9], s[14], 13u, 28u, 5u, 13u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of 8 x mixer + cache line s[0] & mask; 8 x final mixer.
|
||||
static inline void mh_item(__global const uint* cache, __global const uint* leaf, uint t, uint* s) {
|
||||
s[0] = 0xe23f9008u;
|
||||
s[1] = 0xde33e763u;
|
||||
s[2] = 0xc5ba415du;
|
||||
s[3] = 0x8ddf6786u;
|
||||
s[4] = 0x59f4a4beu;
|
||||
s[5] = 0x8a3bc680u;
|
||||
s[6] = 0x701b8e40u;
|
||||
s[7] = 0x3b59025au;
|
||||
s[8] = t * 0xd1d79e7fu + 0xe5bef5a3u;
|
||||
s[9] = t * 0xbcb61f8bu + 0xdb2a7d90u;
|
||||
s[10] = t * 0x970e91abu + 0xcd1fa7e1u;
|
||||
s[11] = t * 0xd749d96du + 0x30289419u;
|
||||
s[12] = t * 0x143d7339u + 0x0a730d58u;
|
||||
s[13] = t * 0x2cde0d69u + 0x432f8579u;
|
||||
s[14] = t * 0x12d0a8c1u + 0x6ab978a5u;
|
||||
s[15] = t * 0x2d335be9u + 0x8c984f49u;
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= leaf[i];
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
for (uint j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 8u + j + 1u));
|
||||
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
for (uint j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (64u + j + 1u));
|
||||
}
|
||||
// Class v5 (docs/design/class-v5-stored-state.md): leaf(t) = leaves[t mod nLeaves], 16 words per leaf (leaves.bin).
|
||||
static inline __global const uint* mh_leaf(__global const uint* leaves, uint nLeaves, uint t) { return leaves + ((t % nLeaves) * 16u); }
|
||||
// Era layout (docs/plans/era-layout.md 1.2): dataset word w holds word mh_j(w) of item mh_t(w); j's bits sit at positions 1 9 13 14 of w.
|
||||
static inline uint mh_j(uint w) { return ((w >> 1u) & 1u) | (((w >> 9u) & 1u) << 1) | (((w >> 13u) & 1u) << 2) | (((w >> 14u) & 1u) << 3); }
|
||||
static inline uint mh_t(uint w) { w = (w & 0x00003fffu) | ((w >> 15u) << 14u); w = (w & 0x00001fffu) | ((w >> 14u) << 13u); w = (w & 0x000001ffu) | ((w >> 10u) << 9u); w = (w & 0x00000001u) | ((w >> 2u) << 1u); return w; }
|
||||
static inline uint mh_addr(uint t, uint j) { uint w = t; w = ((w >> 1u) << 2u) | (w & 0x00000001u) | (((j >> 0u) & 1u) << 1u); w = ((w >> 9u) << 10u) | (w & 0x000001ffu) | (((j >> 1u) & 1u) << 9u); w = ((w >> 13u) << 14u) | (w & 0x00001fffu) | (((j >> 2u) & 1u) << 13u); w = ((w >> 14u) << 15u) | (w & 0x00003fffu) | (((j >> 3u) & 1u) << 14u); return w; }
|
||||
// dataset[w] without the dataset: derive item mh_t(w) and take word mh_j(w).
|
||||
static inline uint mh_word(__global const uint* cache, __global const uint* leaves, uint nLeaves, uint w) { uint s[16]; mh_item(cache, mh_leaf(leaves, nLeaves, mh_t(w)), mh_t(w), s); return s[mh_j(w)]; }
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
|
||||
// The same constants as memhard.h in this pack (one emitter, three dialects).
|
||||
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
|
||||
uint seg = (uint)get_global_id(0);
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
// Class v5: the window's leaves (leaves.bin, IGNEUM_STATE_LEAVES x 16 words) and their count.
|
||||
__kernel void igneum_build(__global uint* ds, __global const uint* cache, __global const uint* leaves, uint nLeaves, uint nItems) {
|
||||
uint t = (uint)get_global_id(0);
|
||||
if (t < nItems) {
|
||||
uint s[16];
|
||||
mh_item(cache, mh_leaf(leaves, nLeaves, t), t, s);
|
||||
for (uint i = 0u; i < 16u; ++i) ds[(ulong)mh_addr(t, i)] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask) {
|
||||
uint gid = (uint)get_global_id(0);
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ 0xbe8c5a0cu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x7646e626u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint x = nonce ^ 0x7646e626u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0x508e29d0u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint x = nonce ^ 0x508e29d0u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0xfb8a5b4au; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint x = nonce ^ 0xfb8a5b4au; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0x53c50955u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint x = nonce ^ 0x53c50955u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0x685d62fcu; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint x = nonce ^ 0x685d62fcu; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x6065c013u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint x = nonce ^ 0x6065c013u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x881096ebu; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint x = nonce ^ 0x881096ebu; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0xbe8c5a0cu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r6 = r6 - r3; // 0 sub
|
||||
r2 = r2 | r6; // 1 or
|
||||
r4 = r4 + r0 + ((((sel >> 4u) & 1u) != 0u) ? 0x7731324bu : 0x1a579b38u); // 2 add
|
||||
r3 = r3 ^ ds[((rotl_imm(r6 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 3 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r0 = r0 ^ t_; } // 4 shfl
|
||||
r4 = r4 ^ r1; // 5 xor
|
||||
r3 = r5 * r3 + r3; // 6 mad
|
||||
r1 = r1 * r6; // 7 mul
|
||||
r2 = r2 ^ ds[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x03ffffffu) | 0x08000000u) & mask]; // 8 load
|
||||
r0 = r0 ^ r4; // 9 xor
|
||||
r0 = r0 + r7 + ((((sel >> 30u) & 1u) != 0u) ? 0xb10fcef8u : 0xee083919u); // 10 add
|
||||
r0 = mul_hi(r0, r2); // 11 mulhi
|
||||
r5 = r5 ^ r4; // 12 xor
|
||||
r7 = r3 * r0 + r7; // 13 mad
|
||||
r3 = r3 ^ ds[((rotl_imm(r4 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x00000000u) & mask]; // 14 load
|
||||
r2 = r2 ^ ds[((rotl_imm(r5 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 15 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r7 = r7 ^ t_; } // 16 shfl
|
||||
r7 = r7 * r6; // 17 mul
|
||||
r2 = r3 * r1 + r2; // 18 mad
|
||||
r5 = rotr_var(r5, r2); // 19 rotr
|
||||
r5 = r5 ^ ds[((rotl_imm(r3 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 20 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r6 = r6 ^ t_; } // 21 shfl
|
||||
r1 = mul_hi(r1, r2); // 22 mulhi
|
||||
r6 = r6 ^ r5; // 23 xor
|
||||
r5 = r2 * r0 + r5; // 24 mad
|
||||
r6 = r3 * r7 + r6; // 25 mad
|
||||
r6 = r6 ^ ds[((rotl_imm(r5 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 26 load
|
||||
r1 = r5 * r5 + r1; // 27 mad
|
||||
r0 = r0 ^ ds[((rotl_imm(r6 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 28 load
|
||||
r7 = r7 + r4 + ((((sel >> 7u) & 1u) != 0u) ? 0x0b1f02c7u : 0xd5e6c37cu); // 29 add
|
||||
r7 = r7 + r0 + ((((sel >> 0u) & 1u) != 0u) ? 0x2d6e8bd1u : 0x4e45ba51u); // 30 add
|
||||
r0 = r0 ^ r2; // 31 xor
|
||||
r6 = r6 + r2 + ((((sel >> 13u) & 1u) != 0u) ? 0x351422d2u : 0x1c91b2a1u); // 32 add
|
||||
r4 = mul_hi(r4, r5); // 33 mulhi
|
||||
r3 = rotr_var(r3, r5); // 34 rotr
|
||||
r3 = r3 ^ ds[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 35 load
|
||||
r4 = rotl_imm(r4, 13u); // 36 rotl
|
||||
r6 = r6 + r1 + ((((sel >> 7u) & 1u) != 0u) ? 0x64a28251u : 0x3f2970f7u); // 37 add
|
||||
r2 = r2 * r0; // 38 mul
|
||||
r3 = r3 + r4 + ((((sel >> 15u) & 1u) != 0u) ? 0x7378c955u : 0x3b7b3317u); // 39 add
|
||||
r0 = r0 ^ ds[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x03ffffffu) | 0x0c000000u) & mask]; // 40 load
|
||||
r6 = r6 ^ r5; // 41 xor
|
||||
r3 = r3 + r0 + ((((sel >> 2u) & 1u) != 0u) ? 0xb31f6a64u : 0x6678b059u); // 42 add
|
||||
r7 = r7 ^ ds[((rotl_imm(r3 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x00000000u) & mask]; // 43 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 4u); r3 = r3 ^ t_; } // 44 shfl
|
||||
r1 = mul_hi(r1, r5); // 45 mulhi
|
||||
r0 = rotl_imm(r0, 6u); // 46 rotl
|
||||
r1 = r1 ^ ds[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 47 load
|
||||
r4 = r4 + r0 + ((((sel >> 17u) & 1u) != 0u) ? 0x5feccee6u : 0x43095946u); // 48 add
|
||||
r2 = r2 ^ ds[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 49 load
|
||||
r2 = r2 - r3; // 50 sub
|
||||
r7 = r7 + r2 + ((((sel >> 24u) & 1u) != 0u) ? 0x26e2b582u : 0xf45ecdf8u); // 51 add
|
||||
r5 = r5 ^ ds[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x03ffffffu) | 0x0c000000u) & mask]; // 52 load
|
||||
r6 = r6 ^ ds[((rotl_imm(r3 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 53 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r7 = r7 ^ t_; } // 54 shfl
|
||||
r1 = rotr_var(r1, r2); // 55 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r0 = r0 ^ t_; } // 56 shfl
|
||||
r5 = rotl_imm(r5, 12u); // 57 rotl
|
||||
r5 = r5 + r3 + ((((sel >> 4u) & 1u) != 0u) ? 0x473b1718u : 0x6b4f35e8u); // 58 add
|
||||
r4 = rotr_var(r4, r1); // 59 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 1u); r4 = r4 ^ t_; } // 60 shfl
|
||||
r7 = r7 ^ ds[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 61 load
|
||||
r5 = r5 ^ ds[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 62 load
|
||||
r4 = r4 * r1; // 63 mul
|
||||
// latency-shadow block (Counter ASIC 3.0 item 8): 256 ALU instructions x 27 passes after instruction 63, no load
|
||||
for (uint sh = 0u; sh < 27u; ++sh) {
|
||||
r6 = r6 | r0; // s0 or
|
||||
r7 = r7 * r3; // s1 mul
|
||||
r3 = r5 * r5 + r3; // s2 mad
|
||||
r7 = r7 + r4 + ((((sel >> 6u) & 1u) != 0u) ? 0x38e58a06u : 0x97d3d105u); // s3 add
|
||||
r6 = r6 + r0 + ((((sel >> 20u) & 1u) != 0u) ? 0x9629673du : 0x29b87b9au); // s4 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 1u); r6 = r6 ^ t_; } // s5 shfl
|
||||
r5 = r3 * r5 + r5; // s6 mad
|
||||
r3 = rotl_imm(r3, 5u); // s7 rotl
|
||||
r0 = r0 + r7 + ((((sel >> 12u) & 1u) != 0u) ? 0x4a5f55e1u : 0xa7c1fb0fu); // s8 add
|
||||
r5 = r5 * r7; // s9 mul
|
||||
r4 = r4 ^ r5; // s10 xor
|
||||
r6 = mul_hi(r6, r4); // s11 mulhi
|
||||
r5 = rotl_imm(r5, 13u); // s12 rotl
|
||||
r0 = r2 * r1 + r0; // s13 mad
|
||||
r2 = mul_hi(r2, r4); // s14 mulhi
|
||||
r3 = r3 | r1; // s15 or
|
||||
r0 = r0 - r5; // s16 sub
|
||||
r0 = r0 * r6; // s17 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 4u); r5 = r5 ^ t_; } // s18 shfl
|
||||
r3 = r3 | r4; // s19 or
|
||||
r6 = r3 * r2 + r6; // s20 mad
|
||||
r0 = rotl_imm(r0, 11u); // s21 rotl
|
||||
r3 = rotl_imm(r3, 25u); // s22 rotl
|
||||
r6 = r6 ^ r4; // s23 xor
|
||||
r4 = r4 | r2; // s24 or
|
||||
r1 = r1 + r6 + ((((sel >> 7u) & 1u) != 0u) ? 0xfc2f6d17u : 0x53e74799u); // s25 add
|
||||
r7 = rotr_var(r7, r6); // s26 rotr
|
||||
r7 = r7 + r5 + ((((sel >> 27u) & 1u) != 0u) ? 0x787048dcu : 0xb8fae90eu); // s27 add
|
||||
r4 = r4 ^ r1; // s28 xor
|
||||
r6 = r6 ^ r1; // s29 xor
|
||||
r2 = mul_hi(r2, r0); // s30 mulhi
|
||||
r1 = rotr_var(r1, r4); // s31 rotr
|
||||
r7 = r7 ^ r4; // s32 xor
|
||||
r3 = r2 * r6 + r3; // s33 mad
|
||||
r3 = r3 + r1 + ((((sel >> 12u) & 1u) != 0u) ? 0x9ffd510bu : 0xd2dc42ebu); // s34 add
|
||||
r4 = r4 + r7 + ((((sel >> 7u) & 1u) != 0u) ? 0xf871ba37u : 0xad8eda6fu); // s35 add
|
||||
r7 = r7 + r4 + ((((sel >> 3u) & 1u) != 0u) ? 0xd02d30cau : 0xb0cef8f4u); // s36 add
|
||||
r4 = r3 * r2 + r4; // s37 mad
|
||||
r5 = r7 * r7 + r5; // s38 mad
|
||||
r3 = r3 + r6 + ((((sel >> 26u) & 1u) != 0u) ? 0x82e98a22u : 0x2bd506e2u); // s39 add
|
||||
r0 = rotl_imm(r0, 21u); // s40 rotl
|
||||
r1 = r1 | r0; // s41 or
|
||||
r7 = r7 + r5 + ((((sel >> 28u) & 1u) != 0u) ? 0x6435f6d8u : 0xc0386e49u); // s42 add
|
||||
r1 = mul_hi(r1, r0); // s43 mulhi
|
||||
r5 = r5 + r6 + ((((sel >> 12u) & 1u) != 0u) ? 0x1f682c68u : 0x3a8921a4u); // s44 add
|
||||
r4 = r4 + r5 + ((((sel >> 1u) & 1u) != 0u) ? 0xf246e180u : 0xeeca4334u); // s45 add
|
||||
r2 = r2 ^ r3; // s46 xor
|
||||
r2 = r2 ^ r5; // s47 xor
|
||||
r7 = r4 * r1 + r7; // s48 mad
|
||||
r2 = r6 * r1 + r2; // s49 mad
|
||||
r7 = r0 * r7 + r7; // s50 mad
|
||||
r5 = r5 + r6 + ((((sel >> 26u) & 1u) != 0u) ? 0x41bd68ccu : 0x745776d8u); // s51 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 1u); r3 = r3 ^ t_; } // s52 shfl
|
||||
r4 = r4 | r0; // s53 or
|
||||
r0 = r2 * r4 + r0; // s54 mad
|
||||
r5 = rotl_imm(r5, 9u); // s55 rotl
|
||||
r5 = r5 * r7; // s56 mul
|
||||
r7 = rotl_imm(r7, 27u); // s57 rotl
|
||||
r7 = r7 - r6; // s58 sub
|
||||
r2 = rotl_imm(r2, 22u); // s59 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 2u); r7 = r7 ^ t_; } // s60 shfl
|
||||
r4 = r4 + r0 + ((((sel >> 7u) & 1u) != 0u) ? 0xcba6a161u : 0x60d0ff55u); // s61 add
|
||||
r1 = r1 * r5; // s62 mul
|
||||
r7 = mul_hi(r7, r6); // s63 mulhi
|
||||
r5 = r5 * r0; // s64 mul
|
||||
r2 = rotr_var(r2, r5); // s65 rotr
|
||||
r0 = mul_hi(r0, r3); // s66 mulhi
|
||||
r0 = r0 | r7; // s67 or
|
||||
r0 = rotl_imm(r0, 19u); // s68 rotl
|
||||
r0 = rotl_imm(r0, 1u); // s69 rotl
|
||||
r2 = r2 | r3; // s70 or
|
||||
r3 = r7 * r5 + r3; // s71 mad
|
||||
r7 = r0 * r4 + r7; // s72 mad
|
||||
r4 = r4 | r3; // s73 or
|
||||
r3 = r3 ^ r1; // s74 xor
|
||||
r1 = r1 + r3 + ((((sel >> 15u) & 1u) != 0u) ? 0x8e35924eu : 0x3f4ffea2u); // s75 add
|
||||
r1 = r1 * r3; // s76 mul
|
||||
r0 = r3 * r2 + r0; // s77 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 16u); r2 = r2 ^ t_; } // s78 shfl
|
||||
r1 = mul_hi(r1, r0); // s79 mulhi
|
||||
r0 = mul_hi(r0, r4); // s80 mulhi
|
||||
r4 = r4 * r2; // s81 mul
|
||||
r5 = r5 - r7; // s82 sub
|
||||
r0 = r2 * r3 + r0; // s83 mad
|
||||
r7 = r7 - r2; // s84 sub
|
||||
r1 = r1 ^ r2; // s85 xor
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 16u); r6 = r6 ^ t_; } // s86 shfl
|
||||
r3 = r0 * r0 + r3; // s87 mad
|
||||
r0 = rotl_imm(r0, 20u); // s88 rotl
|
||||
r1 = r1 - r5; // s89 sub
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r6 = r6 ^ t_; } // s90 shfl
|
||||
r5 = r5 ^ r1; // s91 xor
|
||||
r5 = rotl_imm(r5, 12u); // s92 rotl
|
||||
r6 = r1 * r2 + r6; // s93 mad
|
||||
r2 = rotr_var(r2, r5); // s94 rotr
|
||||
r0 = r0 | r4; // s95 or
|
||||
r7 = r7 * r3; // s96 mul
|
||||
r5 = r5 | r0; // s97 or
|
||||
r3 = r3 + r1 + ((((sel >> 13u) & 1u) != 0u) ? 0x386bb642u : 0xbcfdd8a7u); // s98 add
|
||||
r0 = r0 + r6 + ((((sel >> 9u) & 1u) != 0u) ? 0xa3962d03u : 0xc921a021u); // s99 add
|
||||
r1 = rotr_var(r1, r0); // s100 rotr
|
||||
r2 = r2 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0xc5990c5cu : 0x70da4067u); // s101 add
|
||||
r2 = rotl_imm(r2, 3u); // s102 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 4u); r1 = r1 ^ t_; } // s103 shfl
|
||||
r0 = r0 ^ r2; // s104 xor
|
||||
r4 = r4 + r3 + ((((sel >> 28u) & 1u) != 0u) ? 0xb15aec75u : 0x1a5880cbu); // s105 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 2u); r4 = r4 ^ t_; } // s106 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 4u); r7 = r7 ^ t_; } // s107 shfl
|
||||
r1 = r7 * r3 + r1; // s108 mad
|
||||
r0 = r2 * r3 + r0; // s109 mad
|
||||
r2 = r2 | r4; // s110 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 1u); r6 = r6 ^ t_; } // s111 shfl
|
||||
r2 = r5 * r3 + r2; // s112 mad
|
||||
r4 = mul_hi(r4, r0); // s113 mulhi
|
||||
r7 = rotl_imm(r7, 29u); // s114 rotl
|
||||
r3 = r3 - r7; // s115 sub
|
||||
r0 = mul_hi(r0, r1); // s116 mulhi
|
||||
r4 = rotl_imm(r4, 12u); // s117 rotl
|
||||
r1 = r1 + r4 + ((((sel >> 7u) & 1u) != 0u) ? 0xc5e98ee5u : 0x24494ad3u); // s118 add
|
||||
r0 = r0 + r4 + ((((sel >> 4u) & 1u) != 0u) ? 0xa97bc949u : 0x1b2d1c83u); // s119 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 4u); r5 = r5 ^ t_; } // s120 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r3 = r3 ^ t_; } // s121 shfl
|
||||
r7 = rotl_imm(r7, 3u); // s122 rotl
|
||||
r4 = rotr_var(r4, r6); // s123 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 1u); r6 = r6 ^ t_; } // s124 shfl
|
||||
r5 = rotl_imm(r5, 21u); // s125 rotl
|
||||
r0 = r3 * r1 + r0; // s126 mad
|
||||
r0 = r0 + r3 + ((((sel >> 24u) & 1u) != 0u) ? 0xa7f6bb54u : 0xff14c08du); // s127 add
|
||||
r7 = r7 ^ r4; // s128 xor
|
||||
r3 = mul_hi(r3, r5); // s129 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 8u); r5 = r5 ^ t_; } // s130 shfl
|
||||
r6 = mul_hi(r6, r4); // s131 mulhi
|
||||
r5 = r5 ^ r3; // s132 xor
|
||||
r0 = rotl_imm(r0, 19u); // s133 rotl
|
||||
r4 = rotr_var(r4, r3); // s134 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 16u); r1 = r1 ^ t_; } // s135 shfl
|
||||
r5 = r5 - r2; // s136 sub
|
||||
r3 = r3 ^ r5; // s137 xor
|
||||
r0 = r0 - r2; // s138 sub
|
||||
r3 = r3 * r7; // s139 mul
|
||||
r5 = r5 + r3 + ((((sel >> 24u) & 1u) != 0u) ? 0x2d465ca8u : 0x82ab98f4u); // s140 add
|
||||
r4 = r4 + r1 + ((((sel >> 11u) & 1u) != 0u) ? 0x237a9d8eu : 0x80594390u); // s141 add
|
||||
r3 = r7 * r0 + r3; // s142 mad
|
||||
r0 = r0 | r3; // s143 or
|
||||
r0 = mul_hi(r0, r1); // s144 mulhi
|
||||
r6 = r6 ^ r2; // s145 xor
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 16u); r7 = r7 ^ t_; } // s146 shfl
|
||||
r0 = r0 + r1 + ((((sel >> 18u) & 1u) != 0u) ? 0xe68cf57eu : 0x07f15148u); // s147 add
|
||||
r7 = r7 | r3; // s148 or
|
||||
r4 = r4 ^ r2; // s149 xor
|
||||
r7 = r7 ^ r1; // s150 xor
|
||||
r5 = rotr_var(r5, r2); // s151 rotr
|
||||
r0 = r0 * r7; // s152 mul
|
||||
r4 = rotl_imm(r4, 16u); // s153 rotl
|
||||
r6 = rotr_var(r6, r4); // s154 rotr
|
||||
r6 = r6 ^ r1; // s155 xor
|
||||
r6 = r6 | r3; // s156 or
|
||||
r1 = r1 + r6 + ((((sel >> 19u) & 1u) != 0u) ? 0x84c2a09du : 0x14878c5au); // s157 add
|
||||
r4 = r4 - r1; // s158 sub
|
||||
r4 = r6 * r7 + r4; // s159 mad
|
||||
r1 = r1 * r5; // s160 mul
|
||||
r4 = r4 | r0; // s161 or
|
||||
r7 = rotl_imm(r7, 21u); // s162 rotl
|
||||
r0 = r0 - r1; // s163 sub
|
||||
r1 = r1 * r0; // s164 mul
|
||||
r2 = r2 - r1; // s165 sub
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r0 = r0 ^ t_; } // s166 shfl
|
||||
r0 = r0 - r4; // s167 sub
|
||||
r1 = rotr_var(r1, r6); // s168 rotr
|
||||
r7 = rotr_var(r7, r1); // s169 rotr
|
||||
r3 = r1 * r3 + r3; // s170 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 4u); r2 = r2 ^ t_; } // s171 shfl
|
||||
r2 = r2 - r1; // s172 sub
|
||||
r7 = r7 ^ r6; // s173 xor
|
||||
r4 = rotr_var(r4, r1); // s174 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 2u); r4 = r4 ^ t_; } // s175 shfl
|
||||
r6 = r6 * r2; // s176 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r0 = r0 ^ t_; } // s177 shfl
|
||||
r6 = r6 + r4 + ((((sel >> 21u) & 1u) != 0u) ? 0x509871e6u : 0x4fa43965u); // s178 add
|
||||
r0 = r0 ^ r6; // s179 xor
|
||||
r5 = r5 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0xb95b8a53u : 0xfb3c4bd8u); // s180 add
|
||||
r6 = r6 * r3; // s181 mul
|
||||
r7 = rotl_imm(r7, 1u); // s182 rotl
|
||||
r0 = r0 + r7 + ((((sel >> 11u) & 1u) != 0u) ? 0x7047b7cfu : 0xc448a197u); // s183 add
|
||||
r3 = r3 * r5; // s184 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 1u); r5 = r5 ^ t_; } // s185 shfl
|
||||
r0 = r0 * r2; // s186 mul
|
||||
r4 = mul_hi(r4, r7); // s187 mulhi
|
||||
r2 = r7 * r2 + r2; // s188 mad
|
||||
r7 = r7 - r2; // s189 sub
|
||||
r2 = r2 + r5 + ((((sel >> 19u) & 1u) != 0u) ? 0x5caccc5du : 0x2fceaf49u); // s190 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 1u); r1 = r1 ^ t_; } // s191 shfl
|
||||
r7 = r7 + r1 + ((((sel >> 20u) & 1u) != 0u) ? 0x857bb5feu : 0xba0cee62u); // s192 add
|
||||
r6 = rotl_imm(r6, 22u); // s193 rotl
|
||||
r3 = r3 ^ r7; // s194 xor
|
||||
r7 = r7 * r0; // s195 mul
|
||||
r3 = r3 ^ r5; // s196 xor
|
||||
r5 = r5 + r1 + ((((sel >> 29u) & 1u) != 0u) ? 0xeae84577u : 0x8519428cu); // s197 add
|
||||
r7 = r4 * r4 + r7; // s198 mad
|
||||
r3 = r3 + r6 + ((((sel >> 8u) & 1u) != 0u) ? 0x0bfdbfa1u : 0x468639d3u); // s199 add
|
||||
r5 = r3 * r4 + r5; // s200 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 4u); r2 = r2 ^ t_; } // s201 shfl
|
||||
r7 = r7 + r4 + ((((sel >> 23u) & 1u) != 0u) ? 0x53ee8e50u : 0x18ec9a69u); // s202 add
|
||||
r6 = r6 ^ r0; // s203 xor
|
||||
r4 = r4 + r5 + ((((sel >> 4u) & 1u) != 0u) ? 0xac63376eu : 0x3e81485bu); // s204 add
|
||||
r7 = r1 * r4 + r7; // s205 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r1 = r1 ^ t_; } // s206 shfl
|
||||
r4 = r4 * r0; // s207 mul
|
||||
r1 = r1 | r6; // s208 or
|
||||
r6 = r6 + r4 + ((((sel >> 28u) & 1u) != 0u) ? 0x5b745519u : 0x66ccb75du); // s209 add
|
||||
r1 = rotr_var(r1, r3); // s210 rotr
|
||||
r5 = r5 - r4; // s211 sub
|
||||
r4 = r4 ^ r7; // s212 xor
|
||||
r1 = r1 + r7 + ((((sel >> 8u) & 1u) != 0u) ? 0x1c3dccf7u : 0x0ce0553du); // s213 add
|
||||
r0 = mul_hi(r0, r1); // s214 mulhi
|
||||
r3 = r3 + r6 + ((((sel >> 9u) & 1u) != 0u) ? 0xfadc9205u : 0x96bfec89u); // s215 add
|
||||
r0 = rotl_imm(r0, 4u); // s216 rotl
|
||||
r6 = r6 ^ r4; // s217 xor
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 2u); r6 = r6 ^ t_; } // s218 shfl
|
||||
r7 = r0 * r1 + r7; // s219 mad
|
||||
r4 = r4 * r0; // s220 mul
|
||||
r2 = r6 * r0 + r2; // s221 mad
|
||||
r5 = r5 ^ r4; // s222 xor
|
||||
r0 = r0 * r5; // s223 mul
|
||||
r2 = r2 + r0 + ((((sel >> 9u) & 1u) != 0u) ? 0x66b15c04u : 0x61495681u); // s224 add
|
||||
r3 = rotr_var(r3, r2); // s225 rotr
|
||||
r2 = r3 * r3 + r2; // s226 mad
|
||||
r6 = r6 ^ r0; // s227 xor
|
||||
r4 = rotl_imm(r4, 30u); // s228 rotl
|
||||
r2 = r2 + r3 + ((((sel >> 29u) & 1u) != 0u) ? 0x25bc17c2u : 0x70d05c34u); // s229 add
|
||||
r1 = rotr_var(r1, r5); // s230 rotr
|
||||
r1 = r1 + r4 + ((((sel >> 8u) & 1u) != 0u) ? 0x2d8728f3u : 0x32a28384u); // s231 add
|
||||
r1 = rotl_imm(r1, 6u); // s232 rotl
|
||||
r2 = r2 + r0 + ((((sel >> 21u) & 1u) != 0u) ? 0x61e81fe6u : 0xcf0949f1u); // s233 add
|
||||
r4 = rotl_imm(r4, 2u); // s234 rotl
|
||||
r2 = r2 - r6; // s235 sub
|
||||
r6 = r6 | r3; // s236 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r7 = r7 ^ t_; } // s237 shfl
|
||||
r1 = r1 ^ r3; // s238 xor
|
||||
r2 = rotl_imm(r2, 1u); // s239 rotl
|
||||
r5 = rotr_var(r5, r2); // s240 rotr
|
||||
r4 = rotl_imm(r4, 10u); // s241 rotl
|
||||
r6 = r6 + r2 + ((((sel >> 16u) & 1u) != 0u) ? 0x0a67565du : 0xe204fb50u); // s242 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 2u); r1 = r1 ^ t_; } // s243 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 16u); r0 = r0 ^ t_; } // s244 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 2u); r5 = r5 ^ t_; } // s245 shfl
|
||||
r4 = r4 ^ r5; // s246 xor
|
||||
r7 = r7 ^ r2; // s247 xor
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 16u); r0 = r0 ^ t_; } // s248 shfl
|
||||
r1 = mul_hi(r1, r3); // s249 mulhi
|
||||
r7 = r7 + r5 + ((((sel >> 31u) & 1u) != 0u) ? 0x5a79fd6fu : 0x3d6daffdu); // s250 add
|
||||
r6 = r6 + r0 + ((((sel >> 28u) & 1u) != 0u) ? 0x6beb28c2u : 0x84334aeau); // s251 add
|
||||
r7 = r7 - r1; // s252 sub
|
||||
r5 = rotr_var(r5, r2); // s253 rotr
|
||||
r2 = mul_hi(r2, r5); // s254 mulhi
|
||||
r6 = r6 * r3; // s255 mul
|
||||
}
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
|
||||
#if IGNEUM_EXCHANGE != 0
|
||||
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
|
||||
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
|
||||
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
|
||||
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
|
||||
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
|
||||
}
|
||||
#endif
|
||||
|
||||
// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash.
|
||||
IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw) {
|
||||
uint gid = (uint)get_global_id(0);
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; }
|
||||
{ uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; }
|
||||
{ uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; }
|
||||
{ uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; }
|
||||
{ uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; }
|
||||
{ uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; }
|
||||
{ uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; }
|
||||
{ uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r6 = r6 - r3; // 0 sub
|
||||
r2 = r2 | r6; // 1 or
|
||||
r4 = r4 + r0 + ((((sel >> 4u) & 1u) != 0u) ? 0x7731324bu : 0x1a579b38u); // 2 add
|
||||
r3 = r3 ^ ds[((rotl_imm(r6 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 3 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r0 = r0 ^ t_; } // 4 shfl
|
||||
r4 = r4 ^ r1; // 5 xor
|
||||
r3 = r5 * r3 + r3; // 6 mad
|
||||
r1 = r1 * r6; // 7 mul
|
||||
r2 = r2 ^ ds[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x03ffffffu) | 0x08000000u) & mask]; // 8 load
|
||||
r0 = r0 ^ r4; // 9 xor
|
||||
r0 = r0 + r7 + ((((sel >> 30u) & 1u) != 0u) ? 0xb10fcef8u : 0xee083919u); // 10 add
|
||||
r0 = mul_hi(r0, r2); // 11 mulhi
|
||||
r5 = r5 ^ r4; // 12 xor
|
||||
r7 = r3 * r0 + r7; // 13 mad
|
||||
r3 = r3 ^ ds[((rotl_imm(r4 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x00000000u) & mask]; // 14 load
|
||||
r2 = r2 ^ ds[((rotl_imm(r5 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 15 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r7 = r7 ^ t_; } // 16 shfl
|
||||
r7 = r7 * r6; // 17 mul
|
||||
r2 = r3 * r1 + r2; // 18 mad
|
||||
r5 = rotr_var(r5, r2); // 19 rotr
|
||||
r5 = r5 ^ ds[((rotl_imm(r3 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 20 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r6 = r6 ^ t_; } // 21 shfl
|
||||
r1 = mul_hi(r1, r2); // 22 mulhi
|
||||
r6 = r6 ^ r5; // 23 xor
|
||||
r5 = r2 * r0 + r5; // 24 mad
|
||||
r6 = r3 * r7 + r6; // 25 mad
|
||||
r6 = r6 ^ ds[((rotl_imm(r5 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 26 load
|
||||
r1 = r5 * r5 + r1; // 27 mad
|
||||
r0 = r0 ^ ds[((rotl_imm(r6 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 28 load
|
||||
r7 = r7 + r4 + ((((sel >> 7u) & 1u) != 0u) ? 0x0b1f02c7u : 0xd5e6c37cu); // 29 add
|
||||
r7 = r7 + r0 + ((((sel >> 0u) & 1u) != 0u) ? 0x2d6e8bd1u : 0x4e45ba51u); // 30 add
|
||||
r0 = r0 ^ r2; // 31 xor
|
||||
r6 = r6 + r2 + ((((sel >> 13u) & 1u) != 0u) ? 0x351422d2u : 0x1c91b2a1u); // 32 add
|
||||
r4 = mul_hi(r4, r5); // 33 mulhi
|
||||
r3 = rotr_var(r3, r5); // 34 rotr
|
||||
r3 = r3 ^ ds[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 35 load
|
||||
r4 = rotl_imm(r4, 13u); // 36 rotl
|
||||
r6 = r6 + r1 + ((((sel >> 7u) & 1u) != 0u) ? 0x64a28251u : 0x3f2970f7u); // 37 add
|
||||
r2 = r2 * r0; // 38 mul
|
||||
r3 = r3 + r4 + ((((sel >> 15u) & 1u) != 0u) ? 0x7378c955u : 0x3b7b3317u); // 39 add
|
||||
r0 = r0 ^ ds[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x03ffffffu) | 0x0c000000u) & mask]; // 40 load
|
||||
r6 = r6 ^ r5; // 41 xor
|
||||
r3 = r3 + r0 + ((((sel >> 2u) & 1u) != 0u) ? 0xb31f6a64u : 0x6678b059u); // 42 add
|
||||
r7 = r7 ^ ds[((rotl_imm(r3 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x00000000u) & mask]; // 43 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 4u); r3 = r3 ^ t_; } // 44 shfl
|
||||
r1 = mul_hi(r1, r5); // 45 mulhi
|
||||
r0 = rotl_imm(r0, 6u); // 46 rotl
|
||||
r1 = r1 ^ ds[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 47 load
|
||||
r4 = r4 + r0 + ((((sel >> 17u) & 1u) != 0u) ? 0x5feccee6u : 0x43095946u); // 48 add
|
||||
r2 = r2 ^ ds[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 49 load
|
||||
r2 = r2 - r3; // 50 sub
|
||||
r7 = r7 + r2 + ((((sel >> 24u) & 1u) != 0u) ? 0x26e2b582u : 0xf45ecdf8u); // 51 add
|
||||
r5 = r5 ^ ds[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x03ffffffu) | 0x0c000000u) & mask]; // 52 load
|
||||
r6 = r6 ^ ds[((rotl_imm(r3 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 53 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r7 = r7 ^ t_; } // 54 shfl
|
||||
r1 = rotr_var(r1, r2); // 55 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r0 = r0 ^ t_; } // 56 shfl
|
||||
r5 = rotl_imm(r5, 12u); // 57 rotl
|
||||
r5 = r5 + r3 + ((((sel >> 4u) & 1u) != 0u) ? 0x473b1718u : 0x6b4f35e8u); // 58 add
|
||||
r4 = rotr_var(r4, r1); // 59 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 1u); r4 = r4 ^ t_; } // 60 shfl
|
||||
r7 = r7 ^ ds[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 61 load
|
||||
r5 = r5 ^ ds[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 62 load
|
||||
r4 = r4 * r1; // 63 mul
|
||||
// latency-shadow block (Counter ASIC 3.0 item 8): 256 ALU instructions x 27 passes after instruction 63, no load
|
||||
for (uint sh = 0u; sh < 27u; ++sh) {
|
||||
r6 = r6 | r0; // s0 or
|
||||
r7 = r7 * r3; // s1 mul
|
||||
r3 = r5 * r5 + r3; // s2 mad
|
||||
r7 = r7 + r4 + ((((sel >> 6u) & 1u) != 0u) ? 0x38e58a06u : 0x97d3d105u); // s3 add
|
||||
r6 = r6 + r0 + ((((sel >> 20u) & 1u) != 0u) ? 0x9629673du : 0x29b87b9au); // s4 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 1u); r6 = r6 ^ t_; } // s5 shfl
|
||||
r5 = r3 * r5 + r5; // s6 mad
|
||||
r3 = rotl_imm(r3, 5u); // s7 rotl
|
||||
r0 = r0 + r7 + ((((sel >> 12u) & 1u) != 0u) ? 0x4a5f55e1u : 0xa7c1fb0fu); // s8 add
|
||||
r5 = r5 * r7; // s9 mul
|
||||
r4 = r4 ^ r5; // s10 xor
|
||||
r6 = mul_hi(r6, r4); // s11 mulhi
|
||||
r5 = rotl_imm(r5, 13u); // s12 rotl
|
||||
r0 = r2 * r1 + r0; // s13 mad
|
||||
r2 = mul_hi(r2, r4); // s14 mulhi
|
||||
r3 = r3 | r1; // s15 or
|
||||
r0 = r0 - r5; // s16 sub
|
||||
r0 = r0 * r6; // s17 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 4u); r5 = r5 ^ t_; } // s18 shfl
|
||||
r3 = r3 | r4; // s19 or
|
||||
r6 = r3 * r2 + r6; // s20 mad
|
||||
r0 = rotl_imm(r0, 11u); // s21 rotl
|
||||
r3 = rotl_imm(r3, 25u); // s22 rotl
|
||||
r6 = r6 ^ r4; // s23 xor
|
||||
r4 = r4 | r2; // s24 or
|
||||
r1 = r1 + r6 + ((((sel >> 7u) & 1u) != 0u) ? 0xfc2f6d17u : 0x53e74799u); // s25 add
|
||||
r7 = rotr_var(r7, r6); // s26 rotr
|
||||
r7 = r7 + r5 + ((((sel >> 27u) & 1u) != 0u) ? 0x787048dcu : 0xb8fae90eu); // s27 add
|
||||
r4 = r4 ^ r1; // s28 xor
|
||||
r6 = r6 ^ r1; // s29 xor
|
||||
r2 = mul_hi(r2, r0); // s30 mulhi
|
||||
r1 = rotr_var(r1, r4); // s31 rotr
|
||||
r7 = r7 ^ r4; // s32 xor
|
||||
r3 = r2 * r6 + r3; // s33 mad
|
||||
r3 = r3 + r1 + ((((sel >> 12u) & 1u) != 0u) ? 0x9ffd510bu : 0xd2dc42ebu); // s34 add
|
||||
r4 = r4 + r7 + ((((sel >> 7u) & 1u) != 0u) ? 0xf871ba37u : 0xad8eda6fu); // s35 add
|
||||
r7 = r7 + r4 + ((((sel >> 3u) & 1u) != 0u) ? 0xd02d30cau : 0xb0cef8f4u); // s36 add
|
||||
r4 = r3 * r2 + r4; // s37 mad
|
||||
r5 = r7 * r7 + r5; // s38 mad
|
||||
r3 = r3 + r6 + ((((sel >> 26u) & 1u) != 0u) ? 0x82e98a22u : 0x2bd506e2u); // s39 add
|
||||
r0 = rotl_imm(r0, 21u); // s40 rotl
|
||||
r1 = r1 | r0; // s41 or
|
||||
r7 = r7 + r5 + ((((sel >> 28u) & 1u) != 0u) ? 0x6435f6d8u : 0xc0386e49u); // s42 add
|
||||
r1 = mul_hi(r1, r0); // s43 mulhi
|
||||
r5 = r5 + r6 + ((((sel >> 12u) & 1u) != 0u) ? 0x1f682c68u : 0x3a8921a4u); // s44 add
|
||||
r4 = r4 + r5 + ((((sel >> 1u) & 1u) != 0u) ? 0xf246e180u : 0xeeca4334u); // s45 add
|
||||
r2 = r2 ^ r3; // s46 xor
|
||||
r2 = r2 ^ r5; // s47 xor
|
||||
r7 = r4 * r1 + r7; // s48 mad
|
||||
r2 = r6 * r1 + r2; // s49 mad
|
||||
r7 = r0 * r7 + r7; // s50 mad
|
||||
r5 = r5 + r6 + ((((sel >> 26u) & 1u) != 0u) ? 0x41bd68ccu : 0x745776d8u); // s51 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 1u); r3 = r3 ^ t_; } // s52 shfl
|
||||
r4 = r4 | r0; // s53 or
|
||||
r0 = r2 * r4 + r0; // s54 mad
|
||||
r5 = rotl_imm(r5, 9u); // s55 rotl
|
||||
r5 = r5 * r7; // s56 mul
|
||||
r7 = rotl_imm(r7, 27u); // s57 rotl
|
||||
r7 = r7 - r6; // s58 sub
|
||||
r2 = rotl_imm(r2, 22u); // s59 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 2u); r7 = r7 ^ t_; } // s60 shfl
|
||||
r4 = r4 + r0 + ((((sel >> 7u) & 1u) != 0u) ? 0xcba6a161u : 0x60d0ff55u); // s61 add
|
||||
r1 = r1 * r5; // s62 mul
|
||||
r7 = mul_hi(r7, r6); // s63 mulhi
|
||||
r5 = r5 * r0; // s64 mul
|
||||
r2 = rotr_var(r2, r5); // s65 rotr
|
||||
r0 = mul_hi(r0, r3); // s66 mulhi
|
||||
r0 = r0 | r7; // s67 or
|
||||
r0 = rotl_imm(r0, 19u); // s68 rotl
|
||||
r0 = rotl_imm(r0, 1u); // s69 rotl
|
||||
r2 = r2 | r3; // s70 or
|
||||
r3 = r7 * r5 + r3; // s71 mad
|
||||
r7 = r0 * r4 + r7; // s72 mad
|
||||
r4 = r4 | r3; // s73 or
|
||||
r3 = r3 ^ r1; // s74 xor
|
||||
r1 = r1 + r3 + ((((sel >> 15u) & 1u) != 0u) ? 0x8e35924eu : 0x3f4ffea2u); // s75 add
|
||||
r1 = r1 * r3; // s76 mul
|
||||
r0 = r3 * r2 + r0; // s77 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 16u); r2 = r2 ^ t_; } // s78 shfl
|
||||
r1 = mul_hi(r1, r0); // s79 mulhi
|
||||
r0 = mul_hi(r0, r4); // s80 mulhi
|
||||
r4 = r4 * r2; // s81 mul
|
||||
r5 = r5 - r7; // s82 sub
|
||||
r0 = r2 * r3 + r0; // s83 mad
|
||||
r7 = r7 - r2; // s84 sub
|
||||
r1 = r1 ^ r2; // s85 xor
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 16u); r6 = r6 ^ t_; } // s86 shfl
|
||||
r3 = r0 * r0 + r3; // s87 mad
|
||||
r0 = rotl_imm(r0, 20u); // s88 rotl
|
||||
r1 = r1 - r5; // s89 sub
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r6 = r6 ^ t_; } // s90 shfl
|
||||
r5 = r5 ^ r1; // s91 xor
|
||||
r5 = rotl_imm(r5, 12u); // s92 rotl
|
||||
r6 = r1 * r2 + r6; // s93 mad
|
||||
r2 = rotr_var(r2, r5); // s94 rotr
|
||||
r0 = r0 | r4; // s95 or
|
||||
r7 = r7 * r3; // s96 mul
|
||||
r5 = r5 | r0; // s97 or
|
||||
r3 = r3 + r1 + ((((sel >> 13u) & 1u) != 0u) ? 0x386bb642u : 0xbcfdd8a7u); // s98 add
|
||||
r0 = r0 + r6 + ((((sel >> 9u) & 1u) != 0u) ? 0xa3962d03u : 0xc921a021u); // s99 add
|
||||
r1 = rotr_var(r1, r0); // s100 rotr
|
||||
r2 = r2 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0xc5990c5cu : 0x70da4067u); // s101 add
|
||||
r2 = rotl_imm(r2, 3u); // s102 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 4u); r1 = r1 ^ t_; } // s103 shfl
|
||||
r0 = r0 ^ r2; // s104 xor
|
||||
r4 = r4 + r3 + ((((sel >> 28u) & 1u) != 0u) ? 0xb15aec75u : 0x1a5880cbu); // s105 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 2u); r4 = r4 ^ t_; } // s106 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 4u); r7 = r7 ^ t_; } // s107 shfl
|
||||
r1 = r7 * r3 + r1; // s108 mad
|
||||
r0 = r2 * r3 + r0; // s109 mad
|
||||
r2 = r2 | r4; // s110 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 1u); r6 = r6 ^ t_; } // s111 shfl
|
||||
r2 = r5 * r3 + r2; // s112 mad
|
||||
r4 = mul_hi(r4, r0); // s113 mulhi
|
||||
r7 = rotl_imm(r7, 29u); // s114 rotl
|
||||
r3 = r3 - r7; // s115 sub
|
||||
r0 = mul_hi(r0, r1); // s116 mulhi
|
||||
r4 = rotl_imm(r4, 12u); // s117 rotl
|
||||
r1 = r1 + r4 + ((((sel >> 7u) & 1u) != 0u) ? 0xc5e98ee5u : 0x24494ad3u); // s118 add
|
||||
r0 = r0 + r4 + ((((sel >> 4u) & 1u) != 0u) ? 0xa97bc949u : 0x1b2d1c83u); // s119 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 4u); r5 = r5 ^ t_; } // s120 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r3 = r3 ^ t_; } // s121 shfl
|
||||
r7 = rotl_imm(r7, 3u); // s122 rotl
|
||||
r4 = rotr_var(r4, r6); // s123 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 1u); r6 = r6 ^ t_; } // s124 shfl
|
||||
r5 = rotl_imm(r5, 21u); // s125 rotl
|
||||
r0 = r3 * r1 + r0; // s126 mad
|
||||
r0 = r0 + r3 + ((((sel >> 24u) & 1u) != 0u) ? 0xa7f6bb54u : 0xff14c08du); // s127 add
|
||||
r7 = r7 ^ r4; // s128 xor
|
||||
r3 = mul_hi(r3, r5); // s129 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 8u); r5 = r5 ^ t_; } // s130 shfl
|
||||
r6 = mul_hi(r6, r4); // s131 mulhi
|
||||
r5 = r5 ^ r3; // s132 xor
|
||||
r0 = rotl_imm(r0, 19u); // s133 rotl
|
||||
r4 = rotr_var(r4, r3); // s134 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 16u); r1 = r1 ^ t_; } // s135 shfl
|
||||
r5 = r5 - r2; // s136 sub
|
||||
r3 = r3 ^ r5; // s137 xor
|
||||
r0 = r0 - r2; // s138 sub
|
||||
r3 = r3 * r7; // s139 mul
|
||||
r5 = r5 + r3 + ((((sel >> 24u) & 1u) != 0u) ? 0x2d465ca8u : 0x82ab98f4u); // s140 add
|
||||
r4 = r4 + r1 + ((((sel >> 11u) & 1u) != 0u) ? 0x237a9d8eu : 0x80594390u); // s141 add
|
||||
r3 = r7 * r0 + r3; // s142 mad
|
||||
r0 = r0 | r3; // s143 or
|
||||
r0 = mul_hi(r0, r1); // s144 mulhi
|
||||
r6 = r6 ^ r2; // s145 xor
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 16u); r7 = r7 ^ t_; } // s146 shfl
|
||||
r0 = r0 + r1 + ((((sel >> 18u) & 1u) != 0u) ? 0xe68cf57eu : 0x07f15148u); // s147 add
|
||||
r7 = r7 | r3; // s148 or
|
||||
r4 = r4 ^ r2; // s149 xor
|
||||
r7 = r7 ^ r1; // s150 xor
|
||||
r5 = rotr_var(r5, r2); // s151 rotr
|
||||
r0 = r0 * r7; // s152 mul
|
||||
r4 = rotl_imm(r4, 16u); // s153 rotl
|
||||
r6 = rotr_var(r6, r4); // s154 rotr
|
||||
r6 = r6 ^ r1; // s155 xor
|
||||
r6 = r6 | r3; // s156 or
|
||||
r1 = r1 + r6 + ((((sel >> 19u) & 1u) != 0u) ? 0x84c2a09du : 0x14878c5au); // s157 add
|
||||
r4 = r4 - r1; // s158 sub
|
||||
r4 = r6 * r7 + r4; // s159 mad
|
||||
r1 = r1 * r5; // s160 mul
|
||||
r4 = r4 | r0; // s161 or
|
||||
r7 = rotl_imm(r7, 21u); // s162 rotl
|
||||
r0 = r0 - r1; // s163 sub
|
||||
r1 = r1 * r0; // s164 mul
|
||||
r2 = r2 - r1; // s165 sub
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r0 = r0 ^ t_; } // s166 shfl
|
||||
r0 = r0 - r4; // s167 sub
|
||||
r1 = rotr_var(r1, r6); // s168 rotr
|
||||
r7 = rotr_var(r7, r1); // s169 rotr
|
||||
r3 = r1 * r3 + r3; // s170 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 4u); r2 = r2 ^ t_; } // s171 shfl
|
||||
r2 = r2 - r1; // s172 sub
|
||||
r7 = r7 ^ r6; // s173 xor
|
||||
r4 = rotr_var(r4, r1); // s174 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 2u); r4 = r4 ^ t_; } // s175 shfl
|
||||
r6 = r6 * r2; // s176 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r0 = r0 ^ t_; } // s177 shfl
|
||||
r6 = r6 + r4 + ((((sel >> 21u) & 1u) != 0u) ? 0x509871e6u : 0x4fa43965u); // s178 add
|
||||
r0 = r0 ^ r6; // s179 xor
|
||||
r5 = r5 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0xb95b8a53u : 0xfb3c4bd8u); // s180 add
|
||||
r6 = r6 * r3; // s181 mul
|
||||
r7 = rotl_imm(r7, 1u); // s182 rotl
|
||||
r0 = r0 + r7 + ((((sel >> 11u) & 1u) != 0u) ? 0x7047b7cfu : 0xc448a197u); // s183 add
|
||||
r3 = r3 * r5; // s184 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 1u); r5 = r5 ^ t_; } // s185 shfl
|
||||
r0 = r0 * r2; // s186 mul
|
||||
r4 = mul_hi(r4, r7); // s187 mulhi
|
||||
r2 = r7 * r2 + r2; // s188 mad
|
||||
r7 = r7 - r2; // s189 sub
|
||||
r2 = r2 + r5 + ((((sel >> 19u) & 1u) != 0u) ? 0x5caccc5du : 0x2fceaf49u); // s190 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 1u); r1 = r1 ^ t_; } // s191 shfl
|
||||
r7 = r7 + r1 + ((((sel >> 20u) & 1u) != 0u) ? 0x857bb5feu : 0xba0cee62u); // s192 add
|
||||
r6 = rotl_imm(r6, 22u); // s193 rotl
|
||||
r3 = r3 ^ r7; // s194 xor
|
||||
r7 = r7 * r0; // s195 mul
|
||||
r3 = r3 ^ r5; // s196 xor
|
||||
r5 = r5 + r1 + ((((sel >> 29u) & 1u) != 0u) ? 0xeae84577u : 0x8519428cu); // s197 add
|
||||
r7 = r4 * r4 + r7; // s198 mad
|
||||
r3 = r3 + r6 + ((((sel >> 8u) & 1u) != 0u) ? 0x0bfdbfa1u : 0x468639d3u); // s199 add
|
||||
r5 = r3 * r4 + r5; // s200 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 4u); r2 = r2 ^ t_; } // s201 shfl
|
||||
r7 = r7 + r4 + ((((sel >> 23u) & 1u) != 0u) ? 0x53ee8e50u : 0x18ec9a69u); // s202 add
|
||||
r6 = r6 ^ r0; // s203 xor
|
||||
r4 = r4 + r5 + ((((sel >> 4u) & 1u) != 0u) ? 0xac63376eu : 0x3e81485bu); // s204 add
|
||||
r7 = r1 * r4 + r7; // s205 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r1 = r1 ^ t_; } // s206 shfl
|
||||
r4 = r4 * r0; // s207 mul
|
||||
r1 = r1 | r6; // s208 or
|
||||
r6 = r6 + r4 + ((((sel >> 28u) & 1u) != 0u) ? 0x5b745519u : 0x66ccb75du); // s209 add
|
||||
r1 = rotr_var(r1, r3); // s210 rotr
|
||||
r5 = r5 - r4; // s211 sub
|
||||
r4 = r4 ^ r7; // s212 xor
|
||||
r1 = r1 + r7 + ((((sel >> 8u) & 1u) != 0u) ? 0x1c3dccf7u : 0x0ce0553du); // s213 add
|
||||
r0 = mul_hi(r0, r1); // s214 mulhi
|
||||
r3 = r3 + r6 + ((((sel >> 9u) & 1u) != 0u) ? 0xfadc9205u : 0x96bfec89u); // s215 add
|
||||
r0 = rotl_imm(r0, 4u); // s216 rotl
|
||||
r6 = r6 ^ r4; // s217 xor
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 2u); r6 = r6 ^ t_; } // s218 shfl
|
||||
r7 = r0 * r1 + r7; // s219 mad
|
||||
r4 = r4 * r0; // s220 mul
|
||||
r2 = r6 * r0 + r2; // s221 mad
|
||||
r5 = r5 ^ r4; // s222 xor
|
||||
r0 = r0 * r5; // s223 mul
|
||||
r2 = r2 + r0 + ((((sel >> 9u) & 1u) != 0u) ? 0x66b15c04u : 0x61495681u); // s224 add
|
||||
r3 = rotr_var(r3, r2); // s225 rotr
|
||||
r2 = r3 * r3 + r2; // s226 mad
|
||||
r6 = r6 ^ r0; // s227 xor
|
||||
r4 = rotl_imm(r4, 30u); // s228 rotl
|
||||
r2 = r2 + r3 + ((((sel >> 29u) & 1u) != 0u) ? 0x25bc17c2u : 0x70d05c34u); // s229 add
|
||||
r1 = rotr_var(r1, r5); // s230 rotr
|
||||
r1 = r1 + r4 + ((((sel >> 8u) & 1u) != 0u) ? 0x2d8728f3u : 0x32a28384u); // s231 add
|
||||
r1 = rotl_imm(r1, 6u); // s232 rotl
|
||||
r2 = r2 + r0 + ((((sel >> 21u) & 1u) != 0u) ? 0x61e81fe6u : 0xcf0949f1u); // s233 add
|
||||
r4 = rotl_imm(r4, 2u); // s234 rotl
|
||||
r2 = r2 - r6; // s235 sub
|
||||
r6 = r6 | r3; // s236 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r7 = r7 ^ t_; } // s237 shfl
|
||||
r1 = r1 ^ r3; // s238 xor
|
||||
r2 = rotl_imm(r2, 1u); // s239 rotl
|
||||
r5 = rotr_var(r5, r2); // s240 rotr
|
||||
r4 = rotl_imm(r4, 10u); // s241 rotl
|
||||
r6 = r6 + r2 + ((((sel >> 16u) & 1u) != 0u) ? 0x0a67565du : 0xe204fb50u); // s242 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 2u); r1 = r1 ^ t_; } // s243 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 16u); r0 = r0 ^ t_; } // s244 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 2u); r5 = r5 ^ t_; } // s245 shfl
|
||||
r4 = r4 ^ r5; // s246 xor
|
||||
r7 = r7 ^ r2; // s247 xor
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 16u); r0 = r0 ^ t_; } // s248 shfl
|
||||
r1 = mul_hi(r1, r3); // s249 mulhi
|
||||
r7 = r7 + r5 + ((((sel >> 31u) & 1u) != 0u) ? 0x5a79fd6fu : 0x3d6daffdu); // s250 add
|
||||
r6 = r6 + r0 + ((((sel >> 28u) & 1u) != 0u) ? 0x6beb28c2u : 0x84334aeau); // s251 add
|
||||
r7 = r7 - r1; // s252 sub
|
||||
r5 = rotr_var(r5, r2); // s253 rotr
|
||||
r2 = mul_hi(r2, r5); // s254 mulhi
|
||||
r6 = r6 * r3; // s255 mul
|
||||
}
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
382
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/kernel_bound.cu
Normal file
382
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/kernel_bound.cu
Normal file
|
|
@ -0,0 +1,382 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925/day/69676e65756d2d6461792ffd50000000000000". Do not edit by hand.
|
||||
// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW.
|
||||
// Host declarations (also in program_bound.h if present):
|
||||
// struct IgneumInitWords { uint32_t w[8]; };
|
||||
// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps);
|
||||
// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
|
||||
#include <cuda_runtime.h>
|
||||
#include <cstdint>
|
||||
#include "program.h"
|
||||
|
||||
struct IgneumInitWords { uint32_t w[8]; };
|
||||
|
||||
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
|
||||
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
|
||||
__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw) {
|
||||
uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
uint32_t nonce = baseNonce + gid;
|
||||
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; }
|
||||
{ uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; }
|
||||
{ uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; }
|
||||
{ uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; }
|
||||
{ uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; }
|
||||
{ uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; }
|
||||
{ uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; }
|
||||
{ uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; }
|
||||
|
||||
for (uint32_t it = 0u; it < 8u; ++it) {
|
||||
uint32_t sel = r0;
|
||||
r6 = r6 - r3; // 0 sub
|
||||
r2 = r2 | r6; // 1 or
|
||||
r4 = r4 + r0 + ((((sel >> 4u) & 1u) != 0u) ? 0x7731324bu : 0x1a579b38u); // 2 add
|
||||
r3 = r3 ^ ds[((rotl_imm(r6 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 3 load
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 1); // 4 shfl
|
||||
r4 = r4 ^ r1; // 5 xor
|
||||
r3 = r5 * r3 + r3; // 6 mad
|
||||
r1 = r1 * r6; // 7 mul
|
||||
r2 = r2 ^ ds[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x03ffffffu) | 0x08000000u) & mask]; // 8 load
|
||||
r0 = r0 ^ r4; // 9 xor
|
||||
r0 = r0 + r7 + ((((sel >> 30u) & 1u) != 0u) ? 0xb10fcef8u : 0xee083919u); // 10 add
|
||||
r0 = __umulhi(r0, r2); // 11 mulhi
|
||||
r5 = r5 ^ r4; // 12 xor
|
||||
r7 = r3 * r0 + r7; // 13 mad
|
||||
r3 = r3 ^ ds[((rotl_imm(r4 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x00000000u) & mask]; // 14 load
|
||||
r2 = r2 ^ ds[((rotl_imm(r5 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 15 load
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r5, 16); // 16 shfl
|
||||
r7 = r7 * r6; // 17 mul
|
||||
r2 = r3 * r1 + r2; // 18 mad
|
||||
r5 = rotr_var(r5, r2); // 19 rotr
|
||||
r5 = r5 ^ ds[((rotl_imm(r3 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 20 load
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r4, 16); // 21 shfl
|
||||
r1 = __umulhi(r1, r2); // 22 mulhi
|
||||
r6 = r6 ^ r5; // 23 xor
|
||||
r5 = r2 * r0 + r5; // 24 mad
|
||||
r6 = r3 * r7 + r6; // 25 mad
|
||||
r6 = r6 ^ ds[((rotl_imm(r5 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 26 load
|
||||
r1 = r5 * r5 + r1; // 27 mad
|
||||
r0 = r0 ^ ds[((rotl_imm(r6 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 28 load
|
||||
r7 = r7 + r4 + ((((sel >> 7u) & 1u) != 0u) ? 0x0b1f02c7u : 0xd5e6c37cu); // 29 add
|
||||
r7 = r7 + r0 + ((((sel >> 0u) & 1u) != 0u) ? 0x2d6e8bd1u : 0x4e45ba51u); // 30 add
|
||||
r0 = r0 ^ r2; // 31 xor
|
||||
r6 = r6 + r2 + ((((sel >> 13u) & 1u) != 0u) ? 0x351422d2u : 0x1c91b2a1u); // 32 add
|
||||
r4 = __umulhi(r4, r5); // 33 mulhi
|
||||
r3 = rotr_var(r3, r5); // 34 rotr
|
||||
r3 = r3 ^ ds[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 35 load
|
||||
r4 = rotl_imm(r4, 13u); // 36 rotl
|
||||
r6 = r6 + r1 + ((((sel >> 7u) & 1u) != 0u) ? 0x64a28251u : 0x3f2970f7u); // 37 add
|
||||
r2 = r2 * r0; // 38 mul
|
||||
r3 = r3 + r4 + ((((sel >> 15u) & 1u) != 0u) ? 0x7378c955u : 0x3b7b3317u); // 39 add
|
||||
r0 = r0 ^ ds[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x03ffffffu) | 0x0c000000u) & mask]; // 40 load
|
||||
r6 = r6 ^ r5; // 41 xor
|
||||
r3 = r3 + r0 + ((((sel >> 2u) & 1u) != 0u) ? 0xb31f6a64u : 0x6678b059u); // 42 add
|
||||
r7 = r7 ^ ds[((rotl_imm(r3 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x00000000u) & mask]; // 43 load
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r7, 4); // 44 shfl
|
||||
r1 = __umulhi(r1, r5); // 45 mulhi
|
||||
r0 = rotl_imm(r0, 6u); // 46 rotl
|
||||
r1 = r1 ^ ds[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 47 load
|
||||
r4 = r4 + r0 + ((((sel >> 17u) & 1u) != 0u) ? 0x5feccee6u : 0x43095946u); // 48 add
|
||||
r2 = r2 ^ ds[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 49 load
|
||||
r2 = r2 - r3; // 50 sub
|
||||
r7 = r7 + r2 + ((((sel >> 24u) & 1u) != 0u) ? 0x26e2b582u : 0xf45ecdf8u); // 51 add
|
||||
r5 = r5 ^ ds[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x03ffffffu) | 0x0c000000u) & mask]; // 52 load
|
||||
r6 = r6 ^ ds[((rotl_imm(r3 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & mask]; // 53 load
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r5, 4); // 54 shfl
|
||||
r1 = rotr_var(r1, r2); // 55 rotr
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r5, 16); // 56 shfl
|
||||
r5 = rotl_imm(r5, 12u); // 57 rotl
|
||||
r5 = r5 + r3 + ((((sel >> 4u) & 1u) != 0u) ? 0x473b1718u : 0x6b4f35e8u); // 58 add
|
||||
r4 = rotr_var(r4, r1); // 59 rotr
|
||||
r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r5, 1); // 60 shfl
|
||||
r7 = r7 ^ ds[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 61 load
|
||||
r5 = r5 ^ ds[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & mask]; // 62 load
|
||||
r4 = r4 * r1; // 63 mul
|
||||
// latency-shadow block (Counter ASIC 3.0 item 8): 256 ALU instructions x 27 passes after instruction 63, no load
|
||||
for (uint32_t sh = 0u; sh < 27u; ++sh) {
|
||||
r6 = r6 | r0; // s0 or
|
||||
r7 = r7 * r3; // s1 mul
|
||||
r3 = r5 * r5 + r3; // s2 mad
|
||||
r7 = r7 + r4 + ((((sel >> 6u) & 1u) != 0u) ? 0x38e58a06u : 0x97d3d105u); // s3 add
|
||||
r6 = r6 + r0 + ((((sel >> 20u) & 1u) != 0u) ? 0x9629673du : 0x29b87b9au); // s4 add
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r2, 1); // s5 shfl
|
||||
r5 = r3 * r5 + r5; // s6 mad
|
||||
r3 = rotl_imm(r3, 5u); // s7 rotl
|
||||
r0 = r0 + r7 + ((((sel >> 12u) & 1u) != 0u) ? 0x4a5f55e1u : 0xa7c1fb0fu); // s8 add
|
||||
r5 = r5 * r7; // s9 mul
|
||||
r4 = r4 ^ r5; // s10 xor
|
||||
r6 = __umulhi(r6, r4); // s11 mulhi
|
||||
r5 = rotl_imm(r5, 13u); // s12 rotl
|
||||
r0 = r2 * r1 + r0; // s13 mad
|
||||
r2 = __umulhi(r2, r4); // s14 mulhi
|
||||
r3 = r3 | r1; // s15 or
|
||||
r0 = r0 - r5; // s16 sub
|
||||
r0 = r0 * r6; // s17 mul
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r1, 4); // s18 shfl
|
||||
r3 = r3 | r4; // s19 or
|
||||
r6 = r3 * r2 + r6; // s20 mad
|
||||
r0 = rotl_imm(r0, 11u); // s21 rotl
|
||||
r3 = rotl_imm(r3, 25u); // s22 rotl
|
||||
r6 = r6 ^ r4; // s23 xor
|
||||
r4 = r4 | r2; // s24 or
|
||||
r1 = r1 + r6 + ((((sel >> 7u) & 1u) != 0u) ? 0xfc2f6d17u : 0x53e74799u); // s25 add
|
||||
r7 = rotr_var(r7, r6); // s26 rotr
|
||||
r7 = r7 + r5 + ((((sel >> 27u) & 1u) != 0u) ? 0x787048dcu : 0xb8fae90eu); // s27 add
|
||||
r4 = r4 ^ r1; // s28 xor
|
||||
r6 = r6 ^ r1; // s29 xor
|
||||
r2 = __umulhi(r2, r0); // s30 mulhi
|
||||
r1 = rotr_var(r1, r4); // s31 rotr
|
||||
r7 = r7 ^ r4; // s32 xor
|
||||
r3 = r2 * r6 + r3; // s33 mad
|
||||
r3 = r3 + r1 + ((((sel >> 12u) & 1u) != 0u) ? 0x9ffd510bu : 0xd2dc42ebu); // s34 add
|
||||
r4 = r4 + r7 + ((((sel >> 7u) & 1u) != 0u) ? 0xf871ba37u : 0xad8eda6fu); // s35 add
|
||||
r7 = r7 + r4 + ((((sel >> 3u) & 1u) != 0u) ? 0xd02d30cau : 0xb0cef8f4u); // s36 add
|
||||
r4 = r3 * r2 + r4; // s37 mad
|
||||
r5 = r7 * r7 + r5; // s38 mad
|
||||
r3 = r3 + r6 + ((((sel >> 26u) & 1u) != 0u) ? 0x82e98a22u : 0x2bd506e2u); // s39 add
|
||||
r0 = rotl_imm(r0, 21u); // s40 rotl
|
||||
r1 = r1 | r0; // s41 or
|
||||
r7 = r7 + r5 + ((((sel >> 28u) & 1u) != 0u) ? 0x6435f6d8u : 0xc0386e49u); // s42 add
|
||||
r1 = __umulhi(r1, r0); // s43 mulhi
|
||||
r5 = r5 + r6 + ((((sel >> 12u) & 1u) != 0u) ? 0x1f682c68u : 0x3a8921a4u); // s44 add
|
||||
r4 = r4 + r5 + ((((sel >> 1u) & 1u) != 0u) ? 0xf246e180u : 0xeeca4334u); // s45 add
|
||||
r2 = r2 ^ r3; // s46 xor
|
||||
r2 = r2 ^ r5; // s47 xor
|
||||
r7 = r4 * r1 + r7; // s48 mad
|
||||
r2 = r6 * r1 + r2; // s49 mad
|
||||
r7 = r0 * r7 + r7; // s50 mad
|
||||
r5 = r5 + r6 + ((((sel >> 26u) & 1u) != 0u) ? 0x41bd68ccu : 0x745776d8u); // s51 add
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r0, 1); // s52 shfl
|
||||
r4 = r4 | r0; // s53 or
|
||||
r0 = r2 * r4 + r0; // s54 mad
|
||||
r5 = rotl_imm(r5, 9u); // s55 rotl
|
||||
r5 = r5 * r7; // s56 mul
|
||||
r7 = rotl_imm(r7, 27u); // s57 rotl
|
||||
r7 = r7 - r6; // s58 sub
|
||||
r2 = rotl_imm(r2, 22u); // s59 rotl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r1, 2); // s60 shfl
|
||||
r4 = r4 + r0 + ((((sel >> 7u) & 1u) != 0u) ? 0xcba6a161u : 0x60d0ff55u); // s61 add
|
||||
r1 = r1 * r5; // s62 mul
|
||||
r7 = __umulhi(r7, r6); // s63 mulhi
|
||||
r5 = r5 * r0; // s64 mul
|
||||
r2 = rotr_var(r2, r5); // s65 rotr
|
||||
r0 = __umulhi(r0, r3); // s66 mulhi
|
||||
r0 = r0 | r7; // s67 or
|
||||
r0 = rotl_imm(r0, 19u); // s68 rotl
|
||||
r0 = rotl_imm(r0, 1u); // s69 rotl
|
||||
r2 = r2 | r3; // s70 or
|
||||
r3 = r7 * r5 + r3; // s71 mad
|
||||
r7 = r0 * r4 + r7; // s72 mad
|
||||
r4 = r4 | r3; // s73 or
|
||||
r3 = r3 ^ r1; // s74 xor
|
||||
r1 = r1 + r3 + ((((sel >> 15u) & 1u) != 0u) ? 0x8e35924eu : 0x3f4ffea2u); // s75 add
|
||||
r1 = r1 * r3; // s76 mul
|
||||
r0 = r3 * r2 + r0; // s77 mad
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r0, 16); // s78 shfl
|
||||
r1 = __umulhi(r1, r0); // s79 mulhi
|
||||
r0 = __umulhi(r0, r4); // s80 mulhi
|
||||
r4 = r4 * r2; // s81 mul
|
||||
r5 = r5 - r7; // s82 sub
|
||||
r0 = r2 * r3 + r0; // s83 mad
|
||||
r7 = r7 - r2; // s84 sub
|
||||
r1 = r1 ^ r2; // s85 xor
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 16); // s86 shfl
|
||||
r3 = r0 * r0 + r3; // s87 mad
|
||||
r0 = rotl_imm(r0, 20u); // s88 rotl
|
||||
r1 = r1 - r5; // s89 sub
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r4, 16); // s90 shfl
|
||||
r5 = r5 ^ r1; // s91 xor
|
||||
r5 = rotl_imm(r5, 12u); // s92 rotl
|
||||
r6 = r1 * r2 + r6; // s93 mad
|
||||
r2 = rotr_var(r2, r5); // s94 rotr
|
||||
r0 = r0 | r4; // s95 or
|
||||
r7 = r7 * r3; // s96 mul
|
||||
r5 = r5 | r0; // s97 or
|
||||
r3 = r3 + r1 + ((((sel >> 13u) & 1u) != 0u) ? 0x386bb642u : 0xbcfdd8a7u); // s98 add
|
||||
r0 = r0 + r6 + ((((sel >> 9u) & 1u) != 0u) ? 0xa3962d03u : 0xc921a021u); // s99 add
|
||||
r1 = rotr_var(r1, r0); // s100 rotr
|
||||
r2 = r2 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0xc5990c5cu : 0x70da4067u); // s101 add
|
||||
r2 = rotl_imm(r2, 3u); // s102 rotl
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r4, 4); // s103 shfl
|
||||
r0 = r0 ^ r2; // s104 xor
|
||||
r4 = r4 + r3 + ((((sel >> 28u) & 1u) != 0u) ? 0xb15aec75u : 0x1a5880cbu); // s105 add
|
||||
r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r0, 2); // s106 shfl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r0, 4); // s107 shfl
|
||||
r1 = r7 * r3 + r1; // s108 mad
|
||||
r0 = r2 * r3 + r0; // s109 mad
|
||||
r2 = r2 | r4; // s110 or
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r1, 1); // s111 shfl
|
||||
r2 = r5 * r3 + r2; // s112 mad
|
||||
r4 = __umulhi(r4, r0); // s113 mulhi
|
||||
r7 = rotl_imm(r7, 29u); // s114 rotl
|
||||
r3 = r3 - r7; // s115 sub
|
||||
r0 = __umulhi(r0, r1); // s116 mulhi
|
||||
r4 = rotl_imm(r4, 12u); // s117 rotl
|
||||
r1 = r1 + r4 + ((((sel >> 7u) & 1u) != 0u) ? 0xc5e98ee5u : 0x24494ad3u); // s118 add
|
||||
r0 = r0 + r4 + ((((sel >> 4u) & 1u) != 0u) ? 0xa97bc949u : 0x1b2d1c83u); // s119 add
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r4, 4); // s120 shfl
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r2, 8); // s121 shfl
|
||||
r7 = rotl_imm(r7, 3u); // s122 rotl
|
||||
r4 = rotr_var(r4, r6); // s123 rotr
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r5, 1); // s124 shfl
|
||||
r5 = rotl_imm(r5, 21u); // s125 rotl
|
||||
r0 = r3 * r1 + r0; // s126 mad
|
||||
r0 = r0 + r3 + ((((sel >> 24u) & 1u) != 0u) ? 0xa7f6bb54u : 0xff14c08du); // s127 add
|
||||
r7 = r7 ^ r4; // s128 xor
|
||||
r3 = __umulhi(r3, r5); // s129 mulhi
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r7, 8); // s130 shfl
|
||||
r6 = __umulhi(r6, r4); // s131 mulhi
|
||||
r5 = r5 ^ r3; // s132 xor
|
||||
r0 = rotl_imm(r0, 19u); // s133 rotl
|
||||
r4 = rotr_var(r4, r3); // s134 rotr
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r2, 16); // s135 shfl
|
||||
r5 = r5 - r2; // s136 sub
|
||||
r3 = r3 ^ r5; // s137 xor
|
||||
r0 = r0 - r2; // s138 sub
|
||||
r3 = r3 * r7; // s139 mul
|
||||
r5 = r5 + r3 + ((((sel >> 24u) & 1u) != 0u) ? 0x2d465ca8u : 0x82ab98f4u); // s140 add
|
||||
r4 = r4 + r1 + ((((sel >> 11u) & 1u) != 0u) ? 0x237a9d8eu : 0x80594390u); // s141 add
|
||||
r3 = r7 * r0 + r3; // s142 mad
|
||||
r0 = r0 | r3; // s143 or
|
||||
r0 = __umulhi(r0, r1); // s144 mulhi
|
||||
r6 = r6 ^ r2; // s145 xor
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r1, 16); // s146 shfl
|
||||
r0 = r0 + r1 + ((((sel >> 18u) & 1u) != 0u) ? 0xe68cf57eu : 0x07f15148u); // s147 add
|
||||
r7 = r7 | r3; // s148 or
|
||||
r4 = r4 ^ r2; // s149 xor
|
||||
r7 = r7 ^ r1; // s150 xor
|
||||
r5 = rotr_var(r5, r2); // s151 rotr
|
||||
r0 = r0 * r7; // s152 mul
|
||||
r4 = rotl_imm(r4, 16u); // s153 rotl
|
||||
r6 = rotr_var(r6, r4); // s154 rotr
|
||||
r6 = r6 ^ r1; // s155 xor
|
||||
r6 = r6 | r3; // s156 or
|
||||
r1 = r1 + r6 + ((((sel >> 19u) & 1u) != 0u) ? 0x84c2a09du : 0x14878c5au); // s157 add
|
||||
r4 = r4 - r1; // s158 sub
|
||||
r4 = r6 * r7 + r4; // s159 mad
|
||||
r1 = r1 * r5; // s160 mul
|
||||
r4 = r4 | r0; // s161 or
|
||||
r7 = rotl_imm(r7, 21u); // s162 rotl
|
||||
r0 = r0 - r1; // s163 sub
|
||||
r1 = r1 * r0; // s164 mul
|
||||
r2 = r2 - r1; // s165 sub
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 2); // s166 shfl
|
||||
r0 = r0 - r4; // s167 sub
|
||||
r1 = rotr_var(r1, r6); // s168 rotr
|
||||
r7 = rotr_var(r7, r1); // s169 rotr
|
||||
r3 = r1 * r3 + r3; // s170 mad
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r0, 4); // s171 shfl
|
||||
r2 = r2 - r1; // s172 sub
|
||||
r7 = r7 ^ r6; // s173 xor
|
||||
r4 = rotr_var(r4, r1); // s174 rotr
|
||||
r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r7, 2); // s175 shfl
|
||||
r6 = r6 * r2; // s176 mul
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // s177 shfl
|
||||
r6 = r6 + r4 + ((((sel >> 21u) & 1u) != 0u) ? 0x509871e6u : 0x4fa43965u); // s178 add
|
||||
r0 = r0 ^ r6; // s179 xor
|
||||
r5 = r5 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0xb95b8a53u : 0xfb3c4bd8u); // s180 add
|
||||
r6 = r6 * r3; // s181 mul
|
||||
r7 = rotl_imm(r7, 1u); // s182 rotl
|
||||
r0 = r0 + r7 + ((((sel >> 11u) & 1u) != 0u) ? 0x7047b7cfu : 0xc448a197u); // s183 add
|
||||
r3 = r3 * r5; // s184 mul
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r7, 1); // s185 shfl
|
||||
r0 = r0 * r2; // s186 mul
|
||||
r4 = __umulhi(r4, r7); // s187 mulhi
|
||||
r2 = r7 * r2 + r2; // s188 mad
|
||||
r7 = r7 - r2; // s189 sub
|
||||
r2 = r2 + r5 + ((((sel >> 19u) & 1u) != 0u) ? 0x5caccc5du : 0x2fceaf49u); // s190 add
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r2, 1); // s191 shfl
|
||||
r7 = r7 + r1 + ((((sel >> 20u) & 1u) != 0u) ? 0x857bb5feu : 0xba0cee62u); // s192 add
|
||||
r6 = rotl_imm(r6, 22u); // s193 rotl
|
||||
r3 = r3 ^ r7; // s194 xor
|
||||
r7 = r7 * r0; // s195 mul
|
||||
r3 = r3 ^ r5; // s196 xor
|
||||
r5 = r5 + r1 + ((((sel >> 29u) & 1u) != 0u) ? 0xeae84577u : 0x8519428cu); // s197 add
|
||||
r7 = r4 * r4 + r7; // s198 mad
|
||||
r3 = r3 + r6 + ((((sel >> 8u) & 1u) != 0u) ? 0x0bfdbfa1u : 0x468639d3u); // s199 add
|
||||
r5 = r3 * r4 + r5; // s200 mad
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r7, 4); // s201 shfl
|
||||
r7 = r7 + r4 + ((((sel >> 23u) & 1u) != 0u) ? 0x53ee8e50u : 0x18ec9a69u); // s202 add
|
||||
r6 = r6 ^ r0; // s203 xor
|
||||
r4 = r4 + r5 + ((((sel >> 4u) & 1u) != 0u) ? 0xac63376eu : 0x3e81485bu); // s204 add
|
||||
r7 = r1 * r4 + r7; // s205 mad
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 16); // s206 shfl
|
||||
r4 = r4 * r0; // s207 mul
|
||||
r1 = r1 | r6; // s208 or
|
||||
r6 = r6 + r4 + ((((sel >> 28u) & 1u) != 0u) ? 0x5b745519u : 0x66ccb75du); // s209 add
|
||||
r1 = rotr_var(r1, r3); // s210 rotr
|
||||
r5 = r5 - r4; // s211 sub
|
||||
r4 = r4 ^ r7; // s212 xor
|
||||
r1 = r1 + r7 + ((((sel >> 8u) & 1u) != 0u) ? 0x1c3dccf7u : 0x0ce0553du); // s213 add
|
||||
r0 = __umulhi(r0, r1); // s214 mulhi
|
||||
r3 = r3 + r6 + ((((sel >> 9u) & 1u) != 0u) ? 0xfadc9205u : 0x96bfec89u); // s215 add
|
||||
r0 = rotl_imm(r0, 4u); // s216 rotl
|
||||
r6 = r6 ^ r4; // s217 xor
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r7, 2); // s218 shfl
|
||||
r7 = r0 * r1 + r7; // s219 mad
|
||||
r4 = r4 * r0; // s220 mul
|
||||
r2 = r6 * r0 + r2; // s221 mad
|
||||
r5 = r5 ^ r4; // s222 xor
|
||||
r0 = r0 * r5; // s223 mul
|
||||
r2 = r2 + r0 + ((((sel >> 9u) & 1u) != 0u) ? 0x66b15c04u : 0x61495681u); // s224 add
|
||||
r3 = rotr_var(r3, r2); // s225 rotr
|
||||
r2 = r3 * r3 + r2; // s226 mad
|
||||
r6 = r6 ^ r0; // s227 xor
|
||||
r4 = rotl_imm(r4, 30u); // s228 rotl
|
||||
r2 = r2 + r3 + ((((sel >> 29u) & 1u) != 0u) ? 0x25bc17c2u : 0x70d05c34u); // s229 add
|
||||
r1 = rotr_var(r1, r5); // s230 rotr
|
||||
r1 = r1 + r4 + ((((sel >> 8u) & 1u) != 0u) ? 0x2d8728f3u : 0x32a28384u); // s231 add
|
||||
r1 = rotl_imm(r1, 6u); // s232 rotl
|
||||
r2 = r2 + r0 + ((((sel >> 21u) & 1u) != 0u) ? 0x61e81fe6u : 0xcf0949f1u); // s233 add
|
||||
r4 = rotl_imm(r4, 2u); // s234 rotl
|
||||
r2 = r2 - r6; // s235 sub
|
||||
r6 = r6 | r3; // s236 or
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r5, 4); // s237 shfl
|
||||
r1 = r1 ^ r3; // s238 xor
|
||||
r2 = rotl_imm(r2, 1u); // s239 rotl
|
||||
r5 = rotr_var(r5, r2); // s240 rotr
|
||||
r4 = rotl_imm(r4, 10u); // s241 rotl
|
||||
r6 = r6 + r2 + ((((sel >> 16u) & 1u) != 0u) ? 0x0a67565du : 0xe204fb50u); // s242 add
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r3, 2); // s243 shfl
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r1, 16); // s244 shfl
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r7, 2); // s245 shfl
|
||||
r4 = r4 ^ r5; // s246 xor
|
||||
r7 = r7 ^ r2; // s247 xor
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 16); // s248 shfl
|
||||
r1 = __umulhi(r1, r3); // s249 mulhi
|
||||
r7 = r7 + r5 + ((((sel >> 31u) & 1u) != 0u) ? 0x5a79fd6fu : 0x3d6daffdu); // s250 add
|
||||
r6 = r6 + r0 + ((((sel >> 28u) & 1u) != 0u) ? 0x6beb28c2u : 0x84334aeau); // s251 add
|
||||
r7 = r7 - r1; // s252 sub
|
||||
r5 = rotr_var(r5, r2); // s253 rotr
|
||||
r2 = __umulhi(r2, r5); // s254 mulhi
|
||||
r6 = r6 * r3; // s255 mul
|
||||
}
|
||||
}
|
||||
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
|
||||
}
|
||||
|
||||
cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps) {
|
||||
if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 32u * blockWarps;
|
||||
if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;
|
||||
igneum_hash_bound<<<nonces / block, block>>>(ds, out, baseNonce, mask, iw);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
|
||||
cudaFuncAttributes attr;
|
||||
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound);
|
||||
if (e != cudaSuccess) return e;
|
||||
*numRegs = attr.numRegs;
|
||||
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0);
|
||||
}
|
||||
BIN
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/leaves.bin
Normal file
BIN
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/leaves.bin
Normal file
Binary file not shown.
116
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/memhard.h
Normal file
116
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/memhard.h
Normal file
|
|
@ -0,0 +1,116 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925/day/69676e65756d2d6461792ffd50000000000000". Do not edit by hand.
|
||||
// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against.
|
||||
// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference).
|
||||
// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C.
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
#if defined(__CUDACC__)
|
||||
#define IGNEUM_HD __host__ __device__ __forceinline__
|
||||
#elif defined(_MSC_VER) && !defined(__cplusplus)
|
||||
#define IGNEUM_HD static __inline
|
||||
#else
|
||||
#define IGNEUM_HD static inline
|
||||
#endif
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of 8 x seed-parameterised mixer + one 64-byte cache read, then 8 x final mixer (class v3, mixer multiplier 8,
|
||||
// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) {
|
||||
for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint32_t r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) {
|
||||
uint32_t prev[16]; uint32_t x[16]; uint32_t y[16];
|
||||
for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0xe23f9008u ^ prev[4];
|
||||
x[5] = 0xde33e763u ^ prev[5];
|
||||
x[6] = 0xc5ba415du ^ prev[6];
|
||||
x[7] = 0x8ddf6786u ^ prev[7];
|
||||
x[8] = 0x59f4a4beu ^ prev[8];
|
||||
x[9] = 0x8a3bc680u ^ prev[9];
|
||||
x[10] = 0x701b8e40u ^ prev[10];
|
||||
x[11] = 0x3b59025au ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) {
|
||||
s[0] = (s[0] ^ (0xe5bef5a3u + rk)) * 0xd1d79e7fu;
|
||||
s[1] = (s[1] ^ (0xdb2a7d90u + rk)) * 0xbcb61f8bu;
|
||||
s[2] = (s[2] ^ (0xcd1fa7e1u + rk)) * 0x970e91abu;
|
||||
s[3] = (s[3] ^ (0x30289419u + rk)) * 0xd749d96du;
|
||||
s[4] = (s[4] ^ (0x0a730d58u + rk)) * 0x143d7339u;
|
||||
s[5] = (s[5] ^ (0x432f8579u + rk)) * 0x2cde0d69u;
|
||||
s[6] = (s[6] ^ (0x6ab978a5u + rk)) * 0x12d0a8c1u;
|
||||
s[7] = (s[7] ^ (0x8c984f49u + rk)) * 0x2d335be9u;
|
||||
s[8] = (s[8] ^ (0x788c3c9eu + rk)) * 0x80a8aae9u;
|
||||
s[9] = (s[9] ^ (0x051eef02u + rk)) * 0x7ac896c7u;
|
||||
s[10] = (s[10] ^ (0x05e77db9u + rk)) * 0x9de23db7u;
|
||||
s[11] = (s[11] ^ (0x8b3cd4f3u + rk)) * 0xc362827du;
|
||||
s[12] = (s[12] ^ (0x6948d6cfu + rk)) * 0x5f4cdb5bu;
|
||||
s[13] = (s[13] ^ (0x0d8677f6u + rk)) * 0xfc6c5097u;
|
||||
s[14] = (s[14] ^ (0xf9505c8cu + rk)) * 0x6f547f83u;
|
||||
s[15] = (s[15] ^ (0x4513c163u + rk)) * 0x1ac31b47u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 28u, 11u, 18u, 18u) MH_QR(s[1], s[5], s[9], s[13], 28u, 11u, 18u, 18u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 28u, 11u, 18u, 18u) MH_QR(s[3], s[7], s[11], s[15], 28u, 11u, 18u, 18u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 13u, 28u, 5u, 13u) MH_QR(s[1], s[6], s[11], s[12], 13u, 28u, 5u, 13u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 13u, 28u, 5u, 13u) MH_QR(s[3], s[4], s[9], s[14], 13u, 28u, 5u, 13u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of 8 x mixer + cache line s[0] & mask; 8 x final mixer.
|
||||
IGNEUM_HD void mh_item(const uint32_t* cache, const uint32_t* leaf, uint32_t t, uint32_t* s) {
|
||||
s[0] = 0xe23f9008u;
|
||||
s[1] = 0xde33e763u;
|
||||
s[2] = 0xc5ba415du;
|
||||
s[3] = 0x8ddf6786u;
|
||||
s[4] = 0x59f4a4beu;
|
||||
s[5] = 0x8a3bc680u;
|
||||
s[6] = 0x701b8e40u;
|
||||
s[7] = 0x3b59025au;
|
||||
s[8] = t * 0xd1d79e7fu + 0xe5bef5a3u;
|
||||
s[9] = t * 0xbcb61f8bu + 0xdb2a7d90u;
|
||||
s[10] = t * 0x970e91abu + 0xcd1fa7e1u;
|
||||
s[11] = t * 0xd749d96du + 0x30289419u;
|
||||
s[12] = t * 0x143d7339u + 0x0a730d58u;
|
||||
s[13] = t * 0x2cde0d69u + 0x432f8579u;
|
||||
s[14] = t * 0x12d0a8c1u + 0x6ab978a5u;
|
||||
s[15] = t * 0x2d335be9u + 0x8c984f49u;
|
||||
for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= leaf[i];
|
||||
for (uint32_t r = 0u; r < 8u; ++r) {
|
||||
for (uint32_t j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 8u + j + 1u));
|
||||
const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
for (uint32_t j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (64u + j + 1u));
|
||||
}
|
||||
// Class v5 (docs/design/class-v5-stored-state.md): leaf(t) = leaves[t mod nLeaves], 16 words per leaf (leaves.bin).
|
||||
IGNEUM_HD const uint32_t* mh_leaf(const uint32_t* leaves, uint32_t nLeaves, uint32_t t) { return leaves + ((t % nLeaves) * 16u); }
|
||||
// Era layout (docs/plans/era-layout.md 1.2): dataset word w holds word mh_j(w) of item mh_t(w); j's bits sit at positions 1 9 13 14 of w.
|
||||
IGNEUM_HD uint32_t mh_j(uint32_t w) { return ((w >> 1u) & 1u) | (((w >> 9u) & 1u) << 1) | (((w >> 13u) & 1u) << 2) | (((w >> 14u) & 1u) << 3); }
|
||||
IGNEUM_HD uint32_t mh_t(uint32_t w) { w = (w & 0x00003fffu) | ((w >> 15u) << 14u); w = (w & 0x00001fffu) | ((w >> 14u) << 13u); w = (w & 0x000001ffu) | ((w >> 10u) << 9u); w = (w & 0x00000001u) | ((w >> 2u) << 1u); return w; }
|
||||
IGNEUM_HD uint32_t mh_addr(uint32_t t, uint32_t j) { uint32_t w = t; w = ((w >> 1u) << 2u) | (w & 0x00000001u) | (((j >> 0u) & 1u) << 1u); w = ((w >> 9u) << 10u) | (w & 0x000001ffu) | (((j >> 1u) & 1u) << 9u); w = ((w >> 13u) << 14u) | (w & 0x00001fffu) | (((j >> 2u) & 1u) << 13u); w = ((w >> 14u) << 15u) | (w & 0x00003fffu) | (((j >> 3u) & 1u) << 14u); return w; }
|
||||
// dataset[w] without the dataset: derive item mh_t(w) and take word mh_j(w).
|
||||
IGNEUM_HD uint32_t mh_word(const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t w) { uint32_t s[16]; mh_item(cache, mh_leaf(leaves, nLeaves, mh_t(w)), mh_t(w), s); return s[mh_j(w)]; }
|
||||
115
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/memhard.metal
Normal file
115
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/memhard.metal
Normal file
|
|
@ -0,0 +1,115 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of 8 x seed-parameterised mixer + one 64-byte cache read, then 8 x final mixer (class v3, mixer multiplier 8,
|
||||
// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
inline void mh_chacha_block(const thread uint* x, thread uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
inline void mh_cache_segment(device uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0xe23f9008u ^ prev[4];
|
||||
x[5] = 0xde33e763u ^ prev[5];
|
||||
x[6] = 0xc5ba415du ^ prev[6];
|
||||
x[7] = 0x8ddf6786u ^ prev[7];
|
||||
x[8] = 0x59f4a4beu ^ prev[8];
|
||||
x[9] = 0x8a3bc680u ^ prev[9];
|
||||
x[10] = 0x701b8e40u ^ prev[10];
|
||||
x[11] = 0x3b59025au ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
inline void mh_mixer(thread uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xe5bef5a3u + rk)) * 0xd1d79e7fu;
|
||||
s[1] = (s[1] ^ (0xdb2a7d90u + rk)) * 0xbcb61f8bu;
|
||||
s[2] = (s[2] ^ (0xcd1fa7e1u + rk)) * 0x970e91abu;
|
||||
s[3] = (s[3] ^ (0x30289419u + rk)) * 0xd749d96du;
|
||||
s[4] = (s[4] ^ (0x0a730d58u + rk)) * 0x143d7339u;
|
||||
s[5] = (s[5] ^ (0x432f8579u + rk)) * 0x2cde0d69u;
|
||||
s[6] = (s[6] ^ (0x6ab978a5u + rk)) * 0x12d0a8c1u;
|
||||
s[7] = (s[7] ^ (0x8c984f49u + rk)) * 0x2d335be9u;
|
||||
s[8] = (s[8] ^ (0x788c3c9eu + rk)) * 0x80a8aae9u;
|
||||
s[9] = (s[9] ^ (0x051eef02u + rk)) * 0x7ac896c7u;
|
||||
s[10] = (s[10] ^ (0x05e77db9u + rk)) * 0x9de23db7u;
|
||||
s[11] = (s[11] ^ (0x8b3cd4f3u + rk)) * 0xc362827du;
|
||||
s[12] = (s[12] ^ (0x6948d6cfu + rk)) * 0x5f4cdb5bu;
|
||||
s[13] = (s[13] ^ (0x0d8677f6u + rk)) * 0xfc6c5097u;
|
||||
s[14] = (s[14] ^ (0xf9505c8cu + rk)) * 0x6f547f83u;
|
||||
s[15] = (s[15] ^ (0x4513c163u + rk)) * 0x1ac31b47u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 28u, 11u, 18u, 18u) MH_QR(s[1], s[5], s[9], s[13], 28u, 11u, 18u, 18u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 28u, 11u, 18u, 18u) MH_QR(s[3], s[7], s[11], s[15], 28u, 11u, 18u, 18u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 13u, 28u, 5u, 13u) MH_QR(s[1], s[6], s[11], s[12], 13u, 28u, 5u, 13u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 13u, 28u, 5u, 13u) MH_QR(s[3], s[4], s[9], s[14], 13u, 28u, 5u, 13u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of 8 x mixer + cache line s[0] & mask; 8 x final mixer.
|
||||
inline void mh_item(device const uint* cache, device const uint* leaf, uint t, thread uint* s) {
|
||||
s[0] = 0xe23f9008u;
|
||||
s[1] = 0xde33e763u;
|
||||
s[2] = 0xc5ba415du;
|
||||
s[3] = 0x8ddf6786u;
|
||||
s[4] = 0x59f4a4beu;
|
||||
s[5] = 0x8a3bc680u;
|
||||
s[6] = 0x701b8e40u;
|
||||
s[7] = 0x3b59025au;
|
||||
s[8] = t * 0xd1d79e7fu + 0xe5bef5a3u;
|
||||
s[9] = t * 0xbcb61f8bu + 0xdb2a7d90u;
|
||||
s[10] = t * 0x970e91abu + 0xcd1fa7e1u;
|
||||
s[11] = t * 0xd749d96du + 0x30289419u;
|
||||
s[12] = t * 0x143d7339u + 0x0a730d58u;
|
||||
s[13] = t * 0x2cde0d69u + 0x432f8579u;
|
||||
s[14] = t * 0x12d0a8c1u + 0x6ab978a5u;
|
||||
s[15] = t * 0x2d335be9u + 0x8c984f49u;
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= leaf[i];
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
for (uint j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 8u + j + 1u));
|
||||
device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
for (uint j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (64u + j + 1u));
|
||||
}
|
||||
// Class v5 (docs/design/class-v5-stored-state.md): leaf(t) = leaves[t mod nLeaves], 16 words per leaf (leaves.bin).
|
||||
inline device const uint* mh_leaf(device const uint* leaves, uint nLeaves, uint t) { return leaves + ((t % nLeaves) * 16u); }
|
||||
// Era layout (docs/plans/era-layout.md 1.2): dataset word w holds word mh_j(w) of item mh_t(w); j's bits sit at positions 1 9 13 14 of w.
|
||||
inline uint mh_j(uint w) { return ((w >> 1u) & 1u) | (((w >> 9u) & 1u) << 1) | (((w >> 13u) & 1u) << 2) | (((w >> 14u) & 1u) << 3); }
|
||||
inline uint mh_t(uint w) { w = (w & 0x00003fffu) | ((w >> 15u) << 14u); w = (w & 0x00001fffu) | ((w >> 14u) << 13u); w = (w & 0x000001ffu) | ((w >> 10u) << 9u); w = (w & 0x00000001u) | ((w >> 2u) << 1u); return w; }
|
||||
inline uint mh_addr(uint t, uint j) { uint w = t; w = ((w >> 1u) << 2u) | (w & 0x00000001u) | (((j >> 0u) & 1u) << 1u); w = ((w >> 9u) << 10u) | (w & 0x000001ffu) | (((j >> 1u) & 1u) << 9u); w = ((w >> 13u) << 14u) | (w & 0x00001fffu) | (((j >> 2u) & 1u) << 13u); w = ((w >> 14u) << 15u) | (w & 0x00003fffu) | (((j >> 3u) & 1u) << 14u); return w; }
|
||||
// dataset[w] without the dataset: derive item mh_t(w) and take word mh_j(w).
|
||||
inline uint mh_word(device const uint* cache, device const uint* leaves, uint nLeaves, uint w) { uint s[16]; mh_item(cache, mh_leaf(leaves, nLeaves, mh_t(w)), mh_t(w), s); return s[mh_j(w)]; }
|
||||
|
||||
// One thread per segment (2^16 threads).
|
||||
kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) {
|
||||
mh_cache_segment(cache, gid);
|
||||
}
|
||||
// One thread per 64-byte item (dataset words / 16 threads).
|
||||
// Class v5: the window's leaves (leaves.bin, IGNEUM_STATE_LEAVES x 16 words) in buffer 2, their count in buffer 3.
|
||||
kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],
|
||||
device const uint* leaves [[buffer(2)]], constant uint& nLeaves [[buffer(3)]],
|
||||
uint gid [[thread_position_in_grid]]) {
|
||||
uint s[16];
|
||||
mh_item(cache, mh_leaf(leaves, nLeaves, gid), gid, s);
|
||||
for (uint i = 0u; i < 16u; ++i) dataset[mh_addr(gid, i)] = s[i];
|
||||
}
|
||||
97
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/program.h
Normal file
97
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/program.h
Normal file
|
|
@ -0,0 +1,97 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925/day/69676e65756d2d6461792ffd50000000000000". Do not edit by hand.
|
||||
// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.
|
||||
// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros.
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
#ifndef IGNEUM_NO_CUDA
|
||||
#include <cuda_runtime.h>
|
||||
#endif
|
||||
|
||||
#define IGNEUM_SEED_STRING "igneum-epoch/4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925/day/69676e65756d2d6461792ffd50000000000000"
|
||||
#define IGNEUM_SEED_BYTES_HEX "4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925"
|
||||
#define IGNEUM_GENERATOR 5
|
||||
#define IGNEUM_PROGRAM_ATTEMPT 0
|
||||
#define IGNEUM_PROGRAM_ID 0xe5a4ac5978462156ull
|
||||
#define IGNEUM_DAY_STRING "bytes:69676e65756d2d6461792ffd50000000000000"
|
||||
#define IGNEUM_DAY_BYTES_HEX "69676e65756d2d6461792ffd50000000000000"
|
||||
#define IGNEUM_DAY0 0xe23f9008u
|
||||
#define IGNEUM_DAY1 0xde33e763u
|
||||
#define IGNEUM_DATASET_LOG2 28
|
||||
#define IGNEUM_MASK 0x0fffffffu
|
||||
#define IGNEUM_LANES 32
|
||||
#define IGNEUM_ITERATIONS 8
|
||||
#define IGNEUM_INSTR_COUNT 64
|
||||
#define IGNEUM_LOADS_PER_HASH 128
|
||||
#define IGNEUM_WIDE_LOADS_PER_HASH 0
|
||||
#define IGNEUM_OP_MIX "load=16 add=11 shfl=7 mad=6 xor=6 mul=4 mulhi=4 rotr=4 rotl=3 sub=2 or=1"
|
||||
// Program class v5 (proof of stored state and of following, docs/design/class-v5-stored-state.md): generator version 5,
|
||||
// class v4 over a dataset whose every item is keyed by the window's execution state (IGNEUM_STATE_* below, leaves.bin);
|
||||
// a worker that runs another class refuses this pack, and a job line names the class it wants (class=v5 era=<hex>).
|
||||
#define IGNEUM_PROGRAM_CLASS "v5"
|
||||
#define IGNEUM_ERA_SEED_HEX "4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925"
|
||||
// Class v3 construction (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): version 2 loads; the dataset item
|
||||
// derivation applies the mixer IGNEUM_MIXER_MULT times per round (memhard.h), and the cache follows the growth rule.
|
||||
#define IGNEUM_LOAD_CLASS "mx8-eraaf3a9139+sh256x27+state"
|
||||
#define IGNEUM_CLASS_MIXER_MULT 8
|
||||
#define IGNEUM_CACHE_GROWTH 1 // 1: cache words = 2^(26 + doublings(day)), doublings = floor(log2(1 + day / 1460))
|
||||
#define IGNEUM_LOAD_SLOTS 16
|
||||
#define IGNEUM_LOAD_MIX { 100, 0, 0 }
|
||||
#define IGNEUM_LOAD_WIDTH_COUNTS { 16, 0, 0 } // loads of 4, 16, 64 bytes per program
|
||||
#define IGNEUM_BYTES_PER_HASH 512
|
||||
#define IGNEUM_FOLD_ROT 11
|
||||
#define IGNEUM_FOLD_MUL 0x9e3779b1u
|
||||
// Class v5 state (docs/design/class-v5-stored-state.md): the window's reference chain block and the state root after it;
|
||||
// leaves.bin holds IGNEUM_STATE_LEAVES leaves of 16 little-endian words, leaf(t) = leaves[t mod IGNEUM_STATE_LEAVES].
|
||||
#define IGNEUM_STATE_BLOCK_HEX "4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925"
|
||||
#define IGNEUM_STATE_BLOCK_NUMBER 0
|
||||
#define IGNEUM_STATE_ROOT_HEX "7e37a9fb19b154d32daf5bf30a50d339a75029fbc9eec9ea20e95439dba5a311"
|
||||
#define IGNEUM_STATE_LEAVES 11
|
||||
#define IGNEUM_STATE_RECORDS 11
|
||||
#define IGNEUM_STATE_SAMPLED 0
|
||||
#define IGNEUM_STATE_LEAVES_FNV64 0xf141bfee2a8b11b0ull
|
||||
#define IGNEUM_STATE_LEAVES_FILE "leaves.bin"
|
||||
// Era layout (5 October 2026, docs/plans/era-layout.md): NOT the lottery hash. Every dataset load reads
|
||||
// idx = ((rotl(src * STRIDE_MUL, STRIDE_ROT) & window mask) | window offset) & MASK; the window of a load site is the
|
||||
// dataset, a half or a quarter of it (IGNEUM_ERA_WINDOWS: site:shrink:offset); dataset word w holds word j(w) of item
|
||||
// t(w) with j's bits at the INTERLEAVE positions (memhard.h: mh_t, mh_j, mh_addr).
|
||||
#define IGNEUM_ERA_LABEL "af3a9139"
|
||||
#define IGNEUM_ERA_SEED_WORDS { 0xaf3a9139u, 0x894db311u, 0xd6173573u, 0xc346bc0fu, 0x2d547cecu, 0xd5634ab8u, 0xb9584827u, 0x305d030eu }
|
||||
#define IGNEUM_ERA_ALLOWED_WIDTHS { 1, 0, 0 } // words, ascending, 0 = unused; one entry pins the width
|
||||
#define IGNEUM_ERA_WIDTH_WORDS 1
|
||||
#define IGNEUM_ERA_STRIDE_MUL 0x2cb18b35u
|
||||
#define IGNEUM_ERA_STRIDE_ROT 9
|
||||
#define IGNEUM_ERA_INTERLEAVE { 1, 9, 13, 14 }
|
||||
#define IGNEUM_ERA_WINDOWS "3:0:0 8:2:2 14:1:0 15:0:0 20:1:1 26:1:1 28:1:1 35:0:0 40:2:3 43:1:0 47:1:1 49:0:0 52:2:3 53:0:0 61:1:1 62:1:1"
|
||||
// Latency-shadow block (Counter ASIC 3.0 item 8, docs/analysis/latency-shadow-2026-10-06.md): NOT the lottery hash. A block of
|
||||
// IGNEUM_SHADOW_INSTRS ALU instructions (the ten non-load families) runs IGNEUM_SHADOW_REPS times at the end of every
|
||||
// iteration; the 16 loads, the acceptance rule and the base program are the class's without the shadow.
|
||||
#define IGNEUM_SHADOW_INSTRS 256
|
||||
#define IGNEUM_SHADOW_REPS 27
|
||||
#define IGNEUM_SHADOW_INSTRS_PER_HASH 55296
|
||||
#define IGNEUM_SHADOW_OP_MIX "add=45 mad=31 shfl=31 xor=30 rotl=28 mul=22 or=18 mulhi=17 rotr=17 sub=17"
|
||||
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
|
||||
#define IGNEUM_DATASET_MODE 1
|
||||
|
||||
#define IGNEUM_SEEDW_INIT { 0xbe8c5a0cu, 0x7646e626u, 0x508e29d0u, 0xfb8a5b4au, 0x53c50955u, 0x685d62fcu, 0x6065c013u, 0x881096ebu }
|
||||
#define IGNEUM_KEY_INIT { 0xe23f9008u, 0xde33e763u, 0xc5ba415du, 0x8ddf6786u, 0x59f4a4beu, 0x8a3bc680u, 0x701b8e40u, 0x3b59025au }
|
||||
#define IGNEUM_CACHE_LOG2_WORDS 26
|
||||
#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6
|
||||
#define IGNEUM_CACHE_SEGMENTS 65536u
|
||||
#define IGNEUM_ITEM_ROUNDS 8
|
||||
#define IGNEUM_MIXER_MULT 8 // mixer applications per round and after the last read (class v3, docs/plans/mixer-x4.md)
|
||||
#define IGNEUM_MIX_ROT_INIT { 28u, 11u, 18u, 18u, 13u, 28u, 5u, 13u }
|
||||
#define IGNEUM_MIX_MUL_INIT { 0xd1d79e7fu, 0xbcb61f8bu, 0x970e91abu, 0xd749d96du, 0x143d7339u, 0x2cde0d69u, 0x12d0a8c1u, 0x2d335be9u, 0x80a8aae9u, 0x7ac896c7u, 0x9de23db7u, 0xc362827du, 0x5f4cdb5bu, 0xfc6c5097u, 0x6f547f83u, 0x1ac31b47u }
|
||||
#define IGNEUM_MIX_RC_INIT { 0xe5bef5a3u, 0xdb2a7d90u, 0xcd1fa7e1u, 0x30289419u, 0x0a730d58u, 0x432f8579u, 0x6ab978a5u, 0x8c984f49u, 0x788c3c9eu, 0x051eef02u, 0x05e77db9u, 0x8b3cd4f3u, 0x6948d6cfu, 0x0d8677f6u, 0xf9505c8cu, 0x4513c163u }
|
||||
|
||||
#ifndef IGNEUM_NO_CUDA
|
||||
// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().
|
||||
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);
|
||||
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems);
|
||||
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
uint32_t nonces, uint32_t blockWarps);
|
||||
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
|
||||
#endif
|
||||
416
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/program.json
Normal file
416
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/program.json
Normal file
|
|
@ -0,0 +1,416 @@
|
|||
{
|
||||
"format": "igneum-program-pack-3",
|
||||
"generator": 5,
|
||||
"attempt": 0,
|
||||
"program_id": "0xe5a4ac5978462156",
|
||||
"program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32",
|
||||
"dataset_mode": "memory-hard",
|
||||
"seed": "igneum-epoch/4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925/day/69676e65756d2d6461792ffd50000000000000",
|
||||
"seed_bytes": "4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925",
|
||||
"seed_words": ["0xbe8c5a0c", "0x7646e626", "0x508e29d0", "0xfb8a5b4a", "0x53c50955", "0x685d62fc", "0x6065c013", "0x881096eb"],
|
||||
"seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32",
|
||||
"generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried",
|
||||
"lanes": 32,
|
||||
"registers": 8,
|
||||
"iterations": 8,
|
||||
"instruction_count": 64,
|
||||
"loads_per_hash": 128,
|
||||
"program_class": "v5",
|
||||
"era_seed_bytes": "4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925",
|
||||
"state": {
|
||||
"block": "4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925",
|
||||
"block_number": 0,
|
||||
"root": "7e37a9fb19b154d32daf5bf30a50d339a75029fbc9eec9ea20e95439dba5a311",
|
||||
"leaves": 11,
|
||||
"records": 11,
|
||||
"sampled": false,
|
||||
"leaves_fnv1a64": "0xf141bfee2a8b11b0",
|
||||
"leaf_derivation": "leaves[i] = Blake2b-512('igneum-sd1/' || root || i_le32 || record_i) as 16 little-endian words; item t XORs leaves[t mod leaves] into its 16 initial words before the first mixer",
|
||||
"file": "leaves.bin"
|
||||
},
|
||||
"load_class": "mx8-eraaf3a9139+sh256x27+state",
|
||||
"mixer_mult": 8,
|
||||
"cache_growth": true,
|
||||
"mixer": "class v3 (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): every mixer application of the item derivation is 8 applications with round keys (r * 8 + j + 1) * 0x9E3779B9, the 8 dependent cache reads per item unchanged; cache growth rule option C: cache words = 2^(26 + doublings(day)), dataset words = 2^(genesis_log2 + doublings(day)), doublings(day) = floor(log2(1 + day / 1460)) for day = days since genesis",
|
||||
"load_slots": 16,
|
||||
"load_mix_percent_4_16_64": [100, 0, 0],
|
||||
"load_width_counts_4_16_64": [16, 0, 0],
|
||||
"bytes_per_hash": 512,
|
||||
"wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots",
|
||||
"era": {
|
||||
"label": "af3a9139",
|
||||
"seed_words": ["0xaf3a9139", "0x894db311", "0xd6173573", "0xc346bc0f", "0x2d547cec", "0xd5634ab8", "0xb9584827", "0x305d030e"],
|
||||
"draw": "docs/plans/era-layout.md 1.1: SplitMix64 seeded with seed_words[0] | seed_words[1] << 32 of seed_words_from_bytes('igneum-era/' || n_le64 || E_n); width = allowed[below(|allowed|)], stride_mul = low32(next()) | 1, stride_rot = 1 + below(31), then four next() draws for a partial Fisher-Yates over positions log2(W)..15 of which 4 - log2(W) are used",
|
||||
"allowed_widths": [1],
|
||||
"width_words": 1,
|
||||
"stride_mul": "0x2cb18b35",
|
||||
"stride_rot": 9,
|
||||
"interleave": [1, 9, 13, 14],
|
||||
"address": "y = rotl(src * stride_mul, stride_rot); k = min(win, D - 26); idx = ((y & (mask >> k)) | ((off & (2^k - 1)) << (D - k))) & mask; a wide load aligns idx down to W words",
|
||||
"windows": "per instruction, after the width roll: win = below(3), off = low32(next()) & (2^win - 1); used on a load slot (the instruction's win and off fields)",
|
||||
"dataset_word": "dataset[w] = item(t(w))[j(w)]: j(w) gathers the bits of w at the interleave positions, t(w) is w with those bits removed",
|
||||
"program_id_suffix": "'era/' || allowed[3] || width_words || stride_mul_le32 || stride_rot_le32 || interleave[4]"
|
||||
},
|
||||
"op_mix": {"load": 16, "add": 11, "shfl": 7, "mad": 6, "xor": 6, "mul": 4, "mulhi": 4, "rotr": 4, "rotl": 3, "sub": 2, "or": 1},
|
||||
"register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]",
|
||||
"splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16",
|
||||
"iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order",
|
||||
"output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo",
|
||||
"op_semantics": {
|
||||
"add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)",
|
||||
"sub": "dst = dst - src",
|
||||
"mul": "dst = dst * src (low 32)",
|
||||
"mulhi": "dst = high 32 bits of dst * src",
|
||||
"xor": "dst = dst ^ src",
|
||||
"or": "dst = dst | src",
|
||||
"rotl": "dst = rotl(dst, rot), rot in 1..31",
|
||||
"rotr": "dst = rotr(dst, src & 31)",
|
||||
"mad": "dst = src * src2 + dst",
|
||||
"shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp",
|
||||
"load": "dst = dst ^ dataset[src & dataset.mask]",
|
||||
"wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)"
|
||||
},
|
||||
"dataset": {
|
||||
"log2_words": 28,
|
||||
"bytes": 1073741824,
|
||||
"mask": "0x0fffffff",
|
||||
"day": "bytes:69676e65756d2d6461792ffd50000000000000",
|
||||
"day_bytes": "69676e65756d2d6461792ffd50000000000000",
|
||||
"day_words_from": "seed_words_from_bytes(day_bytes)",
|
||||
"d0": "0xe23f9008",
|
||||
"d1": "0xde33e763",
|
||||
"mode": "memory-hard",
|
||||
"spec": "proto-metal/MEMHARD.md",
|
||||
"key": ["0xe23f9008", "0xde33e763", "0xc5ba415d", "0x8ddf6786", "0x59f4a4be", "0x8a3bc680", "0x701b8e40", "0x3b59025a"],
|
||||
"key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]",
|
||||
"cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"},
|
||||
"mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [28, 11, 18, 18, 13, 28, 5, 13], "mul": ["0xd1d79e7f", "0xbcb61f8b", "0x970e91ab", "0xd749d96d", "0x143d7339", "0x2cde0d69", "0x12d0a8c1", "0x2d335be9", "0x80a8aae9", "0x7ac896c7", "0x9de23db7", "0xc362827d", "0x5f4cdb5b", "0xfc6c5097", "0x6f547f83", "0x1ac31b47"], "rc": ["0xe5bef5a3", "0xdb2a7d90", "0xcd1fa7e1", "0x30289419", "0x0a730d58", "0x432f8579", "0x6ab978a5", "0x8c984f49", "0x788c3c9e", "0x051eef02", "0x05e77db9", "0x8b3cd4f3", "0x6948d6cf", "0x0d8677f6", "0xf9505c8c", "0x4513c163"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"},
|
||||
"mixer_mult": 8,
|
||||
"item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: for j in 0..7: s = M(s, rk = (r * 8 + j + 1) * 0x9E3779B9); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then for j in 0..7: s = M(s, rk = (64 + j + 1) * 0x9E3779B9); item(t) = s",
|
||||
"word": "dataset[w] = item(w >> 4)[w & 15]"
|
||||
},
|
||||
"shadow": {"instrs": 256, "reps": 27, "instrs_per_hash": 55296, "op_mix": {"add": 45, "mad": 31, "shfl": 31, "xor": 30, "rotl": 28, "mul": 22, "or": 18, "mulhi": 17, "rotr": 17, "sub": 17}, "rule": "Counter ASIC 3.0 item 8 (docs/analysis/latency-shadow-2026-10-06.md): after the 64 base instructions the program stream draws instrs more ALU instructions (the non-load table, nine draws each, the source as on an ALU slot); the block runs reps times at the end of every iteration with the iteration's sel; the acceptance rule interprets the base program only", "program_id_suffix": "'shadow/' || instrs_le16 || reps_le16", "instructions": [
|
||||
{"i": 0, "op": "or", "dst": 6, "src": 0, "src2": 2, "imm": "0xc1eb1523", "imm2": "0x925958ea", "rot": 14, "bit": 14, "mask": 8},
|
||||
{"i": 1, "op": "mul", "dst": 7, "src": 3, "src2": 4, "imm": "0xc7044334", "imm2": "0x755566b6", "rot": 19, "bit": 17, "mask": 8},
|
||||
{"i": 2, "op": "mad", "dst": 3, "src": 5, "src2": 5, "imm": "0x33b52f17", "imm2": "0xcc459a7a", "rot": 5, "bit": 19, "mask": 16},
|
||||
{"i": 3, "op": "add", "dst": 7, "src": 4, "src2": 4, "imm": "0x97d3d105", "imm2": "0x38e58a06", "rot": 11, "bit": 6, "mask": 1},
|
||||
{"i": 4, "op": "add", "dst": 6, "src": 0, "src2": 6, "imm": "0x29b87b9a", "imm2": "0x9629673d", "rot": 16, "bit": 20, "mask": 8},
|
||||
{"i": 5, "op": "shfl", "dst": 6, "src": 2, "src2": 5, "imm": "0xb61f2e77", "imm2": "0x30e3807c", "rot": 1, "bit": 7, "mask": 1},
|
||||
{"i": 6, "op": "mad", "dst": 5, "src": 3, "src2": 5, "imm": "0x0de66253", "imm2": "0xfdfdd03e", "rot": 4, "bit": 27, "mask": 4},
|
||||
{"i": 7, "op": "rotl", "dst": 3, "src": 7, "src2": 1, "imm": "0x688514dd", "imm2": "0x139d08b5", "rot": 5, "bit": 1, "mask": 1},
|
||||
{"i": 8, "op": "add", "dst": 0, "src": 7, "src2": 3, "imm": "0xa7c1fb0f", "imm2": "0x4a5f55e1", "rot": 15, "bit": 12, "mask": 1},
|
||||
{"i": 9, "op": "mul", "dst": 5, "src": 7, "src2": 2, "imm": "0x1a64de5a", "imm2": "0xe37b0875", "rot": 5, "bit": 27, "mask": 16},
|
||||
{"i": 10, "op": "xor", "dst": 4, "src": 5, "src2": 3, "imm": "0x750d4f1a", "imm2": "0x2f3d9402", "rot": 3, "bit": 14, "mask": 2},
|
||||
{"i": 11, "op": "mulhi", "dst": 6, "src": 4, "src2": 5, "imm": "0x2eaa86c6", "imm2": "0x9fd5b22f", "rot": 30, "bit": 10, "mask": 1},
|
||||
{"i": 12, "op": "rotl", "dst": 5, "src": 2, "src2": 4, "imm": "0x8df50473", "imm2": "0x23d780ce", "rot": 13, "bit": 9, "mask": 1},
|
||||
{"i": 13, "op": "mad", "dst": 0, "src": 2, "src2": 1, "imm": "0xa956a7e6", "imm2": "0x58dbb356", "rot": 31, "bit": 11, "mask": 8},
|
||||
{"i": 14, "op": "mulhi", "dst": 2, "src": 4, "src2": 6, "imm": "0x6fbfb145", "imm2": "0x064eb015", "rot": 20, "bit": 2, "mask": 4},
|
||||
{"i": 15, "op": "or", "dst": 3, "src": 1, "src2": 5, "imm": "0xd313bc27", "imm2": "0x9f0fc235", "rot": 17, "bit": 16, "mask": 1},
|
||||
{"i": 16, "op": "sub", "dst": 0, "src": 5, "src2": 0, "imm": "0x86cbd336", "imm2": "0x94715167", "rot": 17, "bit": 31, "mask": 8},
|
||||
{"i": 17, "op": "mul", "dst": 0, "src": 6, "src2": 3, "imm": "0x95911c6b", "imm2": "0x816cea32", "rot": 23, "bit": 1, "mask": 2},
|
||||
{"i": 18, "op": "shfl", "dst": 5, "src": 1, "src2": 3, "imm": "0xf563dcad", "imm2": "0xb2889856", "rot": 14, "bit": 24, "mask": 4},
|
||||
{"i": 19, "op": "or", "dst": 3, "src": 4, "src2": 2, "imm": "0x7bb48d19", "imm2": "0xbf95ad6f", "rot": 8, "bit": 30, "mask": 16},
|
||||
{"i": 20, "op": "mad", "dst": 6, "src": 3, "src2": 2, "imm": "0xeb686a8d", "imm2": "0x8bda9f8a", "rot": 11, "bit": 18, "mask": 1},
|
||||
{"i": 21, "op": "rotl", "dst": 0, "src": 6, "src2": 0, "imm": "0x0915df51", "imm2": "0xa1365cda", "rot": 11, "bit": 25, "mask": 1},
|
||||
{"i": 22, "op": "rotl", "dst": 3, "src": 0, "src2": 3, "imm": "0xc09e4cdc", "imm2": "0x50c1369d", "rot": 25, "bit": 31, "mask": 2},
|
||||
{"i": 23, "op": "xor", "dst": 6, "src": 4, "src2": 1, "imm": "0xce01abec", "imm2": "0x5ff8c52e", "rot": 30, "bit": 19, "mask": 8},
|
||||
{"i": 24, "op": "or", "dst": 4, "src": 2, "src2": 4, "imm": "0xdacf9e4b", "imm2": "0x086a5684", "rot": 2, "bit": 10, "mask": 2},
|
||||
{"i": 25, "op": "add", "dst": 1, "src": 6, "src2": 3, "imm": "0x53e74799", "imm2": "0xfc2f6d17", "rot": 8, "bit": 7, "mask": 16},
|
||||
{"i": 26, "op": "rotr", "dst": 7, "src": 6, "src2": 7, "imm": "0xc89a75a0", "imm2": "0x67813253", "rot": 24, "bit": 13, "mask": 1},
|
||||
{"i": 27, "op": "add", "dst": 7, "src": 5, "src2": 4, "imm": "0xb8fae90e", "imm2": "0x787048dc", "rot": 26, "bit": 27, "mask": 1},
|
||||
{"i": 28, "op": "xor", "dst": 4, "src": 1, "src2": 0, "imm": "0x4c90c2ea", "imm2": "0x2ce7865b", "rot": 4, "bit": 7, "mask": 2},
|
||||
{"i": 29, "op": "xor", "dst": 6, "src": 1, "src2": 6, "imm": "0x6131ab08", "imm2": "0xd1825676", "rot": 4, "bit": 30, "mask": 2},
|
||||
{"i": 30, "op": "mulhi", "dst": 2, "src": 0, "src2": 1, "imm": "0xbe4a545f", "imm2": "0xdc29e823", "rot": 17, "bit": 5, "mask": 8},
|
||||
{"i": 31, "op": "rotr", "dst": 1, "src": 4, "src2": 1, "imm": "0x4d2f9cdc", "imm2": "0xdee1cc73", "rot": 5, "bit": 16, "mask": 4},
|
||||
{"i": 32, "op": "xor", "dst": 7, "src": 4, "src2": 5, "imm": "0x9d277aaa", "imm2": "0xa902fca5", "rot": 11, "bit": 3, "mask": 1},
|
||||
{"i": 33, "op": "mad", "dst": 3, "src": 2, "src2": 6, "imm": "0x7db09644", "imm2": "0x077bf3d9", "rot": 30, "bit": 19, "mask": 2},
|
||||
{"i": 34, "op": "add", "dst": 3, "src": 1, "src2": 1, "imm": "0xd2dc42eb", "imm2": "0x9ffd510b", "rot": 17, "bit": 12, "mask": 2},
|
||||
{"i": 35, "op": "add", "dst": 4, "src": 7, "src2": 0, "imm": "0xad8eda6f", "imm2": "0xf871ba37", "rot": 24, "bit": 7, "mask": 4},
|
||||
{"i": 36, "op": "add", "dst": 7, "src": 4, "src2": 3, "imm": "0xb0cef8f4", "imm2": "0xd02d30ca", "rot": 22, "bit": 3, "mask": 16},
|
||||
{"i": 37, "op": "mad", "dst": 4, "src": 3, "src2": 2, "imm": "0x6ed99a66", "imm2": "0x83dc97c0", "rot": 4, "bit": 1, "mask": 4},
|
||||
{"i": 38, "op": "mad", "dst": 5, "src": 7, "src2": 7, "imm": "0xe24fdd24", "imm2": "0x360f5c8b", "rot": 19, "bit": 9, "mask": 8},
|
||||
{"i": 39, "op": "add", "dst": 3, "src": 6, "src2": 4, "imm": "0x2bd506e2", "imm2": "0x82e98a22", "rot": 28, "bit": 26, "mask": 1},
|
||||
{"i": 40, "op": "rotl", "dst": 0, "src": 2, "src2": 1, "imm": "0xdb1faed3", "imm2": "0xf968a20d", "rot": 21, "bit": 30, "mask": 16},
|
||||
{"i": 41, "op": "or", "dst": 1, "src": 0, "src2": 1, "imm": "0xe6dfe522", "imm2": "0xb75813da", "rot": 10, "bit": 13, "mask": 1},
|
||||
{"i": 42, "op": "add", "dst": 7, "src": 5, "src2": 7, "imm": "0xc0386e49", "imm2": "0x6435f6d8", "rot": 26, "bit": 28, "mask": 16},
|
||||
{"i": 43, "op": "mulhi", "dst": 1, "src": 0, "src2": 4, "imm": "0xda497f7d", "imm2": "0xc4b748f4", "rot": 30, "bit": 10, "mask": 4},
|
||||
{"i": 44, "op": "add", "dst": 5, "src": 6, "src2": 0, "imm": "0x3a8921a4", "imm2": "0x1f682c68", "rot": 12, "bit": 12, "mask": 1},
|
||||
{"i": 45, "op": "add", "dst": 4, "src": 5, "src2": 6, "imm": "0xeeca4334", "imm2": "0xf246e180", "rot": 26, "bit": 1, "mask": 1},
|
||||
{"i": 46, "op": "xor", "dst": 2, "src": 3, "src2": 5, "imm": "0x052178c3", "imm2": "0xaab31f10", "rot": 28, "bit": 16, "mask": 8},
|
||||
{"i": 47, "op": "xor", "dst": 2, "src": 5, "src2": 0, "imm": "0x91fd7273", "imm2": "0xacc42b07", "rot": 22, "bit": 27, "mask": 4},
|
||||
{"i": 48, "op": "mad", "dst": 7, "src": 4, "src2": 1, "imm": "0x05a2970f", "imm2": "0x14876824", "rot": 10, "bit": 5, "mask": 1},
|
||||
{"i": 49, "op": "mad", "dst": 2, "src": 6, "src2": 1, "imm": "0x4440d6de", "imm2": "0x352023f9", "rot": 10, "bit": 25, "mask": 8},
|
||||
{"i": 50, "op": "mad", "dst": 7, "src": 0, "src2": 7, "imm": "0xa6f82633", "imm2": "0x78465772", "rot": 13, "bit": 8, "mask": 4},
|
||||
{"i": 51, "op": "add", "dst": 5, "src": 6, "src2": 5, "imm": "0x745776d8", "imm2": "0x41bd68cc", "rot": 29, "bit": 26, "mask": 1},
|
||||
{"i": 52, "op": "shfl", "dst": 3, "src": 0, "src2": 6, "imm": "0x8f94ebac", "imm2": "0xe4a208a0", "rot": 5, "bit": 16, "mask": 1},
|
||||
{"i": 53, "op": "or", "dst": 4, "src": 0, "src2": 5, "imm": "0x7b1235f3", "imm2": "0x408b4ea9", "rot": 27, "bit": 19, "mask": 8},
|
||||
{"i": 54, "op": "mad", "dst": 0, "src": 2, "src2": 4, "imm": "0x30114050", "imm2": "0xe71dff8f", "rot": 6, "bit": 30, "mask": 1},
|
||||
{"i": 55, "op": "rotl", "dst": 5, "src": 1, "src2": 7, "imm": "0x69f92198", "imm2": "0x65496f29", "rot": 9, "bit": 20, "mask": 4},
|
||||
{"i": 56, "op": "mul", "dst": 5, "src": 7, "src2": 6, "imm": "0x88b884d4", "imm2": "0x041b67f7", "rot": 25, "bit": 8, "mask": 16},
|
||||
{"i": 57, "op": "rotl", "dst": 7, "src": 5, "src2": 6, "imm": "0x7d1e18bc", "imm2": "0xf54ade27", "rot": 27, "bit": 20, "mask": 1},
|
||||
{"i": 58, "op": "sub", "dst": 7, "src": 6, "src2": 4, "imm": "0xa457bbab", "imm2": "0x4b05eb29", "rot": 6, "bit": 6, "mask": 4},
|
||||
{"i": 59, "op": "rotl", "dst": 2, "src": 7, "src2": 7, "imm": "0x4890bdda", "imm2": "0xadcb3d99", "rot": 22, "bit": 26, "mask": 16},
|
||||
{"i": 60, "op": "shfl", "dst": 7, "src": 1, "src2": 7, "imm": "0xe4da58eb", "imm2": "0x1d50080d", "rot": 22, "bit": 26, "mask": 2},
|
||||
{"i": 61, "op": "add", "dst": 4, "src": 0, "src2": 5, "imm": "0x60d0ff55", "imm2": "0xcba6a161", "rot": 6, "bit": 7, "mask": 8},
|
||||
{"i": 62, "op": "mul", "dst": 1, "src": 5, "src2": 1, "imm": "0x0fdc702e", "imm2": "0x6b1bd03a", "rot": 9, "bit": 23, "mask": 4},
|
||||
{"i": 63, "op": "mulhi", "dst": 7, "src": 6, "src2": 1, "imm": "0xcc44ed62", "imm2": "0x931dd006", "rot": 1, "bit": 0, "mask": 2},
|
||||
{"i": 64, "op": "mul", "dst": 5, "src": 0, "src2": 7, "imm": "0x448c3aba", "imm2": "0x1136818a", "rot": 23, "bit": 7, "mask": 1},
|
||||
{"i": 65, "op": "rotr", "dst": 2, "src": 5, "src2": 1, "imm": "0xb520dfb8", "imm2": "0xdce7d8f4", "rot": 22, "bit": 10, "mask": 8},
|
||||
{"i": 66, "op": "mulhi", "dst": 0, "src": 3, "src2": 4, "imm": "0x19fada48", "imm2": "0x3cc04e0e", "rot": 1, "bit": 24, "mask": 16},
|
||||
{"i": 67, "op": "or", "dst": 0, "src": 7, "src2": 3, "imm": "0x1c5905b4", "imm2": "0xf896dfb7", "rot": 31, "bit": 30, "mask": 8},
|
||||
{"i": 68, "op": "rotl", "dst": 0, "src": 1, "src2": 5, "imm": "0xa28511c1", "imm2": "0xbf649695", "rot": 19, "bit": 27, "mask": 1},
|
||||
{"i": 69, "op": "rotl", "dst": 0, "src": 7, "src2": 5, "imm": "0x22e2fef3", "imm2": "0x468bec9e", "rot": 1, "bit": 28, "mask": 1},
|
||||
{"i": 70, "op": "or", "dst": 2, "src": 3, "src2": 6, "imm": "0xcb542821", "imm2": "0xe084cf11", "rot": 15, "bit": 5, "mask": 8},
|
||||
{"i": 71, "op": "mad", "dst": 3, "src": 7, "src2": 5, "imm": "0xf06b8f6a", "imm2": "0x7382e15f", "rot": 6, "bit": 28, "mask": 16},
|
||||
{"i": 72, "op": "mad", "dst": 7, "src": 0, "src2": 4, "imm": "0x4f64c588", "imm2": "0x47b63ee9", "rot": 8, "bit": 10, "mask": 2},
|
||||
{"i": 73, "op": "or", "dst": 4, "src": 3, "src2": 2, "imm": "0xff138eea", "imm2": "0x6fd90eda", "rot": 23, "bit": 3, "mask": 16},
|
||||
{"i": 74, "op": "xor", "dst": 3, "src": 1, "src2": 0, "imm": "0x6011ece1", "imm2": "0x89579f15", "rot": 11, "bit": 2, "mask": 4},
|
||||
{"i": 75, "op": "add", "dst": 1, "src": 3, "src2": 4, "imm": "0x3f4ffea2", "imm2": "0x8e35924e", "rot": 12, "bit": 15, "mask": 2},
|
||||
{"i": 76, "op": "mul", "dst": 1, "src": 3, "src2": 7, "imm": "0x15efa846", "imm2": "0xa1c971f1", "rot": 14, "bit": 20, "mask": 1},
|
||||
{"i": 77, "op": "mad", "dst": 0, "src": 3, "src2": 2, "imm": "0x8660530a", "imm2": "0xa013524b", "rot": 14, "bit": 12, "mask": 8},
|
||||
{"i": 78, "op": "shfl", "dst": 2, "src": 0, "src2": 0, "imm": "0xcc2e96bd", "imm2": "0xe32c6f87", "rot": 10, "bit": 11, "mask": 16},
|
||||
{"i": 79, "op": "mulhi", "dst": 1, "src": 0, "src2": 6, "imm": "0x61931e3f", "imm2": "0x86383e7f", "rot": 17, "bit": 3, "mask": 1},
|
||||
{"i": 80, "op": "mulhi", "dst": 0, "src": 4, "src2": 7, "imm": "0xd9991afc", "imm2": "0xd00e8678", "rot": 24, "bit": 15, "mask": 16},
|
||||
{"i": 81, "op": "mul", "dst": 4, "src": 2, "src2": 3, "imm": "0x472f70f4", "imm2": "0x9fde4236", "rot": 10, "bit": 5, "mask": 4},
|
||||
{"i": 82, "op": "sub", "dst": 5, "src": 7, "src2": 6, "imm": "0x55a7acb1", "imm2": "0x9fde8df1", "rot": 30, "bit": 29, "mask": 16},
|
||||
{"i": 83, "op": "mad", "dst": 0, "src": 2, "src2": 3, "imm": "0x1c487c40", "imm2": "0xd9d9a1c3", "rot": 9, "bit": 5, "mask": 4},
|
||||
{"i": 84, "op": "sub", "dst": 7, "src": 2, "src2": 4, "imm": "0x515c0a72", "imm2": "0x6df5c83d", "rot": 28, "bit": 21, "mask": 4},
|
||||
{"i": 85, "op": "xor", "dst": 1, "src": 2, "src2": 4, "imm": "0x917f1174", "imm2": "0x8e72e519", "rot": 9, "bit": 18, "mask": 16},
|
||||
{"i": 86, "op": "shfl", "dst": 6, "src": 3, "src2": 3, "imm": "0xda856cc2", "imm2": "0x932ae8d8", "rot": 28, "bit": 1, "mask": 16},
|
||||
{"i": 87, "op": "mad", "dst": 3, "src": 0, "src2": 0, "imm": "0x4ef22321", "imm2": "0x5666ada3", "rot": 4, "bit": 4, "mask": 1},
|
||||
{"i": 88, "op": "rotl", "dst": 0, "src": 1, "src2": 3, "imm": "0xbb3ec00c", "imm2": "0xffe78937", "rot": 20, "bit": 19, "mask": 4},
|
||||
{"i": 89, "op": "sub", "dst": 1, "src": 5, "src2": 4, "imm": "0x8f295708", "imm2": "0x6608577f", "rot": 17, "bit": 18, "mask": 16},
|
||||
{"i": 90, "op": "shfl", "dst": 6, "src": 4, "src2": 5, "imm": "0xd953aa15", "imm2": "0x4ef9f062", "rot": 19, "bit": 9, "mask": 16},
|
||||
{"i": 91, "op": "xor", "dst": 5, "src": 1, "src2": 5, "imm": "0x6c02cde2", "imm2": "0x118dbd2b", "rot": 2, "bit": 20, "mask": 16},
|
||||
{"i": 92, "op": "rotl", "dst": 5, "src": 0, "src2": 5, "imm": "0x9f0b0c61", "imm2": "0xe26dbe7a", "rot": 12, "bit": 17, "mask": 2},
|
||||
{"i": 93, "op": "mad", "dst": 6, "src": 1, "src2": 2, "imm": "0x78a44a91", "imm2": "0x3ce118a9", "rot": 18, "bit": 16, "mask": 16},
|
||||
{"i": 94, "op": "rotr", "dst": 2, "src": 5, "src2": 3, "imm": "0x02295fcb", "imm2": "0xc3353c64", "rot": 5, "bit": 31, "mask": 16},
|
||||
{"i": 95, "op": "or", "dst": 0, "src": 4, "src2": 4, "imm": "0x727120c3", "imm2": "0x528f4416", "rot": 6, "bit": 20, "mask": 8},
|
||||
{"i": 96, "op": "mul", "dst": 7, "src": 3, "src2": 4, "imm": "0x919c42ea", "imm2": "0x87526aa5", "rot": 28, "bit": 1, "mask": 8},
|
||||
{"i": 97, "op": "or", "dst": 5, "src": 0, "src2": 7, "imm": "0x8a0a100c", "imm2": "0xb2c65cc3", "rot": 13, "bit": 7, "mask": 4},
|
||||
{"i": 98, "op": "add", "dst": 3, "src": 1, "src2": 1, "imm": "0xbcfdd8a7", "imm2": "0x386bb642", "rot": 8, "bit": 13, "mask": 8},
|
||||
{"i": 99, "op": "add", "dst": 0, "src": 6, "src2": 7, "imm": "0xc921a021", "imm2": "0xa3962d03", "rot": 31, "bit": 9, "mask": 1},
|
||||
{"i": 100, "op": "rotr", "dst": 1, "src": 0, "src2": 3, "imm": "0x918c6f69", "imm2": "0x1a125801", "rot": 18, "bit": 29, "mask": 16},
|
||||
{"i": 101, "op": "add", "dst": 2, "src": 4, "src2": 1, "imm": "0x70da4067", "imm2": "0xc5990c5c", "rot": 3, "bit": 18, "mask": 4},
|
||||
{"i": 102, "op": "rotl", "dst": 2, "src": 7, "src2": 0, "imm": "0x9048d081", "imm2": "0xc25a9f56", "rot": 3, "bit": 25, "mask": 16},
|
||||
{"i": 103, "op": "shfl", "dst": 1, "src": 4, "src2": 7, "imm": "0x70ca1427", "imm2": "0x6632093d", "rot": 18, "bit": 28, "mask": 4},
|
||||
{"i": 104, "op": "xor", "dst": 0, "src": 2, "src2": 0, "imm": "0x0637204c", "imm2": "0x3b1b7f94", "rot": 1, "bit": 12, "mask": 2},
|
||||
{"i": 105, "op": "add", "dst": 4, "src": 3, "src2": 5, "imm": "0x1a5880cb", "imm2": "0xb15aec75", "rot": 31, "bit": 28, "mask": 1},
|
||||
{"i": 106, "op": "shfl", "dst": 4, "src": 0, "src2": 4, "imm": "0x7c22cb00", "imm2": "0xc4df7949", "rot": 3, "bit": 15, "mask": 2},
|
||||
{"i": 107, "op": "shfl", "dst": 7, "src": 0, "src2": 0, "imm": "0xd43c76ca", "imm2": "0x8d5b9751", "rot": 31, "bit": 28, "mask": 4},
|
||||
{"i": 108, "op": "mad", "dst": 1, "src": 7, "src2": 3, "imm": "0x69879993", "imm2": "0x3c4e18b0", "rot": 19, "bit": 25, "mask": 4},
|
||||
{"i": 109, "op": "mad", "dst": 0, "src": 2, "src2": 3, "imm": "0x8ce3a4a6", "imm2": "0x1b3f0b74", "rot": 19, "bit": 8, "mask": 2},
|
||||
{"i": 110, "op": "or", "dst": 2, "src": 4, "src2": 5, "imm": "0xcdbdac98", "imm2": "0x8d0e9f92", "rot": 19, "bit": 13, "mask": 1},
|
||||
{"i": 111, "op": "shfl", "dst": 6, "src": 1, "src2": 1, "imm": "0xe74157f6", "imm2": "0xf92dead6", "rot": 4, "bit": 14, "mask": 1},
|
||||
{"i": 112, "op": "mad", "dst": 2, "src": 5, "src2": 3, "imm": "0x200831bc", "imm2": "0x1ee98048", "rot": 29, "bit": 14, "mask": 1},
|
||||
{"i": 113, "op": "mulhi", "dst": 4, "src": 0, "src2": 7, "imm": "0xd6827ead", "imm2": "0xc0ee1933", "rot": 26, "bit": 23, "mask": 16},
|
||||
{"i": 114, "op": "rotl", "dst": 7, "src": 2, "src2": 2, "imm": "0x6ae1be2b", "imm2": "0x83216a4e", "rot": 29, "bit": 25, "mask": 16},
|
||||
{"i": 115, "op": "sub", "dst": 3, "src": 7, "src2": 1, "imm": "0xa1d1646e", "imm2": "0x6fa998ce", "rot": 15, "bit": 29, "mask": 2},
|
||||
{"i": 116, "op": "mulhi", "dst": 0, "src": 1, "src2": 4, "imm": "0xd7921c15", "imm2": "0x8b2345a7", "rot": 10, "bit": 19, "mask": 4},
|
||||
{"i": 117, "op": "rotl", "dst": 4, "src": 0, "src2": 4, "imm": "0xc1b1b0ab", "imm2": "0xd19b3f39", "rot": 12, "bit": 16, "mask": 1},
|
||||
{"i": 118, "op": "add", "dst": 1, "src": 4, "src2": 5, "imm": "0x24494ad3", "imm2": "0xc5e98ee5", "rot": 22, "bit": 7, "mask": 16},
|
||||
{"i": 119, "op": "add", "dst": 0, "src": 4, "src2": 6, "imm": "0x1b2d1c83", "imm2": "0xa97bc949", "rot": 8, "bit": 4, "mask": 16},
|
||||
{"i": 120, "op": "shfl", "dst": 5, "src": 4, "src2": 7, "imm": "0xb93015ff", "imm2": "0xc22366cd", "rot": 23, "bit": 16, "mask": 4},
|
||||
{"i": 121, "op": "shfl", "dst": 3, "src": 2, "src2": 5, "imm": "0xc0aaacdb", "imm2": "0xaaa7d229", "rot": 31, "bit": 29, "mask": 8},
|
||||
{"i": 122, "op": "rotl", "dst": 7, "src": 1, "src2": 3, "imm": "0xcce6a73f", "imm2": "0x517b6b86", "rot": 3, "bit": 16, "mask": 8},
|
||||
{"i": 123, "op": "rotr", "dst": 4, "src": 6, "src2": 0, "imm": "0x8e156aca", "imm2": "0x0a894b3e", "rot": 2, "bit": 12, "mask": 2},
|
||||
{"i": 124, "op": "shfl", "dst": 6, "src": 5, "src2": 2, "imm": "0xd88c8446", "imm2": "0xf49fedf3", "rot": 3, "bit": 23, "mask": 1},
|
||||
{"i": 125, "op": "rotl", "dst": 5, "src": 2, "src2": 0, "imm": "0xd10b27cf", "imm2": "0x2dde1faa", "rot": 21, "bit": 7, "mask": 2},
|
||||
{"i": 126, "op": "mad", "dst": 0, "src": 3, "src2": 1, "imm": "0x619e1818", "imm2": "0x62add310", "rot": 26, "bit": 1, "mask": 4},
|
||||
{"i": 127, "op": "add", "dst": 0, "src": 3, "src2": 6, "imm": "0xff14c08d", "imm2": "0xa7f6bb54", "rot": 25, "bit": 24, "mask": 4},
|
||||
{"i": 128, "op": "xor", "dst": 7, "src": 4, "src2": 4, "imm": "0x34224086", "imm2": "0x5e9f858e", "rot": 4, "bit": 16, "mask": 8},
|
||||
{"i": 129, "op": "mulhi", "dst": 3, "src": 5, "src2": 6, "imm": "0x89c3cbdd", "imm2": "0x4e80334c", "rot": 29, "bit": 4, "mask": 1},
|
||||
{"i": 130, "op": "shfl", "dst": 5, "src": 7, "src2": 1, "imm": "0x37d4c3e2", "imm2": "0xbd54378c", "rot": 14, "bit": 26, "mask": 8},
|
||||
{"i": 131, "op": "mulhi", "dst": 6, "src": 4, "src2": 3, "imm": "0x4f5fac7c", "imm2": "0x74070028", "rot": 5, "bit": 25, "mask": 8},
|
||||
{"i": 132, "op": "xor", "dst": 5, "src": 3, "src2": 0, "imm": "0x690eb003", "imm2": "0x8516a581", "rot": 20, "bit": 21, "mask": 2},
|
||||
{"i": 133, "op": "rotl", "dst": 0, "src": 5, "src2": 7, "imm": "0x42b31885", "imm2": "0x08249acb", "rot": 19, "bit": 6, "mask": 8},
|
||||
{"i": 134, "op": "rotr", "dst": 4, "src": 3, "src2": 3, "imm": "0x07d729ed", "imm2": "0xf564da8a", "rot": 17, "bit": 9, "mask": 16},
|
||||
{"i": 135, "op": "shfl", "dst": 1, "src": 2, "src2": 0, "imm": "0x19130ef2", "imm2": "0xf59815e5", "rot": 17, "bit": 12, "mask": 16},
|
||||
{"i": 136, "op": "sub", "dst": 5, "src": 2, "src2": 5, "imm": "0xdeb56168", "imm2": "0x871104c4", "rot": 13, "bit": 17, "mask": 16},
|
||||
{"i": 137, "op": "xor", "dst": 3, "src": 5, "src2": 4, "imm": "0x884fc7da", "imm2": "0x83a4a5bf", "rot": 23, "bit": 12, "mask": 4},
|
||||
{"i": 138, "op": "sub", "dst": 0, "src": 2, "src2": 6, "imm": "0x3e67a6fd", "imm2": "0x844d2039", "rot": 3, "bit": 24, "mask": 4},
|
||||
{"i": 139, "op": "mul", "dst": 3, "src": 7, "src2": 7, "imm": "0x965933b4", "imm2": "0xf37ef93d", "rot": 6, "bit": 9, "mask": 4},
|
||||
{"i": 140, "op": "add", "dst": 5, "src": 3, "src2": 3, "imm": "0x82ab98f4", "imm2": "0x2d465ca8", "rot": 11, "bit": 24, "mask": 1},
|
||||
{"i": 141, "op": "add", "dst": 4, "src": 1, "src2": 7, "imm": "0x80594390", "imm2": "0x237a9d8e", "rot": 14, "bit": 11, "mask": 8},
|
||||
{"i": 142, "op": "mad", "dst": 3, "src": 7, "src2": 0, "imm": "0xd37861a1", "imm2": "0x2fe70eae", "rot": 2, "bit": 11, "mask": 16},
|
||||
{"i": 143, "op": "or", "dst": 0, "src": 3, "src2": 5, "imm": "0x1a018e2c", "imm2": "0x301aa5f6", "rot": 15, "bit": 26, "mask": 1},
|
||||
{"i": 144, "op": "mulhi", "dst": 0, "src": 1, "src2": 5, "imm": "0x69d22dfc", "imm2": "0x4a90cb1e", "rot": 21, "bit": 9, "mask": 16},
|
||||
{"i": 145, "op": "xor", "dst": 6, "src": 2, "src2": 0, "imm": "0xf3f41f47", "imm2": "0xdf8d33e9", "rot": 22, "bit": 0, "mask": 8},
|
||||
{"i": 146, "op": "shfl", "dst": 7, "src": 1, "src2": 1, "imm": "0xf8a52377", "imm2": "0x0613ef41", "rot": 24, "bit": 12, "mask": 16},
|
||||
{"i": 147, "op": "add", "dst": 0, "src": 1, "src2": 3, "imm": "0x07f15148", "imm2": "0xe68cf57e", "rot": 2, "bit": 18, "mask": 4},
|
||||
{"i": 148, "op": "or", "dst": 7, "src": 3, "src2": 0, "imm": "0xbfd39b49", "imm2": "0xc7f6e97c", "rot": 13, "bit": 27, "mask": 2},
|
||||
{"i": 149, "op": "xor", "dst": 4, "src": 2, "src2": 7, "imm": "0x47409e43", "imm2": "0x04ef7db9", "rot": 8, "bit": 6, "mask": 4},
|
||||
{"i": 150, "op": "xor", "dst": 7, "src": 1, "src2": 1, "imm": "0x5d42a156", "imm2": "0xc2f77c48", "rot": 27, "bit": 17, "mask": 4},
|
||||
{"i": 151, "op": "rotr", "dst": 5, "src": 2, "src2": 7, "imm": "0x7c170030", "imm2": "0xa3102916", "rot": 5, "bit": 6, "mask": 1},
|
||||
{"i": 152, "op": "mul", "dst": 0, "src": 7, "src2": 5, "imm": "0x16d009af", "imm2": "0x2a89533b", "rot": 26, "bit": 8, "mask": 4},
|
||||
{"i": 153, "op": "rotl", "dst": 4, "src": 5, "src2": 0, "imm": "0x6e54eb71", "imm2": "0xe2d27256", "rot": 16, "bit": 18, "mask": 1},
|
||||
{"i": 154, "op": "rotr", "dst": 6, "src": 4, "src2": 5, "imm": "0x128f24a4", "imm2": "0x59888463", "rot": 15, "bit": 17, "mask": 16},
|
||||
{"i": 155, "op": "xor", "dst": 6, "src": 1, "src2": 2, "imm": "0x34f276ff", "imm2": "0xef5086b4", "rot": 2, "bit": 10, "mask": 4},
|
||||
{"i": 156, "op": "or", "dst": 6, "src": 3, "src2": 4, "imm": "0x16cc2c2f", "imm2": "0xfae5a09b", "rot": 12, "bit": 31, "mask": 4},
|
||||
{"i": 157, "op": "add", "dst": 1, "src": 6, "src2": 7, "imm": "0x14878c5a", "imm2": "0x84c2a09d", "rot": 30, "bit": 19, "mask": 8},
|
||||
{"i": 158, "op": "sub", "dst": 4, "src": 1, "src2": 0, "imm": "0xfe18e4b0", "imm2": "0x239df0cc", "rot": 14, "bit": 20, "mask": 2},
|
||||
{"i": 159, "op": "mad", "dst": 4, "src": 6, "src2": 7, "imm": "0xbdfcc4e0", "imm2": "0x0870dee8", "rot": 1, "bit": 7, "mask": 8},
|
||||
{"i": 160, "op": "mul", "dst": 1, "src": 5, "src2": 1, "imm": "0xc837b43d", "imm2": "0x027ac950", "rot": 4, "bit": 22, "mask": 16},
|
||||
{"i": 161, "op": "or", "dst": 4, "src": 0, "src2": 2, "imm": "0x9c28f8f6", "imm2": "0x3db71fe6", "rot": 11, "bit": 6, "mask": 16},
|
||||
{"i": 162, "op": "rotl", "dst": 7, "src": 3, "src2": 6, "imm": "0xb6e56f55", "imm2": "0x5fcbd27c", "rot": 21, "bit": 10, "mask": 1},
|
||||
{"i": 163, "op": "sub", "dst": 0, "src": 1, "src2": 4, "imm": "0x8d95ae70", "imm2": "0xd4f10284", "rot": 15, "bit": 12, "mask": 8},
|
||||
{"i": 164, "op": "mul", "dst": 1, "src": 0, "src2": 6, "imm": "0xa9bdf99e", "imm2": "0x1f3dceb6", "rot": 21, "bit": 20, "mask": 16},
|
||||
{"i": 165, "op": "sub", "dst": 2, "src": 1, "src2": 4, "imm": "0xda245058", "imm2": "0x2038e7f6", "rot": 30, "bit": 11, "mask": 2},
|
||||
{"i": 166, "op": "shfl", "dst": 0, "src": 6, "src2": 3, "imm": "0xbbc8eee8", "imm2": "0xa73472fb", "rot": 23, "bit": 4, "mask": 2},
|
||||
{"i": 167, "op": "sub", "dst": 0, "src": 4, "src2": 2, "imm": "0x43540d69", "imm2": "0xb3f116ca", "rot": 28, "bit": 2, "mask": 2},
|
||||
{"i": 168, "op": "rotr", "dst": 1, "src": 6, "src2": 6, "imm": "0x57cea752", "imm2": "0xc1136e71", "rot": 25, "bit": 3, "mask": 8},
|
||||
{"i": 169, "op": "rotr", "dst": 7, "src": 1, "src2": 5, "imm": "0x4550a465", "imm2": "0x14362935", "rot": 25, "bit": 9, "mask": 16},
|
||||
{"i": 170, "op": "mad", "dst": 3, "src": 1, "src2": 3, "imm": "0x8c961d80", "imm2": "0x7de2e1db", "rot": 10, "bit": 6, "mask": 4},
|
||||
{"i": 171, "op": "shfl", "dst": 2, "src": 0, "src2": 0, "imm": "0xda047fca", "imm2": "0x459e17ab", "rot": 22, "bit": 26, "mask": 4},
|
||||
{"i": 172, "op": "sub", "dst": 2, "src": 1, "src2": 2, "imm": "0xbcc341fd", "imm2": "0x17a3b229", "rot": 8, "bit": 25, "mask": 8},
|
||||
{"i": 173, "op": "xor", "dst": 7, "src": 6, "src2": 7, "imm": "0x2302b424", "imm2": "0x78780982", "rot": 23, "bit": 7, "mask": 8},
|
||||
{"i": 174, "op": "rotr", "dst": 4, "src": 1, "src2": 2, "imm": "0xcad6fb5f", "imm2": "0x1807628e", "rot": 8, "bit": 21, "mask": 4},
|
||||
{"i": 175, "op": "shfl", "dst": 4, "src": 7, "src2": 7, "imm": "0xf5057e8e", "imm2": "0xd14d6cc5", "rot": 3, "bit": 12, "mask": 2},
|
||||
{"i": 176, "op": "mul", "dst": 6, "src": 2, "src2": 1, "imm": "0xc2926356", "imm2": "0xae74dc32", "rot": 1, "bit": 21, "mask": 8},
|
||||
{"i": 177, "op": "shfl", "dst": 0, "src": 2, "src2": 7, "imm": "0x8d94d98e", "imm2": "0x7bd6f4dc", "rot": 27, "bit": 14, "mask": 4},
|
||||
{"i": 178, "op": "add", "dst": 6, "src": 4, "src2": 6, "imm": "0x4fa43965", "imm2": "0x509871e6", "rot": 28, "bit": 21, "mask": 4},
|
||||
{"i": 179, "op": "xor", "dst": 0, "src": 6, "src2": 6, "imm": "0xe278fe6c", "imm2": "0x75d8a63f", "rot": 21, "bit": 15, "mask": 2},
|
||||
{"i": 180, "op": "add", "dst": 5, "src": 7, "src2": 7, "imm": "0xfb3c4bd8", "imm2": "0xb95b8a53", "rot": 22, "bit": 0, "mask": 4},
|
||||
{"i": 181, "op": "mul", "dst": 6, "src": 3, "src2": 2, "imm": "0x5a1347ec", "imm2": "0x3ccc5389", "rot": 27, "bit": 26, "mask": 4},
|
||||
{"i": 182, "op": "rotl", "dst": 7, "src": 4, "src2": 7, "imm": "0xdc6db2f7", "imm2": "0x7cdff0c6", "rot": 1, "bit": 2, "mask": 4},
|
||||
{"i": 183, "op": "add", "dst": 0, "src": 7, "src2": 7, "imm": "0xc448a197", "imm2": "0x7047b7cf", "rot": 24, "bit": 11, "mask": 8},
|
||||
{"i": 184, "op": "mul", "dst": 3, "src": 5, "src2": 0, "imm": "0xa5c0e492", "imm2": "0xaf7d8a86", "rot": 11, "bit": 16, "mask": 8},
|
||||
{"i": 185, "op": "shfl", "dst": 5, "src": 7, "src2": 1, "imm": "0x5965794b", "imm2": "0x36ddca6b", "rot": 7, "bit": 25, "mask": 1},
|
||||
{"i": 186, "op": "mul", "dst": 0, "src": 2, "src2": 5, "imm": "0xdfdbcc0f", "imm2": "0xac7b7091", "rot": 11, "bit": 31, "mask": 8},
|
||||
{"i": 187, "op": "mulhi", "dst": 4, "src": 7, "src2": 5, "imm": "0x3d8999e9", "imm2": "0xf1848db8", "rot": 15, "bit": 21, "mask": 8},
|
||||
{"i": 188, "op": "mad", "dst": 2, "src": 7, "src2": 2, "imm": "0xda3af48c", "imm2": "0x8e33a97c", "rot": 6, "bit": 30, "mask": 4},
|
||||
{"i": 189, "op": "sub", "dst": 7, "src": 2, "src2": 4, "imm": "0xa32da8c1", "imm2": "0xf7486d06", "rot": 25, "bit": 16, "mask": 4},
|
||||
{"i": 190, "op": "add", "dst": 2, "src": 5, "src2": 1, "imm": "0x2fceaf49", "imm2": "0x5caccc5d", "rot": 24, "bit": 19, "mask": 2},
|
||||
{"i": 191, "op": "shfl", "dst": 1, "src": 2, "src2": 2, "imm": "0x2a71f5e6", "imm2": "0xb9b51e89", "rot": 31, "bit": 25, "mask": 1},
|
||||
{"i": 192, "op": "add", "dst": 7, "src": 1, "src2": 7, "imm": "0xba0cee62", "imm2": "0x857bb5fe", "rot": 4, "bit": 20, "mask": 2},
|
||||
{"i": 193, "op": "rotl", "dst": 6, "src": 3, "src2": 3, "imm": "0xc7fe35cf", "imm2": "0x515c1201", "rot": 22, "bit": 7, "mask": 16},
|
||||
{"i": 194, "op": "xor", "dst": 3, "src": 7, "src2": 4, "imm": "0xe496d1f1", "imm2": "0x11046e3e", "rot": 16, "bit": 9, "mask": 4},
|
||||
{"i": 195, "op": "mul", "dst": 7, "src": 0, "src2": 2, "imm": "0xc7d5c4c2", "imm2": "0x077e940b", "rot": 23, "bit": 29, "mask": 4},
|
||||
{"i": 196, "op": "xor", "dst": 3, "src": 5, "src2": 3, "imm": "0xbe6db14a", "imm2": "0xf736da5a", "rot": 19, "bit": 6, "mask": 2},
|
||||
{"i": 197, "op": "add", "dst": 5, "src": 1, "src2": 5, "imm": "0x8519428c", "imm2": "0xeae84577", "rot": 18, "bit": 29, "mask": 1},
|
||||
{"i": 198, "op": "mad", "dst": 7, "src": 4, "src2": 4, "imm": "0xb272a029", "imm2": "0x071f6a5c", "rot": 5, "bit": 28, "mask": 8},
|
||||
{"i": 199, "op": "add", "dst": 3, "src": 6, "src2": 3, "imm": "0x468639d3", "imm2": "0x0bfdbfa1", "rot": 20, "bit": 8, "mask": 16},
|
||||
{"i": 200, "op": "mad", "dst": 5, "src": 3, "src2": 4, "imm": "0xffba18b7", "imm2": "0xbdcf9e81", "rot": 13, "bit": 31, "mask": 8},
|
||||
{"i": 201, "op": "shfl", "dst": 2, "src": 7, "src2": 7, "imm": "0xb7533191", "imm2": "0x8d58cc8c", "rot": 7, "bit": 29, "mask": 4},
|
||||
{"i": 202, "op": "add", "dst": 7, "src": 4, "src2": 6, "imm": "0x18ec9a69", "imm2": "0x53ee8e50", "rot": 12, "bit": 23, "mask": 16},
|
||||
{"i": 203, "op": "xor", "dst": 6, "src": 0, "src2": 5, "imm": "0x12011ff1", "imm2": "0xfbb57ac6", "rot": 16, "bit": 25, "mask": 16},
|
||||
{"i": 204, "op": "add", "dst": 4, "src": 5, "src2": 3, "imm": "0x3e81485b", "imm2": "0xac63376e", "rot": 1, "bit": 4, "mask": 1},
|
||||
{"i": 205, "op": "mad", "dst": 7, "src": 1, "src2": 4, "imm": "0xeb9eee5c", "imm2": "0x3f1daecd", "rot": 19, "bit": 29, "mask": 4},
|
||||
{"i": 206, "op": "shfl", "dst": 1, "src": 5, "src2": 2, "imm": "0x4fc50d06", "imm2": "0xb5533b31", "rot": 22, "bit": 27, "mask": 16},
|
||||
{"i": 207, "op": "mul", "dst": 4, "src": 0, "src2": 5, "imm": "0x202d6001", "imm2": "0x50388f85", "rot": 5, "bit": 18, "mask": 16},
|
||||
{"i": 208, "op": "or", "dst": 1, "src": 6, "src2": 7, "imm": "0x8743c0d4", "imm2": "0x9df69539", "rot": 14, "bit": 16, "mask": 1},
|
||||
{"i": 209, "op": "add", "dst": 6, "src": 4, "src2": 6, "imm": "0x66ccb75d", "imm2": "0x5b745519", "rot": 28, "bit": 28, "mask": 8},
|
||||
{"i": 210, "op": "rotr", "dst": 1, "src": 3, "src2": 1, "imm": "0xe9933132", "imm2": "0x6702fe85", "rot": 16, "bit": 19, "mask": 1},
|
||||
{"i": 211, "op": "sub", "dst": 5, "src": 4, "src2": 4, "imm": "0xfe04b942", "imm2": "0x40dd2ce8", "rot": 24, "bit": 8, "mask": 4},
|
||||
{"i": 212, "op": "xor", "dst": 4, "src": 7, "src2": 7, "imm": "0x10827878", "imm2": "0xef0de8fc", "rot": 17, "bit": 11, "mask": 8},
|
||||
{"i": 213, "op": "add", "dst": 1, "src": 7, "src2": 1, "imm": "0x0ce0553d", "imm2": "0x1c3dccf7", "rot": 26, "bit": 8, "mask": 1},
|
||||
{"i": 214, "op": "mulhi", "dst": 0, "src": 1, "src2": 1, "imm": "0xa71d7581", "imm2": "0xe567c74c", "rot": 2, "bit": 29, "mask": 2},
|
||||
{"i": 215, "op": "add", "dst": 3, "src": 6, "src2": 2, "imm": "0x96bfec89", "imm2": "0xfadc9205", "rot": 30, "bit": 9, "mask": 16},
|
||||
{"i": 216, "op": "rotl", "dst": 0, "src": 2, "src2": 5, "imm": "0x947ca474", "imm2": "0xd2a1b650", "rot": 4, "bit": 1, "mask": 1},
|
||||
{"i": 217, "op": "xor", "dst": 6, "src": 4, "src2": 5, "imm": "0x0b7617de", "imm2": "0xc6c0a9a2", "rot": 31, "bit": 15, "mask": 2},
|
||||
{"i": 218, "op": "shfl", "dst": 6, "src": 7, "src2": 1, "imm": "0x738f0a8f", "imm2": "0xfc868560", "rot": 2, "bit": 20, "mask": 2},
|
||||
{"i": 219, "op": "mad", "dst": 7, "src": 0, "src2": 1, "imm": "0x2960f2bc", "imm2": "0x7e9ef9b7", "rot": 21, "bit": 4, "mask": 16},
|
||||
{"i": 220, "op": "mul", "dst": 4, "src": 0, "src2": 2, "imm": "0x6014d800", "imm2": "0xd19ac7d1", "rot": 1, "bit": 13, "mask": 1},
|
||||
{"i": 221, "op": "mad", "dst": 2, "src": 6, "src2": 0, "imm": "0x4b558fc2", "imm2": "0x2d48bd3c", "rot": 18, "bit": 31, "mask": 2},
|
||||
{"i": 222, "op": "xor", "dst": 5, "src": 4, "src2": 6, "imm": "0x0260f3ab", "imm2": "0x8fce22f1", "rot": 21, "bit": 26, "mask": 16},
|
||||
{"i": 223, "op": "mul", "dst": 0, "src": 5, "src2": 0, "imm": "0xe07ab58f", "imm2": "0x5349a41d", "rot": 20, "bit": 3, "mask": 1},
|
||||
{"i": 224, "op": "add", "dst": 2, "src": 0, "src2": 5, "imm": "0x61495681", "imm2": "0x66b15c04", "rot": 24, "bit": 9, "mask": 16},
|
||||
{"i": 225, "op": "rotr", "dst": 3, "src": 2, "src2": 5, "imm": "0x291268bd", "imm2": "0xf046aa79", "rot": 27, "bit": 15, "mask": 4},
|
||||
{"i": 226, "op": "mad", "dst": 2, "src": 3, "src2": 3, "imm": "0xfd6dfa6c", "imm2": "0x2c0db93d", "rot": 29, "bit": 2, "mask": 16},
|
||||
{"i": 227, "op": "xor", "dst": 6, "src": 0, "src2": 4, "imm": "0xd4ae1ee2", "imm2": "0xe6a67972", "rot": 26, "bit": 1, "mask": 4},
|
||||
{"i": 228, "op": "rotl", "dst": 4, "src": 7, "src2": 3, "imm": "0xf6627a9d", "imm2": "0x99078da3", "rot": 30, "bit": 5, "mask": 16},
|
||||
{"i": 229, "op": "add", "dst": 2, "src": 3, "src2": 7, "imm": "0x70d05c34", "imm2": "0x25bc17c2", "rot": 20, "bit": 29, "mask": 4},
|
||||
{"i": 230, "op": "rotr", "dst": 1, "src": 5, "src2": 2, "imm": "0xef874bea", "imm2": "0x3bfbfecc", "rot": 17, "bit": 2, "mask": 2},
|
||||
{"i": 231, "op": "add", "dst": 1, "src": 4, "src2": 5, "imm": "0x32a28384", "imm2": "0x2d8728f3", "rot": 1, "bit": 8, "mask": 8},
|
||||
{"i": 232, "op": "rotl", "dst": 1, "src": 6, "src2": 3, "imm": "0x877c54ab", "imm2": "0x1d5e5014", "rot": 6, "bit": 24, "mask": 2},
|
||||
{"i": 233, "op": "add", "dst": 2, "src": 0, "src2": 5, "imm": "0xcf0949f1", "imm2": "0x61e81fe6", "rot": 2, "bit": 21, "mask": 16},
|
||||
{"i": 234, "op": "rotl", "dst": 4, "src": 5, "src2": 5, "imm": "0x5017c32c", "imm2": "0x1ab34dfe", "rot": 2, "bit": 1, "mask": 2},
|
||||
{"i": 235, "op": "sub", "dst": 2, "src": 6, "src2": 2, "imm": "0xadbdb811", "imm2": "0x2c2d4f2b", "rot": 30, "bit": 29, "mask": 8},
|
||||
{"i": 236, "op": "or", "dst": 6, "src": 3, "src2": 4, "imm": "0x700e204f", "imm2": "0x27294e7b", "rot": 13, "bit": 0, "mask": 4},
|
||||
{"i": 237, "op": "shfl", "dst": 7, "src": 5, "src2": 4, "imm": "0xc39f71ba", "imm2": "0x9a90263d", "rot": 28, "bit": 9, "mask": 4},
|
||||
{"i": 238, "op": "xor", "dst": 1, "src": 3, "src2": 6, "imm": "0x9f84a5b3", "imm2": "0x95691d15", "rot": 14, "bit": 22, "mask": 1},
|
||||
{"i": 239, "op": "rotl", "dst": 2, "src": 4, "src2": 0, "imm": "0xa3b79ed5", "imm2": "0x05f568bc", "rot": 1, "bit": 15, "mask": 4},
|
||||
{"i": 240, "op": "rotr", "dst": 5, "src": 2, "src2": 7, "imm": "0xaaabbb0d", "imm2": "0x71ce4cd6", "rot": 2, "bit": 26, "mask": 4},
|
||||
{"i": 241, "op": "rotl", "dst": 4, "src": 1, "src2": 7, "imm": "0x8146fbd3", "imm2": "0xde65fc31", "rot": 10, "bit": 9, "mask": 4},
|
||||
{"i": 242, "op": "add", "dst": 6, "src": 2, "src2": 7, "imm": "0xe204fb50", "imm2": "0x0a67565d", "rot": 5, "bit": 16, "mask": 1},
|
||||
{"i": 243, "op": "shfl", "dst": 1, "src": 3, "src2": 7, "imm": "0xcb7a3c1b", "imm2": "0x863f002c", "rot": 18, "bit": 1, "mask": 2},
|
||||
{"i": 244, "op": "shfl", "dst": 0, "src": 1, "src2": 3, "imm": "0x92eb030c", "imm2": "0x3cff4c6d", "rot": 1, "bit": 22, "mask": 16},
|
||||
{"i": 245, "op": "shfl", "dst": 5, "src": 7, "src2": 6, "imm": "0x046c969c", "imm2": "0xd21d7e76", "rot": 5, "bit": 31, "mask": 2},
|
||||
{"i": 246, "op": "xor", "dst": 4, "src": 5, "src2": 1, "imm": "0x2e3015f0", "imm2": "0x1b305a5c", "rot": 22, "bit": 16, "mask": 8},
|
||||
{"i": 247, "op": "xor", "dst": 7, "src": 2, "src2": 0, "imm": "0x8238cf71", "imm2": "0xb7874d94", "rot": 9, "bit": 1, "mask": 2},
|
||||
{"i": 248, "op": "shfl", "dst": 0, "src": 6, "src2": 7, "imm": "0x437cbc99", "imm2": "0x7fb41fed", "rot": 17, "bit": 1, "mask": 16},
|
||||
{"i": 249, "op": "mulhi", "dst": 1, "src": 3, "src2": 2, "imm": "0xeb293d88", "imm2": "0xb92f5968", "rot": 3, "bit": 20, "mask": 2},
|
||||
{"i": 250, "op": "add", "dst": 7, "src": 5, "src2": 3, "imm": "0x3d6daffd", "imm2": "0x5a79fd6f", "rot": 6, "bit": 31, "mask": 2},
|
||||
{"i": 251, "op": "add", "dst": 6, "src": 0, "src2": 3, "imm": "0x84334aea", "imm2": "0x6beb28c2", "rot": 27, "bit": 28, "mask": 2},
|
||||
{"i": 252, "op": "sub", "dst": 7, "src": 1, "src2": 6, "imm": "0xe6177638", "imm2": "0x814d3fc5", "rot": 25, "bit": 2, "mask": 2},
|
||||
{"i": 253, "op": "rotr", "dst": 5, "src": 2, "src2": 7, "imm": "0x95560bac", "imm2": "0xbf8b01a9", "rot": 14, "bit": 5, "mask": 16},
|
||||
{"i": 254, "op": "mulhi", "dst": 2, "src": 5, "src2": 1, "imm": "0x318956e5", "imm2": "0x7c20b373", "rot": 13, "bit": 30, "mask": 2},
|
||||
{"i": 255, "op": "mul", "dst": 6, "src": 3, "src2": 3, "imm": "0xe1998e49", "imm2": "0x59353e5a", "rot": 15, "bit": 2, "mask": 4}
|
||||
]},
|
||||
"instructions": [
|
||||
{"i": 0, "op": "sub", "dst": 6, "src": 3, "src2": 3, "imm": "0x77dfbe60", "imm2": "0x404c8b9c", "rot": 24, "bit": 8, "mask": 2, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 1, "op": "or", "dst": 2, "src": 6, "src2": 7, "imm": "0xb2e79058", "imm2": "0xc2485073", "rot": 21, "bit": 23, "mask": 8, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 2, "op": "add", "dst": 4, "src": 0, "src2": 6, "imm": "0x1a579b38", "imm2": "0x7731324b", "rot": 6, "bit": 4, "mask": 2, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 3, "op": "load", "dst": 3, "src": 6, "src2": 0, "imm": "0x683f5bc5", "imm2": "0xc8a218fe", "rot": 25, "bit": 4, "mask": 1, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 4, "op": "shfl", "dst": 0, "src": 6, "src2": 4, "imm": "0x4d7cdcbb", "imm2": "0x1a5bf1a4", "rot": 8, "bit": 28, "mask": 1, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 5, "op": "xor", "dst": 4, "src": 1, "src2": 0, "imm": "0xa2b86827", "imm2": "0x94a44439", "rot": 21, "bit": 22, "mask": 2, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 6, "op": "mad", "dst": 3, "src": 5, "src2": 3, "imm": "0x58a4ea3f", "imm2": "0x467879fa", "rot": 6, "bit": 31, "mask": 8, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 7, "op": "mul", "dst": 1, "src": 6, "src2": 1, "imm": "0x31fcc19c", "imm2": "0x176cb88f", "rot": 1, "bit": 11, "mask": 1, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 8, "op": "load", "dst": 2, "src": 0, "src2": 3, "imm": "0x89b12386", "imm2": "0x661ed5e5", "rot": 11, "bit": 13, "mask": 2, "width": 1, "win": 2, "off": 2},
|
||||
{"i": 9, "op": "xor", "dst": 0, "src": 4, "src2": 1, "imm": "0x69150105", "imm2": "0x8e1f04b9", "rot": 3, "bit": 30, "mask": 2, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 10, "op": "add", "dst": 0, "src": 7, "src2": 5, "imm": "0xee083919", "imm2": "0xb10fcef8", "rot": 9, "bit": 30, "mask": 16, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 11, "op": "mulhi", "dst": 0, "src": 2, "src2": 3, "imm": "0x99ac7c83", "imm2": "0x34eb2845", "rot": 18, "bit": 12, "mask": 16, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 12, "op": "xor", "dst": 5, "src": 4, "src2": 0, "imm": "0x97cfc887", "imm2": "0x0458f053", "rot": 28, "bit": 24, "mask": 2, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 13, "op": "mad", "dst": 7, "src": 3, "src2": 0, "imm": "0x503e87b4", "imm2": "0x2c5d2617", "rot": 3, "bit": 15, "mask": 1, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 14, "op": "load", "dst": 3, "src": 4, "src2": 4, "imm": "0xfae98035", "imm2": "0x6c127281", "rot": 19, "bit": 29, "mask": 4, "width": 1, "win": 1, "off": 0},
|
||||
{"i": 15, "op": "load", "dst": 2, "src": 5, "src2": 2, "imm": "0x67601ed5", "imm2": "0x3c546ed1", "rot": 28, "bit": 28, "mask": 4, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 16, "op": "shfl", "dst": 7, "src": 5, "src2": 6, "imm": "0x7789be79", "imm2": "0x87496b8e", "rot": 22, "bit": 2, "mask": 16, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 17, "op": "mul", "dst": 7, "src": 6, "src2": 2, "imm": "0x559099bb", "imm2": "0x6a72e1a6", "rot": 1, "bit": 11, "mask": 16, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 18, "op": "mad", "dst": 2, "src": 3, "src2": 1, "imm": "0x4f5bae5e", "imm2": "0x4c8c5762", "rot": 27, "bit": 31, "mask": 4, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 19, "op": "rotr", "dst": 5, "src": 2, "src2": 4, "imm": "0x5dc5e6bf", "imm2": "0x36bc207d", "rot": 9, "bit": 4, "mask": 2, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 20, "op": "load", "dst": 5, "src": 3, "src2": 7, "imm": "0x4a278885", "imm2": "0xcd1044b0", "rot": 30, "bit": 0, "mask": 4, "width": 1, "win": 1, "off": 1},
|
||||
{"i": 21, "op": "shfl", "dst": 6, "src": 4, "src2": 3, "imm": "0x48b94ffc", "imm2": "0xd8e14d80", "rot": 2, "bit": 31, "mask": 16, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 22, "op": "mulhi", "dst": 1, "src": 2, "src2": 7, "imm": "0x47bd8492", "imm2": "0x28fa8d0b", "rot": 19, "bit": 26, "mask": 4, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 23, "op": "xor", "dst": 6, "src": 5, "src2": 1, "imm": "0x4ed6baf4", "imm2": "0xa8c31f74", "rot": 30, "bit": 2, "mask": 2, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 24, "op": "mad", "dst": 5, "src": 2, "src2": 0, "imm": "0x6174747d", "imm2": "0xfc912524", "rot": 18, "bit": 29, "mask": 2, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 25, "op": "mad", "dst": 6, "src": 3, "src2": 7, "imm": "0x113a6442", "imm2": "0xbe52ec57", "rot": 15, "bit": 15, "mask": 1, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 26, "op": "load", "dst": 6, "src": 5, "src2": 1, "imm": "0x9d3eb1cc", "imm2": "0x375c73bb", "rot": 29, "bit": 7, "mask": 8, "width": 1, "win": 1, "off": 1},
|
||||
{"i": 27, "op": "mad", "dst": 1, "src": 5, "src2": 5, "imm": "0x150088c3", "imm2": "0x7f17e789", "rot": 8, "bit": 25, "mask": 8, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 28, "op": "load", "dst": 0, "src": 6, "src2": 3, "imm": "0x4b412223", "imm2": "0x9192ceeb", "rot": 16, "bit": 28, "mask": 1, "width": 1, "win": 1, "off": 1},
|
||||
{"i": 29, "op": "add", "dst": 7, "src": 4, "src2": 6, "imm": "0xd5e6c37c", "imm2": "0x0b1f02c7", "rot": 2, "bit": 7, "mask": 8, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 30, "op": "add", "dst": 7, "src": 0, "src2": 6, "imm": "0x4e45ba51", "imm2": "0x2d6e8bd1", "rot": 7, "bit": 0, "mask": 8, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 31, "op": "xor", "dst": 0, "src": 2, "src2": 7, "imm": "0xef601d89", "imm2": "0x805402db", "rot": 7, "bit": 28, "mask": 2, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 32, "op": "add", "dst": 6, "src": 2, "src2": 3, "imm": "0x1c91b2a1", "imm2": "0x351422d2", "rot": 1, "bit": 13, "mask": 2, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 33, "op": "mulhi", "dst": 4, "src": 5, "src2": 7, "imm": "0x7cf21567", "imm2": "0x1075ee0d", "rot": 26, "bit": 17, "mask": 16, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 34, "op": "rotr", "dst": 3, "src": 5, "src2": 6, "imm": "0x60d8f191", "imm2": "0x40523707", "rot": 3, "bit": 10, "mask": 4, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 35, "op": "load", "dst": 3, "src": 0, "src2": 5, "imm": "0xb1dc3f6e", "imm2": "0x74440531", "rot": 21, "bit": 10, "mask": 1, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 36, "op": "rotl", "dst": 4, "src": 6, "src2": 4, "imm": "0xf0d17569", "imm2": "0xd281fd01", "rot": 13, "bit": 4, "mask": 8, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 37, "op": "add", "dst": 6, "src": 1, "src2": 6, "imm": "0x3f2970f7", "imm2": "0x64a28251", "rot": 4, "bit": 7, "mask": 8, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 38, "op": "mul", "dst": 2, "src": 0, "src2": 3, "imm": "0x6463770c", "imm2": "0xb85d603f", "rot": 27, "bit": 1, "mask": 1, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 39, "op": "add", "dst": 3, "src": 4, "src2": 3, "imm": "0x3b7b3317", "imm2": "0x7378c955", "rot": 28, "bit": 15, "mask": 2, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 40, "op": "load", "dst": 0, "src": 7, "src2": 3, "imm": "0xd0fb5982", "imm2": "0x2e4d31a8", "rot": 25, "bit": 27, "mask": 4, "width": 1, "win": 2, "off": 3},
|
||||
{"i": 41, "op": "xor", "dst": 6, "src": 5, "src2": 0, "imm": "0xe96a9e6e", "imm2": "0x8d73cc3e", "rot": 17, "bit": 11, "mask": 16, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 42, "op": "add", "dst": 3, "src": 0, "src2": 6, "imm": "0x6678b059", "imm2": "0xb31f6a64", "rot": 11, "bit": 2, "mask": 2, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 43, "op": "load", "dst": 7, "src": 3, "src2": 2, "imm": "0x65a841c5", "imm2": "0x6aae1403", "rot": 9, "bit": 9, "mask": 1, "width": 1, "win": 1, "off": 0},
|
||||
{"i": 44, "op": "shfl", "dst": 3, "src": 7, "src2": 6, "imm": "0x650ba38d", "imm2": "0x07218508", "rot": 12, "bit": 13, "mask": 4, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 45, "op": "mulhi", "dst": 1, "src": 5, "src2": 7, "imm": "0x18736788", "imm2": "0x9f820eed", "rot": 13, "bit": 29, "mask": 1, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 46, "op": "rotl", "dst": 0, "src": 1, "src2": 5, "imm": "0x48deba75", "imm2": "0xefc508fa", "rot": 6, "bit": 6, "mask": 2, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 47, "op": "load", "dst": 1, "src": 7, "src2": 4, "imm": "0x28a04384", "imm2": "0xc67cb0ac", "rot": 5, "bit": 5, "mask": 2, "width": 1, "win": 1, "off": 1},
|
||||
{"i": 48, "op": "add", "dst": 4, "src": 0, "src2": 5, "imm": "0x43095946", "imm2": "0x5feccee6", "rot": 22, "bit": 17, "mask": 1, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 49, "op": "load", "dst": 2, "src": 0, "src2": 2, "imm": "0x3e78cdd8", "imm2": "0x7c170311", "rot": 10, "bit": 5, "mask": 16, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 50, "op": "sub", "dst": 2, "src": 3, "src2": 7, "imm": "0xaff0cdb2", "imm2": "0xde0b7ee0", "rot": 21, "bit": 11, "mask": 8, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 51, "op": "add", "dst": 7, "src": 2, "src2": 1, "imm": "0xf45ecdf8", "imm2": "0x26e2b582", "rot": 28, "bit": 24, "mask": 4, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 52, "op": "load", "dst": 5, "src": 7, "src2": 7, "imm": "0x67353e69", "imm2": "0xb2d56082", "rot": 27, "bit": 14, "mask": 2, "width": 1, "win": 2, "off": 3},
|
||||
{"i": 53, "op": "load", "dst": 6, "src": 3, "src2": 1, "imm": "0x5a58c787", "imm2": "0x53a0467b", "rot": 6, "bit": 16, "mask": 16, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 54, "op": "shfl", "dst": 7, "src": 5, "src2": 4, "imm": "0x3c237ad5", "imm2": "0xf2681187", "rot": 14, "bit": 26, "mask": 4, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 55, "op": "rotr", "dst": 1, "src": 2, "src2": 3, "imm": "0x1c25c077", "imm2": "0x125de1ff", "rot": 13, "bit": 13, "mask": 2, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 56, "op": "shfl", "dst": 0, "src": 5, "src2": 0, "imm": "0x6465afe2", "imm2": "0x23b41d56", "rot": 31, "bit": 28, "mask": 16, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 57, "op": "rotl", "dst": 5, "src": 0, "src2": 2, "imm": "0xa427ff6e", "imm2": "0x014a0deb", "rot": 12, "bit": 28, "mask": 8, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 58, "op": "add", "dst": 5, "src": 3, "src2": 7, "imm": "0x6b4f35e8", "imm2": "0x473b1718", "rot": 25, "bit": 4, "mask": 8, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 59, "op": "rotr", "dst": 4, "src": 1, "src2": 3, "imm": "0x66cc96ef", "imm2": "0xcb8c8e56", "rot": 7, "bit": 6, "mask": 4, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 60, "op": "shfl", "dst": 4, "src": 5, "src2": 5, "imm": "0x461bbf9f", "imm2": "0xc1ffd350", "rot": 19, "bit": 2, "mask": 1, "width": 1, "win": 0, "off": 0},
|
||||
{"i": 61, "op": "load", "dst": 7, "src": 0, "src2": 1, "imm": "0x4d893571", "imm2": "0x5583e644", "rot": 19, "bit": 19, "mask": 8, "width": 1, "win": 1, "off": 1},
|
||||
{"i": 62, "op": "load", "dst": 5, "src": 7, "src2": 3, "imm": "0x7a9049b4", "imm2": "0xbb7ebdf8", "rot": 15, "bit": 11, "mask": 8, "width": 1, "win": 1, "off": 1},
|
||||
{"i": 63, "op": "mul", "dst": 4, "src": 1, "src2": 7, "imm": "0xe7ae8a83", "imm2": "0x1ded64e8", "rot": 21, "bit": 1, "mask": 8, "width": 1, "win": 0, "off": 0}
|
||||
]
|
||||
}
|
||||
368
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/program.metal
Normal file
368
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/program.metal
Normal file
|
|
@ -0,0 +1,368 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
|
||||
#define MASK 0x0fffffffu
|
||||
constant uint SEEDW[8] = { 0xbe8c5a0cu, 0x7646e626u, 0x508e29d0u, 0xfb8a5b4au, 0x53c50955u, 0x685d62fcu, 0x6065c013u, 0x881096ebu };
|
||||
|
||||
inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
|
||||
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
|
||||
device ulong* out [[buffer(1)]],
|
||||
constant uint& baseNonce [[buffer(2)]],
|
||||
uint gid [[thread_position_in_grid]]) {
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint x = nonce ^ SEEDW[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ SEEDW[1]; }
|
||||
{ uint x = nonce ^ SEEDW[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ SEEDW[2]; }
|
||||
{ uint x = nonce ^ SEEDW[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ SEEDW[3]; }
|
||||
{ uint x = nonce ^ SEEDW[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ SEEDW[4]; }
|
||||
{ uint x = nonce ^ SEEDW[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ SEEDW[5]; }
|
||||
{ uint x = nonce ^ SEEDW[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ SEEDW[6]; }
|
||||
{ uint x = nonce ^ SEEDW[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ SEEDW[7]; }
|
||||
{ uint x = nonce ^ SEEDW[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ SEEDW[0]; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r6 = r6 - r3; // 0
|
||||
r2 = r2 | r6; // 1
|
||||
r4 = r4 + r0 + select(0x1a579b38u, 0x7731324bu, ((sel >> 4u) & 1u) != 0u); // 2
|
||||
r3 = r3 ^ dataset[((rotl_imm(r6 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & MASK]; // 3
|
||||
r0 = r0 ^ simd_shuffle_xor(r6, (ushort)1); // 4
|
||||
r4 = r4 ^ r1; // 5
|
||||
r3 = r5 * r3 + r3; // 6
|
||||
r1 = r1 * r6; // 7
|
||||
r2 = r2 ^ dataset[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x03ffffffu) | 0x08000000u) & MASK]; // 8
|
||||
r0 = r0 ^ r4; // 9
|
||||
r0 = r0 + r7 + select(0xee083919u, 0xb10fcef8u, ((sel >> 30u) & 1u) != 0u); // 10
|
||||
r0 = mulhi(r0, r2); // 11
|
||||
r5 = r5 ^ r4; // 12
|
||||
r7 = r3 * r0 + r7; // 13
|
||||
r3 = r3 ^ dataset[((rotl_imm(r4 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x00000000u) & MASK]; // 14
|
||||
r2 = r2 ^ dataset[((rotl_imm(r5 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & MASK]; // 15
|
||||
r7 = r7 ^ simd_shuffle_xor(r5, (ushort)16); // 16
|
||||
r7 = r7 * r6; // 17
|
||||
r2 = r3 * r1 + r2; // 18
|
||||
r5 = rotr_var(r5, r2); // 19
|
||||
r5 = r5 ^ dataset[((rotl_imm(r3 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 20
|
||||
r6 = r6 ^ simd_shuffle_xor(r4, (ushort)16); // 21
|
||||
r1 = mulhi(r1, r2); // 22
|
||||
r6 = r6 ^ r5; // 23
|
||||
r5 = r2 * r0 + r5; // 24
|
||||
r6 = r3 * r7 + r6; // 25
|
||||
r6 = r6 ^ dataset[((rotl_imm(r5 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 26
|
||||
r1 = r5 * r5 + r1; // 27
|
||||
r0 = r0 ^ dataset[((rotl_imm(r6 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 28
|
||||
r7 = r7 + r4 + select(0xd5e6c37cu, 0x0b1f02c7u, ((sel >> 7u) & 1u) != 0u); // 29
|
||||
r7 = r7 + r0 + select(0x4e45ba51u, 0x2d6e8bd1u, ((sel >> 0u) & 1u) != 0u); // 30
|
||||
r0 = r0 ^ r2; // 31
|
||||
r6 = r6 + r2 + select(0x1c91b2a1u, 0x351422d2u, ((sel >> 13u) & 1u) != 0u); // 32
|
||||
r4 = mulhi(r4, r5); // 33
|
||||
r3 = rotr_var(r3, r5); // 34
|
||||
r3 = r3 ^ dataset[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & MASK]; // 35
|
||||
r4 = rotl_imm(r4, 13u); // 36
|
||||
r6 = r6 + r1 + select(0x3f2970f7u, 0x64a28251u, ((sel >> 7u) & 1u) != 0u); // 37
|
||||
r2 = r2 * r0; // 38
|
||||
r3 = r3 + r4 + select(0x3b7b3317u, 0x7378c955u, ((sel >> 15u) & 1u) != 0u); // 39
|
||||
r0 = r0 ^ dataset[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x03ffffffu) | 0x0c000000u) & MASK]; // 40
|
||||
r6 = r6 ^ r5; // 41
|
||||
r3 = r3 + r0 + select(0x6678b059u, 0xb31f6a64u, ((sel >> 2u) & 1u) != 0u); // 42
|
||||
r7 = r7 ^ dataset[((rotl_imm(r3 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x00000000u) & MASK]; // 43
|
||||
r3 = r3 ^ simd_shuffle_xor(r7, (ushort)4); // 44
|
||||
r1 = mulhi(r1, r5); // 45
|
||||
r0 = rotl_imm(r0, 6u); // 46
|
||||
r1 = r1 ^ dataset[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 47
|
||||
r4 = r4 + r0 + select(0x43095946u, 0x5feccee6u, ((sel >> 17u) & 1u) != 0u); // 48
|
||||
r2 = r2 ^ dataset[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & MASK]; // 49
|
||||
r2 = r2 - r3; // 50
|
||||
r7 = r7 + r2 + select(0xf45ecdf8u, 0x26e2b582u, ((sel >> 24u) & 1u) != 0u); // 51
|
||||
r5 = r5 ^ dataset[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x03ffffffu) | 0x0c000000u) & MASK]; // 52
|
||||
r6 = r6 ^ dataset[((rotl_imm(r3 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & MASK]; // 53
|
||||
r7 = r7 ^ simd_shuffle_xor(r5, (ushort)4); // 54
|
||||
r1 = rotr_var(r1, r2); // 55
|
||||
r0 = r0 ^ simd_shuffle_xor(r5, (ushort)16); // 56
|
||||
r5 = rotl_imm(r5, 12u); // 57
|
||||
r5 = r5 + r3 + select(0x6b4f35e8u, 0x473b1718u, ((sel >> 4u) & 1u) != 0u); // 58
|
||||
r4 = rotr_var(r4, r1); // 59
|
||||
r4 = r4 ^ simd_shuffle_xor(r5, (ushort)1); // 60
|
||||
r7 = r7 ^ dataset[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 61
|
||||
r5 = r5 ^ dataset[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 62
|
||||
r4 = r4 * r1; // 63
|
||||
// latency-shadow block (Counter ASIC 3.0 item 8): 256 ALU instructions x 27 passes after instruction 63, no load
|
||||
for (uint sh = 0u; sh < 27u; ++sh) {
|
||||
r6 = r6 | r0; // s0 or
|
||||
r7 = r7 * r3; // s1 mul
|
||||
r3 = r5 * r5 + r3; // s2 mad
|
||||
r7 = r7 + r4 + select(0x97d3d105u, 0x38e58a06u, ((sel >> 6u) & 1u) != 0u); // s3 add
|
||||
r6 = r6 + r0 + select(0x29b87b9au, 0x9629673du, ((sel >> 20u) & 1u) != 0u); // s4 add
|
||||
r6 = r6 ^ simd_shuffle_xor(r2, (ushort)1); // s5 shfl
|
||||
r5 = r3 * r5 + r5; // s6 mad
|
||||
r3 = rotl_imm(r3, 5u); // s7 rotl
|
||||
r0 = r0 + r7 + select(0xa7c1fb0fu, 0x4a5f55e1u, ((sel >> 12u) & 1u) != 0u); // s8 add
|
||||
r5 = r5 * r7; // s9 mul
|
||||
r4 = r4 ^ r5; // s10 xor
|
||||
r6 = mulhi(r6, r4); // s11 mulhi
|
||||
r5 = rotl_imm(r5, 13u); // s12 rotl
|
||||
r0 = r2 * r1 + r0; // s13 mad
|
||||
r2 = mulhi(r2, r4); // s14 mulhi
|
||||
r3 = r3 | r1; // s15 or
|
||||
r0 = r0 - r5; // s16 sub
|
||||
r0 = r0 * r6; // s17 mul
|
||||
r5 = r5 ^ simd_shuffle_xor(r1, (ushort)4); // s18 shfl
|
||||
r3 = r3 | r4; // s19 or
|
||||
r6 = r3 * r2 + r6; // s20 mad
|
||||
r0 = rotl_imm(r0, 11u); // s21 rotl
|
||||
r3 = rotl_imm(r3, 25u); // s22 rotl
|
||||
r6 = r6 ^ r4; // s23 xor
|
||||
r4 = r4 | r2; // s24 or
|
||||
r1 = r1 + r6 + select(0x53e74799u, 0xfc2f6d17u, ((sel >> 7u) & 1u) != 0u); // s25 add
|
||||
r7 = rotr_var(r7, r6); // s26 rotr
|
||||
r7 = r7 + r5 + select(0xb8fae90eu, 0x787048dcu, ((sel >> 27u) & 1u) != 0u); // s27 add
|
||||
r4 = r4 ^ r1; // s28 xor
|
||||
r6 = r6 ^ r1; // s29 xor
|
||||
r2 = mulhi(r2, r0); // s30 mulhi
|
||||
r1 = rotr_var(r1, r4); // s31 rotr
|
||||
r7 = r7 ^ r4; // s32 xor
|
||||
r3 = r2 * r6 + r3; // s33 mad
|
||||
r3 = r3 + r1 + select(0xd2dc42ebu, 0x9ffd510bu, ((sel >> 12u) & 1u) != 0u); // s34 add
|
||||
r4 = r4 + r7 + select(0xad8eda6fu, 0xf871ba37u, ((sel >> 7u) & 1u) != 0u); // s35 add
|
||||
r7 = r7 + r4 + select(0xb0cef8f4u, 0xd02d30cau, ((sel >> 3u) & 1u) != 0u); // s36 add
|
||||
r4 = r3 * r2 + r4; // s37 mad
|
||||
r5 = r7 * r7 + r5; // s38 mad
|
||||
r3 = r3 + r6 + select(0x2bd506e2u, 0x82e98a22u, ((sel >> 26u) & 1u) != 0u); // s39 add
|
||||
r0 = rotl_imm(r0, 21u); // s40 rotl
|
||||
r1 = r1 | r0; // s41 or
|
||||
r7 = r7 + r5 + select(0xc0386e49u, 0x6435f6d8u, ((sel >> 28u) & 1u) != 0u); // s42 add
|
||||
r1 = mulhi(r1, r0); // s43 mulhi
|
||||
r5 = r5 + r6 + select(0x3a8921a4u, 0x1f682c68u, ((sel >> 12u) & 1u) != 0u); // s44 add
|
||||
r4 = r4 + r5 + select(0xeeca4334u, 0xf246e180u, ((sel >> 1u) & 1u) != 0u); // s45 add
|
||||
r2 = r2 ^ r3; // s46 xor
|
||||
r2 = r2 ^ r5; // s47 xor
|
||||
r7 = r4 * r1 + r7; // s48 mad
|
||||
r2 = r6 * r1 + r2; // s49 mad
|
||||
r7 = r0 * r7 + r7; // s50 mad
|
||||
r5 = r5 + r6 + select(0x745776d8u, 0x41bd68ccu, ((sel >> 26u) & 1u) != 0u); // s51 add
|
||||
r3 = r3 ^ simd_shuffle_xor(r0, (ushort)1); // s52 shfl
|
||||
r4 = r4 | r0; // s53 or
|
||||
r0 = r2 * r4 + r0; // s54 mad
|
||||
r5 = rotl_imm(r5, 9u); // s55 rotl
|
||||
r5 = r5 * r7; // s56 mul
|
||||
r7 = rotl_imm(r7, 27u); // s57 rotl
|
||||
r7 = r7 - r6; // s58 sub
|
||||
r2 = rotl_imm(r2, 22u); // s59 rotl
|
||||
r7 = r7 ^ simd_shuffle_xor(r1, (ushort)2); // s60 shfl
|
||||
r4 = r4 + r0 + select(0x60d0ff55u, 0xcba6a161u, ((sel >> 7u) & 1u) != 0u); // s61 add
|
||||
r1 = r1 * r5; // s62 mul
|
||||
r7 = mulhi(r7, r6); // s63 mulhi
|
||||
r5 = r5 * r0; // s64 mul
|
||||
r2 = rotr_var(r2, r5); // s65 rotr
|
||||
r0 = mulhi(r0, r3); // s66 mulhi
|
||||
r0 = r0 | r7; // s67 or
|
||||
r0 = rotl_imm(r0, 19u); // s68 rotl
|
||||
r0 = rotl_imm(r0, 1u); // s69 rotl
|
||||
r2 = r2 | r3; // s70 or
|
||||
r3 = r7 * r5 + r3; // s71 mad
|
||||
r7 = r0 * r4 + r7; // s72 mad
|
||||
r4 = r4 | r3; // s73 or
|
||||
r3 = r3 ^ r1; // s74 xor
|
||||
r1 = r1 + r3 + select(0x3f4ffea2u, 0x8e35924eu, ((sel >> 15u) & 1u) != 0u); // s75 add
|
||||
r1 = r1 * r3; // s76 mul
|
||||
r0 = r3 * r2 + r0; // s77 mad
|
||||
r2 = r2 ^ simd_shuffle_xor(r0, (ushort)16); // s78 shfl
|
||||
r1 = mulhi(r1, r0); // s79 mulhi
|
||||
r0 = mulhi(r0, r4); // s80 mulhi
|
||||
r4 = r4 * r2; // s81 mul
|
||||
r5 = r5 - r7; // s82 sub
|
||||
r0 = r2 * r3 + r0; // s83 mad
|
||||
r7 = r7 - r2; // s84 sub
|
||||
r1 = r1 ^ r2; // s85 xor
|
||||
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)16); // s86 shfl
|
||||
r3 = r0 * r0 + r3; // s87 mad
|
||||
r0 = rotl_imm(r0, 20u); // s88 rotl
|
||||
r1 = r1 - r5; // s89 sub
|
||||
r6 = r6 ^ simd_shuffle_xor(r4, (ushort)16); // s90 shfl
|
||||
r5 = r5 ^ r1; // s91 xor
|
||||
r5 = rotl_imm(r5, 12u); // s92 rotl
|
||||
r6 = r1 * r2 + r6; // s93 mad
|
||||
r2 = rotr_var(r2, r5); // s94 rotr
|
||||
r0 = r0 | r4; // s95 or
|
||||
r7 = r7 * r3; // s96 mul
|
||||
r5 = r5 | r0; // s97 or
|
||||
r3 = r3 + r1 + select(0xbcfdd8a7u, 0x386bb642u, ((sel >> 13u) & 1u) != 0u); // s98 add
|
||||
r0 = r0 + r6 + select(0xc921a021u, 0xa3962d03u, ((sel >> 9u) & 1u) != 0u); // s99 add
|
||||
r1 = rotr_var(r1, r0); // s100 rotr
|
||||
r2 = r2 + r4 + select(0x70da4067u, 0xc5990c5cu, ((sel >> 18u) & 1u) != 0u); // s101 add
|
||||
r2 = rotl_imm(r2, 3u); // s102 rotl
|
||||
r1 = r1 ^ simd_shuffle_xor(r4, (ushort)4); // s103 shfl
|
||||
r0 = r0 ^ r2; // s104 xor
|
||||
r4 = r4 + r3 + select(0x1a5880cbu, 0xb15aec75u, ((sel >> 28u) & 1u) != 0u); // s105 add
|
||||
r4 = r4 ^ simd_shuffle_xor(r0, (ushort)2); // s106 shfl
|
||||
r7 = r7 ^ simd_shuffle_xor(r0, (ushort)4); // s107 shfl
|
||||
r1 = r7 * r3 + r1; // s108 mad
|
||||
r0 = r2 * r3 + r0; // s109 mad
|
||||
r2 = r2 | r4; // s110 or
|
||||
r6 = r6 ^ simd_shuffle_xor(r1, (ushort)1); // s111 shfl
|
||||
r2 = r5 * r3 + r2; // s112 mad
|
||||
r4 = mulhi(r4, r0); // s113 mulhi
|
||||
r7 = rotl_imm(r7, 29u); // s114 rotl
|
||||
r3 = r3 - r7; // s115 sub
|
||||
r0 = mulhi(r0, r1); // s116 mulhi
|
||||
r4 = rotl_imm(r4, 12u); // s117 rotl
|
||||
r1 = r1 + r4 + select(0x24494ad3u, 0xc5e98ee5u, ((sel >> 7u) & 1u) != 0u); // s118 add
|
||||
r0 = r0 + r4 + select(0x1b2d1c83u, 0xa97bc949u, ((sel >> 4u) & 1u) != 0u); // s119 add
|
||||
r5 = r5 ^ simd_shuffle_xor(r4, (ushort)4); // s120 shfl
|
||||
r3 = r3 ^ simd_shuffle_xor(r2, (ushort)8); // s121 shfl
|
||||
r7 = rotl_imm(r7, 3u); // s122 rotl
|
||||
r4 = rotr_var(r4, r6); // s123 rotr
|
||||
r6 = r6 ^ simd_shuffle_xor(r5, (ushort)1); // s124 shfl
|
||||
r5 = rotl_imm(r5, 21u); // s125 rotl
|
||||
r0 = r3 * r1 + r0; // s126 mad
|
||||
r0 = r0 + r3 + select(0xff14c08du, 0xa7f6bb54u, ((sel >> 24u) & 1u) != 0u); // s127 add
|
||||
r7 = r7 ^ r4; // s128 xor
|
||||
r3 = mulhi(r3, r5); // s129 mulhi
|
||||
r5 = r5 ^ simd_shuffle_xor(r7, (ushort)8); // s130 shfl
|
||||
r6 = mulhi(r6, r4); // s131 mulhi
|
||||
r5 = r5 ^ r3; // s132 xor
|
||||
r0 = rotl_imm(r0, 19u); // s133 rotl
|
||||
r4 = rotr_var(r4, r3); // s134 rotr
|
||||
r1 = r1 ^ simd_shuffle_xor(r2, (ushort)16); // s135 shfl
|
||||
r5 = r5 - r2; // s136 sub
|
||||
r3 = r3 ^ r5; // s137 xor
|
||||
r0 = r0 - r2; // s138 sub
|
||||
r3 = r3 * r7; // s139 mul
|
||||
r5 = r5 + r3 + select(0x82ab98f4u, 0x2d465ca8u, ((sel >> 24u) & 1u) != 0u); // s140 add
|
||||
r4 = r4 + r1 + select(0x80594390u, 0x237a9d8eu, ((sel >> 11u) & 1u) != 0u); // s141 add
|
||||
r3 = r7 * r0 + r3; // s142 mad
|
||||
r0 = r0 | r3; // s143 or
|
||||
r0 = mulhi(r0, r1); // s144 mulhi
|
||||
r6 = r6 ^ r2; // s145 xor
|
||||
r7 = r7 ^ simd_shuffle_xor(r1, (ushort)16); // s146 shfl
|
||||
r0 = r0 + r1 + select(0x07f15148u, 0xe68cf57eu, ((sel >> 18u) & 1u) != 0u); // s147 add
|
||||
r7 = r7 | r3; // s148 or
|
||||
r4 = r4 ^ r2; // s149 xor
|
||||
r7 = r7 ^ r1; // s150 xor
|
||||
r5 = rotr_var(r5, r2); // s151 rotr
|
||||
r0 = r0 * r7; // s152 mul
|
||||
r4 = rotl_imm(r4, 16u); // s153 rotl
|
||||
r6 = rotr_var(r6, r4); // s154 rotr
|
||||
r6 = r6 ^ r1; // s155 xor
|
||||
r6 = r6 | r3; // s156 or
|
||||
r1 = r1 + r6 + select(0x14878c5au, 0x84c2a09du, ((sel >> 19u) & 1u) != 0u); // s157 add
|
||||
r4 = r4 - r1; // s158 sub
|
||||
r4 = r6 * r7 + r4; // s159 mad
|
||||
r1 = r1 * r5; // s160 mul
|
||||
r4 = r4 | r0; // s161 or
|
||||
r7 = rotl_imm(r7, 21u); // s162 rotl
|
||||
r0 = r0 - r1; // s163 sub
|
||||
r1 = r1 * r0; // s164 mul
|
||||
r2 = r2 - r1; // s165 sub
|
||||
r0 = r0 ^ simd_shuffle_xor(r6, (ushort)2); // s166 shfl
|
||||
r0 = r0 - r4; // s167 sub
|
||||
r1 = rotr_var(r1, r6); // s168 rotr
|
||||
r7 = rotr_var(r7, r1); // s169 rotr
|
||||
r3 = r1 * r3 + r3; // s170 mad
|
||||
r2 = r2 ^ simd_shuffle_xor(r0, (ushort)4); // s171 shfl
|
||||
r2 = r2 - r1; // s172 sub
|
||||
r7 = r7 ^ r6; // s173 xor
|
||||
r4 = rotr_var(r4, r1); // s174 rotr
|
||||
r4 = r4 ^ simd_shuffle_xor(r7, (ushort)2); // s175 shfl
|
||||
r6 = r6 * r2; // s176 mul
|
||||
r0 = r0 ^ simd_shuffle_xor(r2, (ushort)4); // s177 shfl
|
||||
r6 = r6 + r4 + select(0x4fa43965u, 0x509871e6u, ((sel >> 21u) & 1u) != 0u); // s178 add
|
||||
r0 = r0 ^ r6; // s179 xor
|
||||
r5 = r5 + r7 + select(0xfb3c4bd8u, 0xb95b8a53u, ((sel >> 0u) & 1u) != 0u); // s180 add
|
||||
r6 = r6 * r3; // s181 mul
|
||||
r7 = rotl_imm(r7, 1u); // s182 rotl
|
||||
r0 = r0 + r7 + select(0xc448a197u, 0x7047b7cfu, ((sel >> 11u) & 1u) != 0u); // s183 add
|
||||
r3 = r3 * r5; // s184 mul
|
||||
r5 = r5 ^ simd_shuffle_xor(r7, (ushort)1); // s185 shfl
|
||||
r0 = r0 * r2; // s186 mul
|
||||
r4 = mulhi(r4, r7); // s187 mulhi
|
||||
r2 = r7 * r2 + r2; // s188 mad
|
||||
r7 = r7 - r2; // s189 sub
|
||||
r2 = r2 + r5 + select(0x2fceaf49u, 0x5caccc5du, ((sel >> 19u) & 1u) != 0u); // s190 add
|
||||
r1 = r1 ^ simd_shuffle_xor(r2, (ushort)1); // s191 shfl
|
||||
r7 = r7 + r1 + select(0xba0cee62u, 0x857bb5feu, ((sel >> 20u) & 1u) != 0u); // s192 add
|
||||
r6 = rotl_imm(r6, 22u); // s193 rotl
|
||||
r3 = r3 ^ r7; // s194 xor
|
||||
r7 = r7 * r0; // s195 mul
|
||||
r3 = r3 ^ r5; // s196 xor
|
||||
r5 = r5 + r1 + select(0x8519428cu, 0xeae84577u, ((sel >> 29u) & 1u) != 0u); // s197 add
|
||||
r7 = r4 * r4 + r7; // s198 mad
|
||||
r3 = r3 + r6 + select(0x468639d3u, 0x0bfdbfa1u, ((sel >> 8u) & 1u) != 0u); // s199 add
|
||||
r5 = r3 * r4 + r5; // s200 mad
|
||||
r2 = r2 ^ simd_shuffle_xor(r7, (ushort)4); // s201 shfl
|
||||
r7 = r7 + r4 + select(0x18ec9a69u, 0x53ee8e50u, ((sel >> 23u) & 1u) != 0u); // s202 add
|
||||
r6 = r6 ^ r0; // s203 xor
|
||||
r4 = r4 + r5 + select(0x3e81485bu, 0xac63376eu, ((sel >> 4u) & 1u) != 0u); // s204 add
|
||||
r7 = r1 * r4 + r7; // s205 mad
|
||||
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)16); // s206 shfl
|
||||
r4 = r4 * r0; // s207 mul
|
||||
r1 = r1 | r6; // s208 or
|
||||
r6 = r6 + r4 + select(0x66ccb75du, 0x5b745519u, ((sel >> 28u) & 1u) != 0u); // s209 add
|
||||
r1 = rotr_var(r1, r3); // s210 rotr
|
||||
r5 = r5 - r4; // s211 sub
|
||||
r4 = r4 ^ r7; // s212 xor
|
||||
r1 = r1 + r7 + select(0x0ce0553du, 0x1c3dccf7u, ((sel >> 8u) & 1u) != 0u); // s213 add
|
||||
r0 = mulhi(r0, r1); // s214 mulhi
|
||||
r3 = r3 + r6 + select(0x96bfec89u, 0xfadc9205u, ((sel >> 9u) & 1u) != 0u); // s215 add
|
||||
r0 = rotl_imm(r0, 4u); // s216 rotl
|
||||
r6 = r6 ^ r4; // s217 xor
|
||||
r6 = r6 ^ simd_shuffle_xor(r7, (ushort)2); // s218 shfl
|
||||
r7 = r0 * r1 + r7; // s219 mad
|
||||
r4 = r4 * r0; // s220 mul
|
||||
r2 = r6 * r0 + r2; // s221 mad
|
||||
r5 = r5 ^ r4; // s222 xor
|
||||
r0 = r0 * r5; // s223 mul
|
||||
r2 = r2 + r0 + select(0x61495681u, 0x66b15c04u, ((sel >> 9u) & 1u) != 0u); // s224 add
|
||||
r3 = rotr_var(r3, r2); // s225 rotr
|
||||
r2 = r3 * r3 + r2; // s226 mad
|
||||
r6 = r6 ^ r0; // s227 xor
|
||||
r4 = rotl_imm(r4, 30u); // s228 rotl
|
||||
r2 = r2 + r3 + select(0x70d05c34u, 0x25bc17c2u, ((sel >> 29u) & 1u) != 0u); // s229 add
|
||||
r1 = rotr_var(r1, r5); // s230 rotr
|
||||
r1 = r1 + r4 + select(0x32a28384u, 0x2d8728f3u, ((sel >> 8u) & 1u) != 0u); // s231 add
|
||||
r1 = rotl_imm(r1, 6u); // s232 rotl
|
||||
r2 = r2 + r0 + select(0xcf0949f1u, 0x61e81fe6u, ((sel >> 21u) & 1u) != 0u); // s233 add
|
||||
r4 = rotl_imm(r4, 2u); // s234 rotl
|
||||
r2 = r2 - r6; // s235 sub
|
||||
r6 = r6 | r3; // s236 or
|
||||
r7 = r7 ^ simd_shuffle_xor(r5, (ushort)4); // s237 shfl
|
||||
r1 = r1 ^ r3; // s238 xor
|
||||
r2 = rotl_imm(r2, 1u); // s239 rotl
|
||||
r5 = rotr_var(r5, r2); // s240 rotr
|
||||
r4 = rotl_imm(r4, 10u); // s241 rotl
|
||||
r6 = r6 + r2 + select(0xe204fb50u, 0x0a67565du, ((sel >> 16u) & 1u) != 0u); // s242 add
|
||||
r1 = r1 ^ simd_shuffle_xor(r3, (ushort)2); // s243 shfl
|
||||
r0 = r0 ^ simd_shuffle_xor(r1, (ushort)16); // s244 shfl
|
||||
r5 = r5 ^ simd_shuffle_xor(r7, (ushort)2); // s245 shfl
|
||||
r4 = r4 ^ r5; // s246 xor
|
||||
r7 = r7 ^ r2; // s247 xor
|
||||
r0 = r0 ^ simd_shuffle_xor(r6, (ushort)16); // s248 shfl
|
||||
r1 = mulhi(r1, r3); // s249 mulhi
|
||||
r7 = r7 + r5 + select(0x3d6daffdu, 0x5a79fd6fu, ((sel >> 31u) & 1u) != 0u); // s250 add
|
||||
r6 = r6 + r0 + select(0x84334aeau, 0x6beb28c2u, ((sel >> 28u) & 1u) != 0u); // s251 add
|
||||
r7 = r7 - r1; // s252 sub
|
||||
r5 = rotr_var(r5, r2); // s253 rotr
|
||||
r2 = mulhi(r2, r5); // s254 mulhi
|
||||
r6 = r6 * r3; // s255 mul
|
||||
}
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
370
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/program_bound.metal
Normal file
370
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/program_bound.metal
Normal file
|
|
@ -0,0 +1,370 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
|
||||
#define MASK 0x0fffffffu
|
||||
constant uint SEEDW[8] = { 0xbe8c5a0cu, 0x7646e626u, 0x508e29d0u, 0xfb8a5b4au, 0x53c50955u, 0x685d62fcu, 0x6065c013u, 0x881096ebu };
|
||||
|
||||
inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
|
||||
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW.
|
||||
kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
|
||||
device ulong* out [[buffer(1)]],
|
||||
constant uint& baseNonce [[buffer(2)]],
|
||||
constant uint* initw [[buffer(3)]],
|
||||
uint gid [[thread_position_in_grid]]) {
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint x = nonce ^ initw[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ initw[1]; }
|
||||
{ uint x = nonce ^ initw[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ initw[2]; }
|
||||
{ uint x = nonce ^ initw[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ initw[3]; }
|
||||
{ uint x = nonce ^ initw[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ initw[4]; }
|
||||
{ uint x = nonce ^ initw[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ initw[5]; }
|
||||
{ uint x = nonce ^ initw[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ initw[6]; }
|
||||
{ uint x = nonce ^ initw[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ initw[7]; }
|
||||
{ uint x = nonce ^ initw[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ initw[0]; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r6 = r6 - r3; // 0
|
||||
r2 = r2 | r6; // 1
|
||||
r4 = r4 + r0 + select(0x1a579b38u, 0x7731324bu, ((sel >> 4u) & 1u) != 0u); // 2
|
||||
r3 = r3 ^ dataset[((rotl_imm(r6 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & MASK]; // 3
|
||||
r0 = r0 ^ simd_shuffle_xor(r6, (ushort)1); // 4
|
||||
r4 = r4 ^ r1; // 5
|
||||
r3 = r5 * r3 + r3; // 6
|
||||
r1 = r1 * r6; // 7
|
||||
r2 = r2 ^ dataset[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x03ffffffu) | 0x08000000u) & MASK]; // 8
|
||||
r0 = r0 ^ r4; // 9
|
||||
r0 = r0 + r7 + select(0xee083919u, 0xb10fcef8u, ((sel >> 30u) & 1u) != 0u); // 10
|
||||
r0 = mulhi(r0, r2); // 11
|
||||
r5 = r5 ^ r4; // 12
|
||||
r7 = r3 * r0 + r7; // 13
|
||||
r3 = r3 ^ dataset[((rotl_imm(r4 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x00000000u) & MASK]; // 14
|
||||
r2 = r2 ^ dataset[((rotl_imm(r5 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & MASK]; // 15
|
||||
r7 = r7 ^ simd_shuffle_xor(r5, (ushort)16); // 16
|
||||
r7 = r7 * r6; // 17
|
||||
r2 = r3 * r1 + r2; // 18
|
||||
r5 = rotr_var(r5, r2); // 19
|
||||
r5 = r5 ^ dataset[((rotl_imm(r3 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 20
|
||||
r6 = r6 ^ simd_shuffle_xor(r4, (ushort)16); // 21
|
||||
r1 = mulhi(r1, r2); // 22
|
||||
r6 = r6 ^ r5; // 23
|
||||
r5 = r2 * r0 + r5; // 24
|
||||
r6 = r3 * r7 + r6; // 25
|
||||
r6 = r6 ^ dataset[((rotl_imm(r5 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 26
|
||||
r1 = r5 * r5 + r1; // 27
|
||||
r0 = r0 ^ dataset[((rotl_imm(r6 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 28
|
||||
r7 = r7 + r4 + select(0xd5e6c37cu, 0x0b1f02c7u, ((sel >> 7u) & 1u) != 0u); // 29
|
||||
r7 = r7 + r0 + select(0x4e45ba51u, 0x2d6e8bd1u, ((sel >> 0u) & 1u) != 0u); // 30
|
||||
r0 = r0 ^ r2; // 31
|
||||
r6 = r6 + r2 + select(0x1c91b2a1u, 0x351422d2u, ((sel >> 13u) & 1u) != 0u); // 32
|
||||
r4 = mulhi(r4, r5); // 33
|
||||
r3 = rotr_var(r3, r5); // 34
|
||||
r3 = r3 ^ dataset[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & MASK]; // 35
|
||||
r4 = rotl_imm(r4, 13u); // 36
|
||||
r6 = r6 + r1 + select(0x3f2970f7u, 0x64a28251u, ((sel >> 7u) & 1u) != 0u); // 37
|
||||
r2 = r2 * r0; // 38
|
||||
r3 = r3 + r4 + select(0x3b7b3317u, 0x7378c955u, ((sel >> 15u) & 1u) != 0u); // 39
|
||||
r0 = r0 ^ dataset[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x03ffffffu) | 0x0c000000u) & MASK]; // 40
|
||||
r6 = r6 ^ r5; // 41
|
||||
r3 = r3 + r0 + select(0x6678b059u, 0xb31f6a64u, ((sel >> 2u) & 1u) != 0u); // 42
|
||||
r7 = r7 ^ dataset[((rotl_imm(r3 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x00000000u) & MASK]; // 43
|
||||
r3 = r3 ^ simd_shuffle_xor(r7, (ushort)4); // 44
|
||||
r1 = mulhi(r1, r5); // 45
|
||||
r0 = rotl_imm(r0, 6u); // 46
|
||||
r1 = r1 ^ dataset[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 47
|
||||
r4 = r4 + r0 + select(0x43095946u, 0x5feccee6u, ((sel >> 17u) & 1u) != 0u); // 48
|
||||
r2 = r2 ^ dataset[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & MASK]; // 49
|
||||
r2 = r2 - r3; // 50
|
||||
r7 = r7 + r2 + select(0xf45ecdf8u, 0x26e2b582u, ((sel >> 24u) & 1u) != 0u); // 51
|
||||
r5 = r5 ^ dataset[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x03ffffffu) | 0x0c000000u) & MASK]; // 52
|
||||
r6 = r6 ^ dataset[((rotl_imm(r3 * 0x2cb18b35u, 9u) & 0x0fffffffu) | 0x00000000u) & MASK]; // 53
|
||||
r7 = r7 ^ simd_shuffle_xor(r5, (ushort)4); // 54
|
||||
r1 = rotr_var(r1, r2); // 55
|
||||
r0 = r0 ^ simd_shuffle_xor(r5, (ushort)16); // 56
|
||||
r5 = rotl_imm(r5, 12u); // 57
|
||||
r5 = r5 + r3 + select(0x6b4f35e8u, 0x473b1718u, ((sel >> 4u) & 1u) != 0u); // 58
|
||||
r4 = rotr_var(r4, r1); // 59
|
||||
r4 = r4 ^ simd_shuffle_xor(r5, (ushort)1); // 60
|
||||
r7 = r7 ^ dataset[((rotl_imm(r0 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 61
|
||||
r5 = r5 ^ dataset[((rotl_imm(r7 * 0x2cb18b35u, 9u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 62
|
||||
r4 = r4 * r1; // 63
|
||||
// latency-shadow block (Counter ASIC 3.0 item 8): 256 ALU instructions x 27 passes after instruction 63, no load
|
||||
for (uint sh = 0u; sh < 27u; ++sh) {
|
||||
r6 = r6 | r0; // s0 or
|
||||
r7 = r7 * r3; // s1 mul
|
||||
r3 = r5 * r5 + r3; // s2 mad
|
||||
r7 = r7 + r4 + select(0x97d3d105u, 0x38e58a06u, ((sel >> 6u) & 1u) != 0u); // s3 add
|
||||
r6 = r6 + r0 + select(0x29b87b9au, 0x9629673du, ((sel >> 20u) & 1u) != 0u); // s4 add
|
||||
r6 = r6 ^ simd_shuffle_xor(r2, (ushort)1); // s5 shfl
|
||||
r5 = r3 * r5 + r5; // s6 mad
|
||||
r3 = rotl_imm(r3, 5u); // s7 rotl
|
||||
r0 = r0 + r7 + select(0xa7c1fb0fu, 0x4a5f55e1u, ((sel >> 12u) & 1u) != 0u); // s8 add
|
||||
r5 = r5 * r7; // s9 mul
|
||||
r4 = r4 ^ r5; // s10 xor
|
||||
r6 = mulhi(r6, r4); // s11 mulhi
|
||||
r5 = rotl_imm(r5, 13u); // s12 rotl
|
||||
r0 = r2 * r1 + r0; // s13 mad
|
||||
r2 = mulhi(r2, r4); // s14 mulhi
|
||||
r3 = r3 | r1; // s15 or
|
||||
r0 = r0 - r5; // s16 sub
|
||||
r0 = r0 * r6; // s17 mul
|
||||
r5 = r5 ^ simd_shuffle_xor(r1, (ushort)4); // s18 shfl
|
||||
r3 = r3 | r4; // s19 or
|
||||
r6 = r3 * r2 + r6; // s20 mad
|
||||
r0 = rotl_imm(r0, 11u); // s21 rotl
|
||||
r3 = rotl_imm(r3, 25u); // s22 rotl
|
||||
r6 = r6 ^ r4; // s23 xor
|
||||
r4 = r4 | r2; // s24 or
|
||||
r1 = r1 + r6 + select(0x53e74799u, 0xfc2f6d17u, ((sel >> 7u) & 1u) != 0u); // s25 add
|
||||
r7 = rotr_var(r7, r6); // s26 rotr
|
||||
r7 = r7 + r5 + select(0xb8fae90eu, 0x787048dcu, ((sel >> 27u) & 1u) != 0u); // s27 add
|
||||
r4 = r4 ^ r1; // s28 xor
|
||||
r6 = r6 ^ r1; // s29 xor
|
||||
r2 = mulhi(r2, r0); // s30 mulhi
|
||||
r1 = rotr_var(r1, r4); // s31 rotr
|
||||
r7 = r7 ^ r4; // s32 xor
|
||||
r3 = r2 * r6 + r3; // s33 mad
|
||||
r3 = r3 + r1 + select(0xd2dc42ebu, 0x9ffd510bu, ((sel >> 12u) & 1u) != 0u); // s34 add
|
||||
r4 = r4 + r7 + select(0xad8eda6fu, 0xf871ba37u, ((sel >> 7u) & 1u) != 0u); // s35 add
|
||||
r7 = r7 + r4 + select(0xb0cef8f4u, 0xd02d30cau, ((sel >> 3u) & 1u) != 0u); // s36 add
|
||||
r4 = r3 * r2 + r4; // s37 mad
|
||||
r5 = r7 * r7 + r5; // s38 mad
|
||||
r3 = r3 + r6 + select(0x2bd506e2u, 0x82e98a22u, ((sel >> 26u) & 1u) != 0u); // s39 add
|
||||
r0 = rotl_imm(r0, 21u); // s40 rotl
|
||||
r1 = r1 | r0; // s41 or
|
||||
r7 = r7 + r5 + select(0xc0386e49u, 0x6435f6d8u, ((sel >> 28u) & 1u) != 0u); // s42 add
|
||||
r1 = mulhi(r1, r0); // s43 mulhi
|
||||
r5 = r5 + r6 + select(0x3a8921a4u, 0x1f682c68u, ((sel >> 12u) & 1u) != 0u); // s44 add
|
||||
r4 = r4 + r5 + select(0xeeca4334u, 0xf246e180u, ((sel >> 1u) & 1u) != 0u); // s45 add
|
||||
r2 = r2 ^ r3; // s46 xor
|
||||
r2 = r2 ^ r5; // s47 xor
|
||||
r7 = r4 * r1 + r7; // s48 mad
|
||||
r2 = r6 * r1 + r2; // s49 mad
|
||||
r7 = r0 * r7 + r7; // s50 mad
|
||||
r5 = r5 + r6 + select(0x745776d8u, 0x41bd68ccu, ((sel >> 26u) & 1u) != 0u); // s51 add
|
||||
r3 = r3 ^ simd_shuffle_xor(r0, (ushort)1); // s52 shfl
|
||||
r4 = r4 | r0; // s53 or
|
||||
r0 = r2 * r4 + r0; // s54 mad
|
||||
r5 = rotl_imm(r5, 9u); // s55 rotl
|
||||
r5 = r5 * r7; // s56 mul
|
||||
r7 = rotl_imm(r7, 27u); // s57 rotl
|
||||
r7 = r7 - r6; // s58 sub
|
||||
r2 = rotl_imm(r2, 22u); // s59 rotl
|
||||
r7 = r7 ^ simd_shuffle_xor(r1, (ushort)2); // s60 shfl
|
||||
r4 = r4 + r0 + select(0x60d0ff55u, 0xcba6a161u, ((sel >> 7u) & 1u) != 0u); // s61 add
|
||||
r1 = r1 * r5; // s62 mul
|
||||
r7 = mulhi(r7, r6); // s63 mulhi
|
||||
r5 = r5 * r0; // s64 mul
|
||||
r2 = rotr_var(r2, r5); // s65 rotr
|
||||
r0 = mulhi(r0, r3); // s66 mulhi
|
||||
r0 = r0 | r7; // s67 or
|
||||
r0 = rotl_imm(r0, 19u); // s68 rotl
|
||||
r0 = rotl_imm(r0, 1u); // s69 rotl
|
||||
r2 = r2 | r3; // s70 or
|
||||
r3 = r7 * r5 + r3; // s71 mad
|
||||
r7 = r0 * r4 + r7; // s72 mad
|
||||
r4 = r4 | r3; // s73 or
|
||||
r3 = r3 ^ r1; // s74 xor
|
||||
r1 = r1 + r3 + select(0x3f4ffea2u, 0x8e35924eu, ((sel >> 15u) & 1u) != 0u); // s75 add
|
||||
r1 = r1 * r3; // s76 mul
|
||||
r0 = r3 * r2 + r0; // s77 mad
|
||||
r2 = r2 ^ simd_shuffle_xor(r0, (ushort)16); // s78 shfl
|
||||
r1 = mulhi(r1, r0); // s79 mulhi
|
||||
r0 = mulhi(r0, r4); // s80 mulhi
|
||||
r4 = r4 * r2; // s81 mul
|
||||
r5 = r5 - r7; // s82 sub
|
||||
r0 = r2 * r3 + r0; // s83 mad
|
||||
r7 = r7 - r2; // s84 sub
|
||||
r1 = r1 ^ r2; // s85 xor
|
||||
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)16); // s86 shfl
|
||||
r3 = r0 * r0 + r3; // s87 mad
|
||||
r0 = rotl_imm(r0, 20u); // s88 rotl
|
||||
r1 = r1 - r5; // s89 sub
|
||||
r6 = r6 ^ simd_shuffle_xor(r4, (ushort)16); // s90 shfl
|
||||
r5 = r5 ^ r1; // s91 xor
|
||||
r5 = rotl_imm(r5, 12u); // s92 rotl
|
||||
r6 = r1 * r2 + r6; // s93 mad
|
||||
r2 = rotr_var(r2, r5); // s94 rotr
|
||||
r0 = r0 | r4; // s95 or
|
||||
r7 = r7 * r3; // s96 mul
|
||||
r5 = r5 | r0; // s97 or
|
||||
r3 = r3 + r1 + select(0xbcfdd8a7u, 0x386bb642u, ((sel >> 13u) & 1u) != 0u); // s98 add
|
||||
r0 = r0 + r6 + select(0xc921a021u, 0xa3962d03u, ((sel >> 9u) & 1u) != 0u); // s99 add
|
||||
r1 = rotr_var(r1, r0); // s100 rotr
|
||||
r2 = r2 + r4 + select(0x70da4067u, 0xc5990c5cu, ((sel >> 18u) & 1u) != 0u); // s101 add
|
||||
r2 = rotl_imm(r2, 3u); // s102 rotl
|
||||
r1 = r1 ^ simd_shuffle_xor(r4, (ushort)4); // s103 shfl
|
||||
r0 = r0 ^ r2; // s104 xor
|
||||
r4 = r4 + r3 + select(0x1a5880cbu, 0xb15aec75u, ((sel >> 28u) & 1u) != 0u); // s105 add
|
||||
r4 = r4 ^ simd_shuffle_xor(r0, (ushort)2); // s106 shfl
|
||||
r7 = r7 ^ simd_shuffle_xor(r0, (ushort)4); // s107 shfl
|
||||
r1 = r7 * r3 + r1; // s108 mad
|
||||
r0 = r2 * r3 + r0; // s109 mad
|
||||
r2 = r2 | r4; // s110 or
|
||||
r6 = r6 ^ simd_shuffle_xor(r1, (ushort)1); // s111 shfl
|
||||
r2 = r5 * r3 + r2; // s112 mad
|
||||
r4 = mulhi(r4, r0); // s113 mulhi
|
||||
r7 = rotl_imm(r7, 29u); // s114 rotl
|
||||
r3 = r3 - r7; // s115 sub
|
||||
r0 = mulhi(r0, r1); // s116 mulhi
|
||||
r4 = rotl_imm(r4, 12u); // s117 rotl
|
||||
r1 = r1 + r4 + select(0x24494ad3u, 0xc5e98ee5u, ((sel >> 7u) & 1u) != 0u); // s118 add
|
||||
r0 = r0 + r4 + select(0x1b2d1c83u, 0xa97bc949u, ((sel >> 4u) & 1u) != 0u); // s119 add
|
||||
r5 = r5 ^ simd_shuffle_xor(r4, (ushort)4); // s120 shfl
|
||||
r3 = r3 ^ simd_shuffle_xor(r2, (ushort)8); // s121 shfl
|
||||
r7 = rotl_imm(r7, 3u); // s122 rotl
|
||||
r4 = rotr_var(r4, r6); // s123 rotr
|
||||
r6 = r6 ^ simd_shuffle_xor(r5, (ushort)1); // s124 shfl
|
||||
r5 = rotl_imm(r5, 21u); // s125 rotl
|
||||
r0 = r3 * r1 + r0; // s126 mad
|
||||
r0 = r0 + r3 + select(0xff14c08du, 0xa7f6bb54u, ((sel >> 24u) & 1u) != 0u); // s127 add
|
||||
r7 = r7 ^ r4; // s128 xor
|
||||
r3 = mulhi(r3, r5); // s129 mulhi
|
||||
r5 = r5 ^ simd_shuffle_xor(r7, (ushort)8); // s130 shfl
|
||||
r6 = mulhi(r6, r4); // s131 mulhi
|
||||
r5 = r5 ^ r3; // s132 xor
|
||||
r0 = rotl_imm(r0, 19u); // s133 rotl
|
||||
r4 = rotr_var(r4, r3); // s134 rotr
|
||||
r1 = r1 ^ simd_shuffle_xor(r2, (ushort)16); // s135 shfl
|
||||
r5 = r5 - r2; // s136 sub
|
||||
r3 = r3 ^ r5; // s137 xor
|
||||
r0 = r0 - r2; // s138 sub
|
||||
r3 = r3 * r7; // s139 mul
|
||||
r5 = r5 + r3 + select(0x82ab98f4u, 0x2d465ca8u, ((sel >> 24u) & 1u) != 0u); // s140 add
|
||||
r4 = r4 + r1 + select(0x80594390u, 0x237a9d8eu, ((sel >> 11u) & 1u) != 0u); // s141 add
|
||||
r3 = r7 * r0 + r3; // s142 mad
|
||||
r0 = r0 | r3; // s143 or
|
||||
r0 = mulhi(r0, r1); // s144 mulhi
|
||||
r6 = r6 ^ r2; // s145 xor
|
||||
r7 = r7 ^ simd_shuffle_xor(r1, (ushort)16); // s146 shfl
|
||||
r0 = r0 + r1 + select(0x07f15148u, 0xe68cf57eu, ((sel >> 18u) & 1u) != 0u); // s147 add
|
||||
r7 = r7 | r3; // s148 or
|
||||
r4 = r4 ^ r2; // s149 xor
|
||||
r7 = r7 ^ r1; // s150 xor
|
||||
r5 = rotr_var(r5, r2); // s151 rotr
|
||||
r0 = r0 * r7; // s152 mul
|
||||
r4 = rotl_imm(r4, 16u); // s153 rotl
|
||||
r6 = rotr_var(r6, r4); // s154 rotr
|
||||
r6 = r6 ^ r1; // s155 xor
|
||||
r6 = r6 | r3; // s156 or
|
||||
r1 = r1 + r6 + select(0x14878c5au, 0x84c2a09du, ((sel >> 19u) & 1u) != 0u); // s157 add
|
||||
r4 = r4 - r1; // s158 sub
|
||||
r4 = r6 * r7 + r4; // s159 mad
|
||||
r1 = r1 * r5; // s160 mul
|
||||
r4 = r4 | r0; // s161 or
|
||||
r7 = rotl_imm(r7, 21u); // s162 rotl
|
||||
r0 = r0 - r1; // s163 sub
|
||||
r1 = r1 * r0; // s164 mul
|
||||
r2 = r2 - r1; // s165 sub
|
||||
r0 = r0 ^ simd_shuffle_xor(r6, (ushort)2); // s166 shfl
|
||||
r0 = r0 - r4; // s167 sub
|
||||
r1 = rotr_var(r1, r6); // s168 rotr
|
||||
r7 = rotr_var(r7, r1); // s169 rotr
|
||||
r3 = r1 * r3 + r3; // s170 mad
|
||||
r2 = r2 ^ simd_shuffle_xor(r0, (ushort)4); // s171 shfl
|
||||
r2 = r2 - r1; // s172 sub
|
||||
r7 = r7 ^ r6; // s173 xor
|
||||
r4 = rotr_var(r4, r1); // s174 rotr
|
||||
r4 = r4 ^ simd_shuffle_xor(r7, (ushort)2); // s175 shfl
|
||||
r6 = r6 * r2; // s176 mul
|
||||
r0 = r0 ^ simd_shuffle_xor(r2, (ushort)4); // s177 shfl
|
||||
r6 = r6 + r4 + select(0x4fa43965u, 0x509871e6u, ((sel >> 21u) & 1u) != 0u); // s178 add
|
||||
r0 = r0 ^ r6; // s179 xor
|
||||
r5 = r5 + r7 + select(0xfb3c4bd8u, 0xb95b8a53u, ((sel >> 0u) & 1u) != 0u); // s180 add
|
||||
r6 = r6 * r3; // s181 mul
|
||||
r7 = rotl_imm(r7, 1u); // s182 rotl
|
||||
r0 = r0 + r7 + select(0xc448a197u, 0x7047b7cfu, ((sel >> 11u) & 1u) != 0u); // s183 add
|
||||
r3 = r3 * r5; // s184 mul
|
||||
r5 = r5 ^ simd_shuffle_xor(r7, (ushort)1); // s185 shfl
|
||||
r0 = r0 * r2; // s186 mul
|
||||
r4 = mulhi(r4, r7); // s187 mulhi
|
||||
r2 = r7 * r2 + r2; // s188 mad
|
||||
r7 = r7 - r2; // s189 sub
|
||||
r2 = r2 + r5 + select(0x2fceaf49u, 0x5caccc5du, ((sel >> 19u) & 1u) != 0u); // s190 add
|
||||
r1 = r1 ^ simd_shuffle_xor(r2, (ushort)1); // s191 shfl
|
||||
r7 = r7 + r1 + select(0xba0cee62u, 0x857bb5feu, ((sel >> 20u) & 1u) != 0u); // s192 add
|
||||
r6 = rotl_imm(r6, 22u); // s193 rotl
|
||||
r3 = r3 ^ r7; // s194 xor
|
||||
r7 = r7 * r0; // s195 mul
|
||||
r3 = r3 ^ r5; // s196 xor
|
||||
r5 = r5 + r1 + select(0x8519428cu, 0xeae84577u, ((sel >> 29u) & 1u) != 0u); // s197 add
|
||||
r7 = r4 * r4 + r7; // s198 mad
|
||||
r3 = r3 + r6 + select(0x468639d3u, 0x0bfdbfa1u, ((sel >> 8u) & 1u) != 0u); // s199 add
|
||||
r5 = r3 * r4 + r5; // s200 mad
|
||||
r2 = r2 ^ simd_shuffle_xor(r7, (ushort)4); // s201 shfl
|
||||
r7 = r7 + r4 + select(0x18ec9a69u, 0x53ee8e50u, ((sel >> 23u) & 1u) != 0u); // s202 add
|
||||
r6 = r6 ^ r0; // s203 xor
|
||||
r4 = r4 + r5 + select(0x3e81485bu, 0xac63376eu, ((sel >> 4u) & 1u) != 0u); // s204 add
|
||||
r7 = r1 * r4 + r7; // s205 mad
|
||||
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)16); // s206 shfl
|
||||
r4 = r4 * r0; // s207 mul
|
||||
r1 = r1 | r6; // s208 or
|
||||
r6 = r6 + r4 + select(0x66ccb75du, 0x5b745519u, ((sel >> 28u) & 1u) != 0u); // s209 add
|
||||
r1 = rotr_var(r1, r3); // s210 rotr
|
||||
r5 = r5 - r4; // s211 sub
|
||||
r4 = r4 ^ r7; // s212 xor
|
||||
r1 = r1 + r7 + select(0x0ce0553du, 0x1c3dccf7u, ((sel >> 8u) & 1u) != 0u); // s213 add
|
||||
r0 = mulhi(r0, r1); // s214 mulhi
|
||||
r3 = r3 + r6 + select(0x96bfec89u, 0xfadc9205u, ((sel >> 9u) & 1u) != 0u); // s215 add
|
||||
r0 = rotl_imm(r0, 4u); // s216 rotl
|
||||
r6 = r6 ^ r4; // s217 xor
|
||||
r6 = r6 ^ simd_shuffle_xor(r7, (ushort)2); // s218 shfl
|
||||
r7 = r0 * r1 + r7; // s219 mad
|
||||
r4 = r4 * r0; // s220 mul
|
||||
r2 = r6 * r0 + r2; // s221 mad
|
||||
r5 = r5 ^ r4; // s222 xor
|
||||
r0 = r0 * r5; // s223 mul
|
||||
r2 = r2 + r0 + select(0x61495681u, 0x66b15c04u, ((sel >> 9u) & 1u) != 0u); // s224 add
|
||||
r3 = rotr_var(r3, r2); // s225 rotr
|
||||
r2 = r3 * r3 + r2; // s226 mad
|
||||
r6 = r6 ^ r0; // s227 xor
|
||||
r4 = rotl_imm(r4, 30u); // s228 rotl
|
||||
r2 = r2 + r3 + select(0x70d05c34u, 0x25bc17c2u, ((sel >> 29u) & 1u) != 0u); // s229 add
|
||||
r1 = rotr_var(r1, r5); // s230 rotr
|
||||
r1 = r1 + r4 + select(0x32a28384u, 0x2d8728f3u, ((sel >> 8u) & 1u) != 0u); // s231 add
|
||||
r1 = rotl_imm(r1, 6u); // s232 rotl
|
||||
r2 = r2 + r0 + select(0xcf0949f1u, 0x61e81fe6u, ((sel >> 21u) & 1u) != 0u); // s233 add
|
||||
r4 = rotl_imm(r4, 2u); // s234 rotl
|
||||
r2 = r2 - r6; // s235 sub
|
||||
r6 = r6 | r3; // s236 or
|
||||
r7 = r7 ^ simd_shuffle_xor(r5, (ushort)4); // s237 shfl
|
||||
r1 = r1 ^ r3; // s238 xor
|
||||
r2 = rotl_imm(r2, 1u); // s239 rotl
|
||||
r5 = rotr_var(r5, r2); // s240 rotr
|
||||
r4 = rotl_imm(r4, 10u); // s241 rotl
|
||||
r6 = r6 + r2 + select(0xe204fb50u, 0x0a67565du, ((sel >> 16u) & 1u) != 0u); // s242 add
|
||||
r1 = r1 ^ simd_shuffle_xor(r3, (ushort)2); // s243 shfl
|
||||
r0 = r0 ^ simd_shuffle_xor(r1, (ushort)16); // s244 shfl
|
||||
r5 = r5 ^ simd_shuffle_xor(r7, (ushort)2); // s245 shfl
|
||||
r4 = r4 ^ r5; // s246 xor
|
||||
r7 = r7 ^ r2; // s247 xor
|
||||
r0 = r0 ^ simd_shuffle_xor(r6, (ushort)16); // s248 shfl
|
||||
r1 = mulhi(r1, r3); // s249 mulhi
|
||||
r7 = r7 + r5 + select(0x3d6daffdu, 0x5a79fd6fu, ((sel >> 31u) & 1u) != 0u); // s250 add
|
||||
r6 = r6 + r0 + select(0x84334aeau, 0x6beb28c2u, ((sel >> 28u) & 1u) != 0u); // s251 add
|
||||
r7 = r7 - r1; // s252 sub
|
||||
r5 = rotr_var(r5, r2); // s253 rotr
|
||||
r2 = mulhi(r2, r5); // s254 mulhi
|
||||
r6 = r6 * r3; // s255 mul
|
||||
}
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
BIN
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/state.igsd1
Normal file
BIN
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/state.igsd1
Normal file
Binary file not shown.
57
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/vectors.h
Normal file
57
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/vectors.h
Normal file
|
|
@ -0,0 +1,57 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925/day/69676e65756d2d6461792ffd50000000000000". Do not edit by hand.
|
||||
// Expected outputs: igneum-pow (Rust) CPU interpreter, generator v5, memory-hard dataset
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
|
||||
#define IGNEUM_VEC_WARPS 3
|
||||
static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u };
|
||||
static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {
|
||||
{ // base nonce 0
|
||||
0xe552166a03298f7full, 0xf526f1c5bbcce3feull, 0xd2599307c7dff26dull, 0x292f1d7c62935cefull, 0x176dd4b8dc247c8eull, 0x916c618e3f222f3cull, 0xc4b4cd32e2b17869ull, 0x083985b4514fdb52ull,
|
||||
0x471c78fc23623557ull, 0x3ef558f8f91fef48ull, 0xc3a23b4f40a9b42eull, 0xe0cad6d14d62a59eull, 0x23be7de4ce7cde95ull, 0x3eb342c0a40d848aull, 0x93bd93cc7f35be92ull, 0xd468fe59ee140d1aull,
|
||||
0x242f02aefcdd4907ull, 0x4a0af1c5c1807a38ull, 0x40cc94e1ba07667dull, 0xb20da649934a63f1ull, 0x172560c509523166ull, 0x7608ac11340f39efull, 0x4b401d953c834cbaull, 0x66017b835f9d80e2ull,
|
||||
0x9150f03bfeac28d3ull, 0xdf54e85028a3591aull, 0x637a68f8ad0a322full, 0x48f87298734fdb52ull, 0x65f3200db1282d59ull, 0x6c48d31c96caa0e4ull, 0x3b06a9d989aa3eb7ull, 0x316f5a49394af8abull
|
||||
},
|
||||
{ // base nonce 4096
|
||||
0xd2d4d57cca870a92ull, 0xc2a46c89e9e8d2e8ull, 0xe9d5a1048e9e357full, 0xa482ea1a68fb0fdfull, 0x9d4c01c50ebafb81ull, 0xe844ac46bf40758aull, 0x4fccaf252c2cbbf7ull, 0x90f96462f44878a4ull,
|
||||
0x07bad307b27ea868ull, 0xf93fba205fd5a9ccull, 0xc08d0da873abfff4ull, 0x5fcb36b12eabd4b3ull, 0x48cbb9465226c503ull, 0x7281def60f0c6982ull, 0x609c4a01b5ff6b09ull, 0x505e0d5083c5ebfaull,
|
||||
0x2e6370a4045ee3daull, 0x16df206ad3197bf6ull, 0xb63552c2e683f34aull, 0xb48d70f0f4fe8c3cull, 0xe4f941f4d5cac0bdull, 0xfe43ac9df2145859ull, 0x5964c115d0d9616full, 0xc2fb078d60b1b21cull,
|
||||
0x663a7905f56d4585ull, 0xfad4000340b1e774ull, 0x0e9cf62f4e5b5fe9ull, 0xa284122fdd4037e3ull, 0x0e24488e1fe86f6eull, 0x563f0ca5d9d78390ull, 0x3b9c6c5bb8ca2771ull, 0xe2dafada40c22022ull
|
||||
},
|
||||
{ // base nonce 1000000
|
||||
0x46d0145d98c681e2ull, 0x2909904cf57e5364ull, 0xb3686f5347a3920dull, 0x48c80e2427513b60ull, 0xae8b429006c2876aull, 0xdfeeddc0b4714cbbull, 0xa839ddb421d4b06bull, 0x77265ae1bde25c0bull,
|
||||
0x666fda9b8a119724ull, 0x2fba7b9e71ddd3b3ull, 0xc81c4e05c94a23feull, 0x49c52ee7e025f9bbull, 0x7951fad6f47e130dull, 0x9769f514ba0c5527ull, 0xe8cf1f573d573d41ull, 0x5f6c19433ba166d0ull,
|
||||
0x70806160896dc78bull, 0xbcbd57d0462c2b4full, 0xce880c91f263b526ull, 0x4bd3f39d60e5b6f3ull, 0x065159ea440706f5ull, 0xe4cbee1294943b62ull, 0xa2ce61093e845bdcull, 0xc6195798c0f2cdd0ull,
|
||||
0x1c4b63a0aea5fd36ull, 0xe5254c6a049e115aull, 0x9bdf1a4ca31ecf17ull, 0x64f8159b9c84ddd4ull, 0x78c6caf1a64a7293ull, 0xe360359048ecdc56ull, 0xf55b82c8ed94c781ull, 0x423ef7066859a0b0ull
|
||||
}
|
||||
};
|
||||
|
||||
// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (268435455).
|
||||
static const uint32_t IGNEUM_DS_HEAD[16] = {
|
||||
0x0790a3c8u, 0xf061f4c1u, 0xc93a88c3u, 0x0b6b0fb8u, 0x4dc6214au, 0x7e397ebau, 0x62bdca7cu, 0x2cd12c0cu,
|
||||
0xb8a05ca1u, 0xe4e4e28fu, 0xb8623289u, 0x8244cdd4u, 0x12bb8880u, 0xd8cb0b94u, 0x6c8c05b9u, 0x26fd923bu
|
||||
};
|
||||
static const uint32_t IGNEUM_DS_LAST_INDEX = 268435455u;
|
||||
static const uint32_t IGNEUM_DS_LAST = 0xc5f96c2eu;
|
||||
// 64 sampled dataset words (index, value) computed on the Mac.
|
||||
#define IGNEUM_DS_SAMPLES 64
|
||||
static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = {
|
||||
59471966u, 217795994u, 208353206u, 42483309u, 172547758u, 148076330u, 183853158u, 214389424u, 267488061u, 169781097u, 184093494u, 153880993u, 84977930u, 46426879u, 3093825u, 225364072u, 44593546u, 260713159u, 168250303u, 52384140u, 223401610u, 45554030u, 95410555u, 175039924u, 79171087u, 267580473u, 24168642u, 37981670u, 171551130u, 195559979u, 204611762u, 140997658u, 138925853u, 86637313u, 20736778u, 219665210u, 160430336u, 264654675u, 8013395u, 228945585u, 213884386u, 104419827u, 44185464u, 142737231u, 99284897u, 132475900u, 61861762u, 132056166u, 262388043u, 91878046u, 117353561u, 124768597u, 71352993u, 190698941u, 46055428u, 55281366u, 165145231u, 106810753u, 171985651u, 232085256u, 159510492u, 40072060u, 209107596u, 39023794u
|
||||
};
|
||||
static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = {
|
||||
0x9dead642u, 0x94ef5e4au, 0x74892075u, 0xbc8466d3u, 0x5da3db7au, 0x5f7d336cu, 0xdb6e2d4bu, 0x707871e4u, 0xab1a8c26u, 0x095645e2u, 0x009e1047u, 0x0e043163u, 0x7f7a83f7u, 0xc80d8368u, 0x5c82379du, 0x6f040f0au, 0x9c15d89bu, 0x2e8158f1u, 0x899eb1dbu, 0x6b7b78ceu, 0x87c98f2du, 0x9bf58c2cu, 0xfe7c3758u, 0x135eb1c6u, 0x2241ea1cu, 0x58a5f34bu, 0x96891cc6u, 0xf3e92116u, 0xca2b5299u, 0xc48760fbu, 0xbfc8e6ecu, 0xf10dbadfu, 0x19672a34u, 0x2b38c7f0u, 0x2cc23ef9u, 0x732c2e9au, 0xa733b519u, 0xb4f99cc9u, 0xe0f2ceddu, 0xe405baa6u, 0x2515e05eu, 0xb5fe054eu, 0x62e77ca0u, 0x18a85e70u, 0x8c0deaf1u, 0x42a2f9c8u, 0x06e2ad43u, 0x14a20999u, 0x1d817d84u, 0x3493d381u, 0xc60d470bu, 0x6366a26bu, 0xfa46d099u, 0xa3d3c90eu, 0xcc73ca26u, 0x3ed2d84eu, 0xfca19a4fu, 0xa43b015eu, 0x1f5203b5u, 0x3bfb700cu, 0x1cb4554cu, 0x4c45d8afu, 0x271554b7u, 0xfb3d3fb8u
|
||||
};
|
||||
// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^26 words.
|
||||
static const uint32_t IGNEUM_CACHE_HEAD[16] = {
|
||||
0x47e15959u, 0xf138e163u, 0x4fc0aa19u, 0xbe50897cu, 0xddcc43a9u, 0xfbecd8bau, 0x3be5e048u, 0xc330cb77u,
|
||||
0xc00ff902u, 0x62eb64d7u, 0x825ef196u, 0x9bd34544u, 0x84c52834u, 0x6f7303eau, 0x5546e71fu, 0x3f0b835cu
|
||||
};
|
||||
static const uint32_t IGNEUM_CACHE_LAST[16] = {
|
||||
0xfe146921u, 0x0d972fc0u, 0xa5383d3eu, 0xf751aaf9u, 0x52509092u, 0x447ad742u, 0x18253e5cu, 0x15dae1fbu,
|
||||
0xff5b8c7du, 0x8914179bu, 0xdfec91fau, 0x80d3d462u, 0xec0a5caeu, 0xa8af9a8bu, 0x1b7935a6u, 0x1026962fu
|
||||
};
|
||||
static const uint64_t IGNEUM_CACHE_FNV64 = 0x7334fa46e5d972ebull;
|
||||
36
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/vectors.json
Normal file
36
proto-cuda/packs-ca3-v5/v5-dn3-epoch0/vectors.json
Normal file
|
|
@ -0,0 +1,36 @@
|
|||
{
|
||||
"seed": "igneum-epoch/4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925/day/69676e65756d2d6461792ffd50000000000000",
|
||||
"day": "bytes:69676e65756d2d6461792ffd50000000000000",
|
||||
"dataset_mode": "memory-hard",
|
||||
"dataset_log2_words": 28,
|
||||
"mask": "0x0fffffff",
|
||||
"lanes": 32,
|
||||
"source": "igneum-pow (Rust) CPU interpreter, generator v5, memory-hard dataset",
|
||||
"warps": [
|
||||
{"base_nonce": 0, "expected": [
|
||||
"0xe552166a03298f7f", "0xf526f1c5bbcce3fe", "0xd2599307c7dff26d", "0x292f1d7c62935cef", "0x176dd4b8dc247c8e", "0x916c618e3f222f3c", "0xc4b4cd32e2b17869", "0x083985b4514fdb52",
|
||||
"0x471c78fc23623557", "0x3ef558f8f91fef48", "0xc3a23b4f40a9b42e", "0xe0cad6d14d62a59e", "0x23be7de4ce7cde95", "0x3eb342c0a40d848a", "0x93bd93cc7f35be92", "0xd468fe59ee140d1a",
|
||||
"0x242f02aefcdd4907", "0x4a0af1c5c1807a38", "0x40cc94e1ba07667d", "0xb20da649934a63f1", "0x172560c509523166", "0x7608ac11340f39ef", "0x4b401d953c834cba", "0x66017b835f9d80e2",
|
||||
"0x9150f03bfeac28d3", "0xdf54e85028a3591a", "0x637a68f8ad0a322f", "0x48f87298734fdb52", "0x65f3200db1282d59", "0x6c48d31c96caa0e4", "0x3b06a9d989aa3eb7", "0x316f5a49394af8ab"
|
||||
]},
|
||||
{"base_nonce": 4096, "expected": [
|
||||
"0xd2d4d57cca870a92", "0xc2a46c89e9e8d2e8", "0xe9d5a1048e9e357f", "0xa482ea1a68fb0fdf", "0x9d4c01c50ebafb81", "0xe844ac46bf40758a", "0x4fccaf252c2cbbf7", "0x90f96462f44878a4",
|
||||
"0x07bad307b27ea868", "0xf93fba205fd5a9cc", "0xc08d0da873abfff4", "0x5fcb36b12eabd4b3", "0x48cbb9465226c503", "0x7281def60f0c6982", "0x609c4a01b5ff6b09", "0x505e0d5083c5ebfa",
|
||||
"0x2e6370a4045ee3da", "0x16df206ad3197bf6", "0xb63552c2e683f34a", "0xb48d70f0f4fe8c3c", "0xe4f941f4d5cac0bd", "0xfe43ac9df2145859", "0x5964c115d0d9616f", "0xc2fb078d60b1b21c",
|
||||
"0x663a7905f56d4585", "0xfad4000340b1e774", "0x0e9cf62f4e5b5fe9", "0xa284122fdd4037e3", "0x0e24488e1fe86f6e", "0x563f0ca5d9d78390", "0x3b9c6c5bb8ca2771", "0xe2dafada40c22022"
|
||||
]},
|
||||
{"base_nonce": 1000000, "expected": [
|
||||
"0x46d0145d98c681e2", "0x2909904cf57e5364", "0xb3686f5347a3920d", "0x48c80e2427513b60", "0xae8b429006c2876a", "0xdfeeddc0b4714cbb", "0xa839ddb421d4b06b", "0x77265ae1bde25c0b",
|
||||
"0x666fda9b8a119724", "0x2fba7b9e71ddd3b3", "0xc81c4e05c94a23fe", "0x49c52ee7e025f9bb", "0x7951fad6f47e130d", "0x9769f514ba0c5527", "0xe8cf1f573d573d41", "0x5f6c19433ba166d0",
|
||||
"0x70806160896dc78b", "0xbcbd57d0462c2b4f", "0xce880c91f263b526", "0x4bd3f39d60e5b6f3", "0x065159ea440706f5", "0xe4cbee1294943b62", "0xa2ce61093e845bdc", "0xc6195798c0f2cdd0",
|
||||
"0x1c4b63a0aea5fd36", "0xe5254c6a049e115a", "0x9bdf1a4ca31ecf17", "0x64f8159b9c84ddd4", "0x78c6caf1a64a7293", "0xe360359048ecdc56", "0xf55b82c8ed94c781", "0x423ef7066859a0b0"
|
||||
]}
|
||||
],
|
||||
"dataset_head": ["0x0790a3c8", "0xf061f4c1", "0xc93a88c3", "0x0b6b0fb8", "0x4dc6214a", "0x7e397eba", "0x62bdca7c", "0x2cd12c0c", "0xb8a05ca1", "0xe4e4e28f", "0xb8623289", "0x8244cdd4", "0x12bb8880", "0xd8cb0b94", "0x6c8c05b9", "0x26fd923b"],
|
||||
"dataset_last_index": 268435455,
|
||||
"dataset_last": "0xc5f96c2e",
|
||||
"dataset_samples": [{"index": 59471966, "value": "0x9dead642"}, {"index": 217795994, "value": "0x94ef5e4a"}, {"index": 208353206, "value": "0x74892075"}, {"index": 42483309, "value": "0xbc8466d3"}, {"index": 172547758, "value": "0x5da3db7a"}, {"index": 148076330, "value": "0x5f7d336c"}, {"index": 183853158, "value": "0xdb6e2d4b"}, {"index": 214389424, "value": "0x707871e4"}, {"index": 267488061, "value": "0xab1a8c26"}, {"index": 169781097, "value": "0x095645e2"}, {"index": 184093494, "value": "0x009e1047"}, {"index": 153880993, "value": "0x0e043163"}, {"index": 84977930, "value": "0x7f7a83f7"}, {"index": 46426879, "value": "0xc80d8368"}, {"index": 3093825, "value": "0x5c82379d"}, {"index": 225364072, "value": "0x6f040f0a"}, {"index": 44593546, "value": "0x9c15d89b"}, {"index": 260713159, "value": "0x2e8158f1"}, {"index": 168250303, "value": "0x899eb1db"}, {"index": 52384140, "value": "0x6b7b78ce"}, {"index": 223401610, "value": "0x87c98f2d"}, {"index": 45554030, "value": "0x9bf58c2c"}, {"index": 95410555, "value": "0xfe7c3758"}, {"index": 175039924, "value": "0x135eb1c6"}, {"index": 79171087, "value": "0x2241ea1c"}, {"index": 267580473, "value": "0x58a5f34b"}, {"index": 24168642, "value": "0x96891cc6"}, {"index": 37981670, "value": "0xf3e92116"}, {"index": 171551130, "value": "0xca2b5299"}, {"index": 195559979, "value": "0xc48760fb"}, {"index": 204611762, "value": "0xbfc8e6ec"}, {"index": 140997658, "value": "0xf10dbadf"}, {"index": 138925853, "value": "0x19672a34"}, {"index": 86637313, "value": "0x2b38c7f0"}, {"index": 20736778, "value": "0x2cc23ef9"}, {"index": 219665210, "value": "0x732c2e9a"}, {"index": 160430336, "value": "0xa733b519"}, {"index": 264654675, "value": "0xb4f99cc9"}, {"index": 8013395, "value": "0xe0f2cedd"}, {"index": 228945585, "value": "0xe405baa6"}, {"index": 213884386, "value": "0x2515e05e"}, {"index": 104419827, "value": "0xb5fe054e"}, {"index": 44185464, "value": "0x62e77ca0"}, {"index": 142737231, "value": "0x18a85e70"}, {"index": 99284897, "value": "0x8c0deaf1"}, {"index": 132475900, "value": "0x42a2f9c8"}, {"index": 61861762, "value": "0x06e2ad43"}, {"index": 132056166, "value": "0x14a20999"}, {"index": 262388043, "value": "0x1d817d84"}, {"index": 91878046, "value": "0x3493d381"}, {"index": 117353561, "value": "0xc60d470b"}, {"index": 124768597, "value": "0x6366a26b"}, {"index": 71352993, "value": "0xfa46d099"}, {"index": 190698941, "value": "0xa3d3c90e"}, {"index": 46055428, "value": "0xcc73ca26"}, {"index": 55281366, "value": "0x3ed2d84e"}, {"index": 165145231, "value": "0xfca19a4f"}, {"index": 106810753, "value": "0xa43b015e"}, {"index": 171985651, "value": "0x1f5203b5"}, {"index": 232085256, "value": "0x3bfb700c"}, {"index": 159510492, "value": "0x1cb4554c"}, {"index": 40072060, "value": "0x4c45d8af"}, {"index": 209107596, "value": "0x271554b7"}, {"index": 39023794, "value": "0xfb3d3fb8"}],
|
||||
"cache_head": ["0x47e15959", "0xf138e163", "0x4fc0aa19", "0xbe50897c", "0xddcc43a9", "0xfbecd8ba", "0x3be5e048", "0xc330cb77", "0xc00ff902", "0x62eb64d7", "0x825ef196", "0x9bd34544", "0x84c52834", "0x6f7303ea", "0x5546e71f", "0x3f0b835c"],
|
||||
"cache_last_line": ["0xfe146921", "0x0d972fc0", "0xa5383d3e", "0xf751aaf9", "0x52509092", "0x447ad742", "0x18253e5c", "0x15dae1fb", "0xff5b8c7d", "0x8914179b", "0xdfec91fa", "0x80d3d462", "0xec0a5cae", "0xa8af9a8b", "0x1b7935a6", "0x1026962f"],
|
||||
"cache_fnv1a64": "0x7334fa46e5d972eb"
|
||||
}
|
||||
542
proto-cuda/packs-ca3-v5/v5-genesis/kernel.cl
Normal file
542
proto-cuda/packs-ca3-v5/v5-genesis/kernel.cl
Normal file
|
|
@ -0,0 +1,542 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
|
||||
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
|
||||
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
|
||||
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
|
||||
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
|
||||
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
|
||||
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
|
||||
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
|
||||
#ifndef IGNEUM_GROUP
|
||||
#define IGNEUM_GROUP 32
|
||||
#endif
|
||||
#ifndef IGNEUM_EXCHANGE
|
||||
#define IGNEUM_EXCHANGE 0
|
||||
#endif
|
||||
#ifdef __OPENCL_VERSION__
|
||||
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
|
||||
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#ifdef cl_khr_subgroups
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
|
||||
#endif
|
||||
#ifdef cl_khr_subgroup_shuffle
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
|
||||
#endif
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
|
||||
#endif
|
||||
#else
|
||||
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
|
||||
#include "emu_opencl.h"
|
||||
#endif
|
||||
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#else
|
||||
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
|
||||
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
|
||||
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
|
||||
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
|
||||
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
|
||||
#endif
|
||||
|
||||
static inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
|
||||
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
|
||||
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
|
||||
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
|
||||
static inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of 8 x seed-parameterised mixer + one 64-byte cache read, then 8 x final mixer (class v3, mixer multiplier 8,
|
||||
// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
static inline void mh_chacha_block(const uint* x, uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
static inline void mh_cache_segment(__global uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
static inline void mh_mixer(uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of 8 x mixer + cache line s[0] & mask; 8 x final mixer.
|
||||
static inline void mh_item(__global const uint* cache, __global const uint* leaf, uint t, uint* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= leaf[i];
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
for (uint j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 8u + j + 1u));
|
||||
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
for (uint j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (64u + j + 1u));
|
||||
}
|
||||
// Class v5 (docs/design/class-v5-stored-state.md): leaf(t) = leaves[t mod nLeaves], 16 words per leaf (leaves.bin).
|
||||
static inline __global const uint* mh_leaf(__global const uint* leaves, uint nLeaves, uint t) { return leaves + ((t % nLeaves) * 16u); }
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
static inline uint mh_word(__global const uint* cache, __global const uint* leaves, uint nLeaves, uint w) { uint s[16]; mh_item(cache, mh_leaf(leaves, nLeaves, w >> 4u), w >> 4u, s); return s[w & 15u]; }
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
|
||||
// The same constants as memhard.h in this pack (one emitter, three dialects).
|
||||
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
|
||||
uint seg = (uint)get_global_id(0);
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
// Class v5: the window's leaves (leaves.bin, IGNEUM_STATE_LEAVES x 16 words) and their count.
|
||||
__kernel void igneum_build(__global uint* ds, __global const uint* cache, __global const uint* leaves, uint nLeaves, uint nItems) {
|
||||
uint t = (uint)get_global_id(0);
|
||||
if (t < nItems) {
|
||||
uint s[16];
|
||||
mh_item(cache, mh_leaf(leaves, nLeaves, t), t, s);
|
||||
__global uint* d = ds + ((ulong)t * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask) {
|
||||
uint gid = (uint)get_global_id(0);
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r2 = r1 * r1 + r2; // 1 mad
|
||||
r2 = r3 * r2 + r2; // 2 mad
|
||||
r3 = r3 ^ r5; // 3 xor
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r5 = r5 ^ ds[r7 & mask]; // 5 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl
|
||||
r1 = mul_hi(r1, r5); // 8 mulhi
|
||||
r6 = rotr_var(r6, r3); // 9 rotr
|
||||
r3 = r3 | r4; // 10 or
|
||||
r4 = r4 ^ ds[r6 & mask]; // 11 load
|
||||
r0 = mul_hi(r0, r4); // 12 mulhi
|
||||
r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add
|
||||
r0 = r0 ^ ds[r5 & mask]; // 14 load
|
||||
r2 = r2 - r4; // 15 sub
|
||||
r2 = r2 ^ ds[r7 & mask]; // 16 load
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl
|
||||
r5 = r5 * r0; // 19 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl
|
||||
r6 = mul_hi(r6, r2); // 22 mulhi
|
||||
r6 = r6 ^ ds[r3 & mask]; // 23 load
|
||||
r5 = r5 * r0; // 24 mul
|
||||
r5 = rotl_imm(r5, 19u); // 25 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl
|
||||
r0 = r0 ^ r5; // 27 xor
|
||||
r0 = r0 ^ r4; // 28 xor
|
||||
r3 = r3 - r0; // 29 sub
|
||||
r5 = r5 * r1; // 30 mul
|
||||
r7 = r7 ^ ds[r3 & mask]; // 31 load
|
||||
r1 = r1 ^ ds[r2 & mask]; // 32 load
|
||||
r5 = r5 ^ r6; // 33 xor
|
||||
r5 = r5 ^ ds[r0 & mask]; // 34 load
|
||||
r0 = mul_hi(r0, r5); // 35 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl
|
||||
r7 = r7 ^ ds[r6 & mask]; // 37 load
|
||||
r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl
|
||||
r2 = r2 ^ r5; // 40 xor
|
||||
r3 = r6 * r3 + r3; // 41 mad
|
||||
r6 = r6 - r7; // 42 sub
|
||||
r7 = r7 ^ r0; // 43 xor
|
||||
r1 = r1 ^ ds[r7 & mask]; // 44 load
|
||||
r2 = r2 * r3; // 45 mul
|
||||
r1 = mul_hi(r1, r5); // 46 mulhi
|
||||
r4 = r4 - r3; // 47 sub
|
||||
r2 = rotr_var(r2, r6); // 48 rotr
|
||||
r3 = r3 ^ ds[r4 & mask]; // 49 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add
|
||||
r0 = r0 * r2; // 51 mul
|
||||
r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add
|
||||
r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add
|
||||
r7 = rotl_imm(r7, 14u); // 54 rotl
|
||||
r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add
|
||||
r6 = r6 ^ ds[r7 & mask]; // 56 load
|
||||
r1 = rotr_var(r1, r5); // 57 rotr
|
||||
r5 = r5 ^ ds[r3 & mask]; // 58 load
|
||||
r6 = r6 ^ ds[r1 & mask]; // 59 load
|
||||
r3 = r5 * r0 + r3; // 60 mad
|
||||
r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add
|
||||
r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add
|
||||
r5 = rotl_imm(r5, 19u); // 63 rotl
|
||||
// latency-shadow block (Counter ASIC 3.0 item 8): 256 ALU instructions x 27 passes after instruction 63, no load
|
||||
for (uint sh = 0u; sh < 27u; ++sh) {
|
||||
r5 = r5 + r2 + ((((sel >> 18u) & 1u) != 0u) ? 0x8bc12da9u : 0x92199f99u); // s0 add
|
||||
r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x63079e5au : 0x8d72d3adu); // s1 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 2u); r6 = r6 ^ t_; } // s2 shfl
|
||||
r4 = r4 - r2; // s3 sub
|
||||
r7 = r7 + r0 + ((((sel >> 31u) & 1u) != 0u) ? 0x5d4c7a60u : 0xb21b4babu); // s4 add
|
||||
r0 = rotl_imm(r0, 11u); // s5 rotl
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // s6 add
|
||||
r7 = r7 ^ r1; // s7 xor
|
||||
r1 = r6 * r5 + r1; // s8 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 4u); r6 = r6 ^ t_; } // s9 shfl
|
||||
r1 = r2 * r2 + r1; // s10 mad
|
||||
r5 = r0 * r3 + r5; // s11 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 4u); r2 = r2 ^ t_; } // s12 shfl
|
||||
r1 = r1 + r5 + ((((sel >> 25u) & 1u) != 0u) ? 0xbc48c63eu : 0xbdf8f9a5u); // s13 add
|
||||
r1 = rotl_imm(r1, 29u); // s14 rotl
|
||||
r1 = r1 - r4; // s15 sub
|
||||
r7 = r7 | r1; // s16 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r2 = r2 ^ t_; } // s17 shfl
|
||||
r7 = r7 ^ r4; // s18 xor
|
||||
r6 = r6 * r1; // s19 mul
|
||||
r5 = r6 * r0 + r5; // s20 mad
|
||||
r3 = r3 - r1; // s21 sub
|
||||
r6 = r6 * r0; // s22 mul
|
||||
r2 = r2 + r0 + ((((sel >> 1u) & 1u) != 0u) ? 0x96e8f127u : 0x45c37cecu); // s23 add
|
||||
r6 = r6 - r4; // s24 sub
|
||||
r7 = r3 * r4 + r7; // s25 mad
|
||||
r3 = rotl_imm(r3, 9u); // s26 rotl
|
||||
r2 = r2 - r1; // s27 sub
|
||||
r6 = mul_hi(r6, r0); // s28 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r2 = r2 ^ t_; } // s29 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r1 = r1 ^ t_; } // s30 shfl
|
||||
r1 = r1 + r6 + ((((sel >> 1u) & 1u) != 0u) ? 0xb1cdb2abu : 0x37985632u); // s31 add
|
||||
r0 = rotr_var(r0, r1); // s32 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // s33 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r0 = r0 ^ t_; } // s34 shfl
|
||||
r7 = r7 * r0; // s35 mul
|
||||
r3 = r3 + r1 + ((((sel >> 0u) & 1u) != 0u) ? 0x856e0180u : 0x804e777eu); // s36 add
|
||||
r1 = r1 ^ r6; // s37 xor
|
||||
r2 = r2 * r7; // s38 mul
|
||||
r6 = r6 + r5 + ((((sel >> 2u) & 1u) != 0u) ? 0xe9bd0cc4u : 0x0b06c8a7u); // s39 add
|
||||
r5 = r5 * r7; // s40 mul
|
||||
r2 = mul_hi(r2, r3); // s41 mulhi
|
||||
r2 = r2 ^ r0; // s42 xor
|
||||
r0 = rotl_imm(r0, 4u); // s43 rotl
|
||||
r4 = mul_hi(r4, r3); // s44 mulhi
|
||||
r6 = r6 + r7 + ((((sel >> 12u) & 1u) != 0u) ? 0xcdcb63f6u : 0xee822e17u); // s45 add
|
||||
r3 = r2 * r0 + r3; // s46 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r4 = r4 ^ t_; } // s47 shfl
|
||||
r6 = mul_hi(r6, r5); // s48 mulhi
|
||||
r0 = r0 - r2; // s49 sub
|
||||
r3 = r3 - r5; // s50 sub
|
||||
r1 = rotr_var(r1, r4); // s51 rotr
|
||||
r6 = r7 * r7 + r6; // s52 mad
|
||||
r5 = r5 ^ r3; // s53 xor
|
||||
r1 = r1 - r0; // s54 sub
|
||||
r5 = r5 - r6; // s55 sub
|
||||
r3 = r3 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0xd4758987u : 0x3dfad1b6u); // s56 add
|
||||
r4 = r4 + r1 + ((((sel >> 0u) & 1u) != 0u) ? 0x0602d6beu : 0xfca75bc2u); // s57 add
|
||||
r5 = mul_hi(r5, r6); // s58 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 2u); r2 = r2 ^ t_; } // s59 shfl
|
||||
r2 = rotl_imm(r2, 9u); // s60 rotl
|
||||
r4 = r4 | r6; // s61 or
|
||||
r6 = rotr_var(r6, r4); // s62 rotr
|
||||
r2 = r2 * r4; // s63 mul
|
||||
r0 = r0 ^ r5; // s64 xor
|
||||
r2 = r2 ^ r7; // s65 xor
|
||||
r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x8596687au : 0xd71c02ffu); // s66 add
|
||||
r1 = rotl_imm(r1, 21u); // s67 rotl
|
||||
r3 = r3 * r2; // s68 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 2u); r7 = r7 ^ t_; } // s69 shfl
|
||||
r3 = r3 * r2; // s70 mul
|
||||
r0 = r2 * r5 + r0; // s71 mad
|
||||
r6 = r6 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0x08ac5733u : 0x3c6fe15du); // s72 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r3 = r3 ^ t_; } // s73 shfl
|
||||
r3 = r3 * r6; // s74 mul
|
||||
r6 = r6 ^ r7; // s75 xor
|
||||
r3 = r3 | r0; // s76 or
|
||||
r2 = r2 + r4 + ((((sel >> 1u) & 1u) != 0u) ? 0xc100b495u : 0x295d5faeu); // s77 add
|
||||
r3 = mul_hi(r3, r7); // s78 mulhi
|
||||
r4 = r4 | r1; // s79 or
|
||||
r4 = rotr_var(r4, r3); // s80 rotr
|
||||
r4 = r4 + r3 + ((((sel >> 15u) & 1u) != 0u) ? 0x5cc59530u : 0x274a9221u); // s81 add
|
||||
r7 = r7 + r3 + ((((sel >> 5u) & 1u) != 0u) ? 0x141479ecu : 0xc8651f8eu); // s82 add
|
||||
r3 = rotr_var(r3, r6); // s83 rotr
|
||||
r2 = r4 * r7 + r2; // s84 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r2 = r2 ^ t_; } // s85 shfl
|
||||
r3 = r3 ^ r2; // s86 xor
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 2u); r5 = r5 ^ t_; } // s87 shfl
|
||||
r0 = r0 ^ r4; // s88 xor
|
||||
r3 = r3 + r2 + ((((sel >> 18u) & 1u) != 0u) ? 0x883c0c92u : 0x53f915c2u); // s89 add
|
||||
r7 = rotr_var(r7, r5); // s90 rotr
|
||||
r3 = r3 + r1 + ((((sel >> 31u) & 1u) != 0u) ? 0x3cc5bd20u : 0xe2be00a5u); // s91 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 1u); r7 = r7 ^ t_; } // s92 shfl
|
||||
r6 = rotr_var(r6, r2); // s93 rotr
|
||||
r7 = rotl_imm(r7, 14u); // s94 rotl
|
||||
r4 = r5 * r5 + r4; // s95 mad
|
||||
r2 = r4 * r5 + r2; // s96 mad
|
||||
r1 = r1 ^ r0; // s97 xor
|
||||
r5 = r5 * r4; // s98 mul
|
||||
r2 = r2 - r0; // s99 sub
|
||||
r7 = rotl_imm(r7, 30u); // s100 rotl
|
||||
r5 = r5 + r6 + ((((sel >> 17u) & 1u) != 0u) ? 0xbfd646cbu : 0x423fd9c9u); // s101 add
|
||||
r7 = r7 + r6 + ((((sel >> 11u) & 1u) != 0u) ? 0x19b74a43u : 0x8d3c011du); // s102 add
|
||||
r5 = rotl_imm(r5, 6u); // s103 rotl
|
||||
r0 = r0 * r4; // s104 mul
|
||||
r0 = r0 | r5; // s105 or
|
||||
r0 = r0 + r1 + ((((sel >> 15u) & 1u) != 0u) ? 0x7dcb7e18u : 0x8ce14721u); // s106 add
|
||||
r0 = rotl_imm(r0, 15u); // s107 rotl
|
||||
r4 = r4 - r2; // s108 sub
|
||||
r2 = mul_hi(r2, r0); // s109 mulhi
|
||||
r1 = mul_hi(r1, r0); // s110 mulhi
|
||||
r0 = r0 + r2 + ((((sel >> 28u) & 1u) != 0u) ? 0x3c179ad8u : 0x2d9fc6b9u); // s111 add
|
||||
r2 = mul_hi(r2, r0); // s112 mulhi
|
||||
r6 = r6 - r3; // s113 sub
|
||||
r7 = r6 * r3 + r7; // s114 mad
|
||||
r5 = rotr_var(r5, r1); // s115 rotr
|
||||
r6 = rotr_var(r6, r4); // s116 rotr
|
||||
r2 = r2 + r1 + ((((sel >> 22u) & 1u) != 0u) ? 0x47359729u : 0x502138e2u); // s117 add
|
||||
r3 = r2 * r0 + r3; // s118 mad
|
||||
r1 = r1 * r3; // s119 mul
|
||||
r2 = r0 * r4 + r2; // s120 mad
|
||||
r0 = r0 * r3; // s121 mul
|
||||
r2 = mul_hi(r2, r0); // s122 mulhi
|
||||
r5 = r5 * r6; // s123 mul
|
||||
r4 = r2 * r4 + r4; // s124 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 2u); r4 = r4 ^ t_; } // s125 shfl
|
||||
r2 = r2 ^ r0; // s126 xor
|
||||
r1 = rotl_imm(r1, 7u); // s127 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r7 = r7 ^ t_; } // s128 shfl
|
||||
r6 = rotl_imm(r6, 17u); // s129 rotl
|
||||
r2 = r2 + r5 + ((((sel >> 15u) & 1u) != 0u) ? 0x6c57e4e7u : 0x33dff776u); // s130 add
|
||||
r0 = r0 | r3; // s131 or
|
||||
r5 = rotr_var(r5, r0); // s132 rotr
|
||||
r2 = r2 + r3 + ((((sel >> 30u) & 1u) != 0u) ? 0xe8212c0cu : 0x5db25b34u); // s133 add
|
||||
r4 = r4 ^ r3; // s134 xor
|
||||
r6 = r6 ^ r1; // s135 xor
|
||||
r1 = r1 - r7; // s136 sub
|
||||
r2 = r6 * r2 + r2; // s137 mad
|
||||
r0 = mul_hi(r0, r5); // s138 mulhi
|
||||
r2 = r2 + r7 + ((((sel >> 10u) & 1u) != 0u) ? 0xd44ff710u : 0x218d4090u); // s139 add
|
||||
r1 = rotr_var(r1, r4); // s140 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r3 = r3 ^ t_; } // s141 shfl
|
||||
r7 = r6 * r2 + r7; // s142 mad
|
||||
r2 = r2 * r3; // s143 mul
|
||||
r7 = r7 + r4 + ((((sel >> 29u) & 1u) != 0u) ? 0xb98942fau : 0xf4b1a8deu); // s144 add
|
||||
r7 = rotl_imm(r7, 15u); // s145 rotl
|
||||
r7 = r7 ^ r5; // s146 xor
|
||||
r4 = r7 * r4 + r4; // s147 mad
|
||||
r6 = rotr_var(r6, r5); // s148 rotr
|
||||
r1 = r2 * r4 + r1; // s149 mad
|
||||
r1 = r1 + r7 + ((((sel >> 17u) & 1u) != 0u) ? 0x0d48ba42u : 0x2bef10f2u); // s150 add
|
||||
r5 = r5 ^ r4; // s151 xor
|
||||
r7 = r7 + r0 + ((((sel >> 30u) & 1u) != 0u) ? 0xc98dea9cu : 0xdc5cc080u); // s152 add
|
||||
r4 = r4 * r6; // s153 mul
|
||||
r0 = r0 * r2; // s154 mul
|
||||
r6 = r6 ^ r5; // s155 xor
|
||||
r4 = rotr_var(r4, r2); // s156 rotr
|
||||
r1 = rotl_imm(r1, 11u); // s157 rotl
|
||||
r5 = r5 + r0 + ((((sel >> 11u) & 1u) != 0u) ? 0x6c752dcbu : 0x18197438u); // s158 add
|
||||
r4 = r1 * r5 + r4; // s159 mad
|
||||
r4 = rotl_imm(r4, 26u); // s160 rotl
|
||||
r3 = r0 * r7 + r3; // s161 mad
|
||||
r3 = rotr_var(r3, r1); // s162 rotr
|
||||
r4 = r4 + r0 + ((((sel >> 3u) & 1u) != 0u) ? 0x8e481727u : 0xf1c46574u); // s163 add
|
||||
r0 = mul_hi(r0, r4); // s164 mulhi
|
||||
r2 = r2 + r6 + ((((sel >> 20u) & 1u) != 0u) ? 0x550ab406u : 0xac578137u); // s165 add
|
||||
r0 = rotl_imm(r0, 13u); // s166 rotl
|
||||
r3 = r3 + r1 + ((((sel >> 14u) & 1u) != 0u) ? 0xaebb5966u : 0xa33e6706u); // s167 add
|
||||
r3 = r3 | r5; // s168 or
|
||||
r6 = rotr_var(r6, r2); // s169 rotr
|
||||
r4 = r4 ^ r6; // s170 xor
|
||||
r6 = r6 - r1; // s171 sub
|
||||
r7 = rotl_imm(r7, 22u); // s172 rotl
|
||||
r5 = rotl_imm(r5, 15u); // s173 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 8u); r7 = r7 ^ t_; } // s174 shfl
|
||||
r0 = r0 ^ r5; // s175 xor
|
||||
r7 = rotl_imm(r7, 6u); // s176 rotl
|
||||
r7 = r7 - r0; // s177 sub
|
||||
r3 = rotl_imm(r3, 30u); // s178 rotl
|
||||
r7 = r6 * r1 + r7; // s179 mad
|
||||
r6 = rotl_imm(r6, 9u); // s180 rotl
|
||||
r2 = r2 ^ r4; // s181 xor
|
||||
r2 = r2 ^ r7; // s182 xor
|
||||
r7 = r7 ^ r2; // s183 xor
|
||||
r1 = r1 + r2 + ((((sel >> 21u) & 1u) != 0u) ? 0x5bb7550fu : 0xd94d55acu); // s184 add
|
||||
r3 = r3 | r5; // s185 or
|
||||
r6 = mul_hi(r6, r3); // s186 mulhi
|
||||
r4 = r4 | r0; // s187 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 2u); r7 = r7 ^ t_; } // s188 shfl
|
||||
r6 = r6 + r5 + ((((sel >> 6u) & 1u) != 0u) ? 0x2a354e2du : 0x89e747feu); // s189 add
|
||||
r1 = r1 ^ r4; // s190 xor
|
||||
r7 = mul_hi(r7, r4); // s191 mulhi
|
||||
r2 = rotl_imm(r2, 26u); // s192 rotl
|
||||
r5 = rotl_imm(r5, 8u); // s193 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r4 = r4 ^ t_; } // s194 shfl
|
||||
r4 = r4 ^ r5; // s195 xor
|
||||
r1 = r1 * r3; // s196 mul
|
||||
r5 = r5 + r2 + ((((sel >> 7u) & 1u) != 0u) ? 0x10cfdc71u : 0xea03e8e7u); // s197 add
|
||||
r7 = r7 + r5 + ((((sel >> 15u) & 1u) != 0u) ? 0x66e148d3u : 0xa7ee0102u); // s198 add
|
||||
r5 = r5 - r2; // s199 sub
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 2u); r5 = r5 ^ t_; } // s200 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 8u); r7 = r7 ^ t_; } // s201 shfl
|
||||
r4 = rotr_var(r4, r5); // s202 rotr
|
||||
r7 = r7 ^ r5; // s203 xor
|
||||
r7 = r7 ^ r0; // s204 xor
|
||||
r6 = r6 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0x8558b619u : 0x8379a4deu); // s205 add
|
||||
r4 = rotl_imm(r4, 13u); // s206 rotl
|
||||
r1 = mul_hi(r1, r3); // s207 mulhi
|
||||
r1 = r4 * r4 + r1; // s208 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 4u); r2 = r2 ^ t_; } // s209 shfl
|
||||
r7 = r7 + r0 + ((((sel >> 11u) & 1u) != 0u) ? 0xf4049c4cu : 0xaa8cb14eu); // s210 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 16u); r7 = r7 ^ t_; } // s211 shfl
|
||||
r1 = r1 ^ r0; // s212 xor
|
||||
r7 = rotl_imm(r7, 3u); // s213 rotl
|
||||
r4 = rotr_var(r4, r7); // s214 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 16u); r3 = r3 ^ t_; } // s215 shfl
|
||||
r5 = r5 | r1; // s216 or
|
||||
r1 = r1 - r2; // s217 sub
|
||||
r6 = r6 - r5; // s218 sub
|
||||
r6 = rotl_imm(r6, 4u); // s219 rotl
|
||||
r2 = mul_hi(r2, r0); // s220 mulhi
|
||||
r2 = r2 | r0; // s221 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r5 = r5 ^ t_; } // s222 shfl
|
||||
r5 = mul_hi(r5, r6); // s223 mulhi
|
||||
r0 = r0 - r6; // s224 sub
|
||||
r7 = rotl_imm(r7, 23u); // s225 rotl
|
||||
r4 = r4 | r2; // s226 or
|
||||
r2 = r2 * r4; // s227 mul
|
||||
r3 = rotl_imm(r3, 12u); // s228 rotl
|
||||
r0 = rotr_var(r0, r4); // s229 rotr
|
||||
r0 = r0 + r6 + ((((sel >> 16u) & 1u) != 0u) ? 0x2a7fecb2u : 0x1d176220u); // s230 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r1 = r1 ^ t_; } // s231 shfl
|
||||
r2 = r2 * r1; // s232 mul
|
||||
r7 = r0 * r1 + r7; // s233 mad
|
||||
r5 = rotl_imm(r5, 22u); // s234 rotl
|
||||
r4 = rotr_var(r4, r6); // s235 rotr
|
||||
r0 = r5 * r1 + r0; // s236 mad
|
||||
r6 = r6 ^ r4; // s237 xor
|
||||
r4 = r4 ^ r6; // s238 xor
|
||||
r6 = rotl_imm(r6, 18u); // s239 rotl
|
||||
r4 = r4 + r7 + ((((sel >> 25u) & 1u) != 0u) ? 0xadce39f3u : 0x17dafb4du); // s240 add
|
||||
r0 = r0 - r7; // s241 sub
|
||||
r1 = rotr_var(r1, r6); // s242 rotr
|
||||
r3 = r3 + r0 + ((((sel >> 29u) & 1u) != 0u) ? 0x0e7033b6u : 0xf2f77d26u); // s243 add
|
||||
r2 = r7 * r4 + r2; // s244 mad
|
||||
r4 = r0 * r6 + r4; // s245 mad
|
||||
r5 = r5 * r7; // s246 mul
|
||||
r3 = r3 + r5 + ((((sel >> 31u) & 1u) != 0u) ? 0x7b2ff6b7u : 0xfed2da4eu); // s247 add
|
||||
r0 = r0 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0xb5fad7b1u : 0x03b2891cu); // s248 add
|
||||
r5 = mul_hi(r5, r4); // s249 mulhi
|
||||
r5 = r5 + r2 + ((((sel >> 11u) & 1u) != 0u) ? 0xa7e71c2fu : 0x042cd6e3u); // s250 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 1u); r6 = r6 ^ t_; } // s251 shfl
|
||||
r6 = r6 ^ r1; // s252 xor
|
||||
r2 = r2 * r5; // s253 mul
|
||||
r0 = r0 + r6 + ((((sel >> 16u) & 1u) != 0u) ? 0xb83d78deu : 0x784a302bu); // s254 add
|
||||
r2 = r2 - r4; // s255 sub
|
||||
}
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
|
||||
#if IGNEUM_EXCHANGE != 0
|
||||
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
|
||||
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
|
||||
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
|
||||
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
|
||||
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
|
||||
}
|
||||
#endif
|
||||
424
proto-cuda/packs-ca3-v5/v5-genesis/kernel.cu
Normal file
424
proto-cuda/packs-ca3-v5/v5-genesis/kernel.cu
Normal file
|
|
@ -0,0 +1,424 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal).
|
||||
// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC.
|
||||
#include <cuda_runtime.h>
|
||||
#include <cstdint>
|
||||
#include "program.h"
|
||||
#include "memhard.h"
|
||||
|
||||
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31.
|
||||
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
|
||||
// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x.
|
||||
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) {
|
||||
uint32_t x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item.
|
||||
// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host.
|
||||
__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) {
|
||||
uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
// Class v5: the window's leaves (leaves.bin, IGNEUM_STATE_LEAVES x 16 words) and their count.
|
||||
__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems) {
|
||||
uint32_t t = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (t < nItems) {
|
||||
uint32_t s[16];
|
||||
mh_item(cache, mh_leaf(leaves, nLeaves, t), t, s);
|
||||
uint32_t* d = ds + (size_t)t * 16u;
|
||||
for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
|
||||
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
|
||||
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
|
||||
__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask) {
|
||||
uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
uint32_t nonce = baseNonce + gid;
|
||||
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint32_t x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint32_t x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint32_t x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint32_t x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint32_t x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint32_t x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint32_t x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint32_t x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint32_t it = 0u; it < 8u; ++it) {
|
||||
uint32_t sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r2 = r1 * r1 + r2; // 1 mad
|
||||
r2 = r3 * r2 + r2; // 2 mad
|
||||
r3 = r3 ^ r5; // 3 xor
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r5 = r5 ^ ds[r7 & mask]; // 5 load
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 6 shfl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 8); // 7 shfl
|
||||
r1 = __umulhi(r1, r5); // 8 mulhi
|
||||
r6 = rotr_var(r6, r3); // 9 rotr
|
||||
r3 = r3 | r4; // 10 or
|
||||
r4 = r4 ^ ds[r6 & mask]; // 11 load
|
||||
r0 = __umulhi(r0, r4); // 12 mulhi
|
||||
r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add
|
||||
r0 = r0 ^ ds[r5 & mask]; // 14 load
|
||||
r2 = r2 - r4; // 15 sub
|
||||
r2 = r2 ^ ds[r7 & mask]; // 16 load
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 18 shfl
|
||||
r5 = r5 * r0; // 19 mul
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 20 shfl
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 16); // 21 shfl
|
||||
r6 = __umulhi(r6, r2); // 22 mulhi
|
||||
r6 = r6 ^ ds[r3 & mask]; // 23 load
|
||||
r5 = r5 * r0; // 24 mul
|
||||
r5 = rotl_imm(r5, 19u); // 25 rotl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r6, 2); // 26 shfl
|
||||
r0 = r0 ^ r5; // 27 xor
|
||||
r0 = r0 ^ r4; // 28 xor
|
||||
r3 = r3 - r0; // 29 sub
|
||||
r5 = r5 * r1; // 30 mul
|
||||
r7 = r7 ^ ds[r3 & mask]; // 31 load
|
||||
r1 = r1 ^ ds[r2 & mask]; // 32 load
|
||||
r5 = r5 ^ r6; // 33 xor
|
||||
r5 = r5 ^ ds[r0 & mask]; // 34 load
|
||||
r0 = __umulhi(r0, r5); // 35 mulhi
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // 36 shfl
|
||||
r7 = r7 ^ ds[r6 & mask]; // 37 load
|
||||
r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 4); // 39 shfl
|
||||
r2 = r2 ^ r5; // 40 xor
|
||||
r3 = r6 * r3 + r3; // 41 mad
|
||||
r6 = r6 - r7; // 42 sub
|
||||
r7 = r7 ^ r0; // 43 xor
|
||||
r1 = r1 ^ ds[r7 & mask]; // 44 load
|
||||
r2 = r2 * r3; // 45 mul
|
||||
r1 = __umulhi(r1, r5); // 46 mulhi
|
||||
r4 = r4 - r3; // 47 sub
|
||||
r2 = rotr_var(r2, r6); // 48 rotr
|
||||
r3 = r3 ^ ds[r4 & mask]; // 49 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add
|
||||
r0 = r0 * r2; // 51 mul
|
||||
r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add
|
||||
r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add
|
||||
r7 = rotl_imm(r7, 14u); // 54 rotl
|
||||
r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add
|
||||
r6 = r6 ^ ds[r7 & mask]; // 56 load
|
||||
r1 = rotr_var(r1, r5); // 57 rotr
|
||||
r5 = r5 ^ ds[r3 & mask]; // 58 load
|
||||
r6 = r6 ^ ds[r1 & mask]; // 59 load
|
||||
r3 = r5 * r0 + r3; // 60 mad
|
||||
r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add
|
||||
r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add
|
||||
r5 = rotl_imm(r5, 19u); // 63 rotl
|
||||
// latency-shadow block (Counter ASIC 3.0 item 8): 256 ALU instructions x 27 passes after instruction 63, no load
|
||||
for (uint32_t sh = 0u; sh < 27u; ++sh) {
|
||||
r5 = r5 + r2 + ((((sel >> 18u) & 1u) != 0u) ? 0x8bc12da9u : 0x92199f99u); // s0 add
|
||||
r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x63079e5au : 0x8d72d3adu); // s1 add
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 2); // s2 shfl
|
||||
r4 = r4 - r2; // s3 sub
|
||||
r7 = r7 + r0 + ((((sel >> 31u) & 1u) != 0u) ? 0x5d4c7a60u : 0xb21b4babu); // s4 add
|
||||
r0 = rotl_imm(r0, 11u); // s5 rotl
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // s6 add
|
||||
r7 = r7 ^ r1; // s7 xor
|
||||
r1 = r6 * r5 + r1; // s8 mad
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r1, 4); // s9 shfl
|
||||
r1 = r2 * r2 + r1; // s10 mad
|
||||
r5 = r0 * r3 + r5; // s11 mad
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r6, 4); // s12 shfl
|
||||
r1 = r1 + r5 + ((((sel >> 25u) & 1u) != 0u) ? 0xbc48c63eu : 0xbdf8f9a5u); // s13 add
|
||||
r1 = rotl_imm(r1, 29u); // s14 rotl
|
||||
r1 = r1 - r4; // s15 sub
|
||||
r7 = r7 | r1; // s16 or
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // s17 shfl
|
||||
r7 = r7 ^ r4; // s18 xor
|
||||
r6 = r6 * r1; // s19 mul
|
||||
r5 = r6 * r0 + r5; // s20 mad
|
||||
r3 = r3 - r1; // s21 sub
|
||||
r6 = r6 * r0; // s22 mul
|
||||
r2 = r2 + r0 + ((((sel >> 1u) & 1u) != 0u) ? 0x96e8f127u : 0x45c37cecu); // s23 add
|
||||
r6 = r6 - r4; // s24 sub
|
||||
r7 = r3 * r4 + r7; // s25 mad
|
||||
r3 = rotl_imm(r3, 9u); // s26 rotl
|
||||
r2 = r2 - r1; // s27 sub
|
||||
r6 = __umulhi(r6, r0); // s28 mulhi
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // s29 shfl
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r6, 1); // s30 shfl
|
||||
r1 = r1 + r6 + ((((sel >> 1u) & 1u) != 0u) ? 0xb1cdb2abu : 0x37985632u); // s31 add
|
||||
r0 = rotr_var(r0, r1); // s32 rotr
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // s33 shfl
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // s34 shfl
|
||||
r7 = r7 * r0; // s35 mul
|
||||
r3 = r3 + r1 + ((((sel >> 0u) & 1u) != 0u) ? 0x856e0180u : 0x804e777eu); // s36 add
|
||||
r1 = r1 ^ r6; // s37 xor
|
||||
r2 = r2 * r7; // s38 mul
|
||||
r6 = r6 + r5 + ((((sel >> 2u) & 1u) != 0u) ? 0xe9bd0cc4u : 0x0b06c8a7u); // s39 add
|
||||
r5 = r5 * r7; // s40 mul
|
||||
r2 = __umulhi(r2, r3); // s41 mulhi
|
||||
r2 = r2 ^ r0; // s42 xor
|
||||
r0 = rotl_imm(r0, 4u); // s43 rotl
|
||||
r4 = __umulhi(r4, r3); // s44 mulhi
|
||||
r6 = r6 + r7 + ((((sel >> 12u) & 1u) != 0u) ? 0xcdcb63f6u : 0xee822e17u); // s45 add
|
||||
r3 = r2 * r0 + r3; // s46 mad
|
||||
r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r3, 8); // s47 shfl
|
||||
r6 = __umulhi(r6, r5); // s48 mulhi
|
||||
r0 = r0 - r2; // s49 sub
|
||||
r3 = r3 - r5; // s50 sub
|
||||
r1 = rotr_var(r1, r4); // s51 rotr
|
||||
r6 = r7 * r7 + r6; // s52 mad
|
||||
r5 = r5 ^ r3; // s53 xor
|
||||
r1 = r1 - r0; // s54 sub
|
||||
r5 = r5 - r6; // s55 sub
|
||||
r3 = r3 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0xd4758987u : 0x3dfad1b6u); // s56 add
|
||||
r4 = r4 + r1 + ((((sel >> 0u) & 1u) != 0u) ? 0x0602d6beu : 0xfca75bc2u); // s57 add
|
||||
r5 = __umulhi(r5, r6); // s58 mulhi
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r1, 2); // s59 shfl
|
||||
r2 = rotl_imm(r2, 9u); // s60 rotl
|
||||
r4 = r4 | r6; // s61 or
|
||||
r6 = rotr_var(r6, r4); // s62 rotr
|
||||
r2 = r2 * r4; // s63 mul
|
||||
r0 = r0 ^ r5; // s64 xor
|
||||
r2 = r2 ^ r7; // s65 xor
|
||||
r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x8596687au : 0xd71c02ffu); // s66 add
|
||||
r1 = rotl_imm(r1, 21u); // s67 rotl
|
||||
r3 = r3 * r2; // s68 mul
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r5, 2); // s69 shfl
|
||||
r3 = r3 * r2; // s70 mul
|
||||
r0 = r2 * r5 + r0; // s71 mad
|
||||
r6 = r6 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0x08ac5733u : 0x3c6fe15du); // s72 add
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r2, 8); // s73 shfl
|
||||
r3 = r3 * r6; // s74 mul
|
||||
r6 = r6 ^ r7; // s75 xor
|
||||
r3 = r3 | r0; // s76 or
|
||||
r2 = r2 + r4 + ((((sel >> 1u) & 1u) != 0u) ? 0xc100b495u : 0x295d5faeu); // s77 add
|
||||
r3 = __umulhi(r3, r7); // s78 mulhi
|
||||
r4 = r4 | r1; // s79 or
|
||||
r4 = rotr_var(r4, r3); // s80 rotr
|
||||
r4 = r4 + r3 + ((((sel >> 15u) & 1u) != 0u) ? 0x5cc59530u : 0x274a9221u); // s81 add
|
||||
r7 = r7 + r3 + ((((sel >> 5u) & 1u) != 0u) ? 0x141479ecu : 0xc8651f8eu); // s82 add
|
||||
r3 = rotr_var(r3, r6); // s83 rotr
|
||||
r2 = r4 * r7 + r2; // s84 mad
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r5, 16); // s85 shfl
|
||||
r3 = r3 ^ r2; // s86 xor
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r7, 2); // s87 shfl
|
||||
r0 = r0 ^ r4; // s88 xor
|
||||
r3 = r3 + r2 + ((((sel >> 18u) & 1u) != 0u) ? 0x883c0c92u : 0x53f915c2u); // s89 add
|
||||
r7 = rotr_var(r7, r5); // s90 rotr
|
||||
r3 = r3 + r1 + ((((sel >> 31u) & 1u) != 0u) ? 0x3cc5bd20u : 0xe2be00a5u); // s91 add
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r2, 1); // s92 shfl
|
||||
r6 = rotr_var(r6, r2); // s93 rotr
|
||||
r7 = rotl_imm(r7, 14u); // s94 rotl
|
||||
r4 = r5 * r5 + r4; // s95 mad
|
||||
r2 = r4 * r5 + r2; // s96 mad
|
||||
r1 = r1 ^ r0; // s97 xor
|
||||
r5 = r5 * r4; // s98 mul
|
||||
r2 = r2 - r0; // s99 sub
|
||||
r7 = rotl_imm(r7, 30u); // s100 rotl
|
||||
r5 = r5 + r6 + ((((sel >> 17u) & 1u) != 0u) ? 0xbfd646cbu : 0x423fd9c9u); // s101 add
|
||||
r7 = r7 + r6 + ((((sel >> 11u) & 1u) != 0u) ? 0x19b74a43u : 0x8d3c011du); // s102 add
|
||||
r5 = rotl_imm(r5, 6u); // s103 rotl
|
||||
r0 = r0 * r4; // s104 mul
|
||||
r0 = r0 | r5; // s105 or
|
||||
r0 = r0 + r1 + ((((sel >> 15u) & 1u) != 0u) ? 0x7dcb7e18u : 0x8ce14721u); // s106 add
|
||||
r0 = rotl_imm(r0, 15u); // s107 rotl
|
||||
r4 = r4 - r2; // s108 sub
|
||||
r2 = __umulhi(r2, r0); // s109 mulhi
|
||||
r1 = __umulhi(r1, r0); // s110 mulhi
|
||||
r0 = r0 + r2 + ((((sel >> 28u) & 1u) != 0u) ? 0x3c179ad8u : 0x2d9fc6b9u); // s111 add
|
||||
r2 = __umulhi(r2, r0); // s112 mulhi
|
||||
r6 = r6 - r3; // s113 sub
|
||||
r7 = r6 * r3 + r7; // s114 mad
|
||||
r5 = rotr_var(r5, r1); // s115 rotr
|
||||
r6 = rotr_var(r6, r4); // s116 rotr
|
||||
r2 = r2 + r1 + ((((sel >> 22u) & 1u) != 0u) ? 0x47359729u : 0x502138e2u); // s117 add
|
||||
r3 = r2 * r0 + r3; // s118 mad
|
||||
r1 = r1 * r3; // s119 mul
|
||||
r2 = r0 * r4 + r2; // s120 mad
|
||||
r0 = r0 * r3; // s121 mul
|
||||
r2 = __umulhi(r2, r0); // s122 mulhi
|
||||
r5 = r5 * r6; // s123 mul
|
||||
r4 = r2 * r4 + r4; // s124 mad
|
||||
r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r2, 2); // s125 shfl
|
||||
r2 = r2 ^ r0; // s126 xor
|
||||
r1 = rotl_imm(r1, 7u); // s127 rotl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // s128 shfl
|
||||
r6 = rotl_imm(r6, 17u); // s129 rotl
|
||||
r2 = r2 + r5 + ((((sel >> 15u) & 1u) != 0u) ? 0x6c57e4e7u : 0x33dff776u); // s130 add
|
||||
r0 = r0 | r3; // s131 or
|
||||
r5 = rotr_var(r5, r0); // s132 rotr
|
||||
r2 = r2 + r3 + ((((sel >> 30u) & 1u) != 0u) ? 0xe8212c0cu : 0x5db25b34u); // s133 add
|
||||
r4 = r4 ^ r3; // s134 xor
|
||||
r6 = r6 ^ r1; // s135 xor
|
||||
r1 = r1 - r7; // s136 sub
|
||||
r2 = r6 * r2 + r2; // s137 mad
|
||||
r0 = __umulhi(r0, r5); // s138 mulhi
|
||||
r2 = r2 + r7 + ((((sel >> 10u) & 1u) != 0u) ? 0xd44ff710u : 0x218d4090u); // s139 add
|
||||
r1 = rotr_var(r1, r4); // s140 rotr
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r6, 1); // s141 shfl
|
||||
r7 = r6 * r2 + r7; // s142 mad
|
||||
r2 = r2 * r3; // s143 mul
|
||||
r7 = r7 + r4 + ((((sel >> 29u) & 1u) != 0u) ? 0xb98942fau : 0xf4b1a8deu); // s144 add
|
||||
r7 = rotl_imm(r7, 15u); // s145 rotl
|
||||
r7 = r7 ^ r5; // s146 xor
|
||||
r4 = r7 * r4 + r4; // s147 mad
|
||||
r6 = rotr_var(r6, r5); // s148 rotr
|
||||
r1 = r2 * r4 + r1; // s149 mad
|
||||
r1 = r1 + r7 + ((((sel >> 17u) & 1u) != 0u) ? 0x0d48ba42u : 0x2bef10f2u); // s150 add
|
||||
r5 = r5 ^ r4; // s151 xor
|
||||
r7 = r7 + r0 + ((((sel >> 30u) & 1u) != 0u) ? 0xc98dea9cu : 0xdc5cc080u); // s152 add
|
||||
r4 = r4 * r6; // s153 mul
|
||||
r0 = r0 * r2; // s154 mul
|
||||
r6 = r6 ^ r5; // s155 xor
|
||||
r4 = rotr_var(r4, r2); // s156 rotr
|
||||
r1 = rotl_imm(r1, 11u); // s157 rotl
|
||||
r5 = r5 + r0 + ((((sel >> 11u) & 1u) != 0u) ? 0x6c752dcbu : 0x18197438u); // s158 add
|
||||
r4 = r1 * r5 + r4; // s159 mad
|
||||
r4 = rotl_imm(r4, 26u); // s160 rotl
|
||||
r3 = r0 * r7 + r3; // s161 mad
|
||||
r3 = rotr_var(r3, r1); // s162 rotr
|
||||
r4 = r4 + r0 + ((((sel >> 3u) & 1u) != 0u) ? 0x8e481727u : 0xf1c46574u); // s163 add
|
||||
r0 = __umulhi(r0, r4); // s164 mulhi
|
||||
r2 = r2 + r6 + ((((sel >> 20u) & 1u) != 0u) ? 0x550ab406u : 0xac578137u); // s165 add
|
||||
r0 = rotl_imm(r0, 13u); // s166 rotl
|
||||
r3 = r3 + r1 + ((((sel >> 14u) & 1u) != 0u) ? 0xaebb5966u : 0xa33e6706u); // s167 add
|
||||
r3 = r3 | r5; // s168 or
|
||||
r6 = rotr_var(r6, r2); // s169 rotr
|
||||
r4 = r4 ^ r6; // s170 xor
|
||||
r6 = r6 - r1; // s171 sub
|
||||
r7 = rotl_imm(r7, 22u); // s172 rotl
|
||||
r5 = rotl_imm(r5, 15u); // s173 rotl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r0, 8); // s174 shfl
|
||||
r0 = r0 ^ r5; // s175 xor
|
||||
r7 = rotl_imm(r7, 6u); // s176 rotl
|
||||
r7 = r7 - r0; // s177 sub
|
||||
r3 = rotl_imm(r3, 30u); // s178 rotl
|
||||
r7 = r6 * r1 + r7; // s179 mad
|
||||
r6 = rotl_imm(r6, 9u); // s180 rotl
|
||||
r2 = r2 ^ r4; // s181 xor
|
||||
r2 = r2 ^ r7; // s182 xor
|
||||
r7 = r7 ^ r2; // s183 xor
|
||||
r1 = r1 + r2 + ((((sel >> 21u) & 1u) != 0u) ? 0x5bb7550fu : 0xd94d55acu); // s184 add
|
||||
r3 = r3 | r5; // s185 or
|
||||
r6 = __umulhi(r6, r3); // s186 mulhi
|
||||
r4 = r4 | r0; // s187 or
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r1, 2); // s188 shfl
|
||||
r6 = r6 + r5 + ((((sel >> 6u) & 1u) != 0u) ? 0x2a354e2du : 0x89e747feu); // s189 add
|
||||
r1 = r1 ^ r4; // s190 xor
|
||||
r7 = __umulhi(r7, r4); // s191 mulhi
|
||||
r2 = rotl_imm(r2, 26u); // s192 rotl
|
||||
r5 = rotl_imm(r5, 8u); // s193 rotl
|
||||
r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r5, 16); // s194 shfl
|
||||
r4 = r4 ^ r5; // s195 xor
|
||||
r1 = r1 * r3; // s196 mul
|
||||
r5 = r5 + r2 + ((((sel >> 7u) & 1u) != 0u) ? 0x10cfdc71u : 0xea03e8e7u); // s197 add
|
||||
r7 = r7 + r5 + ((((sel >> 15u) & 1u) != 0u) ? 0x66e148d3u : 0xa7ee0102u); // s198 add
|
||||
r5 = r5 - r2; // s199 sub
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r1, 2); // s200 shfl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r1, 8); // s201 shfl
|
||||
r4 = rotr_var(r4, r5); // s202 rotr
|
||||
r7 = r7 ^ r5; // s203 xor
|
||||
r7 = r7 ^ r0; // s204 xor
|
||||
r6 = r6 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0x8558b619u : 0x8379a4deu); // s205 add
|
||||
r4 = rotl_imm(r4, 13u); // s206 rotl
|
||||
r1 = __umulhi(r1, r3); // s207 mulhi
|
||||
r1 = r4 * r4 + r1; // s208 mad
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r0, 4); // s209 shfl
|
||||
r7 = r7 + r0 + ((((sel >> 11u) & 1u) != 0u) ? 0xf4049c4cu : 0xaa8cb14eu); // s210 add
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r1, 16); // s211 shfl
|
||||
r1 = r1 ^ r0; // s212 xor
|
||||
r7 = rotl_imm(r7, 3u); // s213 rotl
|
||||
r4 = rotr_var(r4, r7); // s214 rotr
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r2, 16); // s215 shfl
|
||||
r5 = r5 | r1; // s216 or
|
||||
r1 = r1 - r2; // s217 sub
|
||||
r6 = r6 - r5; // s218 sub
|
||||
r6 = rotl_imm(r6, 4u); // s219 rotl
|
||||
r2 = __umulhi(r2, r0); // s220 mulhi
|
||||
r2 = r2 | r0; // s221 or
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r2, 8); // s222 shfl
|
||||
r5 = __umulhi(r5, r6); // s223 mulhi
|
||||
r0 = r0 - r6; // s224 sub
|
||||
r7 = rotl_imm(r7, 23u); // s225 rotl
|
||||
r4 = r4 | r2; // s226 or
|
||||
r2 = r2 * r4; // s227 mul
|
||||
r3 = rotl_imm(r3, 12u); // s228 rotl
|
||||
r0 = rotr_var(r0, r4); // s229 rotr
|
||||
r0 = r0 + r6 + ((((sel >> 16u) & 1u) != 0u) ? 0x2a7fecb2u : 0x1d176220u); // s230 add
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // s231 shfl
|
||||
r2 = r2 * r1; // s232 mul
|
||||
r7 = r0 * r1 + r7; // s233 mad
|
||||
r5 = rotl_imm(r5, 22u); // s234 rotl
|
||||
r4 = rotr_var(r4, r6); // s235 rotr
|
||||
r0 = r5 * r1 + r0; // s236 mad
|
||||
r6 = r6 ^ r4; // s237 xor
|
||||
r4 = r4 ^ r6; // s238 xor
|
||||
r6 = rotl_imm(r6, 18u); // s239 rotl
|
||||
r4 = r4 + r7 + ((((sel >> 25u) & 1u) != 0u) ? 0xadce39f3u : 0x17dafb4du); // s240 add
|
||||
r0 = r0 - r7; // s241 sub
|
||||
r1 = rotr_var(r1, r6); // s242 rotr
|
||||
r3 = r3 + r0 + ((((sel >> 29u) & 1u) != 0u) ? 0x0e7033b6u : 0xf2f77d26u); // s243 add
|
||||
r2 = r7 * r4 + r2; // s244 mad
|
||||
r4 = r0 * r6 + r4; // s245 mad
|
||||
r5 = r5 * r7; // s246 mul
|
||||
r3 = r3 + r5 + ((((sel >> 31u) & 1u) != 0u) ? 0x7b2ff6b7u : 0xfed2da4eu); // s247 add
|
||||
r0 = r0 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0xb5fad7b1u : 0x03b2891cu); // s248 add
|
||||
r5 = __umulhi(r5, r4); // s249 mulhi
|
||||
r5 = r5 + r2 + ((((sel >> 11u) & 1u) != 0u) ? 0xa7e71c2fu : 0x042cd6e3u); // s250 add
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r1, 1); // s251 shfl
|
||||
r6 = r6 ^ r1; // s252 xor
|
||||
r2 = r2 * r5; // s253 mul
|
||||
r0 = r0 + r6 + ((((sel >> 16u) & 1u) != 0u) ? 0xb83d78deu : 0x784a302bu); // s254 add
|
||||
r2 = r2 - r4; // s255 sub
|
||||
}
|
||||
}
|
||||
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
|
||||
}
|
||||
|
||||
// Host-side launch wrappers. Declared in program.h, called from host.cu.
|
||||
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) {
|
||||
if (nSegments == 0u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 256u;
|
||||
uint32_t grid = (nSegments + block - 1u) / block;
|
||||
igneum_cache_fill<<<grid, block>>>(cache, nSegments);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems) {
|
||||
if (nItems == 0u || nLeaves == 0u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 256u;
|
||||
uint32_t grid = (nItems + block - 1u) / block;
|
||||
igneum_build<<<grid, block>>>(ds, cache, leaves, nLeaves, nItems);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
uint32_t nonces, uint32_t blockWarps) {
|
||||
if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 32u * blockWarps;
|
||||
if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;
|
||||
igneum_hash<<<nonces / block, block>>>(ds, out, baseNonce, mask);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
|
||||
cudaFuncAttributes attr;
|
||||
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash);
|
||||
if (e != cudaSuccess) return e;
|
||||
*numRegs = attr.numRegs;
|
||||
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0);
|
||||
}
|
||||
895
proto-cuda/packs-ca3-v5/v5-genesis/kernel_bound.cl
Normal file
895
proto-cuda/packs-ca3-v5/v5-genesis/kernel_bound.cl
Normal file
|
|
@ -0,0 +1,895 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
|
||||
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
|
||||
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
|
||||
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
|
||||
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
|
||||
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
|
||||
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
|
||||
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
|
||||
#ifndef IGNEUM_GROUP
|
||||
#define IGNEUM_GROUP 32
|
||||
#endif
|
||||
#ifndef IGNEUM_EXCHANGE
|
||||
#define IGNEUM_EXCHANGE 0
|
||||
#endif
|
||||
#ifdef __OPENCL_VERSION__
|
||||
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
|
||||
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#ifdef cl_khr_subgroups
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
|
||||
#endif
|
||||
#ifdef cl_khr_subgroup_shuffle
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
|
||||
#endif
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
|
||||
#endif
|
||||
#else
|
||||
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
|
||||
#include "emu_opencl.h"
|
||||
#endif
|
||||
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#else
|
||||
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
|
||||
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
|
||||
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
|
||||
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
|
||||
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
|
||||
#endif
|
||||
|
||||
static inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
|
||||
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
|
||||
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
|
||||
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
|
||||
static inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of 8 x seed-parameterised mixer + one 64-byte cache read, then 8 x final mixer (class v3, mixer multiplier 8,
|
||||
// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
static inline void mh_chacha_block(const uint* x, uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
static inline void mh_cache_segment(__global uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
static inline void mh_mixer(uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of 8 x mixer + cache line s[0] & mask; 8 x final mixer.
|
||||
static inline void mh_item(__global const uint* cache, __global const uint* leaf, uint t, uint* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= leaf[i];
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
for (uint j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 8u + j + 1u));
|
||||
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
for (uint j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (64u + j + 1u));
|
||||
}
|
||||
// Class v5 (docs/design/class-v5-stored-state.md): leaf(t) = leaves[t mod nLeaves], 16 words per leaf (leaves.bin).
|
||||
static inline __global const uint* mh_leaf(__global const uint* leaves, uint nLeaves, uint t) { return leaves + ((t % nLeaves) * 16u); }
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
static inline uint mh_word(__global const uint* cache, __global const uint* leaves, uint nLeaves, uint w) { uint s[16]; mh_item(cache, mh_leaf(leaves, nLeaves, w >> 4u), w >> 4u, s); return s[w & 15u]; }
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
|
||||
// The same constants as memhard.h in this pack (one emitter, three dialects).
|
||||
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
|
||||
uint seg = (uint)get_global_id(0);
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
// Class v5: the window's leaves (leaves.bin, IGNEUM_STATE_LEAVES x 16 words) and their count.
|
||||
__kernel void igneum_build(__global uint* ds, __global const uint* cache, __global const uint* leaves, uint nLeaves, uint nItems) {
|
||||
uint t = (uint)get_global_id(0);
|
||||
if (t < nItems) {
|
||||
uint s[16];
|
||||
mh_item(cache, mh_leaf(leaves, nLeaves, t), t, s);
|
||||
__global uint* d = ds + ((ulong)t * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask) {
|
||||
uint gid = (uint)get_global_id(0);
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r2 = r1 * r1 + r2; // 1 mad
|
||||
r2 = r3 * r2 + r2; // 2 mad
|
||||
r3 = r3 ^ r5; // 3 xor
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r5 = r5 ^ ds[r7 & mask]; // 5 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl
|
||||
r1 = mul_hi(r1, r5); // 8 mulhi
|
||||
r6 = rotr_var(r6, r3); // 9 rotr
|
||||
r3 = r3 | r4; // 10 or
|
||||
r4 = r4 ^ ds[r6 & mask]; // 11 load
|
||||
r0 = mul_hi(r0, r4); // 12 mulhi
|
||||
r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add
|
||||
r0 = r0 ^ ds[r5 & mask]; // 14 load
|
||||
r2 = r2 - r4; // 15 sub
|
||||
r2 = r2 ^ ds[r7 & mask]; // 16 load
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl
|
||||
r5 = r5 * r0; // 19 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl
|
||||
r6 = mul_hi(r6, r2); // 22 mulhi
|
||||
r6 = r6 ^ ds[r3 & mask]; // 23 load
|
||||
r5 = r5 * r0; // 24 mul
|
||||
r5 = rotl_imm(r5, 19u); // 25 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl
|
||||
r0 = r0 ^ r5; // 27 xor
|
||||
r0 = r0 ^ r4; // 28 xor
|
||||
r3 = r3 - r0; // 29 sub
|
||||
r5 = r5 * r1; // 30 mul
|
||||
r7 = r7 ^ ds[r3 & mask]; // 31 load
|
||||
r1 = r1 ^ ds[r2 & mask]; // 32 load
|
||||
r5 = r5 ^ r6; // 33 xor
|
||||
r5 = r5 ^ ds[r0 & mask]; // 34 load
|
||||
r0 = mul_hi(r0, r5); // 35 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl
|
||||
r7 = r7 ^ ds[r6 & mask]; // 37 load
|
||||
r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl
|
||||
r2 = r2 ^ r5; // 40 xor
|
||||
r3 = r6 * r3 + r3; // 41 mad
|
||||
r6 = r6 - r7; // 42 sub
|
||||
r7 = r7 ^ r0; // 43 xor
|
||||
r1 = r1 ^ ds[r7 & mask]; // 44 load
|
||||
r2 = r2 * r3; // 45 mul
|
||||
r1 = mul_hi(r1, r5); // 46 mulhi
|
||||
r4 = r4 - r3; // 47 sub
|
||||
r2 = rotr_var(r2, r6); // 48 rotr
|
||||
r3 = r3 ^ ds[r4 & mask]; // 49 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add
|
||||
r0 = r0 * r2; // 51 mul
|
||||
r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add
|
||||
r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add
|
||||
r7 = rotl_imm(r7, 14u); // 54 rotl
|
||||
r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add
|
||||
r6 = r6 ^ ds[r7 & mask]; // 56 load
|
||||
r1 = rotr_var(r1, r5); // 57 rotr
|
||||
r5 = r5 ^ ds[r3 & mask]; // 58 load
|
||||
r6 = r6 ^ ds[r1 & mask]; // 59 load
|
||||
r3 = r5 * r0 + r3; // 60 mad
|
||||
r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add
|
||||
r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add
|
||||
r5 = rotl_imm(r5, 19u); // 63 rotl
|
||||
// latency-shadow block (Counter ASIC 3.0 item 8): 256 ALU instructions x 27 passes after instruction 63, no load
|
||||
for (uint sh = 0u; sh < 27u; ++sh) {
|
||||
r5 = r5 + r2 + ((((sel >> 18u) & 1u) != 0u) ? 0x8bc12da9u : 0x92199f99u); // s0 add
|
||||
r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x63079e5au : 0x8d72d3adu); // s1 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 2u); r6 = r6 ^ t_; } // s2 shfl
|
||||
r4 = r4 - r2; // s3 sub
|
||||
r7 = r7 + r0 + ((((sel >> 31u) & 1u) != 0u) ? 0x5d4c7a60u : 0xb21b4babu); // s4 add
|
||||
r0 = rotl_imm(r0, 11u); // s5 rotl
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // s6 add
|
||||
r7 = r7 ^ r1; // s7 xor
|
||||
r1 = r6 * r5 + r1; // s8 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 4u); r6 = r6 ^ t_; } // s9 shfl
|
||||
r1 = r2 * r2 + r1; // s10 mad
|
||||
r5 = r0 * r3 + r5; // s11 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 4u); r2 = r2 ^ t_; } // s12 shfl
|
||||
r1 = r1 + r5 + ((((sel >> 25u) & 1u) != 0u) ? 0xbc48c63eu : 0xbdf8f9a5u); // s13 add
|
||||
r1 = rotl_imm(r1, 29u); // s14 rotl
|
||||
r1 = r1 - r4; // s15 sub
|
||||
r7 = r7 | r1; // s16 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r2 = r2 ^ t_; } // s17 shfl
|
||||
r7 = r7 ^ r4; // s18 xor
|
||||
r6 = r6 * r1; // s19 mul
|
||||
r5 = r6 * r0 + r5; // s20 mad
|
||||
r3 = r3 - r1; // s21 sub
|
||||
r6 = r6 * r0; // s22 mul
|
||||
r2 = r2 + r0 + ((((sel >> 1u) & 1u) != 0u) ? 0x96e8f127u : 0x45c37cecu); // s23 add
|
||||
r6 = r6 - r4; // s24 sub
|
||||
r7 = r3 * r4 + r7; // s25 mad
|
||||
r3 = rotl_imm(r3, 9u); // s26 rotl
|
||||
r2 = r2 - r1; // s27 sub
|
||||
r6 = mul_hi(r6, r0); // s28 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r2 = r2 ^ t_; } // s29 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r1 = r1 ^ t_; } // s30 shfl
|
||||
r1 = r1 + r6 + ((((sel >> 1u) & 1u) != 0u) ? 0xb1cdb2abu : 0x37985632u); // s31 add
|
||||
r0 = rotr_var(r0, r1); // s32 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // s33 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r0 = r0 ^ t_; } // s34 shfl
|
||||
r7 = r7 * r0; // s35 mul
|
||||
r3 = r3 + r1 + ((((sel >> 0u) & 1u) != 0u) ? 0x856e0180u : 0x804e777eu); // s36 add
|
||||
r1 = r1 ^ r6; // s37 xor
|
||||
r2 = r2 * r7; // s38 mul
|
||||
r6 = r6 + r5 + ((((sel >> 2u) & 1u) != 0u) ? 0xe9bd0cc4u : 0x0b06c8a7u); // s39 add
|
||||
r5 = r5 * r7; // s40 mul
|
||||
r2 = mul_hi(r2, r3); // s41 mulhi
|
||||
r2 = r2 ^ r0; // s42 xor
|
||||
r0 = rotl_imm(r0, 4u); // s43 rotl
|
||||
r4 = mul_hi(r4, r3); // s44 mulhi
|
||||
r6 = r6 + r7 + ((((sel >> 12u) & 1u) != 0u) ? 0xcdcb63f6u : 0xee822e17u); // s45 add
|
||||
r3 = r2 * r0 + r3; // s46 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r4 = r4 ^ t_; } // s47 shfl
|
||||
r6 = mul_hi(r6, r5); // s48 mulhi
|
||||
r0 = r0 - r2; // s49 sub
|
||||
r3 = r3 - r5; // s50 sub
|
||||
r1 = rotr_var(r1, r4); // s51 rotr
|
||||
r6 = r7 * r7 + r6; // s52 mad
|
||||
r5 = r5 ^ r3; // s53 xor
|
||||
r1 = r1 - r0; // s54 sub
|
||||
r5 = r5 - r6; // s55 sub
|
||||
r3 = r3 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0xd4758987u : 0x3dfad1b6u); // s56 add
|
||||
r4 = r4 + r1 + ((((sel >> 0u) & 1u) != 0u) ? 0x0602d6beu : 0xfca75bc2u); // s57 add
|
||||
r5 = mul_hi(r5, r6); // s58 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 2u); r2 = r2 ^ t_; } // s59 shfl
|
||||
r2 = rotl_imm(r2, 9u); // s60 rotl
|
||||
r4 = r4 | r6; // s61 or
|
||||
r6 = rotr_var(r6, r4); // s62 rotr
|
||||
r2 = r2 * r4; // s63 mul
|
||||
r0 = r0 ^ r5; // s64 xor
|
||||
r2 = r2 ^ r7; // s65 xor
|
||||
r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x8596687au : 0xd71c02ffu); // s66 add
|
||||
r1 = rotl_imm(r1, 21u); // s67 rotl
|
||||
r3 = r3 * r2; // s68 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 2u); r7 = r7 ^ t_; } // s69 shfl
|
||||
r3 = r3 * r2; // s70 mul
|
||||
r0 = r2 * r5 + r0; // s71 mad
|
||||
r6 = r6 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0x08ac5733u : 0x3c6fe15du); // s72 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r3 = r3 ^ t_; } // s73 shfl
|
||||
r3 = r3 * r6; // s74 mul
|
||||
r6 = r6 ^ r7; // s75 xor
|
||||
r3 = r3 | r0; // s76 or
|
||||
r2 = r2 + r4 + ((((sel >> 1u) & 1u) != 0u) ? 0xc100b495u : 0x295d5faeu); // s77 add
|
||||
r3 = mul_hi(r3, r7); // s78 mulhi
|
||||
r4 = r4 | r1; // s79 or
|
||||
r4 = rotr_var(r4, r3); // s80 rotr
|
||||
r4 = r4 + r3 + ((((sel >> 15u) & 1u) != 0u) ? 0x5cc59530u : 0x274a9221u); // s81 add
|
||||
r7 = r7 + r3 + ((((sel >> 5u) & 1u) != 0u) ? 0x141479ecu : 0xc8651f8eu); // s82 add
|
||||
r3 = rotr_var(r3, r6); // s83 rotr
|
||||
r2 = r4 * r7 + r2; // s84 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r2 = r2 ^ t_; } // s85 shfl
|
||||
r3 = r3 ^ r2; // s86 xor
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 2u); r5 = r5 ^ t_; } // s87 shfl
|
||||
r0 = r0 ^ r4; // s88 xor
|
||||
r3 = r3 + r2 + ((((sel >> 18u) & 1u) != 0u) ? 0x883c0c92u : 0x53f915c2u); // s89 add
|
||||
r7 = rotr_var(r7, r5); // s90 rotr
|
||||
r3 = r3 + r1 + ((((sel >> 31u) & 1u) != 0u) ? 0x3cc5bd20u : 0xe2be00a5u); // s91 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 1u); r7 = r7 ^ t_; } // s92 shfl
|
||||
r6 = rotr_var(r6, r2); // s93 rotr
|
||||
r7 = rotl_imm(r7, 14u); // s94 rotl
|
||||
r4 = r5 * r5 + r4; // s95 mad
|
||||
r2 = r4 * r5 + r2; // s96 mad
|
||||
r1 = r1 ^ r0; // s97 xor
|
||||
r5 = r5 * r4; // s98 mul
|
||||
r2 = r2 - r0; // s99 sub
|
||||
r7 = rotl_imm(r7, 30u); // s100 rotl
|
||||
r5 = r5 + r6 + ((((sel >> 17u) & 1u) != 0u) ? 0xbfd646cbu : 0x423fd9c9u); // s101 add
|
||||
r7 = r7 + r6 + ((((sel >> 11u) & 1u) != 0u) ? 0x19b74a43u : 0x8d3c011du); // s102 add
|
||||
r5 = rotl_imm(r5, 6u); // s103 rotl
|
||||
r0 = r0 * r4; // s104 mul
|
||||
r0 = r0 | r5; // s105 or
|
||||
r0 = r0 + r1 + ((((sel >> 15u) & 1u) != 0u) ? 0x7dcb7e18u : 0x8ce14721u); // s106 add
|
||||
r0 = rotl_imm(r0, 15u); // s107 rotl
|
||||
r4 = r4 - r2; // s108 sub
|
||||
r2 = mul_hi(r2, r0); // s109 mulhi
|
||||
r1 = mul_hi(r1, r0); // s110 mulhi
|
||||
r0 = r0 + r2 + ((((sel >> 28u) & 1u) != 0u) ? 0x3c179ad8u : 0x2d9fc6b9u); // s111 add
|
||||
r2 = mul_hi(r2, r0); // s112 mulhi
|
||||
r6 = r6 - r3; // s113 sub
|
||||
r7 = r6 * r3 + r7; // s114 mad
|
||||
r5 = rotr_var(r5, r1); // s115 rotr
|
||||
r6 = rotr_var(r6, r4); // s116 rotr
|
||||
r2 = r2 + r1 + ((((sel >> 22u) & 1u) != 0u) ? 0x47359729u : 0x502138e2u); // s117 add
|
||||
r3 = r2 * r0 + r3; // s118 mad
|
||||
r1 = r1 * r3; // s119 mul
|
||||
r2 = r0 * r4 + r2; // s120 mad
|
||||
r0 = r0 * r3; // s121 mul
|
||||
r2 = mul_hi(r2, r0); // s122 mulhi
|
||||
r5 = r5 * r6; // s123 mul
|
||||
r4 = r2 * r4 + r4; // s124 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 2u); r4 = r4 ^ t_; } // s125 shfl
|
||||
r2 = r2 ^ r0; // s126 xor
|
||||
r1 = rotl_imm(r1, 7u); // s127 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r7 = r7 ^ t_; } // s128 shfl
|
||||
r6 = rotl_imm(r6, 17u); // s129 rotl
|
||||
r2 = r2 + r5 + ((((sel >> 15u) & 1u) != 0u) ? 0x6c57e4e7u : 0x33dff776u); // s130 add
|
||||
r0 = r0 | r3; // s131 or
|
||||
r5 = rotr_var(r5, r0); // s132 rotr
|
||||
r2 = r2 + r3 + ((((sel >> 30u) & 1u) != 0u) ? 0xe8212c0cu : 0x5db25b34u); // s133 add
|
||||
r4 = r4 ^ r3; // s134 xor
|
||||
r6 = r6 ^ r1; // s135 xor
|
||||
r1 = r1 - r7; // s136 sub
|
||||
r2 = r6 * r2 + r2; // s137 mad
|
||||
r0 = mul_hi(r0, r5); // s138 mulhi
|
||||
r2 = r2 + r7 + ((((sel >> 10u) & 1u) != 0u) ? 0xd44ff710u : 0x218d4090u); // s139 add
|
||||
r1 = rotr_var(r1, r4); // s140 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r3 = r3 ^ t_; } // s141 shfl
|
||||
r7 = r6 * r2 + r7; // s142 mad
|
||||
r2 = r2 * r3; // s143 mul
|
||||
r7 = r7 + r4 + ((((sel >> 29u) & 1u) != 0u) ? 0xb98942fau : 0xf4b1a8deu); // s144 add
|
||||
r7 = rotl_imm(r7, 15u); // s145 rotl
|
||||
r7 = r7 ^ r5; // s146 xor
|
||||
r4 = r7 * r4 + r4; // s147 mad
|
||||
r6 = rotr_var(r6, r5); // s148 rotr
|
||||
r1 = r2 * r4 + r1; // s149 mad
|
||||
r1 = r1 + r7 + ((((sel >> 17u) & 1u) != 0u) ? 0x0d48ba42u : 0x2bef10f2u); // s150 add
|
||||
r5 = r5 ^ r4; // s151 xor
|
||||
r7 = r7 + r0 + ((((sel >> 30u) & 1u) != 0u) ? 0xc98dea9cu : 0xdc5cc080u); // s152 add
|
||||
r4 = r4 * r6; // s153 mul
|
||||
r0 = r0 * r2; // s154 mul
|
||||
r6 = r6 ^ r5; // s155 xor
|
||||
r4 = rotr_var(r4, r2); // s156 rotr
|
||||
r1 = rotl_imm(r1, 11u); // s157 rotl
|
||||
r5 = r5 + r0 + ((((sel >> 11u) & 1u) != 0u) ? 0x6c752dcbu : 0x18197438u); // s158 add
|
||||
r4 = r1 * r5 + r4; // s159 mad
|
||||
r4 = rotl_imm(r4, 26u); // s160 rotl
|
||||
r3 = r0 * r7 + r3; // s161 mad
|
||||
r3 = rotr_var(r3, r1); // s162 rotr
|
||||
r4 = r4 + r0 + ((((sel >> 3u) & 1u) != 0u) ? 0x8e481727u : 0xf1c46574u); // s163 add
|
||||
r0 = mul_hi(r0, r4); // s164 mulhi
|
||||
r2 = r2 + r6 + ((((sel >> 20u) & 1u) != 0u) ? 0x550ab406u : 0xac578137u); // s165 add
|
||||
r0 = rotl_imm(r0, 13u); // s166 rotl
|
||||
r3 = r3 + r1 + ((((sel >> 14u) & 1u) != 0u) ? 0xaebb5966u : 0xa33e6706u); // s167 add
|
||||
r3 = r3 | r5; // s168 or
|
||||
r6 = rotr_var(r6, r2); // s169 rotr
|
||||
r4 = r4 ^ r6; // s170 xor
|
||||
r6 = r6 - r1; // s171 sub
|
||||
r7 = rotl_imm(r7, 22u); // s172 rotl
|
||||
r5 = rotl_imm(r5, 15u); // s173 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 8u); r7 = r7 ^ t_; } // s174 shfl
|
||||
r0 = r0 ^ r5; // s175 xor
|
||||
r7 = rotl_imm(r7, 6u); // s176 rotl
|
||||
r7 = r7 - r0; // s177 sub
|
||||
r3 = rotl_imm(r3, 30u); // s178 rotl
|
||||
r7 = r6 * r1 + r7; // s179 mad
|
||||
r6 = rotl_imm(r6, 9u); // s180 rotl
|
||||
r2 = r2 ^ r4; // s181 xor
|
||||
r2 = r2 ^ r7; // s182 xor
|
||||
r7 = r7 ^ r2; // s183 xor
|
||||
r1 = r1 + r2 + ((((sel >> 21u) & 1u) != 0u) ? 0x5bb7550fu : 0xd94d55acu); // s184 add
|
||||
r3 = r3 | r5; // s185 or
|
||||
r6 = mul_hi(r6, r3); // s186 mulhi
|
||||
r4 = r4 | r0; // s187 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 2u); r7 = r7 ^ t_; } // s188 shfl
|
||||
r6 = r6 + r5 + ((((sel >> 6u) & 1u) != 0u) ? 0x2a354e2du : 0x89e747feu); // s189 add
|
||||
r1 = r1 ^ r4; // s190 xor
|
||||
r7 = mul_hi(r7, r4); // s191 mulhi
|
||||
r2 = rotl_imm(r2, 26u); // s192 rotl
|
||||
r5 = rotl_imm(r5, 8u); // s193 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r4 = r4 ^ t_; } // s194 shfl
|
||||
r4 = r4 ^ r5; // s195 xor
|
||||
r1 = r1 * r3; // s196 mul
|
||||
r5 = r5 + r2 + ((((sel >> 7u) & 1u) != 0u) ? 0x10cfdc71u : 0xea03e8e7u); // s197 add
|
||||
r7 = r7 + r5 + ((((sel >> 15u) & 1u) != 0u) ? 0x66e148d3u : 0xa7ee0102u); // s198 add
|
||||
r5 = r5 - r2; // s199 sub
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 2u); r5 = r5 ^ t_; } // s200 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 8u); r7 = r7 ^ t_; } // s201 shfl
|
||||
r4 = rotr_var(r4, r5); // s202 rotr
|
||||
r7 = r7 ^ r5; // s203 xor
|
||||
r7 = r7 ^ r0; // s204 xor
|
||||
r6 = r6 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0x8558b619u : 0x8379a4deu); // s205 add
|
||||
r4 = rotl_imm(r4, 13u); // s206 rotl
|
||||
r1 = mul_hi(r1, r3); // s207 mulhi
|
||||
r1 = r4 * r4 + r1; // s208 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 4u); r2 = r2 ^ t_; } // s209 shfl
|
||||
r7 = r7 + r0 + ((((sel >> 11u) & 1u) != 0u) ? 0xf4049c4cu : 0xaa8cb14eu); // s210 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 16u); r7 = r7 ^ t_; } // s211 shfl
|
||||
r1 = r1 ^ r0; // s212 xor
|
||||
r7 = rotl_imm(r7, 3u); // s213 rotl
|
||||
r4 = rotr_var(r4, r7); // s214 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 16u); r3 = r3 ^ t_; } // s215 shfl
|
||||
r5 = r5 | r1; // s216 or
|
||||
r1 = r1 - r2; // s217 sub
|
||||
r6 = r6 - r5; // s218 sub
|
||||
r6 = rotl_imm(r6, 4u); // s219 rotl
|
||||
r2 = mul_hi(r2, r0); // s220 mulhi
|
||||
r2 = r2 | r0; // s221 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r5 = r5 ^ t_; } // s222 shfl
|
||||
r5 = mul_hi(r5, r6); // s223 mulhi
|
||||
r0 = r0 - r6; // s224 sub
|
||||
r7 = rotl_imm(r7, 23u); // s225 rotl
|
||||
r4 = r4 | r2; // s226 or
|
||||
r2 = r2 * r4; // s227 mul
|
||||
r3 = rotl_imm(r3, 12u); // s228 rotl
|
||||
r0 = rotr_var(r0, r4); // s229 rotr
|
||||
r0 = r0 + r6 + ((((sel >> 16u) & 1u) != 0u) ? 0x2a7fecb2u : 0x1d176220u); // s230 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r1 = r1 ^ t_; } // s231 shfl
|
||||
r2 = r2 * r1; // s232 mul
|
||||
r7 = r0 * r1 + r7; // s233 mad
|
||||
r5 = rotl_imm(r5, 22u); // s234 rotl
|
||||
r4 = rotr_var(r4, r6); // s235 rotr
|
||||
r0 = r5 * r1 + r0; // s236 mad
|
||||
r6 = r6 ^ r4; // s237 xor
|
||||
r4 = r4 ^ r6; // s238 xor
|
||||
r6 = rotl_imm(r6, 18u); // s239 rotl
|
||||
r4 = r4 + r7 + ((((sel >> 25u) & 1u) != 0u) ? 0xadce39f3u : 0x17dafb4du); // s240 add
|
||||
r0 = r0 - r7; // s241 sub
|
||||
r1 = rotr_var(r1, r6); // s242 rotr
|
||||
r3 = r3 + r0 + ((((sel >> 29u) & 1u) != 0u) ? 0x0e7033b6u : 0xf2f77d26u); // s243 add
|
||||
r2 = r7 * r4 + r2; // s244 mad
|
||||
r4 = r0 * r6 + r4; // s245 mad
|
||||
r5 = r5 * r7; // s246 mul
|
||||
r3 = r3 + r5 + ((((sel >> 31u) & 1u) != 0u) ? 0x7b2ff6b7u : 0xfed2da4eu); // s247 add
|
||||
r0 = r0 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0xb5fad7b1u : 0x03b2891cu); // s248 add
|
||||
r5 = mul_hi(r5, r4); // s249 mulhi
|
||||
r5 = r5 + r2 + ((((sel >> 11u) & 1u) != 0u) ? 0xa7e71c2fu : 0x042cd6e3u); // s250 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 1u); r6 = r6 ^ t_; } // s251 shfl
|
||||
r6 = r6 ^ r1; // s252 xor
|
||||
r2 = r2 * r5; // s253 mul
|
||||
r0 = r0 + r6 + ((((sel >> 16u) & 1u) != 0u) ? 0xb83d78deu : 0x784a302bu); // s254 add
|
||||
r2 = r2 - r4; // s255 sub
|
||||
}
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
|
||||
#if IGNEUM_EXCHANGE != 0
|
||||
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
|
||||
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
|
||||
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
|
||||
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
|
||||
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
|
||||
}
|
||||
#endif
|
||||
|
||||
// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash.
|
||||
IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw) {
|
||||
uint gid = (uint)get_global_id(0);
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; }
|
||||
{ uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; }
|
||||
{ uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; }
|
||||
{ uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; }
|
||||
{ uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; }
|
||||
{ uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; }
|
||||
{ uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; }
|
||||
{ uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r2 = r1 * r1 + r2; // 1 mad
|
||||
r2 = r3 * r2 + r2; // 2 mad
|
||||
r3 = r3 ^ r5; // 3 xor
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r5 = r5 ^ ds[r7 & mask]; // 5 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl
|
||||
r1 = mul_hi(r1, r5); // 8 mulhi
|
||||
r6 = rotr_var(r6, r3); // 9 rotr
|
||||
r3 = r3 | r4; // 10 or
|
||||
r4 = r4 ^ ds[r6 & mask]; // 11 load
|
||||
r0 = mul_hi(r0, r4); // 12 mulhi
|
||||
r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add
|
||||
r0 = r0 ^ ds[r5 & mask]; // 14 load
|
||||
r2 = r2 - r4; // 15 sub
|
||||
r2 = r2 ^ ds[r7 & mask]; // 16 load
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl
|
||||
r5 = r5 * r0; // 19 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl
|
||||
r6 = mul_hi(r6, r2); // 22 mulhi
|
||||
r6 = r6 ^ ds[r3 & mask]; // 23 load
|
||||
r5 = r5 * r0; // 24 mul
|
||||
r5 = rotl_imm(r5, 19u); // 25 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl
|
||||
r0 = r0 ^ r5; // 27 xor
|
||||
r0 = r0 ^ r4; // 28 xor
|
||||
r3 = r3 - r0; // 29 sub
|
||||
r5 = r5 * r1; // 30 mul
|
||||
r7 = r7 ^ ds[r3 & mask]; // 31 load
|
||||
r1 = r1 ^ ds[r2 & mask]; // 32 load
|
||||
r5 = r5 ^ r6; // 33 xor
|
||||
r5 = r5 ^ ds[r0 & mask]; // 34 load
|
||||
r0 = mul_hi(r0, r5); // 35 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl
|
||||
r7 = r7 ^ ds[r6 & mask]; // 37 load
|
||||
r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl
|
||||
r2 = r2 ^ r5; // 40 xor
|
||||
r3 = r6 * r3 + r3; // 41 mad
|
||||
r6 = r6 - r7; // 42 sub
|
||||
r7 = r7 ^ r0; // 43 xor
|
||||
r1 = r1 ^ ds[r7 & mask]; // 44 load
|
||||
r2 = r2 * r3; // 45 mul
|
||||
r1 = mul_hi(r1, r5); // 46 mulhi
|
||||
r4 = r4 - r3; // 47 sub
|
||||
r2 = rotr_var(r2, r6); // 48 rotr
|
||||
r3 = r3 ^ ds[r4 & mask]; // 49 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add
|
||||
r0 = r0 * r2; // 51 mul
|
||||
r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add
|
||||
r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add
|
||||
r7 = rotl_imm(r7, 14u); // 54 rotl
|
||||
r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add
|
||||
r6 = r6 ^ ds[r7 & mask]; // 56 load
|
||||
r1 = rotr_var(r1, r5); // 57 rotr
|
||||
r5 = r5 ^ ds[r3 & mask]; // 58 load
|
||||
r6 = r6 ^ ds[r1 & mask]; // 59 load
|
||||
r3 = r5 * r0 + r3; // 60 mad
|
||||
r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add
|
||||
r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add
|
||||
r5 = rotl_imm(r5, 19u); // 63 rotl
|
||||
// latency-shadow block (Counter ASIC 3.0 item 8): 256 ALU instructions x 27 passes after instruction 63, no load
|
||||
for (uint sh = 0u; sh < 27u; ++sh) {
|
||||
r5 = r5 + r2 + ((((sel >> 18u) & 1u) != 0u) ? 0x8bc12da9u : 0x92199f99u); // s0 add
|
||||
r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x63079e5au : 0x8d72d3adu); // s1 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 2u); r6 = r6 ^ t_; } // s2 shfl
|
||||
r4 = r4 - r2; // s3 sub
|
||||
r7 = r7 + r0 + ((((sel >> 31u) & 1u) != 0u) ? 0x5d4c7a60u : 0xb21b4babu); // s4 add
|
||||
r0 = rotl_imm(r0, 11u); // s5 rotl
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // s6 add
|
||||
r7 = r7 ^ r1; // s7 xor
|
||||
r1 = r6 * r5 + r1; // s8 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 4u); r6 = r6 ^ t_; } // s9 shfl
|
||||
r1 = r2 * r2 + r1; // s10 mad
|
||||
r5 = r0 * r3 + r5; // s11 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 4u); r2 = r2 ^ t_; } // s12 shfl
|
||||
r1 = r1 + r5 + ((((sel >> 25u) & 1u) != 0u) ? 0xbc48c63eu : 0xbdf8f9a5u); // s13 add
|
||||
r1 = rotl_imm(r1, 29u); // s14 rotl
|
||||
r1 = r1 - r4; // s15 sub
|
||||
r7 = r7 | r1; // s16 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r2 = r2 ^ t_; } // s17 shfl
|
||||
r7 = r7 ^ r4; // s18 xor
|
||||
r6 = r6 * r1; // s19 mul
|
||||
r5 = r6 * r0 + r5; // s20 mad
|
||||
r3 = r3 - r1; // s21 sub
|
||||
r6 = r6 * r0; // s22 mul
|
||||
r2 = r2 + r0 + ((((sel >> 1u) & 1u) != 0u) ? 0x96e8f127u : 0x45c37cecu); // s23 add
|
||||
r6 = r6 - r4; // s24 sub
|
||||
r7 = r3 * r4 + r7; // s25 mad
|
||||
r3 = rotl_imm(r3, 9u); // s26 rotl
|
||||
r2 = r2 - r1; // s27 sub
|
||||
r6 = mul_hi(r6, r0); // s28 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r2 = r2 ^ t_; } // s29 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r1 = r1 ^ t_; } // s30 shfl
|
||||
r1 = r1 + r6 + ((((sel >> 1u) & 1u) != 0u) ? 0xb1cdb2abu : 0x37985632u); // s31 add
|
||||
r0 = rotr_var(r0, r1); // s32 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // s33 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r0 = r0 ^ t_; } // s34 shfl
|
||||
r7 = r7 * r0; // s35 mul
|
||||
r3 = r3 + r1 + ((((sel >> 0u) & 1u) != 0u) ? 0x856e0180u : 0x804e777eu); // s36 add
|
||||
r1 = r1 ^ r6; // s37 xor
|
||||
r2 = r2 * r7; // s38 mul
|
||||
r6 = r6 + r5 + ((((sel >> 2u) & 1u) != 0u) ? 0xe9bd0cc4u : 0x0b06c8a7u); // s39 add
|
||||
r5 = r5 * r7; // s40 mul
|
||||
r2 = mul_hi(r2, r3); // s41 mulhi
|
||||
r2 = r2 ^ r0; // s42 xor
|
||||
r0 = rotl_imm(r0, 4u); // s43 rotl
|
||||
r4 = mul_hi(r4, r3); // s44 mulhi
|
||||
r6 = r6 + r7 + ((((sel >> 12u) & 1u) != 0u) ? 0xcdcb63f6u : 0xee822e17u); // s45 add
|
||||
r3 = r2 * r0 + r3; // s46 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r4 = r4 ^ t_; } // s47 shfl
|
||||
r6 = mul_hi(r6, r5); // s48 mulhi
|
||||
r0 = r0 - r2; // s49 sub
|
||||
r3 = r3 - r5; // s50 sub
|
||||
r1 = rotr_var(r1, r4); // s51 rotr
|
||||
r6 = r7 * r7 + r6; // s52 mad
|
||||
r5 = r5 ^ r3; // s53 xor
|
||||
r1 = r1 - r0; // s54 sub
|
||||
r5 = r5 - r6; // s55 sub
|
||||
r3 = r3 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0xd4758987u : 0x3dfad1b6u); // s56 add
|
||||
r4 = r4 + r1 + ((((sel >> 0u) & 1u) != 0u) ? 0x0602d6beu : 0xfca75bc2u); // s57 add
|
||||
r5 = mul_hi(r5, r6); // s58 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 2u); r2 = r2 ^ t_; } // s59 shfl
|
||||
r2 = rotl_imm(r2, 9u); // s60 rotl
|
||||
r4 = r4 | r6; // s61 or
|
||||
r6 = rotr_var(r6, r4); // s62 rotr
|
||||
r2 = r2 * r4; // s63 mul
|
||||
r0 = r0 ^ r5; // s64 xor
|
||||
r2 = r2 ^ r7; // s65 xor
|
||||
r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x8596687au : 0xd71c02ffu); // s66 add
|
||||
r1 = rotl_imm(r1, 21u); // s67 rotl
|
||||
r3 = r3 * r2; // s68 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 2u); r7 = r7 ^ t_; } // s69 shfl
|
||||
r3 = r3 * r2; // s70 mul
|
||||
r0 = r2 * r5 + r0; // s71 mad
|
||||
r6 = r6 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0x08ac5733u : 0x3c6fe15du); // s72 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r3 = r3 ^ t_; } // s73 shfl
|
||||
r3 = r3 * r6; // s74 mul
|
||||
r6 = r6 ^ r7; // s75 xor
|
||||
r3 = r3 | r0; // s76 or
|
||||
r2 = r2 + r4 + ((((sel >> 1u) & 1u) != 0u) ? 0xc100b495u : 0x295d5faeu); // s77 add
|
||||
r3 = mul_hi(r3, r7); // s78 mulhi
|
||||
r4 = r4 | r1; // s79 or
|
||||
r4 = rotr_var(r4, r3); // s80 rotr
|
||||
r4 = r4 + r3 + ((((sel >> 15u) & 1u) != 0u) ? 0x5cc59530u : 0x274a9221u); // s81 add
|
||||
r7 = r7 + r3 + ((((sel >> 5u) & 1u) != 0u) ? 0x141479ecu : 0xc8651f8eu); // s82 add
|
||||
r3 = rotr_var(r3, r6); // s83 rotr
|
||||
r2 = r4 * r7 + r2; // s84 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r2 = r2 ^ t_; } // s85 shfl
|
||||
r3 = r3 ^ r2; // s86 xor
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 2u); r5 = r5 ^ t_; } // s87 shfl
|
||||
r0 = r0 ^ r4; // s88 xor
|
||||
r3 = r3 + r2 + ((((sel >> 18u) & 1u) != 0u) ? 0x883c0c92u : 0x53f915c2u); // s89 add
|
||||
r7 = rotr_var(r7, r5); // s90 rotr
|
||||
r3 = r3 + r1 + ((((sel >> 31u) & 1u) != 0u) ? 0x3cc5bd20u : 0xe2be00a5u); // s91 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 1u); r7 = r7 ^ t_; } // s92 shfl
|
||||
r6 = rotr_var(r6, r2); // s93 rotr
|
||||
r7 = rotl_imm(r7, 14u); // s94 rotl
|
||||
r4 = r5 * r5 + r4; // s95 mad
|
||||
r2 = r4 * r5 + r2; // s96 mad
|
||||
r1 = r1 ^ r0; // s97 xor
|
||||
r5 = r5 * r4; // s98 mul
|
||||
r2 = r2 - r0; // s99 sub
|
||||
r7 = rotl_imm(r7, 30u); // s100 rotl
|
||||
r5 = r5 + r6 + ((((sel >> 17u) & 1u) != 0u) ? 0xbfd646cbu : 0x423fd9c9u); // s101 add
|
||||
r7 = r7 + r6 + ((((sel >> 11u) & 1u) != 0u) ? 0x19b74a43u : 0x8d3c011du); // s102 add
|
||||
r5 = rotl_imm(r5, 6u); // s103 rotl
|
||||
r0 = r0 * r4; // s104 mul
|
||||
r0 = r0 | r5; // s105 or
|
||||
r0 = r0 + r1 + ((((sel >> 15u) & 1u) != 0u) ? 0x7dcb7e18u : 0x8ce14721u); // s106 add
|
||||
r0 = rotl_imm(r0, 15u); // s107 rotl
|
||||
r4 = r4 - r2; // s108 sub
|
||||
r2 = mul_hi(r2, r0); // s109 mulhi
|
||||
r1 = mul_hi(r1, r0); // s110 mulhi
|
||||
r0 = r0 + r2 + ((((sel >> 28u) & 1u) != 0u) ? 0x3c179ad8u : 0x2d9fc6b9u); // s111 add
|
||||
r2 = mul_hi(r2, r0); // s112 mulhi
|
||||
r6 = r6 - r3; // s113 sub
|
||||
r7 = r6 * r3 + r7; // s114 mad
|
||||
r5 = rotr_var(r5, r1); // s115 rotr
|
||||
r6 = rotr_var(r6, r4); // s116 rotr
|
||||
r2 = r2 + r1 + ((((sel >> 22u) & 1u) != 0u) ? 0x47359729u : 0x502138e2u); // s117 add
|
||||
r3 = r2 * r0 + r3; // s118 mad
|
||||
r1 = r1 * r3; // s119 mul
|
||||
r2 = r0 * r4 + r2; // s120 mad
|
||||
r0 = r0 * r3; // s121 mul
|
||||
r2 = mul_hi(r2, r0); // s122 mulhi
|
||||
r5 = r5 * r6; // s123 mul
|
||||
r4 = r2 * r4 + r4; // s124 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 2u); r4 = r4 ^ t_; } // s125 shfl
|
||||
r2 = r2 ^ r0; // s126 xor
|
||||
r1 = rotl_imm(r1, 7u); // s127 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r7 = r7 ^ t_; } // s128 shfl
|
||||
r6 = rotl_imm(r6, 17u); // s129 rotl
|
||||
r2 = r2 + r5 + ((((sel >> 15u) & 1u) != 0u) ? 0x6c57e4e7u : 0x33dff776u); // s130 add
|
||||
r0 = r0 | r3; // s131 or
|
||||
r5 = rotr_var(r5, r0); // s132 rotr
|
||||
r2 = r2 + r3 + ((((sel >> 30u) & 1u) != 0u) ? 0xe8212c0cu : 0x5db25b34u); // s133 add
|
||||
r4 = r4 ^ r3; // s134 xor
|
||||
r6 = r6 ^ r1; // s135 xor
|
||||
r1 = r1 - r7; // s136 sub
|
||||
r2 = r6 * r2 + r2; // s137 mad
|
||||
r0 = mul_hi(r0, r5); // s138 mulhi
|
||||
r2 = r2 + r7 + ((((sel >> 10u) & 1u) != 0u) ? 0xd44ff710u : 0x218d4090u); // s139 add
|
||||
r1 = rotr_var(r1, r4); // s140 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r3 = r3 ^ t_; } // s141 shfl
|
||||
r7 = r6 * r2 + r7; // s142 mad
|
||||
r2 = r2 * r3; // s143 mul
|
||||
r7 = r7 + r4 + ((((sel >> 29u) & 1u) != 0u) ? 0xb98942fau : 0xf4b1a8deu); // s144 add
|
||||
r7 = rotl_imm(r7, 15u); // s145 rotl
|
||||
r7 = r7 ^ r5; // s146 xor
|
||||
r4 = r7 * r4 + r4; // s147 mad
|
||||
r6 = rotr_var(r6, r5); // s148 rotr
|
||||
r1 = r2 * r4 + r1; // s149 mad
|
||||
r1 = r1 + r7 + ((((sel >> 17u) & 1u) != 0u) ? 0x0d48ba42u : 0x2bef10f2u); // s150 add
|
||||
r5 = r5 ^ r4; // s151 xor
|
||||
r7 = r7 + r0 + ((((sel >> 30u) & 1u) != 0u) ? 0xc98dea9cu : 0xdc5cc080u); // s152 add
|
||||
r4 = r4 * r6; // s153 mul
|
||||
r0 = r0 * r2; // s154 mul
|
||||
r6 = r6 ^ r5; // s155 xor
|
||||
r4 = rotr_var(r4, r2); // s156 rotr
|
||||
r1 = rotl_imm(r1, 11u); // s157 rotl
|
||||
r5 = r5 + r0 + ((((sel >> 11u) & 1u) != 0u) ? 0x6c752dcbu : 0x18197438u); // s158 add
|
||||
r4 = r1 * r5 + r4; // s159 mad
|
||||
r4 = rotl_imm(r4, 26u); // s160 rotl
|
||||
r3 = r0 * r7 + r3; // s161 mad
|
||||
r3 = rotr_var(r3, r1); // s162 rotr
|
||||
r4 = r4 + r0 + ((((sel >> 3u) & 1u) != 0u) ? 0x8e481727u : 0xf1c46574u); // s163 add
|
||||
r0 = mul_hi(r0, r4); // s164 mulhi
|
||||
r2 = r2 + r6 + ((((sel >> 20u) & 1u) != 0u) ? 0x550ab406u : 0xac578137u); // s165 add
|
||||
r0 = rotl_imm(r0, 13u); // s166 rotl
|
||||
r3 = r3 + r1 + ((((sel >> 14u) & 1u) != 0u) ? 0xaebb5966u : 0xa33e6706u); // s167 add
|
||||
r3 = r3 | r5; // s168 or
|
||||
r6 = rotr_var(r6, r2); // s169 rotr
|
||||
r4 = r4 ^ r6; // s170 xor
|
||||
r6 = r6 - r1; // s171 sub
|
||||
r7 = rotl_imm(r7, 22u); // s172 rotl
|
||||
r5 = rotl_imm(r5, 15u); // s173 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 8u); r7 = r7 ^ t_; } // s174 shfl
|
||||
r0 = r0 ^ r5; // s175 xor
|
||||
r7 = rotl_imm(r7, 6u); // s176 rotl
|
||||
r7 = r7 - r0; // s177 sub
|
||||
r3 = rotl_imm(r3, 30u); // s178 rotl
|
||||
r7 = r6 * r1 + r7; // s179 mad
|
||||
r6 = rotl_imm(r6, 9u); // s180 rotl
|
||||
r2 = r2 ^ r4; // s181 xor
|
||||
r2 = r2 ^ r7; // s182 xor
|
||||
r7 = r7 ^ r2; // s183 xor
|
||||
r1 = r1 + r2 + ((((sel >> 21u) & 1u) != 0u) ? 0x5bb7550fu : 0xd94d55acu); // s184 add
|
||||
r3 = r3 | r5; // s185 or
|
||||
r6 = mul_hi(r6, r3); // s186 mulhi
|
||||
r4 = r4 | r0; // s187 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 2u); r7 = r7 ^ t_; } // s188 shfl
|
||||
r6 = r6 + r5 + ((((sel >> 6u) & 1u) != 0u) ? 0x2a354e2du : 0x89e747feu); // s189 add
|
||||
r1 = r1 ^ r4; // s190 xor
|
||||
r7 = mul_hi(r7, r4); // s191 mulhi
|
||||
r2 = rotl_imm(r2, 26u); // s192 rotl
|
||||
r5 = rotl_imm(r5, 8u); // s193 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r4 = r4 ^ t_; } // s194 shfl
|
||||
r4 = r4 ^ r5; // s195 xor
|
||||
r1 = r1 * r3; // s196 mul
|
||||
r5 = r5 + r2 + ((((sel >> 7u) & 1u) != 0u) ? 0x10cfdc71u : 0xea03e8e7u); // s197 add
|
||||
r7 = r7 + r5 + ((((sel >> 15u) & 1u) != 0u) ? 0x66e148d3u : 0xa7ee0102u); // s198 add
|
||||
r5 = r5 - r2; // s199 sub
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 2u); r5 = r5 ^ t_; } // s200 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 8u); r7 = r7 ^ t_; } // s201 shfl
|
||||
r4 = rotr_var(r4, r5); // s202 rotr
|
||||
r7 = r7 ^ r5; // s203 xor
|
||||
r7 = r7 ^ r0; // s204 xor
|
||||
r6 = r6 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0x8558b619u : 0x8379a4deu); // s205 add
|
||||
r4 = rotl_imm(r4, 13u); // s206 rotl
|
||||
r1 = mul_hi(r1, r3); // s207 mulhi
|
||||
r1 = r4 * r4 + r1; // s208 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 4u); r2 = r2 ^ t_; } // s209 shfl
|
||||
r7 = r7 + r0 + ((((sel >> 11u) & 1u) != 0u) ? 0xf4049c4cu : 0xaa8cb14eu); // s210 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 16u); r7 = r7 ^ t_; } // s211 shfl
|
||||
r1 = r1 ^ r0; // s212 xor
|
||||
r7 = rotl_imm(r7, 3u); // s213 rotl
|
||||
r4 = rotr_var(r4, r7); // s214 rotr
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 16u); r3 = r3 ^ t_; } // s215 shfl
|
||||
r5 = r5 | r1; // s216 or
|
||||
r1 = r1 - r2; // s217 sub
|
||||
r6 = r6 - r5; // s218 sub
|
||||
r6 = rotl_imm(r6, 4u); // s219 rotl
|
||||
r2 = mul_hi(r2, r0); // s220 mulhi
|
||||
r2 = r2 | r0; // s221 or
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r5 = r5 ^ t_; } // s222 shfl
|
||||
r5 = mul_hi(r5, r6); // s223 mulhi
|
||||
r0 = r0 - r6; // s224 sub
|
||||
r7 = rotl_imm(r7, 23u); // s225 rotl
|
||||
r4 = r4 | r2; // s226 or
|
||||
r2 = r2 * r4; // s227 mul
|
||||
r3 = rotl_imm(r3, 12u); // s228 rotl
|
||||
r0 = rotr_var(r0, r4); // s229 rotr
|
||||
r0 = r0 + r6 + ((((sel >> 16u) & 1u) != 0u) ? 0x2a7fecb2u : 0x1d176220u); // s230 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r1 = r1 ^ t_; } // s231 shfl
|
||||
r2 = r2 * r1; // s232 mul
|
||||
r7 = r0 * r1 + r7; // s233 mad
|
||||
r5 = rotl_imm(r5, 22u); // s234 rotl
|
||||
r4 = rotr_var(r4, r6); // s235 rotr
|
||||
r0 = r5 * r1 + r0; // s236 mad
|
||||
r6 = r6 ^ r4; // s237 xor
|
||||
r4 = r4 ^ r6; // s238 xor
|
||||
r6 = rotl_imm(r6, 18u); // s239 rotl
|
||||
r4 = r4 + r7 + ((((sel >> 25u) & 1u) != 0u) ? 0xadce39f3u : 0x17dafb4du); // s240 add
|
||||
r0 = r0 - r7; // s241 sub
|
||||
r1 = rotr_var(r1, r6); // s242 rotr
|
||||
r3 = r3 + r0 + ((((sel >> 29u) & 1u) != 0u) ? 0x0e7033b6u : 0xf2f77d26u); // s243 add
|
||||
r2 = r7 * r4 + r2; // s244 mad
|
||||
r4 = r0 * r6 + r4; // s245 mad
|
||||
r5 = r5 * r7; // s246 mul
|
||||
r3 = r3 + r5 + ((((sel >> 31u) & 1u) != 0u) ? 0x7b2ff6b7u : 0xfed2da4eu); // s247 add
|
||||
r0 = r0 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0xb5fad7b1u : 0x03b2891cu); // s248 add
|
||||
r5 = mul_hi(r5, r4); // s249 mulhi
|
||||
r5 = r5 + r2 + ((((sel >> 11u) & 1u) != 0u) ? 0xa7e71c2fu : 0x042cd6e3u); // s250 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r1, 1u); r6 = r6 ^ t_; } // s251 shfl
|
||||
r6 = r6 ^ r1; // s252 xor
|
||||
r2 = r2 * r5; // s253 mul
|
||||
r0 = r0 + r6 + ((((sel >> 16u) & 1u) != 0u) ? 0xb83d78deu : 0x784a302bu); // s254 add
|
||||
r2 = r2 - r4; // s255 sub
|
||||
}
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
382
proto-cuda/packs-ca3-v5/v5-genesis/kernel_bound.cu
Normal file
382
proto-cuda/packs-ca3-v5/v5-genesis/kernel_bound.cu
Normal file
|
|
@ -0,0 +1,382 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW.
|
||||
// Host declarations (also in program_bound.h if present):
|
||||
// struct IgneumInitWords { uint32_t w[8]; };
|
||||
// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps);
|
||||
// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
|
||||
#include <cuda_runtime.h>
|
||||
#include <cstdint>
|
||||
#include "program.h"
|
||||
|
||||
struct IgneumInitWords { uint32_t w[8]; };
|
||||
|
||||
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
|
||||
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
|
||||
__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw) {
|
||||
uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
uint32_t nonce = baseNonce + gid;
|
||||
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; }
|
||||
{ uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; }
|
||||
{ uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; }
|
||||
{ uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; }
|
||||
{ uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; }
|
||||
{ uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; }
|
||||
{ uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; }
|
||||
{ uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; }
|
||||
|
||||
for (uint32_t it = 0u; it < 8u; ++it) {
|
||||
uint32_t sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r2 = r1 * r1 + r2; // 1 mad
|
||||
r2 = r3 * r2 + r2; // 2 mad
|
||||
r3 = r3 ^ r5; // 3 xor
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r5 = r5 ^ ds[r7 & mask]; // 5 load
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 6 shfl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 8); // 7 shfl
|
||||
r1 = __umulhi(r1, r5); // 8 mulhi
|
||||
r6 = rotr_var(r6, r3); // 9 rotr
|
||||
r3 = r3 | r4; // 10 or
|
||||
r4 = r4 ^ ds[r6 & mask]; // 11 load
|
||||
r0 = __umulhi(r0, r4); // 12 mulhi
|
||||
r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add
|
||||
r0 = r0 ^ ds[r5 & mask]; // 14 load
|
||||
r2 = r2 - r4; // 15 sub
|
||||
r2 = r2 ^ ds[r7 & mask]; // 16 load
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 18 shfl
|
||||
r5 = r5 * r0; // 19 mul
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 20 shfl
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 16); // 21 shfl
|
||||
r6 = __umulhi(r6, r2); // 22 mulhi
|
||||
r6 = r6 ^ ds[r3 & mask]; // 23 load
|
||||
r5 = r5 * r0; // 24 mul
|
||||
r5 = rotl_imm(r5, 19u); // 25 rotl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r6, 2); // 26 shfl
|
||||
r0 = r0 ^ r5; // 27 xor
|
||||
r0 = r0 ^ r4; // 28 xor
|
||||
r3 = r3 - r0; // 29 sub
|
||||
r5 = r5 * r1; // 30 mul
|
||||
r7 = r7 ^ ds[r3 & mask]; // 31 load
|
||||
r1 = r1 ^ ds[r2 & mask]; // 32 load
|
||||
r5 = r5 ^ r6; // 33 xor
|
||||
r5 = r5 ^ ds[r0 & mask]; // 34 load
|
||||
r0 = __umulhi(r0, r5); // 35 mulhi
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // 36 shfl
|
||||
r7 = r7 ^ ds[r6 & mask]; // 37 load
|
||||
r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 4); // 39 shfl
|
||||
r2 = r2 ^ r5; // 40 xor
|
||||
r3 = r6 * r3 + r3; // 41 mad
|
||||
r6 = r6 - r7; // 42 sub
|
||||
r7 = r7 ^ r0; // 43 xor
|
||||
r1 = r1 ^ ds[r7 & mask]; // 44 load
|
||||
r2 = r2 * r3; // 45 mul
|
||||
r1 = __umulhi(r1, r5); // 46 mulhi
|
||||
r4 = r4 - r3; // 47 sub
|
||||
r2 = rotr_var(r2, r6); // 48 rotr
|
||||
r3 = r3 ^ ds[r4 & mask]; // 49 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add
|
||||
r0 = r0 * r2; // 51 mul
|
||||
r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add
|
||||
r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add
|
||||
r7 = rotl_imm(r7, 14u); // 54 rotl
|
||||
r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add
|
||||
r6 = r6 ^ ds[r7 & mask]; // 56 load
|
||||
r1 = rotr_var(r1, r5); // 57 rotr
|
||||
r5 = r5 ^ ds[r3 & mask]; // 58 load
|
||||
r6 = r6 ^ ds[r1 & mask]; // 59 load
|
||||
r3 = r5 * r0 + r3; // 60 mad
|
||||
r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add
|
||||
r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add
|
||||
r5 = rotl_imm(r5, 19u); // 63 rotl
|
||||
// latency-shadow block (Counter ASIC 3.0 item 8): 256 ALU instructions x 27 passes after instruction 63, no load
|
||||
for (uint32_t sh = 0u; sh < 27u; ++sh) {
|
||||
r5 = r5 + r2 + ((((sel >> 18u) & 1u) != 0u) ? 0x8bc12da9u : 0x92199f99u); // s0 add
|
||||
r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x63079e5au : 0x8d72d3adu); // s1 add
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 2); // s2 shfl
|
||||
r4 = r4 - r2; // s3 sub
|
||||
r7 = r7 + r0 + ((((sel >> 31u) & 1u) != 0u) ? 0x5d4c7a60u : 0xb21b4babu); // s4 add
|
||||
r0 = rotl_imm(r0, 11u); // s5 rotl
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // s6 add
|
||||
r7 = r7 ^ r1; // s7 xor
|
||||
r1 = r6 * r5 + r1; // s8 mad
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r1, 4); // s9 shfl
|
||||
r1 = r2 * r2 + r1; // s10 mad
|
||||
r5 = r0 * r3 + r5; // s11 mad
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r6, 4); // s12 shfl
|
||||
r1 = r1 + r5 + ((((sel >> 25u) & 1u) != 0u) ? 0xbc48c63eu : 0xbdf8f9a5u); // s13 add
|
||||
r1 = rotl_imm(r1, 29u); // s14 rotl
|
||||
r1 = r1 - r4; // s15 sub
|
||||
r7 = r7 | r1; // s16 or
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // s17 shfl
|
||||
r7 = r7 ^ r4; // s18 xor
|
||||
r6 = r6 * r1; // s19 mul
|
||||
r5 = r6 * r0 + r5; // s20 mad
|
||||
r3 = r3 - r1; // s21 sub
|
||||
r6 = r6 * r0; // s22 mul
|
||||
r2 = r2 + r0 + ((((sel >> 1u) & 1u) != 0u) ? 0x96e8f127u : 0x45c37cecu); // s23 add
|
||||
r6 = r6 - r4; // s24 sub
|
||||
r7 = r3 * r4 + r7; // s25 mad
|
||||
r3 = rotl_imm(r3, 9u); // s26 rotl
|
||||
r2 = r2 - r1; // s27 sub
|
||||
r6 = __umulhi(r6, r0); // s28 mulhi
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // s29 shfl
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r6, 1); // s30 shfl
|
||||
r1 = r1 + r6 + ((((sel >> 1u) & 1u) != 0u) ? 0xb1cdb2abu : 0x37985632u); // s31 add
|
||||
r0 = rotr_var(r0, r1); // s32 rotr
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // s33 shfl
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // s34 shfl
|
||||
r7 = r7 * r0; // s35 mul
|
||||
r3 = r3 + r1 + ((((sel >> 0u) & 1u) != 0u) ? 0x856e0180u : 0x804e777eu); // s36 add
|
||||
r1 = r1 ^ r6; // s37 xor
|
||||
r2 = r2 * r7; // s38 mul
|
||||
r6 = r6 + r5 + ((((sel >> 2u) & 1u) != 0u) ? 0xe9bd0cc4u : 0x0b06c8a7u); // s39 add
|
||||
r5 = r5 * r7; // s40 mul
|
||||
r2 = __umulhi(r2, r3); // s41 mulhi
|
||||
r2 = r2 ^ r0; // s42 xor
|
||||
r0 = rotl_imm(r0, 4u); // s43 rotl
|
||||
r4 = __umulhi(r4, r3); // s44 mulhi
|
||||
r6 = r6 + r7 + ((((sel >> 12u) & 1u) != 0u) ? 0xcdcb63f6u : 0xee822e17u); // s45 add
|
||||
r3 = r2 * r0 + r3; // s46 mad
|
||||
r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r3, 8); // s47 shfl
|
||||
r6 = __umulhi(r6, r5); // s48 mulhi
|
||||
r0 = r0 - r2; // s49 sub
|
||||
r3 = r3 - r5; // s50 sub
|
||||
r1 = rotr_var(r1, r4); // s51 rotr
|
||||
r6 = r7 * r7 + r6; // s52 mad
|
||||
r5 = r5 ^ r3; // s53 xor
|
||||
r1 = r1 - r0; // s54 sub
|
||||
r5 = r5 - r6; // s55 sub
|
||||
r3 = r3 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0xd4758987u : 0x3dfad1b6u); // s56 add
|
||||
r4 = r4 + r1 + ((((sel >> 0u) & 1u) != 0u) ? 0x0602d6beu : 0xfca75bc2u); // s57 add
|
||||
r5 = __umulhi(r5, r6); // s58 mulhi
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r1, 2); // s59 shfl
|
||||
r2 = rotl_imm(r2, 9u); // s60 rotl
|
||||
r4 = r4 | r6; // s61 or
|
||||
r6 = rotr_var(r6, r4); // s62 rotr
|
||||
r2 = r2 * r4; // s63 mul
|
||||
r0 = r0 ^ r5; // s64 xor
|
||||
r2 = r2 ^ r7; // s65 xor
|
||||
r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x8596687au : 0xd71c02ffu); // s66 add
|
||||
r1 = rotl_imm(r1, 21u); // s67 rotl
|
||||
r3 = r3 * r2; // s68 mul
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r5, 2); // s69 shfl
|
||||
r3 = r3 * r2; // s70 mul
|
||||
r0 = r2 * r5 + r0; // s71 mad
|
||||
r6 = r6 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0x08ac5733u : 0x3c6fe15du); // s72 add
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r2, 8); // s73 shfl
|
||||
r3 = r3 * r6; // s74 mul
|
||||
r6 = r6 ^ r7; // s75 xor
|
||||
r3 = r3 | r0; // s76 or
|
||||
r2 = r2 + r4 + ((((sel >> 1u) & 1u) != 0u) ? 0xc100b495u : 0x295d5faeu); // s77 add
|
||||
r3 = __umulhi(r3, r7); // s78 mulhi
|
||||
r4 = r4 | r1; // s79 or
|
||||
r4 = rotr_var(r4, r3); // s80 rotr
|
||||
r4 = r4 + r3 + ((((sel >> 15u) & 1u) != 0u) ? 0x5cc59530u : 0x274a9221u); // s81 add
|
||||
r7 = r7 + r3 + ((((sel >> 5u) & 1u) != 0u) ? 0x141479ecu : 0xc8651f8eu); // s82 add
|
||||
r3 = rotr_var(r3, r6); // s83 rotr
|
||||
r2 = r4 * r7 + r2; // s84 mad
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r5, 16); // s85 shfl
|
||||
r3 = r3 ^ r2; // s86 xor
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r7, 2); // s87 shfl
|
||||
r0 = r0 ^ r4; // s88 xor
|
||||
r3 = r3 + r2 + ((((sel >> 18u) & 1u) != 0u) ? 0x883c0c92u : 0x53f915c2u); // s89 add
|
||||
r7 = rotr_var(r7, r5); // s90 rotr
|
||||
r3 = r3 + r1 + ((((sel >> 31u) & 1u) != 0u) ? 0x3cc5bd20u : 0xe2be00a5u); // s91 add
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r2, 1); // s92 shfl
|
||||
r6 = rotr_var(r6, r2); // s93 rotr
|
||||
r7 = rotl_imm(r7, 14u); // s94 rotl
|
||||
r4 = r5 * r5 + r4; // s95 mad
|
||||
r2 = r4 * r5 + r2; // s96 mad
|
||||
r1 = r1 ^ r0; // s97 xor
|
||||
r5 = r5 * r4; // s98 mul
|
||||
r2 = r2 - r0; // s99 sub
|
||||
r7 = rotl_imm(r7, 30u); // s100 rotl
|
||||
r5 = r5 + r6 + ((((sel >> 17u) & 1u) != 0u) ? 0xbfd646cbu : 0x423fd9c9u); // s101 add
|
||||
r7 = r7 + r6 + ((((sel >> 11u) & 1u) != 0u) ? 0x19b74a43u : 0x8d3c011du); // s102 add
|
||||
r5 = rotl_imm(r5, 6u); // s103 rotl
|
||||
r0 = r0 * r4; // s104 mul
|
||||
r0 = r0 | r5; // s105 or
|
||||
r0 = r0 + r1 + ((((sel >> 15u) & 1u) != 0u) ? 0x7dcb7e18u : 0x8ce14721u); // s106 add
|
||||
r0 = rotl_imm(r0, 15u); // s107 rotl
|
||||
r4 = r4 - r2; // s108 sub
|
||||
r2 = __umulhi(r2, r0); // s109 mulhi
|
||||
r1 = __umulhi(r1, r0); // s110 mulhi
|
||||
r0 = r0 + r2 + ((((sel >> 28u) & 1u) != 0u) ? 0x3c179ad8u : 0x2d9fc6b9u); // s111 add
|
||||
r2 = __umulhi(r2, r0); // s112 mulhi
|
||||
r6 = r6 - r3; // s113 sub
|
||||
r7 = r6 * r3 + r7; // s114 mad
|
||||
r5 = rotr_var(r5, r1); // s115 rotr
|
||||
r6 = rotr_var(r6, r4); // s116 rotr
|
||||
r2 = r2 + r1 + ((((sel >> 22u) & 1u) != 0u) ? 0x47359729u : 0x502138e2u); // s117 add
|
||||
r3 = r2 * r0 + r3; // s118 mad
|
||||
r1 = r1 * r3; // s119 mul
|
||||
r2 = r0 * r4 + r2; // s120 mad
|
||||
r0 = r0 * r3; // s121 mul
|
||||
r2 = __umulhi(r2, r0); // s122 mulhi
|
||||
r5 = r5 * r6; // s123 mul
|
||||
r4 = r2 * r4 + r4; // s124 mad
|
||||
r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r2, 2); // s125 shfl
|
||||
r2 = r2 ^ r0; // s126 xor
|
||||
r1 = rotl_imm(r1, 7u); // s127 rotl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // s128 shfl
|
||||
r6 = rotl_imm(r6, 17u); // s129 rotl
|
||||
r2 = r2 + r5 + ((((sel >> 15u) & 1u) != 0u) ? 0x6c57e4e7u : 0x33dff776u); // s130 add
|
||||
r0 = r0 | r3; // s131 or
|
||||
r5 = rotr_var(r5, r0); // s132 rotr
|
||||
r2 = r2 + r3 + ((((sel >> 30u) & 1u) != 0u) ? 0xe8212c0cu : 0x5db25b34u); // s133 add
|
||||
r4 = r4 ^ r3; // s134 xor
|
||||
r6 = r6 ^ r1; // s135 xor
|
||||
r1 = r1 - r7; // s136 sub
|
||||
r2 = r6 * r2 + r2; // s137 mad
|
||||
r0 = __umulhi(r0, r5); // s138 mulhi
|
||||
r2 = r2 + r7 + ((((sel >> 10u) & 1u) != 0u) ? 0xd44ff710u : 0x218d4090u); // s139 add
|
||||
r1 = rotr_var(r1, r4); // s140 rotr
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r6, 1); // s141 shfl
|
||||
r7 = r6 * r2 + r7; // s142 mad
|
||||
r2 = r2 * r3; // s143 mul
|
||||
r7 = r7 + r4 + ((((sel >> 29u) & 1u) != 0u) ? 0xb98942fau : 0xf4b1a8deu); // s144 add
|
||||
r7 = rotl_imm(r7, 15u); // s145 rotl
|
||||
r7 = r7 ^ r5; // s146 xor
|
||||
r4 = r7 * r4 + r4; // s147 mad
|
||||
r6 = rotr_var(r6, r5); // s148 rotr
|
||||
r1 = r2 * r4 + r1; // s149 mad
|
||||
r1 = r1 + r7 + ((((sel >> 17u) & 1u) != 0u) ? 0x0d48ba42u : 0x2bef10f2u); // s150 add
|
||||
r5 = r5 ^ r4; // s151 xor
|
||||
r7 = r7 + r0 + ((((sel >> 30u) & 1u) != 0u) ? 0xc98dea9cu : 0xdc5cc080u); // s152 add
|
||||
r4 = r4 * r6; // s153 mul
|
||||
r0 = r0 * r2; // s154 mul
|
||||
r6 = r6 ^ r5; // s155 xor
|
||||
r4 = rotr_var(r4, r2); // s156 rotr
|
||||
r1 = rotl_imm(r1, 11u); // s157 rotl
|
||||
r5 = r5 + r0 + ((((sel >> 11u) & 1u) != 0u) ? 0x6c752dcbu : 0x18197438u); // s158 add
|
||||
r4 = r1 * r5 + r4; // s159 mad
|
||||
r4 = rotl_imm(r4, 26u); // s160 rotl
|
||||
r3 = r0 * r7 + r3; // s161 mad
|
||||
r3 = rotr_var(r3, r1); // s162 rotr
|
||||
r4 = r4 + r0 + ((((sel >> 3u) & 1u) != 0u) ? 0x8e481727u : 0xf1c46574u); // s163 add
|
||||
r0 = __umulhi(r0, r4); // s164 mulhi
|
||||
r2 = r2 + r6 + ((((sel >> 20u) & 1u) != 0u) ? 0x550ab406u : 0xac578137u); // s165 add
|
||||
r0 = rotl_imm(r0, 13u); // s166 rotl
|
||||
r3 = r3 + r1 + ((((sel >> 14u) & 1u) != 0u) ? 0xaebb5966u : 0xa33e6706u); // s167 add
|
||||
r3 = r3 | r5; // s168 or
|
||||
r6 = rotr_var(r6, r2); // s169 rotr
|
||||
r4 = r4 ^ r6; // s170 xor
|
||||
r6 = r6 - r1; // s171 sub
|
||||
r7 = rotl_imm(r7, 22u); // s172 rotl
|
||||
r5 = rotl_imm(r5, 15u); // s173 rotl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r0, 8); // s174 shfl
|
||||
r0 = r0 ^ r5; // s175 xor
|
||||
r7 = rotl_imm(r7, 6u); // s176 rotl
|
||||
r7 = r7 - r0; // s177 sub
|
||||
r3 = rotl_imm(r3, 30u); // s178 rotl
|
||||
r7 = r6 * r1 + r7; // s179 mad
|
||||
r6 = rotl_imm(r6, 9u); // s180 rotl
|
||||
r2 = r2 ^ r4; // s181 xor
|
||||
r2 = r2 ^ r7; // s182 xor
|
||||
r7 = r7 ^ r2; // s183 xor
|
||||
r1 = r1 + r2 + ((((sel >> 21u) & 1u) != 0u) ? 0x5bb7550fu : 0xd94d55acu); // s184 add
|
||||
r3 = r3 | r5; // s185 or
|
||||
r6 = __umulhi(r6, r3); // s186 mulhi
|
||||
r4 = r4 | r0; // s187 or
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r1, 2); // s188 shfl
|
||||
r6 = r6 + r5 + ((((sel >> 6u) & 1u) != 0u) ? 0x2a354e2du : 0x89e747feu); // s189 add
|
||||
r1 = r1 ^ r4; // s190 xor
|
||||
r7 = __umulhi(r7, r4); // s191 mulhi
|
||||
r2 = rotl_imm(r2, 26u); // s192 rotl
|
||||
r5 = rotl_imm(r5, 8u); // s193 rotl
|
||||
r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r5, 16); // s194 shfl
|
||||
r4 = r4 ^ r5; // s195 xor
|
||||
r1 = r1 * r3; // s196 mul
|
||||
r5 = r5 + r2 + ((((sel >> 7u) & 1u) != 0u) ? 0x10cfdc71u : 0xea03e8e7u); // s197 add
|
||||
r7 = r7 + r5 + ((((sel >> 15u) & 1u) != 0u) ? 0x66e148d3u : 0xa7ee0102u); // s198 add
|
||||
r5 = r5 - r2; // s199 sub
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r1, 2); // s200 shfl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r1, 8); // s201 shfl
|
||||
r4 = rotr_var(r4, r5); // s202 rotr
|
||||
r7 = r7 ^ r5; // s203 xor
|
||||
r7 = r7 ^ r0; // s204 xor
|
||||
r6 = r6 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0x8558b619u : 0x8379a4deu); // s205 add
|
||||
r4 = rotl_imm(r4, 13u); // s206 rotl
|
||||
r1 = __umulhi(r1, r3); // s207 mulhi
|
||||
r1 = r4 * r4 + r1; // s208 mad
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r0, 4); // s209 shfl
|
||||
r7 = r7 + r0 + ((((sel >> 11u) & 1u) != 0u) ? 0xf4049c4cu : 0xaa8cb14eu); // s210 add
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r1, 16); // s211 shfl
|
||||
r1 = r1 ^ r0; // s212 xor
|
||||
r7 = rotl_imm(r7, 3u); // s213 rotl
|
||||
r4 = rotr_var(r4, r7); // s214 rotr
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r2, 16); // s215 shfl
|
||||
r5 = r5 | r1; // s216 or
|
||||
r1 = r1 - r2; // s217 sub
|
||||
r6 = r6 - r5; // s218 sub
|
||||
r6 = rotl_imm(r6, 4u); // s219 rotl
|
||||
r2 = __umulhi(r2, r0); // s220 mulhi
|
||||
r2 = r2 | r0; // s221 or
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r2, 8); // s222 shfl
|
||||
r5 = __umulhi(r5, r6); // s223 mulhi
|
||||
r0 = r0 - r6; // s224 sub
|
||||
r7 = rotl_imm(r7, 23u); // s225 rotl
|
||||
r4 = r4 | r2; // s226 or
|
||||
r2 = r2 * r4; // s227 mul
|
||||
r3 = rotl_imm(r3, 12u); // s228 rotl
|
||||
r0 = rotr_var(r0, r4); // s229 rotr
|
||||
r0 = r0 + r6 + ((((sel >> 16u) & 1u) != 0u) ? 0x2a7fecb2u : 0x1d176220u); // s230 add
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // s231 shfl
|
||||
r2 = r2 * r1; // s232 mul
|
||||
r7 = r0 * r1 + r7; // s233 mad
|
||||
r5 = rotl_imm(r5, 22u); // s234 rotl
|
||||
r4 = rotr_var(r4, r6); // s235 rotr
|
||||
r0 = r5 * r1 + r0; // s236 mad
|
||||
r6 = r6 ^ r4; // s237 xor
|
||||
r4 = r4 ^ r6; // s238 xor
|
||||
r6 = rotl_imm(r6, 18u); // s239 rotl
|
||||
r4 = r4 + r7 + ((((sel >> 25u) & 1u) != 0u) ? 0xadce39f3u : 0x17dafb4du); // s240 add
|
||||
r0 = r0 - r7; // s241 sub
|
||||
r1 = rotr_var(r1, r6); // s242 rotr
|
||||
r3 = r3 + r0 + ((((sel >> 29u) & 1u) != 0u) ? 0x0e7033b6u : 0xf2f77d26u); // s243 add
|
||||
r2 = r7 * r4 + r2; // s244 mad
|
||||
r4 = r0 * r6 + r4; // s245 mad
|
||||
r5 = r5 * r7; // s246 mul
|
||||
r3 = r3 + r5 + ((((sel >> 31u) & 1u) != 0u) ? 0x7b2ff6b7u : 0xfed2da4eu); // s247 add
|
||||
r0 = r0 + r7 + ((((sel >> 0u) & 1u) != 0u) ? 0xb5fad7b1u : 0x03b2891cu); // s248 add
|
||||
r5 = __umulhi(r5, r4); // s249 mulhi
|
||||
r5 = r5 + r2 + ((((sel >> 11u) & 1u) != 0u) ? 0xa7e71c2fu : 0x042cd6e3u); // s250 add
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r1, 1); // s251 shfl
|
||||
r6 = r6 ^ r1; // s252 xor
|
||||
r2 = r2 * r5; // s253 mul
|
||||
r0 = r0 + r6 + ((((sel >> 16u) & 1u) != 0u) ? 0xb83d78deu : 0x784a302bu); // s254 add
|
||||
r2 = r2 - r4; // s255 sub
|
||||
}
|
||||
}
|
||||
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
|
||||
}
|
||||
|
||||
cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps) {
|
||||
if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 32u * blockWarps;
|
||||
if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;
|
||||
igneum_hash_bound<<<nonces / block, block>>>(ds, out, baseNonce, mask, iw);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
|
||||
cudaFuncAttributes attr;
|
||||
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound);
|
||||
if (e != cudaSuccess) return e;
|
||||
*numRegs = attr.numRegs;
|
||||
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0);
|
||||
}
|
||||
BIN
proto-cuda/packs-ca3-v5/v5-genesis/leaves.bin
Normal file
BIN
proto-cuda/packs-ca3-v5/v5-genesis/leaves.bin
Normal file
Binary file not shown.
112
proto-cuda/packs-ca3-v5/v5-genesis/memhard.h
Normal file
112
proto-cuda/packs-ca3-v5/v5-genesis/memhard.h
Normal file
|
|
@ -0,0 +1,112 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against.
|
||||
// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference).
|
||||
// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C.
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
#if defined(__CUDACC__)
|
||||
#define IGNEUM_HD __host__ __device__ __forceinline__
|
||||
#elif defined(_MSC_VER) && !defined(__cplusplus)
|
||||
#define IGNEUM_HD static __inline
|
||||
#else
|
||||
#define IGNEUM_HD static inline
|
||||
#endif
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of 8 x seed-parameterised mixer + one 64-byte cache read, then 8 x final mixer (class v3, mixer multiplier 8,
|
||||
// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) {
|
||||
for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint32_t r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) {
|
||||
uint32_t prev[16]; uint32_t x[16]; uint32_t y[16];
|
||||
for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of 8 x mixer + cache line s[0] & mask; 8 x final mixer.
|
||||
IGNEUM_HD void mh_item(const uint32_t* cache, const uint32_t* leaf, uint32_t t, uint32_t* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= leaf[i];
|
||||
for (uint32_t r = 0u; r < 8u; ++r) {
|
||||
for (uint32_t j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 8u + j + 1u));
|
||||
const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
for (uint32_t j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (64u + j + 1u));
|
||||
}
|
||||
// Class v5 (docs/design/class-v5-stored-state.md): leaf(t) = leaves[t mod nLeaves], 16 words per leaf (leaves.bin).
|
||||
IGNEUM_HD const uint32_t* mh_leaf(const uint32_t* leaves, uint32_t nLeaves, uint32_t t) { return leaves + ((t % nLeaves) * 16u); }
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
IGNEUM_HD uint32_t mh_word(const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t w) { uint32_t s[16]; mh_item(cache, mh_leaf(leaves, nLeaves, w >> 4u), w >> 4u, s); return s[w & 15u]; }
|
||||
112
proto-cuda/packs-ca3-v5/v5-genesis/memhard.metal
Normal file
112
proto-cuda/packs-ca3-v5/v5-genesis/memhard.metal
Normal file
|
|
@ -0,0 +1,112 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of 8 x seed-parameterised mixer + one 64-byte cache read, then 8 x final mixer (class v3, mixer multiplier 8,
|
||||
// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
inline void mh_chacha_block(const thread uint* x, thread uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
inline void mh_cache_segment(device uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
inline void mh_mixer(thread uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of 8 x mixer + cache line s[0] & mask; 8 x final mixer.
|
||||
inline void mh_item(device const uint* cache, device const uint* leaf, uint t, thread uint* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= leaf[i];
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
for (uint j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 8u + j + 1u));
|
||||
device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
for (uint j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (64u + j + 1u));
|
||||
}
|
||||
// Class v5 (docs/design/class-v5-stored-state.md): leaf(t) = leaves[t mod nLeaves], 16 words per leaf (leaves.bin).
|
||||
inline device const uint* mh_leaf(device const uint* leaves, uint nLeaves, uint t) { return leaves + ((t % nLeaves) * 16u); }
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
inline uint mh_word(device const uint* cache, device const uint* leaves, uint nLeaves, uint w) { uint s[16]; mh_item(cache, mh_leaf(leaves, nLeaves, w >> 4u), w >> 4u, s); return s[w & 15u]; }
|
||||
|
||||
// One thread per segment (2^16 threads).
|
||||
kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) {
|
||||
mh_cache_segment(cache, gid);
|
||||
}
|
||||
// One thread per 64-byte item (dataset words / 16 threads).
|
||||
// Class v5: the window's leaves (leaves.bin, IGNEUM_STATE_LEAVES x 16 words) in buffer 2, their count in buffer 3.
|
||||
kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],
|
||||
device const uint* leaves [[buffer(2)]], constant uint& nLeaves [[buffer(3)]],
|
||||
uint gid [[thread_position_in_grid]]) {
|
||||
uint s[16];
|
||||
mh_item(cache, mh_leaf(leaves, nLeaves, gid), gid, s);
|
||||
device uint* d = dataset + gid * 16u;
|
||||
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
84
proto-cuda/packs-ca3-v5/v5-genesis/program.h
Normal file
84
proto-cuda/packs-ca3-v5/v5-genesis/program.h
Normal file
|
|
@ -0,0 +1,84 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.
|
||||
// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros.
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
#ifndef IGNEUM_NO_CUDA
|
||||
#include <cuda_runtime.h>
|
||||
#endif
|
||||
|
||||
#define IGNEUM_SEED_STRING "igneum-genesis"
|
||||
#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973"
|
||||
#define IGNEUM_GENERATOR 5
|
||||
#define IGNEUM_PROGRAM_ATTEMPT 0
|
||||
#define IGNEUM_PROGRAM_ID 0x7c54302b487340a1ull
|
||||
#define IGNEUM_DAY_STRING "2026-10-03"
|
||||
#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033"
|
||||
#define IGNEUM_DAY0 0x3067619fu
|
||||
#define IGNEUM_DAY1 0x3c269176u
|
||||
#define IGNEUM_DATASET_LOG2 28
|
||||
#define IGNEUM_MASK 0x0fffffffu
|
||||
#define IGNEUM_LANES 32
|
||||
#define IGNEUM_ITERATIONS 8
|
||||
#define IGNEUM_INSTR_COUNT 64
|
||||
#define IGNEUM_LOADS_PER_HASH 128
|
||||
#define IGNEUM_WIDE_LOADS_PER_HASH 0
|
||||
#define IGNEUM_OP_MIX "load=16 add=8 shfl=8 xor=6 mad=5 mul=5 mulhi=5 sub=4 rotl=3 rotr=3 or=1"
|
||||
// Program class v5 (proof of stored state and of following, docs/design/class-v5-stored-state.md): generator version 5,
|
||||
// class v4 over a dataset whose every item is keyed by the window's execution state (IGNEUM_STATE_* below, leaves.bin);
|
||||
// a worker that runs another class refuses this pack, and a job line names the class it wants (class=v5 era=<hex>).
|
||||
#define IGNEUM_PROGRAM_CLASS "v5"
|
||||
// Class v3 construction (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): version 2 loads; the dataset item
|
||||
// derivation applies the mixer IGNEUM_MIXER_MULT times per round (memhard.h), and the cache follows the growth rule.
|
||||
#define IGNEUM_LOAD_CLASS "mx8+sh256x27+state"
|
||||
#define IGNEUM_CLASS_MIXER_MULT 8
|
||||
#define IGNEUM_CACHE_GROWTH 1 // 1: cache words = 2^(26 + doublings(day)), doublings = floor(log2(1 + day / 1460))
|
||||
#define IGNEUM_LOAD_SLOTS 16
|
||||
#define IGNEUM_LOAD_MIX { 100, 0, 0 }
|
||||
#define IGNEUM_LOAD_WIDTH_COUNTS { 16, 0, 0 } // loads of 4, 16, 64 bytes per program
|
||||
#define IGNEUM_BYTES_PER_HASH 512
|
||||
#define IGNEUM_FOLD_ROT 11
|
||||
#define IGNEUM_FOLD_MUL 0x9e3779b1u
|
||||
// Class v5 state (docs/design/class-v5-stored-state.md): the window's reference chain block and the state root after it;
|
||||
// leaves.bin holds IGNEUM_STATE_LEAVES leaves of 16 little-endian words, leaf(t) = leaves[t mod IGNEUM_STATE_LEAVES].
|
||||
#define IGNEUM_STATE_BLOCK_HEX "af89be5ddbadb6f6b4aee28ac8f249713be5d4c12621e3cea7f83ceada3c66b3"
|
||||
#define IGNEUM_STATE_BLOCK_NUMBER 159357
|
||||
#define IGNEUM_STATE_ROOT_HEX "1c583d352bb9c75a06dadb8d82d42836ebe8afa82d0b413be87bf921741f1526"
|
||||
#define IGNEUM_STATE_LEAVES 93
|
||||
#define IGNEUM_STATE_RECORDS 93
|
||||
#define IGNEUM_STATE_SAMPLED 0
|
||||
#define IGNEUM_STATE_LEAVES_FNV64 0x850ad094a937a5c5ull
|
||||
#define IGNEUM_STATE_LEAVES_FILE "leaves.bin"
|
||||
// Latency-shadow block (Counter ASIC 3.0 item 8, docs/analysis/latency-shadow-2026-10-06.md): NOT the lottery hash. A block of
|
||||
// IGNEUM_SHADOW_INSTRS ALU instructions (the ten non-load families) runs IGNEUM_SHADOW_REPS times at the end of every
|
||||
// iteration; the 16 loads, the acceptance rule and the base program are the class's without the shadow.
|
||||
#define IGNEUM_SHADOW_INSTRS 256
|
||||
#define IGNEUM_SHADOW_REPS 27
|
||||
#define IGNEUM_SHADOW_INSTRS_PER_HASH 55296
|
||||
#define IGNEUM_SHADOW_OP_MIX "add=47 rotl=30 xor=30 shfl=29 mad=27 mul=22 sub=21 rotr=20 mulhi=18 or=12"
|
||||
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
|
||||
#define IGNEUM_DATASET_MODE 1
|
||||
|
||||
#define IGNEUM_SEEDW_INIT { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }
|
||||
#define IGNEUM_KEY_INIT { 0x3067619fu, 0x3c269176u, 0x84a03b03u, 0xf8c63294u, 0xff977c5bu, 0xe60def3eu, 0x63630141u, 0xb8fbcb58u }
|
||||
#define IGNEUM_CACHE_LOG2_WORDS 26
|
||||
#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6
|
||||
#define IGNEUM_CACHE_SEGMENTS 65536u
|
||||
#define IGNEUM_ITEM_ROUNDS 8
|
||||
#define IGNEUM_MIXER_MULT 8 // mixer applications per round and after the last read (class v3, docs/plans/mixer-x4.md)
|
||||
#define IGNEUM_MIX_ROT_INIT { 20u, 20u, 19u, 4u, 26u, 3u, 3u, 27u }
|
||||
#define IGNEUM_MIX_MUL_INIT { 0x42146205u, 0x52cbe0fbu, 0x7ecf4a03u, 0x6728907fu, 0xd81d9751u, 0x132952c3u, 0xf60de277u, 0x05358035u, 0xbaf6499du, 0xe4db9667u, 0x3e98f45du, 0xd0004eddu, 0x2691630du, 0x9beb3bcfu, 0xab310379u, 0x99cfb423u }
|
||||
#define IGNEUM_MIX_RC_INIT { 0xbab68293u, 0xcc162340u, 0x6ce151ccu, 0xe62b8997u, 0xc9c80297u, 0xf74a1654u, 0x3d704af5u, 0x3cf522b7u, 0x2b9cac04u, 0xa880ac10u, 0x13e5dd1du, 0x6fc3e233u, 0x2d83eeacu, 0x9006e8bfu, 0x2c4b5362u, 0x31b49ee2u }
|
||||
|
||||
#ifndef IGNEUM_NO_CUDA
|
||||
// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().
|
||||
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);
|
||||
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems);
|
||||
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
uint32_t nonces, uint32_t blockWarps);
|
||||
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
|
||||
#endif
|
||||
401
proto-cuda/packs-ca3-v5/v5-genesis/program.json
Normal file
401
proto-cuda/packs-ca3-v5/v5-genesis/program.json
Normal file
|
|
@ -0,0 +1,401 @@
|
|||
{
|
||||
"format": "igneum-program-pack-3",
|
||||
"generator": 5,
|
||||
"attempt": 0,
|
||||
"program_id": "0x7c54302b487340a1",
|
||||
"program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32",
|
||||
"dataset_mode": "memory-hard",
|
||||
"seed": "igneum-genesis",
|
||||
"seed_bytes": "69676e65756d2d67656e65736973",
|
||||
"seed_words": ["0x67a9a7be", "0x1a155b25", "0xfddfb732", "0x4b5af2e8", "0xc55caf33", "0xa27c13b7", "0x06628a48", "0x03852469"],
|
||||
"seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32",
|
||||
"generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried",
|
||||
"lanes": 32,
|
||||
"registers": 8,
|
||||
"iterations": 8,
|
||||
"instruction_count": 64,
|
||||
"loads_per_hash": 128,
|
||||
"program_class": "v5",
|
||||
"state": {
|
||||
"block": "af89be5ddbadb6f6b4aee28ac8f249713be5d4c12621e3cea7f83ceada3c66b3",
|
||||
"block_number": 159357,
|
||||
"root": "1c583d352bb9c75a06dadb8d82d42836ebe8afa82d0b413be87bf921741f1526",
|
||||
"leaves": 93,
|
||||
"records": 93,
|
||||
"sampled": false,
|
||||
"leaves_fnv1a64": "0x850ad094a937a5c5",
|
||||
"leaf_derivation": "leaves[i] = Blake2b-512('igneum-sd1/' || root || i_le32 || record_i) as 16 little-endian words; item t XORs leaves[t mod leaves] into its 16 initial words before the first mixer",
|
||||
"file": "leaves.bin"
|
||||
},
|
||||
"load_class": "mx8+sh256x27+state",
|
||||
"mixer_mult": 8,
|
||||
"cache_growth": true,
|
||||
"mixer": "class v3 (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): every mixer application of the item derivation is 8 applications with round keys (r * 8 + j + 1) * 0x9E3779B9, the 8 dependent cache reads per item unchanged; cache growth rule option C: cache words = 2^(26 + doublings(day)), dataset words = 2^(genesis_log2 + doublings(day)), doublings(day) = floor(log2(1 + day / 1460)) for day = days since genesis",
|
||||
"load_slots": 16,
|
||||
"load_mix_percent_4_16_64": [100, 0, 0],
|
||||
"load_width_counts_4_16_64": [16, 0, 0],
|
||||
"bytes_per_hash": 512,
|
||||
"wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots",
|
||||
"op_mix": {"load": 16, "add": 8, "shfl": 8, "xor": 6, "mad": 5, "mul": 5, "mulhi": 5, "sub": 4, "rotl": 3, "rotr": 3, "or": 1},
|
||||
"register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]",
|
||||
"splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16",
|
||||
"iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order",
|
||||
"output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo",
|
||||
"op_semantics": {
|
||||
"add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)",
|
||||
"sub": "dst = dst - src",
|
||||
"mul": "dst = dst * src (low 32)",
|
||||
"mulhi": "dst = high 32 bits of dst * src",
|
||||
"xor": "dst = dst ^ src",
|
||||
"or": "dst = dst | src",
|
||||
"rotl": "dst = rotl(dst, rot), rot in 1..31",
|
||||
"rotr": "dst = rotr(dst, src & 31)",
|
||||
"mad": "dst = src * src2 + dst",
|
||||
"shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp",
|
||||
"load": "dst = dst ^ dataset[src & dataset.mask]",
|
||||
"wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)"
|
||||
},
|
||||
"dataset": {
|
||||
"log2_words": 28,
|
||||
"bytes": 1073741824,
|
||||
"mask": "0x0fffffff",
|
||||
"day": "2026-10-03",
|
||||
"day_bytes": "6461792f323032362d31302d3033",
|
||||
"day_words_from": "seed_words_from_bytes(day_bytes)",
|
||||
"d0": "0x3067619f",
|
||||
"d1": "0x3c269176",
|
||||
"mode": "memory-hard",
|
||||
"spec": "proto-metal/MEMHARD.md",
|
||||
"key": ["0x3067619f", "0x3c269176", "0x84a03b03", "0xf8c63294", "0xff977c5b", "0xe60def3e", "0x63630141", "0xb8fbcb58"],
|
||||
"key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]",
|
||||
"cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"},
|
||||
"mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [20, 20, 19, 4, 26, 3, 3, 27], "mul": ["0x42146205", "0x52cbe0fb", "0x7ecf4a03", "0x6728907f", "0xd81d9751", "0x132952c3", "0xf60de277", "0x05358035", "0xbaf6499d", "0xe4db9667", "0x3e98f45d", "0xd0004edd", "0x2691630d", "0x9beb3bcf", "0xab310379", "0x99cfb423"], "rc": ["0xbab68293", "0xcc162340", "0x6ce151cc", "0xe62b8997", "0xc9c80297", "0xf74a1654", "0x3d704af5", "0x3cf522b7", "0x2b9cac04", "0xa880ac10", "0x13e5dd1d", "0x6fc3e233", "0x2d83eeac", "0x9006e8bf", "0x2c4b5362", "0x31b49ee2"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"},
|
||||
"mixer_mult": 8,
|
||||
"item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: for j in 0..7: s = M(s, rk = (r * 8 + j + 1) * 0x9E3779B9); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then for j in 0..7: s = M(s, rk = (64 + j + 1) * 0x9E3779B9); item(t) = s",
|
||||
"word": "dataset[w] = item(w >> 4)[w & 15]"
|
||||
},
|
||||
"shadow": {"instrs": 256, "reps": 27, "instrs_per_hash": 55296, "op_mix": {"add": 47, "rotl": 30, "xor": 30, "shfl": 29, "mad": 27, "mul": 22, "sub": 21, "rotr": 20, "mulhi": 18, "or": 12}, "rule": "Counter ASIC 3.0 item 8 (docs/analysis/latency-shadow-2026-10-06.md): after the 64 base instructions the program stream draws instrs more ALU instructions (the non-load table, nine draws each, the source as on an ALU slot); the block runs reps times at the end of every iteration with the iteration's sel; the acceptance rule interprets the base program only", "program_id_suffix": "'shadow/' || instrs_le16 || reps_le16", "instructions": [
|
||||
{"i": 0, "op": "add", "dst": 5, "src": 2, "src2": 1, "imm": "0x92199f99", "imm2": "0x8bc12da9", "rot": 9, "bit": 18, "mask": 4},
|
||||
{"i": 1, "op": "add", "dst": 0, "src": 7, "src2": 1, "imm": "0x8d72d3ad", "imm2": "0x63079e5a", "rot": 20, "bit": 26, "mask": 4},
|
||||
{"i": 2, "op": "shfl", "dst": 6, "src": 3, "src2": 6, "imm": "0x9beaeddf", "imm2": "0x744ecb00", "rot": 10, "bit": 4, "mask": 2},
|
||||
{"i": 3, "op": "sub", "dst": 4, "src": 2, "src2": 0, "imm": "0x2a3ddc67", "imm2": "0x72c80241", "rot": 30, "bit": 1, "mask": 16},
|
||||
{"i": 4, "op": "add", "dst": 7, "src": 0, "src2": 5, "imm": "0xb21b4bab", "imm2": "0x5d4c7a60", "rot": 17, "bit": 31, "mask": 4},
|
||||
{"i": 5, "op": "rotl", "dst": 0, "src": 6, "src2": 3, "imm": "0xe69d7919", "imm2": "0xa048c61e", "rot": 11, "bit": 1, "mask": 8},
|
||||
{"i": 6, "op": "add", "dst": 0, "src": 1, "src2": 3, "imm": "0x2fe0e98b", "imm2": "0xc88e2942", "rot": 5, "bit": 16, "mask": 16},
|
||||
{"i": 7, "op": "xor", "dst": 7, "src": 1, "src2": 0, "imm": "0xc16efe98", "imm2": "0x01a638c8", "rot": 8, "bit": 17, "mask": 1},
|
||||
{"i": 8, "op": "mad", "dst": 1, "src": 6, "src2": 5, "imm": "0x76ec7b8b", "imm2": "0x25663feb", "rot": 29, "bit": 30, "mask": 2},
|
||||
{"i": 9, "op": "shfl", "dst": 6, "src": 1, "src2": 0, "imm": "0x6c8ee3cb", "imm2": "0xea93237e", "rot": 27, "bit": 1, "mask": 4},
|
||||
{"i": 10, "op": "mad", "dst": 1, "src": 2, "src2": 2, "imm": "0x6dc4ea18", "imm2": "0x6efde6f5", "rot": 3, "bit": 20, "mask": 2},
|
||||
{"i": 11, "op": "mad", "dst": 5, "src": 0, "src2": 3, "imm": "0x023613fc", "imm2": "0x18c51939", "rot": 19, "bit": 31, "mask": 1},
|
||||
{"i": 12, "op": "shfl", "dst": 2, "src": 6, "src2": 1, "imm": "0x74aec8d2", "imm2": "0x7f7ad29c", "rot": 7, "bit": 8, "mask": 4},
|
||||
{"i": 13, "op": "add", "dst": 1, "src": 5, "src2": 6, "imm": "0xbdf8f9a5", "imm2": "0xbc48c63e", "rot": 6, "bit": 25, "mask": 2},
|
||||
{"i": 14, "op": "rotl", "dst": 1, "src": 2, "src2": 6, "imm": "0x7ad8ca8b", "imm2": "0x14c712ad", "rot": 29, "bit": 25, "mask": 8},
|
||||
{"i": 15, "op": "sub", "dst": 1, "src": 4, "src2": 5, "imm": "0xd3349f69", "imm2": "0x70aac45a", "rot": 29, "bit": 9, "mask": 16},
|
||||
{"i": 16, "op": "or", "dst": 7, "src": 1, "src2": 2, "imm": "0x08168ed1", "imm2": "0x28964037", "rot": 26, "bit": 0, "mask": 2},
|
||||
{"i": 17, "op": "shfl", "dst": 2, "src": 4, "src2": 6, "imm": "0x98d85f72", "imm2": "0x3dc87042", "rot": 29, "bit": 22, "mask": 8},
|
||||
{"i": 18, "op": "xor", "dst": 7, "src": 4, "src2": 0, "imm": "0x651d4a0e", "imm2": "0x93c198bd", "rot": 26, "bit": 12, "mask": 8},
|
||||
{"i": 19, "op": "mul", "dst": 6, "src": 1, "src2": 6, "imm": "0xd38ce89d", "imm2": "0x3dfad388", "rot": 5, "bit": 28, "mask": 8},
|
||||
{"i": 20, "op": "mad", "dst": 5, "src": 6, "src2": 0, "imm": "0x59d78b36", "imm2": "0xc4db274e", "rot": 25, "bit": 24, "mask": 1},
|
||||
{"i": 21, "op": "sub", "dst": 3, "src": 1, "src2": 7, "imm": "0xf6c6a6c7", "imm2": "0x8536f4e6", "rot": 6, "bit": 12, "mask": 4},
|
||||
{"i": 22, "op": "mul", "dst": 6, "src": 0, "src2": 7, "imm": "0x599445b4", "imm2": "0x632f8c32", "rot": 19, "bit": 4, "mask": 8},
|
||||
{"i": 23, "op": "add", "dst": 2, "src": 0, "src2": 7, "imm": "0x45c37cec", "imm2": "0x96e8f127", "rot": 15, "bit": 1, "mask": 4},
|
||||
{"i": 24, "op": "sub", "dst": 6, "src": 4, "src2": 3, "imm": "0xc1c44491", "imm2": "0x92e3ce57", "rot": 20, "bit": 25, "mask": 4},
|
||||
{"i": 25, "op": "mad", "dst": 7, "src": 3, "src2": 4, "imm": "0x89bfb8d3", "imm2": "0x19b5455e", "rot": 22, "bit": 2, "mask": 16},
|
||||
{"i": 26, "op": "rotl", "dst": 3, "src": 5, "src2": 7, "imm": "0x2f47ce8d", "imm2": "0x8b458ec5", "rot": 9, "bit": 0, "mask": 4},
|
||||
{"i": 27, "op": "sub", "dst": 2, "src": 1, "src2": 2, "imm": "0x9f0dce23", "imm2": "0x3cdca814", "rot": 25, "bit": 1, "mask": 2},
|
||||
{"i": 28, "op": "mulhi", "dst": 6, "src": 0, "src2": 3, "imm": "0x61fc9eb8", "imm2": "0x23202e9d", "rot": 30, "bit": 16, "mask": 16},
|
||||
{"i": 29, "op": "shfl", "dst": 2, "src": 4, "src2": 3, "imm": "0x339dbd65", "imm2": "0xc175f639", "rot": 15, "bit": 22, "mask": 2},
|
||||
{"i": 30, "op": "shfl", "dst": 1, "src": 6, "src2": 1, "imm": "0x5bd14589", "imm2": "0xa68a2bed", "rot": 31, "bit": 31, "mask": 1},
|
||||
{"i": 31, "op": "add", "dst": 1, "src": 6, "src2": 1, "imm": "0x37985632", "imm2": "0xb1cdb2ab", "rot": 29, "bit": 1, "mask": 8},
|
||||
{"i": 32, "op": "rotr", "dst": 0, "src": 1, "src2": 6, "imm": "0x4b2058f4", "imm2": "0xf06d8ac9", "rot": 9, "bit": 15, "mask": 16},
|
||||
{"i": 33, "op": "shfl", "dst": 3, "src": 4, "src2": 4, "imm": "0x2081626c", "imm2": "0x08d1bb87", "rot": 24, "bit": 29, "mask": 2},
|
||||
{"i": 34, "op": "shfl", "dst": 0, "src": 3, "src2": 6, "imm": "0xe4fcfe03", "imm2": "0x78a46b13", "rot": 15, "bit": 29, "mask": 4},
|
||||
{"i": 35, "op": "mul", "dst": 7, "src": 0, "src2": 4, "imm": "0x33148d30", "imm2": "0x5780a3d6", "rot": 6, "bit": 11, "mask": 2},
|
||||
{"i": 36, "op": "add", "dst": 3, "src": 1, "src2": 1, "imm": "0x804e777e", "imm2": "0x856e0180", "rot": 22, "bit": 0, "mask": 16},
|
||||
{"i": 37, "op": "xor", "dst": 1, "src": 6, "src2": 0, "imm": "0xc69aa2f6", "imm2": "0xafebe3e8", "rot": 15, "bit": 9, "mask": 16},
|
||||
{"i": 38, "op": "mul", "dst": 2, "src": 7, "src2": 4, "imm": "0xeb4c4082", "imm2": "0x3f1e0da7", "rot": 25, "bit": 20, "mask": 16},
|
||||
{"i": 39, "op": "add", "dst": 6, "src": 5, "src2": 5, "imm": "0x0b06c8a7", "imm2": "0xe9bd0cc4", "rot": 6, "bit": 2, "mask": 4},
|
||||
{"i": 40, "op": "mul", "dst": 5, "src": 7, "src2": 7, "imm": "0x070af1b6", "imm2": "0x6b648e75", "rot": 10, "bit": 6, "mask": 16},
|
||||
{"i": 41, "op": "mulhi", "dst": 2, "src": 3, "src2": 2, "imm": "0x389753b2", "imm2": "0x9e657305", "rot": 30, "bit": 2, "mask": 4},
|
||||
{"i": 42, "op": "xor", "dst": 2, "src": 0, "src2": 4, "imm": "0x0960e5b9", "imm2": "0xa925a406", "rot": 4, "bit": 6, "mask": 16},
|
||||
{"i": 43, "op": "rotl", "dst": 0, "src": 1, "src2": 1, "imm": "0x8eabe09f", "imm2": "0x76ef71e2", "rot": 4, "bit": 19, "mask": 8},
|
||||
{"i": 44, "op": "mulhi", "dst": 4, "src": 3, "src2": 7, "imm": "0x6a34768a", "imm2": "0x2e427c64", "rot": 21, "bit": 5, "mask": 4},
|
||||
{"i": 45, "op": "add", "dst": 6, "src": 7, "src2": 5, "imm": "0xee822e17", "imm2": "0xcdcb63f6", "rot": 27, "bit": 12, "mask": 16},
|
||||
{"i": 46, "op": "mad", "dst": 3, "src": 2, "src2": 0, "imm": "0xc89ead43", "imm2": "0x4d3108b2", "rot": 26, "bit": 17, "mask": 1},
|
||||
{"i": 47, "op": "shfl", "dst": 4, "src": 3, "src2": 0, "imm": "0xdd62be9e", "imm2": "0xf243f6f2", "rot": 8, "bit": 4, "mask": 8},
|
||||
{"i": 48, "op": "mulhi", "dst": 6, "src": 5, "src2": 0, "imm": "0xd3f7fa72", "imm2": "0x0debc83f", "rot": 25, "bit": 5, "mask": 2},
|
||||
{"i": 49, "op": "sub", "dst": 0, "src": 2, "src2": 6, "imm": "0x7afadd15", "imm2": "0xc07fc39c", "rot": 30, "bit": 29, "mask": 8},
|
||||
{"i": 50, "op": "sub", "dst": 3, "src": 5, "src2": 3, "imm": "0x179b78ec", "imm2": "0xaeccde37", "rot": 30, "bit": 17, "mask": 1},
|
||||
{"i": 51, "op": "rotr", "dst": 1, "src": 4, "src2": 1, "imm": "0xa4bfcea6", "imm2": "0xbf63bb2f", "rot": 2, "bit": 21, "mask": 2},
|
||||
{"i": 52, "op": "mad", "dst": 6, "src": 7, "src2": 7, "imm": "0xabf5ed10", "imm2": "0x8a285f51", "rot": 3, "bit": 23, "mask": 2},
|
||||
{"i": 53, "op": "xor", "dst": 5, "src": 3, "src2": 5, "imm": "0x88b416eb", "imm2": "0x36271d87", "rot": 19, "bit": 10, "mask": 2},
|
||||
{"i": 54, "op": "sub", "dst": 1, "src": 0, "src2": 1, "imm": "0xa3417dd3", "imm2": "0xcd98f620", "rot": 28, "bit": 2, "mask": 1},
|
||||
{"i": 55, "op": "sub", "dst": 5, "src": 6, "src2": 2, "imm": "0x45996c6f", "imm2": "0xe3d41087", "rot": 4, "bit": 31, "mask": 1},
|
||||
{"i": 56, "op": "add", "dst": 3, "src": 2, "src2": 0, "imm": "0x3dfad1b6", "imm2": "0xd4758987", "rot": 27, "bit": 10, "mask": 2},
|
||||
{"i": 57, "op": "add", "dst": 4, "src": 1, "src2": 3, "imm": "0xfca75bc2", "imm2": "0x0602d6be", "rot": 19, "bit": 0, "mask": 16},
|
||||
{"i": 58, "op": "mulhi", "dst": 5, "src": 6, "src2": 5, "imm": "0x83250a7b", "imm2": "0x2f93d53b", "rot": 25, "bit": 7, "mask": 2},
|
||||
{"i": 59, "op": "shfl", "dst": 2, "src": 1, "src2": 0, "imm": "0x13f4a089", "imm2": "0x145ea125", "rot": 12, "bit": 3, "mask": 2},
|
||||
{"i": 60, "op": "rotl", "dst": 2, "src": 5, "src2": 6, "imm": "0xad3170e3", "imm2": "0x15db04d1", "rot": 9, "bit": 13, "mask": 2},
|
||||
{"i": 61, "op": "or", "dst": 4, "src": 6, "src2": 2, "imm": "0x4b6305b2", "imm2": "0x6e7b2e9c", "rot": 27, "bit": 16, "mask": 4},
|
||||
{"i": 62, "op": "rotr", "dst": 6, "src": 4, "src2": 6, "imm": "0xfcd4b1c9", "imm2": "0xaf5c733b", "rot": 6, "bit": 21, "mask": 16},
|
||||
{"i": 63, "op": "mul", "dst": 2, "src": 4, "src2": 6, "imm": "0x95cebb3e", "imm2": "0xbba9cdfc", "rot": 19, "bit": 23, "mask": 1},
|
||||
{"i": 64, "op": "xor", "dst": 0, "src": 5, "src2": 7, "imm": "0x5f7579ff", "imm2": "0xb104e01a", "rot": 25, "bit": 12, "mask": 8},
|
||||
{"i": 65, "op": "xor", "dst": 2, "src": 7, "src2": 3, "imm": "0x749c7611", "imm2": "0xf17df3bb", "rot": 27, "bit": 26, "mask": 2},
|
||||
{"i": 66, "op": "add", "dst": 2, "src": 1, "src2": 6, "imm": "0xd71c02ff", "imm2": "0x8596687a", "rot": 4, "bit": 6, "mask": 2},
|
||||
{"i": 67, "op": "rotl", "dst": 1, "src": 3, "src2": 2, "imm": "0xd2864071", "imm2": "0xaa644ef3", "rot": 21, "bit": 5, "mask": 1},
|
||||
{"i": 68, "op": "mul", "dst": 3, "src": 2, "src2": 1, "imm": "0xd0689a5f", "imm2": "0xa3a1f063", "rot": 13, "bit": 28, "mask": 2},
|
||||
{"i": 69, "op": "shfl", "dst": 7, "src": 5, "src2": 6, "imm": "0xd01476eb", "imm2": "0x76abb5c2", "rot": 7, "bit": 30, "mask": 2},
|
||||
{"i": 70, "op": "mul", "dst": 3, "src": 2, "src2": 5, "imm": "0x59a75c10", "imm2": "0x1e1fbb72", "rot": 20, "bit": 16, "mask": 1},
|
||||
{"i": 71, "op": "mad", "dst": 0, "src": 2, "src2": 5, "imm": "0x39cca1df", "imm2": "0x22c057ac", "rot": 5, "bit": 23, "mask": 16},
|
||||
{"i": 72, "op": "add", "dst": 6, "src": 7, "src2": 3, "imm": "0x3c6fe15d", "imm2": "0x08ac5733", "rot": 14, "bit": 0, "mask": 4},
|
||||
{"i": 73, "op": "shfl", "dst": 3, "src": 2, "src2": 6, "imm": "0x2098627d", "imm2": "0xf4fecc81", "rot": 20, "bit": 30, "mask": 8},
|
||||
{"i": 74, "op": "mul", "dst": 3, "src": 6, "src2": 5, "imm": "0xf82e9e23", "imm2": "0x3ad62132", "rot": 10, "bit": 14, "mask": 16},
|
||||
{"i": 75, "op": "xor", "dst": 6, "src": 7, "src2": 4, "imm": "0x70548a91", "imm2": "0xa9715d2e", "rot": 6, "bit": 24, "mask": 1},
|
||||
{"i": 76, "op": "or", "dst": 3, "src": 0, "src2": 7, "imm": "0xa164325f", "imm2": "0xf300838b", "rot": 23, "bit": 23, "mask": 16},
|
||||
{"i": 77, "op": "add", "dst": 2, "src": 4, "src2": 6, "imm": "0x295d5fae", "imm2": "0xc100b495", "rot": 7, "bit": 1, "mask": 2},
|
||||
{"i": 78, "op": "mulhi", "dst": 3, "src": 7, "src2": 4, "imm": "0xb62cca87", "imm2": "0x2ebde415", "rot": 14, "bit": 7, "mask": 2},
|
||||
{"i": 79, "op": "or", "dst": 4, "src": 1, "src2": 7, "imm": "0xfcbc482d", "imm2": "0x8876e6cd", "rot": 29, "bit": 5, "mask": 2},
|
||||
{"i": 80, "op": "rotr", "dst": 4, "src": 3, "src2": 0, "imm": "0x6d64013b", "imm2": "0x675f4a8d", "rot": 26, "bit": 18, "mask": 8},
|
||||
{"i": 81, "op": "add", "dst": 4, "src": 3, "src2": 3, "imm": "0x274a9221", "imm2": "0x5cc59530", "rot": 15, "bit": 15, "mask": 2},
|
||||
{"i": 82, "op": "add", "dst": 7, "src": 3, "src2": 0, "imm": "0xc8651f8e", "imm2": "0x141479ec", "rot": 18, "bit": 5, "mask": 1},
|
||||
{"i": 83, "op": "rotr", "dst": 3, "src": 6, "src2": 0, "imm": "0xcc8a7766", "imm2": "0xc2eb5161", "rot": 4, "bit": 29, "mask": 2},
|
||||
{"i": 84, "op": "mad", "dst": 2, "src": 4, "src2": 7, "imm": "0xbc507d51", "imm2": "0x0b1196fd", "rot": 9, "bit": 7, "mask": 8},
|
||||
{"i": 85, "op": "shfl", "dst": 2, "src": 5, "src2": 5, "imm": "0x5a156c90", "imm2": "0xa6b3fbfa", "rot": 11, "bit": 2, "mask": 16},
|
||||
{"i": 86, "op": "xor", "dst": 3, "src": 2, "src2": 1, "imm": "0x8b042658", "imm2": "0xacf37a8f", "rot": 12, "bit": 23, "mask": 16},
|
||||
{"i": 87, "op": "shfl", "dst": 5, "src": 7, "src2": 2, "imm": "0x1a214238", "imm2": "0x017fdf5d", "rot": 29, "bit": 14, "mask": 2},
|
||||
{"i": 88, "op": "xor", "dst": 0, "src": 4, "src2": 2, "imm": "0xd523e612", "imm2": "0x2158c2ed", "rot": 30, "bit": 14, "mask": 4},
|
||||
{"i": 89, "op": "add", "dst": 3, "src": 2, "src2": 7, "imm": "0x53f915c2", "imm2": "0x883c0c92", "rot": 9, "bit": 18, "mask": 8},
|
||||
{"i": 90, "op": "rotr", "dst": 7, "src": 5, "src2": 5, "imm": "0xbd633b21", "imm2": "0xcf8c356c", "rot": 25, "bit": 31, "mask": 4},
|
||||
{"i": 91, "op": "add", "dst": 3, "src": 1, "src2": 1, "imm": "0xe2be00a5", "imm2": "0x3cc5bd20", "rot": 10, "bit": 31, "mask": 1},
|
||||
{"i": 92, "op": "shfl", "dst": 7, "src": 2, "src2": 4, "imm": "0x3d3da600", "imm2": "0x1f22df89", "rot": 4, "bit": 24, "mask": 1},
|
||||
{"i": 93, "op": "rotr", "dst": 6, "src": 2, "src2": 7, "imm": "0xb0f57471", "imm2": "0x8f2a7eca", "rot": 16, "bit": 25, "mask": 1},
|
||||
{"i": 94, "op": "rotl", "dst": 7, "src": 0, "src2": 4, "imm": "0x00a21815", "imm2": "0xeb4d7218", "rot": 14, "bit": 18, "mask": 16},
|
||||
{"i": 95, "op": "mad", "dst": 4, "src": 5, "src2": 5, "imm": "0xc4c3828e", "imm2": "0xfb7bba17", "rot": 16, "bit": 10, "mask": 16},
|
||||
{"i": 96, "op": "mad", "dst": 2, "src": 4, "src2": 5, "imm": "0xfc5bbc75", "imm2": "0xfc43468f", "rot": 31, "bit": 20, "mask": 16},
|
||||
{"i": 97, "op": "xor", "dst": 1, "src": 0, "src2": 2, "imm": "0x1fe9c249", "imm2": "0x164bb16b", "rot": 15, "bit": 9, "mask": 4},
|
||||
{"i": 98, "op": "mul", "dst": 5, "src": 4, "src2": 1, "imm": "0xeda725fa", "imm2": "0x66f7e9a2", "rot": 21, "bit": 6, "mask": 2},
|
||||
{"i": 99, "op": "sub", "dst": 2, "src": 0, "src2": 7, "imm": "0x8fb29f7d", "imm2": "0xf530eda1", "rot": 27, "bit": 30, "mask": 4},
|
||||
{"i": 100, "op": "rotl", "dst": 7, "src": 2, "src2": 0, "imm": "0x9f4de742", "imm2": "0x9b5ff871", "rot": 30, "bit": 21, "mask": 16},
|
||||
{"i": 101, "op": "add", "dst": 5, "src": 6, "src2": 7, "imm": "0x423fd9c9", "imm2": "0xbfd646cb", "rot": 17, "bit": 17, "mask": 2},
|
||||
{"i": 102, "op": "add", "dst": 7, "src": 6, "src2": 3, "imm": "0x8d3c011d", "imm2": "0x19b74a43", "rot": 12, "bit": 11, "mask": 4},
|
||||
{"i": 103, "op": "rotl", "dst": 5, "src": 6, "src2": 0, "imm": "0xcfc70303", "imm2": "0xf2b3e8ef", "rot": 6, "bit": 24, "mask": 1},
|
||||
{"i": 104, "op": "mul", "dst": 0, "src": 4, "src2": 5, "imm": "0x650475eb", "imm2": "0x11dcbd94", "rot": 15, "bit": 10, "mask": 1},
|
||||
{"i": 105, "op": "or", "dst": 0, "src": 5, "src2": 3, "imm": "0xaacaa145", "imm2": "0x9139d3fe", "rot": 4, "bit": 18, "mask": 2},
|
||||
{"i": 106, "op": "add", "dst": 0, "src": 1, "src2": 7, "imm": "0x8ce14721", "imm2": "0x7dcb7e18", "rot": 24, "bit": 15, "mask": 8},
|
||||
{"i": 107, "op": "rotl", "dst": 0, "src": 3, "src2": 3, "imm": "0xb36d6d98", "imm2": "0x2c3390c8", "rot": 15, "bit": 10, "mask": 16},
|
||||
{"i": 108, "op": "sub", "dst": 4, "src": 2, "src2": 5, "imm": "0xf6bbdaef", "imm2": "0x9db6f65e", "rot": 25, "bit": 10, "mask": 1},
|
||||
{"i": 109, "op": "mulhi", "dst": 2, "src": 0, "src2": 0, "imm": "0xb1721fd6", "imm2": "0xd96d52c9", "rot": 1, "bit": 27, "mask": 8},
|
||||
{"i": 110, "op": "mulhi", "dst": 1, "src": 0, "src2": 5, "imm": "0xfb4ca37f", "imm2": "0xb6ec7dbe", "rot": 27, "bit": 12, "mask": 8},
|
||||
{"i": 111, "op": "add", "dst": 0, "src": 2, "src2": 0, "imm": "0x2d9fc6b9", "imm2": "0x3c179ad8", "rot": 9, "bit": 28, "mask": 16},
|
||||
{"i": 112, "op": "mulhi", "dst": 2, "src": 0, "src2": 6, "imm": "0x71a9f6bd", "imm2": "0xd3417bf5", "rot": 2, "bit": 14, "mask": 8},
|
||||
{"i": 113, "op": "sub", "dst": 6, "src": 3, "src2": 5, "imm": "0xa3d19400", "imm2": "0x15df7330", "rot": 5, "bit": 2, "mask": 8},
|
||||
{"i": 114, "op": "mad", "dst": 7, "src": 6, "src2": 3, "imm": "0x1400d92e", "imm2": "0xfa4a158e", "rot": 16, "bit": 9, "mask": 1},
|
||||
{"i": 115, "op": "rotr", "dst": 5, "src": 1, "src2": 3, "imm": "0xd01684c3", "imm2": "0x6367febb", "rot": 23, "bit": 19, "mask": 2},
|
||||
{"i": 116, "op": "rotr", "dst": 6, "src": 4, "src2": 2, "imm": "0x8cd9eb96", "imm2": "0x98f33b24", "rot": 25, "bit": 1, "mask": 2},
|
||||
{"i": 117, "op": "add", "dst": 2, "src": 1, "src2": 7, "imm": "0x502138e2", "imm2": "0x47359729", "rot": 5, "bit": 22, "mask": 4},
|
||||
{"i": 118, "op": "mad", "dst": 3, "src": 2, "src2": 0, "imm": "0xd94a35c7", "imm2": "0xb83416c3", "rot": 11, "bit": 3, "mask": 4},
|
||||
{"i": 119, "op": "mul", "dst": 1, "src": 3, "src2": 6, "imm": "0x09b30cf7", "imm2": "0xef3b98e7", "rot": 17, "bit": 23, "mask": 8},
|
||||
{"i": 120, "op": "mad", "dst": 2, "src": 0, "src2": 4, "imm": "0x56416fe0", "imm2": "0x5e1dc8f3", "rot": 17, "bit": 14, "mask": 2},
|
||||
{"i": 121, "op": "mul", "dst": 0, "src": 3, "src2": 2, "imm": "0x2892697c", "imm2": "0x9cb3b14e", "rot": 29, "bit": 25, "mask": 2},
|
||||
{"i": 122, "op": "mulhi", "dst": 2, "src": 0, "src2": 3, "imm": "0xc33089a1", "imm2": "0xd6c5b530", "rot": 27, "bit": 13, "mask": 8},
|
||||
{"i": 123, "op": "mul", "dst": 5, "src": 6, "src2": 4, "imm": "0xdfe04e4a", "imm2": "0x3cc40160", "rot": 3, "bit": 14, "mask": 2},
|
||||
{"i": 124, "op": "mad", "dst": 4, "src": 2, "src2": 4, "imm": "0xb33dc1b7", "imm2": "0xab97743a", "rot": 7, "bit": 21, "mask": 4},
|
||||
{"i": 125, "op": "shfl", "dst": 4, "src": 2, "src2": 0, "imm": "0xfd469909", "imm2": "0xfc85dc55", "rot": 28, "bit": 24, "mask": 2},
|
||||
{"i": 126, "op": "xor", "dst": 2, "src": 0, "src2": 5, "imm": "0xa0094578", "imm2": "0xf1bee474", "rot": 31, "bit": 7, "mask": 2},
|
||||
{"i": 127, "op": "rotl", "dst": 1, "src": 7, "src2": 2, "imm": "0xb7ae00a0", "imm2": "0x86cce297", "rot": 7, "bit": 31, "mask": 8},
|
||||
{"i": 128, "op": "shfl", "dst": 7, "src": 2, "src2": 7, "imm": "0x019d1940", "imm2": "0x3fe9e7dd", "rot": 14, "bit": 4, "mask": 4},
|
||||
{"i": 129, "op": "rotl", "dst": 6, "src": 7, "src2": 7, "imm": "0x40a74cc4", "imm2": "0x09eb4adf", "rot": 17, "bit": 30, "mask": 1},
|
||||
{"i": 130, "op": "add", "dst": 2, "src": 5, "src2": 5, "imm": "0x33dff776", "imm2": "0x6c57e4e7", "rot": 27, "bit": 15, "mask": 4},
|
||||
{"i": 131, "op": "or", "dst": 0, "src": 3, "src2": 1, "imm": "0xe0c53bb9", "imm2": "0x124404f6", "rot": 4, "bit": 21, "mask": 16},
|
||||
{"i": 132, "op": "rotr", "dst": 5, "src": 0, "src2": 1, "imm": "0xe4353fce", "imm2": "0x559d0118", "rot": 2, "bit": 29, "mask": 4},
|
||||
{"i": 133, "op": "add", "dst": 2, "src": 3, "src2": 6, "imm": "0x5db25b34", "imm2": "0xe8212c0c", "rot": 21, "bit": 30, "mask": 1},
|
||||
{"i": 134, "op": "xor", "dst": 4, "src": 3, "src2": 6, "imm": "0x7366e50e", "imm2": "0x7cc1ffbd", "rot": 19, "bit": 18, "mask": 16},
|
||||
{"i": 135, "op": "xor", "dst": 6, "src": 1, "src2": 2, "imm": "0xa36decb7", "imm2": "0x79291734", "rot": 26, "bit": 0, "mask": 8},
|
||||
{"i": 136, "op": "sub", "dst": 1, "src": 7, "src2": 0, "imm": "0x225b03e4", "imm2": "0x7183e193", "rot": 16, "bit": 6, "mask": 2},
|
||||
{"i": 137, "op": "mad", "dst": 2, "src": 6, "src2": 2, "imm": "0x6f620d51", "imm2": "0x4e814e19", "rot": 6, "bit": 6, "mask": 2},
|
||||
{"i": 138, "op": "mulhi", "dst": 0, "src": 5, "src2": 6, "imm": "0x919d6bf3", "imm2": "0xc6240638", "rot": 17, "bit": 11, "mask": 16},
|
||||
{"i": 139, "op": "add", "dst": 2, "src": 7, "src2": 7, "imm": "0x218d4090", "imm2": "0xd44ff710", "rot": 1, "bit": 10, "mask": 16},
|
||||
{"i": 140, "op": "rotr", "dst": 1, "src": 4, "src2": 4, "imm": "0x4cbfa722", "imm2": "0x114e9564", "rot": 28, "bit": 9, "mask": 16},
|
||||
{"i": 141, "op": "shfl", "dst": 3, "src": 6, "src2": 2, "imm": "0xf702c6a1", "imm2": "0xf8f3c5d9", "rot": 7, "bit": 24, "mask": 1},
|
||||
{"i": 142, "op": "mad", "dst": 7, "src": 6, "src2": 2, "imm": "0xa2d285be", "imm2": "0x9cc94532", "rot": 15, "bit": 20, "mask": 8},
|
||||
{"i": 143, "op": "mul", "dst": 2, "src": 3, "src2": 7, "imm": "0x08b7ed80", "imm2": "0xa8abced4", "rot": 18, "bit": 10, "mask": 4},
|
||||
{"i": 144, "op": "add", "dst": 7, "src": 4, "src2": 2, "imm": "0xf4b1a8de", "imm2": "0xb98942fa", "rot": 18, "bit": 29, "mask": 8},
|
||||
{"i": 145, "op": "rotl", "dst": 7, "src": 6, "src2": 1, "imm": "0x64b6ba2d", "imm2": "0xf9e86793", "rot": 15, "bit": 24, "mask": 8},
|
||||
{"i": 146, "op": "xor", "dst": 7, "src": 5, "src2": 3, "imm": "0x41c42b5f", "imm2": "0x7c7e0e39", "rot": 8, "bit": 23, "mask": 2},
|
||||
{"i": 147, "op": "mad", "dst": 4, "src": 7, "src2": 4, "imm": "0x3caa807a", "imm2": "0x553a0cec", "rot": 11, "bit": 22, "mask": 8},
|
||||
{"i": 148, "op": "rotr", "dst": 6, "src": 5, "src2": 3, "imm": "0x3cb1289a", "imm2": "0x6b36f78a", "rot": 1, "bit": 9, "mask": 16},
|
||||
{"i": 149, "op": "mad", "dst": 1, "src": 2, "src2": 4, "imm": "0x95310ea8", "imm2": "0xc533aa6a", "rot": 16, "bit": 17, "mask": 1},
|
||||
{"i": 150, "op": "add", "dst": 1, "src": 7, "src2": 7, "imm": "0x2bef10f2", "imm2": "0x0d48ba42", "rot": 8, "bit": 17, "mask": 4},
|
||||
{"i": 151, "op": "xor", "dst": 5, "src": 4, "src2": 7, "imm": "0xda997ab2", "imm2": "0xae31d69e", "rot": 11, "bit": 31, "mask": 2},
|
||||
{"i": 152, "op": "add", "dst": 7, "src": 0, "src2": 6, "imm": "0xdc5cc080", "imm2": "0xc98dea9c", "rot": 16, "bit": 30, "mask": 2},
|
||||
{"i": 153, "op": "mul", "dst": 4, "src": 6, "src2": 7, "imm": "0xbfbf5f6c", "imm2": "0x61f611bb", "rot": 14, "bit": 22, "mask": 2},
|
||||
{"i": 154, "op": "mul", "dst": 0, "src": 2, "src2": 3, "imm": "0xa78008c3", "imm2": "0xbad5eeb1", "rot": 10, "bit": 28, "mask": 8},
|
||||
{"i": 155, "op": "xor", "dst": 6, "src": 5, "src2": 4, "imm": "0x1a73b866", "imm2": "0x1f5a62c9", "rot": 9, "bit": 27, "mask": 8},
|
||||
{"i": 156, "op": "rotr", "dst": 4, "src": 2, "src2": 0, "imm": "0x9c88500c", "imm2": "0xe25dccf9", "rot": 11, "bit": 2, "mask": 16},
|
||||
{"i": 157, "op": "rotl", "dst": 1, "src": 4, "src2": 4, "imm": "0xb05e7669", "imm2": "0x9db704b1", "rot": 11, "bit": 28, "mask": 4},
|
||||
{"i": 158, "op": "add", "dst": 5, "src": 0, "src2": 6, "imm": "0x18197438", "imm2": "0x6c752dcb", "rot": 11, "bit": 11, "mask": 2},
|
||||
{"i": 159, "op": "mad", "dst": 4, "src": 1, "src2": 5, "imm": "0x4796a65e", "imm2": "0x00e08c7a", "rot": 8, "bit": 19, "mask": 4},
|
||||
{"i": 160, "op": "rotl", "dst": 4, "src": 7, "src2": 5, "imm": "0x08a03052", "imm2": "0x0204f0ba", "rot": 26, "bit": 5, "mask": 2},
|
||||
{"i": 161, "op": "mad", "dst": 3, "src": 0, "src2": 7, "imm": "0xa0603b0e", "imm2": "0x7eee83d5", "rot": 28, "bit": 22, "mask": 16},
|
||||
{"i": 162, "op": "rotr", "dst": 3, "src": 1, "src2": 6, "imm": "0x78a3c69d", "imm2": "0x684693a0", "rot": 21, "bit": 23, "mask": 4},
|
||||
{"i": 163, "op": "add", "dst": 4, "src": 0, "src2": 7, "imm": "0xf1c46574", "imm2": "0x8e481727", "rot": 9, "bit": 3, "mask": 4},
|
||||
{"i": 164, "op": "mulhi", "dst": 0, "src": 4, "src2": 3, "imm": "0x04644afa", "imm2": "0x64bda2b5", "rot": 26, "bit": 16, "mask": 16},
|
||||
{"i": 165, "op": "add", "dst": 2, "src": 6, "src2": 2, "imm": "0xac578137", "imm2": "0x550ab406", "rot": 13, "bit": 20, "mask": 16},
|
||||
{"i": 166, "op": "rotl", "dst": 0, "src": 4, "src2": 5, "imm": "0x7f564760", "imm2": "0xb9a8b4f8", "rot": 13, "bit": 29, "mask": 1},
|
||||
{"i": 167, "op": "add", "dst": 3, "src": 1, "src2": 0, "imm": "0xa33e6706", "imm2": "0xaebb5966", "rot": 12, "bit": 14, "mask": 16},
|
||||
{"i": 168, "op": "or", "dst": 3, "src": 5, "src2": 3, "imm": "0x65b2f3eb", "imm2": "0xb1d00d20", "rot": 12, "bit": 2, "mask": 8},
|
||||
{"i": 169, "op": "rotr", "dst": 6, "src": 2, "src2": 3, "imm": "0x7f21faf5", "imm2": "0xbbf0d3f9", "rot": 17, "bit": 13, "mask": 8},
|
||||
{"i": 170, "op": "xor", "dst": 4, "src": 6, "src2": 2, "imm": "0x162c7140", "imm2": "0x90d404ad", "rot": 26, "bit": 1, "mask": 4},
|
||||
{"i": 171, "op": "sub", "dst": 6, "src": 1, "src2": 7, "imm": "0x6ef9c76e", "imm2": "0xfb7ba272", "rot": 31, "bit": 25, "mask": 1},
|
||||
{"i": 172, "op": "rotl", "dst": 7, "src": 5, "src2": 4, "imm": "0xde04eb3b", "imm2": "0xd56caa00", "rot": 22, "bit": 21, "mask": 4},
|
||||
{"i": 173, "op": "rotl", "dst": 5, "src": 3, "src2": 5, "imm": "0xa9eac934", "imm2": "0x2c338e51", "rot": 15, "bit": 22, "mask": 2},
|
||||
{"i": 174, "op": "shfl", "dst": 7, "src": 0, "src2": 3, "imm": "0x700be4e3", "imm2": "0x4bcfc732", "rot": 19, "bit": 6, "mask": 8},
|
||||
{"i": 175, "op": "xor", "dst": 0, "src": 5, "src2": 1, "imm": "0x02ccdba9", "imm2": "0xd0915be0", "rot": 15, "bit": 2, "mask": 16},
|
||||
{"i": 176, "op": "rotl", "dst": 7, "src": 4, "src2": 1, "imm": "0x489c8165", "imm2": "0xf24b5a4f", "rot": 6, "bit": 22, "mask": 1},
|
||||
{"i": 177, "op": "sub", "dst": 7, "src": 0, "src2": 6, "imm": "0x23ad9693", "imm2": "0x9a8c2f7b", "rot": 24, "bit": 2, "mask": 2},
|
||||
{"i": 178, "op": "rotl", "dst": 3, "src": 0, "src2": 6, "imm": "0xafa72a42", "imm2": "0x371d74ee", "rot": 30, "bit": 16, "mask": 1},
|
||||
{"i": 179, "op": "mad", "dst": 7, "src": 6, "src2": 1, "imm": "0x18a2a3f3", "imm2": "0xb811b951", "rot": 2, "bit": 9, "mask": 2},
|
||||
{"i": 180, "op": "rotl", "dst": 6, "src": 4, "src2": 1, "imm": "0xe6c69c0e", "imm2": "0xfc46a951", "rot": 9, "bit": 31, "mask": 4},
|
||||
{"i": 181, "op": "xor", "dst": 2, "src": 4, "src2": 7, "imm": "0xbd1b89e4", "imm2": "0xdf4bce5c", "rot": 15, "bit": 19, "mask": 4},
|
||||
{"i": 182, "op": "xor", "dst": 2, "src": 7, "src2": 5, "imm": "0x3dd12aed", "imm2": "0xd0756a69", "rot": 13, "bit": 16, "mask": 4},
|
||||
{"i": 183, "op": "xor", "dst": 7, "src": 2, "src2": 3, "imm": "0x77ce69d6", "imm2": "0x2b39bdf2", "rot": 8, "bit": 22, "mask": 16},
|
||||
{"i": 184, "op": "add", "dst": 1, "src": 2, "src2": 4, "imm": "0xd94d55ac", "imm2": "0x5bb7550f", "rot": 31, "bit": 21, "mask": 1},
|
||||
{"i": 185, "op": "or", "dst": 3, "src": 5, "src2": 6, "imm": "0x7b1ce846", "imm2": "0xcc3b8509", "rot": 28, "bit": 9, "mask": 4},
|
||||
{"i": 186, "op": "mulhi", "dst": 6, "src": 3, "src2": 0, "imm": "0xa5e24690", "imm2": "0x2200ba81", "rot": 26, "bit": 10, "mask": 16},
|
||||
{"i": 187, "op": "or", "dst": 4, "src": 0, "src2": 3, "imm": "0xf6efe759", "imm2": "0xae1f7118", "rot": 19, "bit": 20, "mask": 4},
|
||||
{"i": 188, "op": "shfl", "dst": 7, "src": 1, "src2": 5, "imm": "0xdb318b45", "imm2": "0xcec459c9", "rot": 20, "bit": 11, "mask": 2},
|
||||
{"i": 189, "op": "add", "dst": 6, "src": 5, "src2": 1, "imm": "0x89e747fe", "imm2": "0x2a354e2d", "rot": 22, "bit": 6, "mask": 1},
|
||||
{"i": 190, "op": "xor", "dst": 1, "src": 4, "src2": 2, "imm": "0xc379e617", "imm2": "0x75e9d63b", "rot": 23, "bit": 3, "mask": 2},
|
||||
{"i": 191, "op": "mulhi", "dst": 7, "src": 4, "src2": 3, "imm": "0x5a710287", "imm2": "0x7fe4ead6", "rot": 10, "bit": 25, "mask": 4},
|
||||
{"i": 192, "op": "rotl", "dst": 2, "src": 1, "src2": 2, "imm": "0x4d5e59a3", "imm2": "0x1e4fef28", "rot": 26, "bit": 0, "mask": 1},
|
||||
{"i": 193, "op": "rotl", "dst": 5, "src": 3, "src2": 7, "imm": "0x7444c47d", "imm2": "0xdad5f8be", "rot": 8, "bit": 30, "mask": 2},
|
||||
{"i": 194, "op": "shfl", "dst": 4, "src": 5, "src2": 7, "imm": "0x6a225bbb", "imm2": "0xd6532cd7", "rot": 13, "bit": 17, "mask": 16},
|
||||
{"i": 195, "op": "xor", "dst": 4, "src": 5, "src2": 3, "imm": "0x68344b9a", "imm2": "0xcb46a38b", "rot": 1, "bit": 27, "mask": 16},
|
||||
{"i": 196, "op": "mul", "dst": 1, "src": 3, "src2": 4, "imm": "0xbebf7359", "imm2": "0x3f0890ba", "rot": 18, "bit": 12, "mask": 4},
|
||||
{"i": 197, "op": "add", "dst": 5, "src": 2, "src2": 3, "imm": "0xea03e8e7", "imm2": "0x10cfdc71", "rot": 31, "bit": 7, "mask": 8},
|
||||
{"i": 198, "op": "add", "dst": 7, "src": 5, "src2": 1, "imm": "0xa7ee0102", "imm2": "0x66e148d3", "rot": 12, "bit": 15, "mask": 1},
|
||||
{"i": 199, "op": "sub", "dst": 5, "src": 2, "src2": 4, "imm": "0xc215584e", "imm2": "0x55fce30f", "rot": 30, "bit": 9, "mask": 2},
|
||||
{"i": 200, "op": "shfl", "dst": 5, "src": 1, "src2": 1, "imm": "0x308f8358", "imm2": "0x489f1f93", "rot": 13, "bit": 13, "mask": 2},
|
||||
{"i": 201, "op": "shfl", "dst": 7, "src": 1, "src2": 5, "imm": "0x713f1957", "imm2": "0x2239f757", "rot": 15, "bit": 7, "mask": 8},
|
||||
{"i": 202, "op": "rotr", "dst": 4, "src": 5, "src2": 1, "imm": "0xb037b6e7", "imm2": "0x49a8b528", "rot": 27, "bit": 13, "mask": 16},
|
||||
{"i": 203, "op": "xor", "dst": 7, "src": 5, "src2": 1, "imm": "0x56110241", "imm2": "0xefc8386d", "rot": 22, "bit": 20, "mask": 1},
|
||||
{"i": 204, "op": "xor", "dst": 7, "src": 0, "src2": 0, "imm": "0x0827fcc7", "imm2": "0xf4f5dd07", "rot": 13, "bit": 7, "mask": 2},
|
||||
{"i": 205, "op": "add", "dst": 6, "src": 2, "src2": 0, "imm": "0x8379a4de", "imm2": "0x8558b619", "rot": 26, "bit": 10, "mask": 1},
|
||||
{"i": 206, "op": "rotl", "dst": 4, "src": 3, "src2": 3, "imm": "0x091ceef0", "imm2": "0x69b0f72f", "rot": 13, "bit": 7, "mask": 1},
|
||||
{"i": 207, "op": "mulhi", "dst": 1, "src": 3, "src2": 4, "imm": "0x758edac1", "imm2": "0x98dd2f21", "rot": 15, "bit": 9, "mask": 1},
|
||||
{"i": 208, "op": "mad", "dst": 1, "src": 4, "src2": 4, "imm": "0x78871a1a", "imm2": "0x5e14e1c8", "rot": 19, "bit": 6, "mask": 2},
|
||||
{"i": 209, "op": "shfl", "dst": 2, "src": 0, "src2": 2, "imm": "0xf7e0b9aa", "imm2": "0xaecfd347", "rot": 21, "bit": 9, "mask": 4},
|
||||
{"i": 210, "op": "add", "dst": 7, "src": 0, "src2": 7, "imm": "0xaa8cb14e", "imm2": "0xf4049c4c", "rot": 24, "bit": 11, "mask": 8},
|
||||
{"i": 211, "op": "shfl", "dst": 7, "src": 1, "src2": 6, "imm": "0xc230d919", "imm2": "0xf76d08fb", "rot": 21, "bit": 13, "mask": 16},
|
||||
{"i": 212, "op": "xor", "dst": 1, "src": 0, "src2": 2, "imm": "0x31a5af90", "imm2": "0x7eef58ee", "rot": 9, "bit": 12, "mask": 4},
|
||||
{"i": 213, "op": "rotl", "dst": 7, "src": 1, "src2": 6, "imm": "0x0db26138", "imm2": "0x8e3c31f9", "rot": 3, "bit": 26, "mask": 2},
|
||||
{"i": 214, "op": "rotr", "dst": 4, "src": 7, "src2": 3, "imm": "0x29558100", "imm2": "0xe4b13ad6", "rot": 29, "bit": 24, "mask": 8},
|
||||
{"i": 215, "op": "shfl", "dst": 3, "src": 2, "src2": 6, "imm": "0x1b48c3d0", "imm2": "0x5c674ff6", "rot": 5, "bit": 9, "mask": 16},
|
||||
{"i": 216, "op": "or", "dst": 5, "src": 1, "src2": 0, "imm": "0xef768632", "imm2": "0x6de9d10d", "rot": 10, "bit": 4, "mask": 4},
|
||||
{"i": 217, "op": "sub", "dst": 1, "src": 2, "src2": 5, "imm": "0x88ca7f5a", "imm2": "0x24718a36", "rot": 18, "bit": 31, "mask": 1},
|
||||
{"i": 218, "op": "sub", "dst": 6, "src": 5, "src2": 0, "imm": "0xc4a06728", "imm2": "0xdc2a4fd8", "rot": 9, "bit": 9, "mask": 2},
|
||||
{"i": 219, "op": "rotl", "dst": 6, "src": 5, "src2": 7, "imm": "0xb7489e47", "imm2": "0xf13795c5", "rot": 4, "bit": 3, "mask": 8},
|
||||
{"i": 220, "op": "mulhi", "dst": 2, "src": 0, "src2": 2, "imm": "0x873cd31b", "imm2": "0x3dfbc55d", "rot": 12, "bit": 23, "mask": 8},
|
||||
{"i": 221, "op": "or", "dst": 2, "src": 0, "src2": 3, "imm": "0xdc21f099", "imm2": "0xee06f01e", "rot": 2, "bit": 17, "mask": 8},
|
||||
{"i": 222, "op": "shfl", "dst": 5, "src": 2, "src2": 6, "imm": "0x36def499", "imm2": "0xa2849d59", "rot": 23, "bit": 4, "mask": 8},
|
||||
{"i": 223, "op": "mulhi", "dst": 5, "src": 6, "src2": 7, "imm": "0xfb95fbca", "imm2": "0xc1aac427", "rot": 14, "bit": 11, "mask": 2},
|
||||
{"i": 224, "op": "sub", "dst": 0, "src": 6, "src2": 7, "imm": "0x3d9d29c4", "imm2": "0x34d0dcc0", "rot": 17, "bit": 6, "mask": 4},
|
||||
{"i": 225, "op": "rotl", "dst": 7, "src": 0, "src2": 5, "imm": "0x93b01b8e", "imm2": "0xfe1d75ac", "rot": 23, "bit": 15, "mask": 1},
|
||||
{"i": 226, "op": "or", "dst": 4, "src": 2, "src2": 4, "imm": "0xfed76e8e", "imm2": "0x1c24ecd8", "rot": 2, "bit": 6, "mask": 16},
|
||||
{"i": 227, "op": "mul", "dst": 2, "src": 4, "src2": 1, "imm": "0x5795f5b0", "imm2": "0x0566ea2a", "rot": 6, "bit": 28, "mask": 4},
|
||||
{"i": 228, "op": "rotl", "dst": 3, "src": 0, "src2": 2, "imm": "0x8c8486de", "imm2": "0xd066aa8f", "rot": 12, "bit": 20, "mask": 1},
|
||||
{"i": 229, "op": "rotr", "dst": 0, "src": 4, "src2": 0, "imm": "0x23a3e882", "imm2": "0xaca23902", "rot": 5, "bit": 22, "mask": 16},
|
||||
{"i": 230, "op": "add", "dst": 0, "src": 6, "src2": 4, "imm": "0x1d176220", "imm2": "0x2a7fecb2", "rot": 20, "bit": 16, "mask": 2},
|
||||
{"i": 231, "op": "shfl", "dst": 1, "src": 2, "src2": 0, "imm": "0x91172787", "imm2": "0xc5d7af28", "rot": 3, "bit": 24, "mask": 4},
|
||||
{"i": 232, "op": "mul", "dst": 2, "src": 1, "src2": 2, "imm": "0x0be68835", "imm2": "0xde692bdb", "rot": 30, "bit": 17, "mask": 1},
|
||||
{"i": 233, "op": "mad", "dst": 7, "src": 0, "src2": 1, "imm": "0xf90d2db5", "imm2": "0x96c4c175", "rot": 28, "bit": 7, "mask": 4},
|
||||
{"i": 234, "op": "rotl", "dst": 5, "src": 2, "src2": 1, "imm": "0x16379736", "imm2": "0x6973b905", "rot": 22, "bit": 17, "mask": 4},
|
||||
{"i": 235, "op": "rotr", "dst": 4, "src": 6, "src2": 2, "imm": "0x84292a13", "imm2": "0x0f897740", "rot": 12, "bit": 6, "mask": 8},
|
||||
{"i": 236, "op": "mad", "dst": 0, "src": 5, "src2": 1, "imm": "0xeb7de837", "imm2": "0x64f0c302", "rot": 4, "bit": 19, "mask": 1},
|
||||
{"i": 237, "op": "xor", "dst": 6, "src": 4, "src2": 4, "imm": "0x6ab4b683", "imm2": "0x20f17adb", "rot": 1, "bit": 0, "mask": 16},
|
||||
{"i": 238, "op": "xor", "dst": 4, "src": 6, "src2": 1, "imm": "0x5836b35c", "imm2": "0x3293cc4a", "rot": 16, "bit": 22, "mask": 16},
|
||||
{"i": 239, "op": "rotl", "dst": 6, "src": 1, "src2": 6, "imm": "0x769d8bc0", "imm2": "0xdc86c9cc", "rot": 18, "bit": 27, "mask": 16},
|
||||
{"i": 240, "op": "add", "dst": 4, "src": 7, "src2": 1, "imm": "0x17dafb4d", "imm2": "0xadce39f3", "rot": 12, "bit": 25, "mask": 8},
|
||||
{"i": 241, "op": "sub", "dst": 0, "src": 7, "src2": 4, "imm": "0xd4690bda", "imm2": "0xdab9b27b", "rot": 30, "bit": 21, "mask": 1},
|
||||
{"i": 242, "op": "rotr", "dst": 1, "src": 6, "src2": 2, "imm": "0x670a2d0a", "imm2": "0x0612e33c", "rot": 31, "bit": 25, "mask": 2},
|
||||
{"i": 243, "op": "add", "dst": 3, "src": 0, "src2": 2, "imm": "0xf2f77d26", "imm2": "0x0e7033b6", "rot": 27, "bit": 29, "mask": 1},
|
||||
{"i": 244, "op": "mad", "dst": 2, "src": 7, "src2": 4, "imm": "0xa9eefc9d", "imm2": "0x16166c85", "rot": 18, "bit": 23, "mask": 16},
|
||||
{"i": 245, "op": "mad", "dst": 4, "src": 0, "src2": 6, "imm": "0xb26f6c0f", "imm2": "0x0630e821", "rot": 23, "bit": 30, "mask": 4},
|
||||
{"i": 246, "op": "mul", "dst": 5, "src": 7, "src2": 2, "imm": "0x269ce4bf", "imm2": "0x1ce28d32", "rot": 30, "bit": 15, "mask": 16},
|
||||
{"i": 247, "op": "add", "dst": 3, "src": 5, "src2": 0, "imm": "0xfed2da4e", "imm2": "0x7b2ff6b7", "rot": 25, "bit": 31, "mask": 2},
|
||||
{"i": 248, "op": "add", "dst": 0, "src": 7, "src2": 5, "imm": "0x03b2891c", "imm2": "0xb5fad7b1", "rot": 2, "bit": 0, "mask": 4},
|
||||
{"i": 249, "op": "mulhi", "dst": 5, "src": 4, "src2": 6, "imm": "0xa69a1e71", "imm2": "0x15f0c0ea", "rot": 4, "bit": 1, "mask": 4},
|
||||
{"i": 250, "op": "add", "dst": 5, "src": 2, "src2": 3, "imm": "0x042cd6e3", "imm2": "0xa7e71c2f", "rot": 24, "bit": 11, "mask": 8},
|
||||
{"i": 251, "op": "shfl", "dst": 6, "src": 1, "src2": 2, "imm": "0x89c6b683", "imm2": "0x10bbf661", "rot": 24, "bit": 5, "mask": 1},
|
||||
{"i": 252, "op": "xor", "dst": 6, "src": 1, "src2": 3, "imm": "0x265c66d6", "imm2": "0xd4a689ed", "rot": 6, "bit": 14, "mask": 8},
|
||||
{"i": 253, "op": "mul", "dst": 2, "src": 5, "src2": 5, "imm": "0x890b8201", "imm2": "0x97c36bf3", "rot": 17, "bit": 22, "mask": 4},
|
||||
{"i": 254, "op": "add", "dst": 0, "src": 6, "src2": 0, "imm": "0x784a302b", "imm2": "0xb83d78de", "rot": 27, "bit": 16, "mask": 4},
|
||||
{"i": 255, "op": "sub", "dst": 2, "src": 4, "src2": 0, "imm": "0x817adb38", "imm2": "0xf3a3534b", "rot": 7, "bit": 22, "mask": 4}
|
||||
]},
|
||||
"instructions": [
|
||||
{"i": 0, "op": "mad", "dst": 2, "src": 3, "src2": 4, "imm": "0xbf7b174d", "imm2": "0x337b762e", "rot": 17, "bit": 2, "mask": 2, "width": 1},
|
||||
{"i": 1, "op": "mad", "dst": 2, "src": 1, "src2": 1, "imm": "0xdd04a5da", "imm2": "0x42da7657", "rot": 15, "bit": 30, "mask": 16, "width": 1},
|
||||
{"i": 2, "op": "mad", "dst": 2, "src": 3, "src2": 2, "imm": "0x734003fa", "imm2": "0x5bb67700", "rot": 3, "bit": 20, "mask": 1, "width": 1},
|
||||
{"i": 3, "op": "xor", "dst": 3, "src": 5, "src2": 5, "imm": "0xc55a1b1c", "imm2": "0xa19720f3", "rot": 7, "bit": 8, "mask": 1, "width": 1},
|
||||
{"i": 4, "op": "load", "dst": 7, "src": 2, "src2": 5, "imm": "0xad572dd7", "imm2": "0x9ceb3ea7", "rot": 18, "bit": 30, "mask": 2, "width": 1},
|
||||
{"i": 5, "op": "load", "dst": 5, "src": 7, "src2": 3, "imm": "0x769a53be", "imm2": "0x80f9067e", "rot": 12, "bit": 22, "mask": 1, "width": 1},
|
||||
{"i": 6, "op": "shfl", "dst": 1, "src": 4, "src2": 0, "imm": "0xd3613d88", "imm2": "0x262fb219", "rot": 10, "bit": 30, "mask": 8, "width": 1},
|
||||
{"i": 7, "op": "shfl", "dst": 7, "src": 3, "src2": 4, "imm": "0xce38e42f", "imm2": "0xb868b818", "rot": 11, "bit": 8, "mask": 8, "width": 1},
|
||||
{"i": 8, "op": "mulhi", "dst": 1, "src": 5, "src2": 2, "imm": "0xa5eebca5", "imm2": "0x5703a72b", "rot": 13, "bit": 13, "mask": 16, "width": 1},
|
||||
{"i": 9, "op": "rotr", "dst": 6, "src": 3, "src2": 4, "imm": "0x17a5a9c7", "imm2": "0xdcfb93a1", "rot": 20, "bit": 27, "mask": 2, "width": 1},
|
||||
{"i": 10, "op": "or", "dst": 3, "src": 4, "src2": 1, "imm": "0xccb7d785", "imm2": "0xc335364c", "rot": 14, "bit": 12, "mask": 4, "width": 1},
|
||||
{"i": 11, "op": "load", "dst": 4, "src": 6, "src2": 2, "imm": "0x88cb9af3", "imm2": "0x4e7dc10d", "rot": 24, "bit": 17, "mask": 4, "width": 1},
|
||||
{"i": 12, "op": "mulhi", "dst": 0, "src": 4, "src2": 4, "imm": "0x45374321", "imm2": "0x3cd91989", "rot": 11, "bit": 4, "mask": 2, "width": 1},
|
||||
{"i": 13, "op": "add", "dst": 5, "src": 1, "src2": 2, "imm": "0xc7934706", "imm2": "0xd3177981", "rot": 16, "bit": 30, "mask": 2, "width": 1},
|
||||
{"i": 14, "op": "load", "dst": 0, "src": 5, "src2": 3, "imm": "0xfd7f56bb", "imm2": "0x65e14f52", "rot": 13, "bit": 22, "mask": 2, "width": 1},
|
||||
{"i": 15, "op": "sub", "dst": 2, "src": 4, "src2": 4, "imm": "0x35a80b49", "imm2": "0x060f2d13", "rot": 16, "bit": 20, "mask": 16, "width": 1},
|
||||
{"i": 16, "op": "load", "dst": 2, "src": 7, "src2": 2, "imm": "0xae0a32c2", "imm2": "0x4c2a4cfe", "rot": 8, "bit": 31, "mask": 16, "width": 1},
|
||||
{"i": 17, "op": "load", "dst": 7, "src": 4, "src2": 6, "imm": "0x82a84cc3", "imm2": "0x21a38d68", "rot": 15, "bit": 21, "mask": 2, "width": 1},
|
||||
{"i": 18, "op": "shfl", "dst": 7, "src": 3, "src2": 3, "imm": "0xa3818806", "imm2": "0x8f66b5c8", "rot": 14, "bit": 6, "mask": 4, "width": 1},
|
||||
{"i": 19, "op": "mul", "dst": 5, "src": 0, "src2": 1, "imm": "0xa00de107", "imm2": "0x77bfcaa5", "rot": 3, "bit": 10, "mask": 2, "width": 1},
|
||||
{"i": 20, "op": "shfl", "dst": 3, "src": 4, "src2": 7, "imm": "0x1d2b8cab", "imm2": "0x80b4f9a2", "rot": 14, "bit": 25, "mask": 2, "width": 1},
|
||||
{"i": 21, "op": "shfl", "dst": 2, "src": 4, "src2": 5, "imm": "0x3ac915d2", "imm2": "0x5fba7bc2", "rot": 16, "bit": 1, "mask": 16, "width": 1},
|
||||
{"i": 22, "op": "mulhi", "dst": 6, "src": 2, "src2": 0, "imm": "0xdc3ec8fd", "imm2": "0x599e2fa3", "rot": 22, "bit": 3, "mask": 2, "width": 1},
|
||||
{"i": 23, "op": "load", "dst": 6, "src": 3, "src2": 5, "imm": "0x2a6b16d5", "imm2": "0xd73e396f", "rot": 28, "bit": 29, "mask": 2, "width": 1},
|
||||
{"i": 24, "op": "mul", "dst": 5, "src": 0, "src2": 7, "imm": "0x376d0223", "imm2": "0xe1c2169a", "rot": 4, "bit": 16, "mask": 16, "width": 1},
|
||||
{"i": 25, "op": "rotl", "dst": 5, "src": 7, "src2": 3, "imm": "0x78ad8c60", "imm2": "0x6f5b77d5", "rot": 19, "bit": 11, "mask": 16, "width": 1},
|
||||
{"i": 26, "op": "shfl", "dst": 7, "src": 6, "src2": 5, "imm": "0x93915b9f", "imm2": "0x1e61fb6b", "rot": 28, "bit": 23, "mask": 2, "width": 1},
|
||||
{"i": 27, "op": "xor", "dst": 0, "src": 5, "src2": 7, "imm": "0x6378fe15", "imm2": "0x66c78f42", "rot": 12, "bit": 31, "mask": 8, "width": 1},
|
||||
{"i": 28, "op": "xor", "dst": 0, "src": 4, "src2": 7, "imm": "0x20a57fda", "imm2": "0x088c848e", "rot": 16, "bit": 13, "mask": 4, "width": 1},
|
||||
{"i": 29, "op": "sub", "dst": 3, "src": 0, "src2": 2, "imm": "0x49d95fd5", "imm2": "0x1a5f946a", "rot": 6, "bit": 12, "mask": 1, "width": 1},
|
||||
{"i": 30, "op": "mul", "dst": 5, "src": 1, "src2": 7, "imm": "0x0a816217", "imm2": "0x405c4f73", "rot": 13, "bit": 27, "mask": 4, "width": 1},
|
||||
{"i": 31, "op": "load", "dst": 7, "src": 3, "src2": 2, "imm": "0x09ed045e", "imm2": "0xd69c4715", "rot": 5, "bit": 9, "mask": 2, "width": 1},
|
||||
{"i": 32, "op": "load", "dst": 1, "src": 2, "src2": 6, "imm": "0xeb79ea49", "imm2": "0xcc587f5a", "rot": 6, "bit": 8, "mask": 16, "width": 1},
|
||||
{"i": 33, "op": "xor", "dst": 5, "src": 6, "src2": 1, "imm": "0x3027401e", "imm2": "0x5f20c27e", "rot": 18, "bit": 9, "mask": 2, "width": 1},
|
||||
{"i": 34, "op": "load", "dst": 5, "src": 0, "src2": 3, "imm": "0x0e1cab07", "imm2": "0x09356c5b", "rot": 19, "bit": 31, "mask": 1, "width": 1},
|
||||
{"i": 35, "op": "mulhi", "dst": 0, "src": 5, "src2": 2, "imm": "0x90e31357", "imm2": "0xabd32484", "rot": 26, "bit": 5, "mask": 8, "width": 1},
|
||||
{"i": 36, "op": "shfl", "dst": 5, "src": 2, "src2": 5, "imm": "0xee9a955f", "imm2": "0x31b3faed", "rot": 8, "bit": 24, "mask": 4, "width": 1},
|
||||
{"i": 37, "op": "load", "dst": 7, "src": 6, "src2": 7, "imm": "0x3ba2f832", "imm2": "0x1160dcd3", "rot": 4, "bit": 29, "mask": 1, "width": 1},
|
||||
{"i": 38, "op": "add", "dst": 3, "src": 1, "src2": 7, "imm": "0x75ba2fad", "imm2": "0x230c005c", "rot": 4, "bit": 27, "mask": 1, "width": 1},
|
||||
{"i": 39, "op": "shfl", "dst": 1, "src": 5, "src2": 3, "imm": "0xdbf37e75", "imm2": "0xb5ac1969", "rot": 30, "bit": 13, "mask": 4, "width": 1},
|
||||
{"i": 40, "op": "xor", "dst": 2, "src": 5, "src2": 5, "imm": "0x47f136c5", "imm2": "0x06ce9153", "rot": 19, "bit": 10, "mask": 2, "width": 1},
|
||||
{"i": 41, "op": "mad", "dst": 3, "src": 6, "src2": 3, "imm": "0xce13eff8", "imm2": "0x04cc1d55", "rot": 3, "bit": 1, "mask": 4, "width": 1},
|
||||
{"i": 42, "op": "sub", "dst": 6, "src": 7, "src2": 1, "imm": "0x6a65ab71", "imm2": "0x8fbc1bcd", "rot": 4, "bit": 1, "mask": 8, "width": 1},
|
||||
{"i": 43, "op": "xor", "dst": 7, "src": 0, "src2": 7, "imm": "0xdaeb4928", "imm2": "0xc0423027", "rot": 24, "bit": 11, "mask": 8, "width": 1},
|
||||
{"i": 44, "op": "load", "dst": 1, "src": 7, "src2": 7, "imm": "0x778f01c9", "imm2": "0x28cedcea", "rot": 12, "bit": 4, "mask": 16, "width": 1},
|
||||
{"i": 45, "op": "mul", "dst": 2, "src": 3, "src2": 2, "imm": "0xf4264f1b", "imm2": "0x0f627d56", "rot": 5, "bit": 28, "mask": 8, "width": 1},
|
||||
{"i": 46, "op": "mulhi", "dst": 1, "src": 5, "src2": 1, "imm": "0xffb2147a", "imm2": "0xccde9b05", "rot": 13, "bit": 9, "mask": 2, "width": 1},
|
||||
{"i": 47, "op": "sub", "dst": 4, "src": 3, "src2": 5, "imm": "0x73b36234", "imm2": "0x3f5d5997", "rot": 7, "bit": 18, "mask": 2, "width": 1},
|
||||
{"i": 48, "op": "rotr", "dst": 2, "src": 6, "src2": 3, "imm": "0x3a4d9aa9", "imm2": "0x212bec7b", "rot": 4, "bit": 29, "mask": 16, "width": 1},
|
||||
{"i": 49, "op": "load", "dst": 3, "src": 4, "src2": 2, "imm": "0x626f11df", "imm2": "0x56cd5bfd", "rot": 7, "bit": 1, "mask": 1, "width": 1},
|
||||
{"i": 50, "op": "add", "dst": 1, "src": 5, "src2": 6, "imm": "0x81b8bc2c", "imm2": "0x1907970c", "rot": 28, "bit": 7, "mask": 4, "width": 1},
|
||||
{"i": 51, "op": "mul", "dst": 0, "src": 2, "src2": 2, "imm": "0xa8848b30", "imm2": "0xef6ac348", "rot": 9, "bit": 15, "mask": 8, "width": 1},
|
||||
{"i": 52, "op": "add", "dst": 0, "src": 2, "src2": 0, "imm": "0x4f92b968", "imm2": "0x699fd448", "rot": 22, "bit": 6, "mask": 4, "width": 1},
|
||||
{"i": 53, "op": "add", "dst": 1, "src": 0, "src2": 2, "imm": "0x2bb965af", "imm2": "0x77b1520d", "rot": 2, "bit": 12, "mask": 8, "width": 1},
|
||||
{"i": 54, "op": "rotl", "dst": 7, "src": 1, "src2": 0, "imm": "0x553e678b", "imm2": "0x3cc8eae0", "rot": 14, "bit": 20, "mask": 2, "width": 1},
|
||||
{"i": 55, "op": "add", "dst": 3, "src": 7, "src2": 2, "imm": "0x7b0fe07a", "imm2": "0xa54c55a0", "rot": 10, "bit": 1, "mask": 1, "width": 1},
|
||||
{"i": 56, "op": "load", "dst": 6, "src": 7, "src2": 4, "imm": "0x01eba9aa", "imm2": "0x2758c0f7", "rot": 14, "bit": 15, "mask": 4, "width": 1},
|
||||
{"i": 57, "op": "rotr", "dst": 1, "src": 5, "src2": 2, "imm": "0x1f5267b3", "imm2": "0x236f5a27", "rot": 2, "bit": 31, "mask": 16, "width": 1},
|
||||
{"i": 58, "op": "load", "dst": 5, "src": 3, "src2": 3, "imm": "0xa9954a9b", "imm2": "0x6a54d4e8", "rot": 11, "bit": 10, "mask": 16, "width": 1},
|
||||
{"i": 59, "op": "load", "dst": 6, "src": 1, "src2": 4, "imm": "0x9923ff88", "imm2": "0x9357254e", "rot": 16, "bit": 1, "mask": 16, "width": 1},
|
||||
{"i": 60, "op": "mad", "dst": 3, "src": 5, "src2": 0, "imm": "0xc0cc51a6", "imm2": "0x3fd7701b", "rot": 20, "bit": 1, "mask": 4, "width": 1},
|
||||
{"i": 61, "op": "add", "dst": 5, "src": 7, "src2": 7, "imm": "0xaf9dd72d", "imm2": "0xad7493e7", "rot": 7, "bit": 31, "mask": 16, "width": 1},
|
||||
{"i": 62, "op": "add", "dst": 4, "src": 6, "src2": 2, "imm": "0x89841d87", "imm2": "0x1e07c3d9", "rot": 6, "bit": 27, "mask": 1, "width": 1},
|
||||
{"i": 63, "op": "rotl", "dst": 5, "src": 4, "src2": 2, "imm": "0xae210f8d", "imm2": "0x8e499ba4", "rot": 19, "bit": 9, "mask": 1, "width": 1}
|
||||
]
|
||||
}
|
||||
368
proto-cuda/packs-ca3-v5/v5-genesis/program.metal
Normal file
368
proto-cuda/packs-ca3-v5/v5-genesis/program.metal
Normal file
|
|
@ -0,0 +1,368 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
|
||||
#define MASK 0x0fffffffu
|
||||
constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u };
|
||||
|
||||
inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
|
||||
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
|
||||
device ulong* out [[buffer(1)]],
|
||||
constant uint& baseNonce [[buffer(2)]],
|
||||
uint gid [[thread_position_in_grid]]) {
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint x = nonce ^ SEEDW[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ SEEDW[1]; }
|
||||
{ uint x = nonce ^ SEEDW[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ SEEDW[2]; }
|
||||
{ uint x = nonce ^ SEEDW[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ SEEDW[3]; }
|
||||
{ uint x = nonce ^ SEEDW[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ SEEDW[4]; }
|
||||
{ uint x = nonce ^ SEEDW[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ SEEDW[5]; }
|
||||
{ uint x = nonce ^ SEEDW[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ SEEDW[6]; }
|
||||
{ uint x = nonce ^ SEEDW[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ SEEDW[7]; }
|
||||
{ uint x = nonce ^ SEEDW[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ SEEDW[0]; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0
|
||||
r2 = r1 * r1 + r2; // 1
|
||||
r2 = r3 * r2 + r2; // 2
|
||||
r3 = r3 ^ r5; // 3
|
||||
r7 = r7 ^ dataset[r2 & MASK]; // 4
|
||||
r5 = r5 ^ dataset[r7 & MASK]; // 5
|
||||
r1 = r1 ^ simd_shuffle_xor(r4, (ushort)8); // 6
|
||||
r7 = r7 ^ simd_shuffle_xor(r3, (ushort)8); // 7
|
||||
r1 = mulhi(r1, r5); // 8
|
||||
r6 = rotr_var(r6, r3); // 9
|
||||
r3 = r3 | r4; // 10
|
||||
r4 = r4 ^ dataset[r6 & MASK]; // 11
|
||||
r0 = mulhi(r0, r4); // 12
|
||||
r5 = r5 + r1 + select(0xc7934706u, 0xd3177981u, ((sel >> 30u) & 1u) != 0u); // 13
|
||||
r0 = r0 ^ dataset[r5 & MASK]; // 14
|
||||
r2 = r2 - r4; // 15
|
||||
r2 = r2 ^ dataset[r7 & MASK]; // 16
|
||||
r7 = r7 ^ dataset[r4 & MASK]; // 17
|
||||
r7 = r7 ^ simd_shuffle_xor(r3, (ushort)4); // 18
|
||||
r5 = r5 * r0; // 19
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 20
|
||||
r2 = r2 ^ simd_shuffle_xor(r4, (ushort)16); // 21
|
||||
r6 = mulhi(r6, r2); // 22
|
||||
r6 = r6 ^ dataset[r3 & MASK]; // 23
|
||||
r5 = r5 * r0; // 24
|
||||
r5 = rotl_imm(r5, 19u); // 25
|
||||
r7 = r7 ^ simd_shuffle_xor(r6, (ushort)2); // 26
|
||||
r0 = r0 ^ r5; // 27
|
||||
r0 = r0 ^ r4; // 28
|
||||
r3 = r3 - r0; // 29
|
||||
r5 = r5 * r1; // 30
|
||||
r7 = r7 ^ dataset[r3 & MASK]; // 31
|
||||
r1 = r1 ^ dataset[r2 & MASK]; // 32
|
||||
r5 = r5 ^ r6; // 33
|
||||
r5 = r5 ^ dataset[r0 & MASK]; // 34
|
||||
r0 = mulhi(r0, r5); // 35
|
||||
r5 = r5 ^ simd_shuffle_xor(r2, (ushort)4); // 36
|
||||
r7 = r7 ^ dataset[r6 & MASK]; // 37
|
||||
r3 = r3 + r1 + select(0x75ba2fadu, 0x230c005cu, ((sel >> 27u) & 1u) != 0u); // 38
|
||||
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)4); // 39
|
||||
r2 = r2 ^ r5; // 40
|
||||
r3 = r6 * r3 + r3; // 41
|
||||
r6 = r6 - r7; // 42
|
||||
r7 = r7 ^ r0; // 43
|
||||
r1 = r1 ^ dataset[r7 & MASK]; // 44
|
||||
r2 = r2 * r3; // 45
|
||||
r1 = mulhi(r1, r5); // 46
|
||||
r4 = r4 - r3; // 47
|
||||
r2 = rotr_var(r2, r6); // 48
|
||||
r3 = r3 ^ dataset[r4 & MASK]; // 49
|
||||
r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 50
|
||||
r0 = r0 * r2; // 51
|
||||
r0 = r0 + r2 + select(0x4f92b968u, 0x699fd448u, ((sel >> 6u) & 1u) != 0u); // 52
|
||||
r1 = r1 + r0 + select(0x2bb965afu, 0x77b1520du, ((sel >> 12u) & 1u) != 0u); // 53
|
||||
r7 = rotl_imm(r7, 14u); // 54
|
||||
r3 = r3 + r7 + select(0x7b0fe07au, 0xa54c55a0u, ((sel >> 1u) & 1u) != 0u); // 55
|
||||
r6 = r6 ^ dataset[r7 & MASK]; // 56
|
||||
r1 = rotr_var(r1, r5); // 57
|
||||
r5 = r5 ^ dataset[r3 & MASK]; // 58
|
||||
r6 = r6 ^ dataset[r1 & MASK]; // 59
|
||||
r3 = r5 * r0 + r3; // 60
|
||||
r5 = r5 + r7 + select(0xaf9dd72du, 0xad7493e7u, ((sel >> 31u) & 1u) != 0u); // 61
|
||||
r4 = r4 + r6 + select(0x89841d87u, 0x1e07c3d9u, ((sel >> 27u) & 1u) != 0u); // 62
|
||||
r5 = rotl_imm(r5, 19u); // 63
|
||||
// latency-shadow block (Counter ASIC 3.0 item 8): 256 ALU instructions x 27 passes after instruction 63, no load
|
||||
for (uint sh = 0u; sh < 27u; ++sh) {
|
||||
r5 = r5 + r2 + select(0x92199f99u, 0x8bc12da9u, ((sel >> 18u) & 1u) != 0u); // s0 add
|
||||
r0 = r0 + r7 + select(0x8d72d3adu, 0x63079e5au, ((sel >> 26u) & 1u) != 0u); // s1 add
|
||||
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)2); // s2 shfl
|
||||
r4 = r4 - r2; // s3 sub
|
||||
r7 = r7 + r0 + select(0xb21b4babu, 0x5d4c7a60u, ((sel >> 31u) & 1u) != 0u); // s4 add
|
||||
r0 = rotl_imm(r0, 11u); // s5 rotl
|
||||
r0 = r0 + r1 + select(0x2fe0e98bu, 0xc88e2942u, ((sel >> 16u) & 1u) != 0u); // s6 add
|
||||
r7 = r7 ^ r1; // s7 xor
|
||||
r1 = r6 * r5 + r1; // s8 mad
|
||||
r6 = r6 ^ simd_shuffle_xor(r1, (ushort)4); // s9 shfl
|
||||
r1 = r2 * r2 + r1; // s10 mad
|
||||
r5 = r0 * r3 + r5; // s11 mad
|
||||
r2 = r2 ^ simd_shuffle_xor(r6, (ushort)4); // s12 shfl
|
||||
r1 = r1 + r5 + select(0xbdf8f9a5u, 0xbc48c63eu, ((sel >> 25u) & 1u) != 0u); // s13 add
|
||||
r1 = rotl_imm(r1, 29u); // s14 rotl
|
||||
r1 = r1 - r4; // s15 sub
|
||||
r7 = r7 | r1; // s16 or
|
||||
r2 = r2 ^ simd_shuffle_xor(r4, (ushort)8); // s17 shfl
|
||||
r7 = r7 ^ r4; // s18 xor
|
||||
r6 = r6 * r1; // s19 mul
|
||||
r5 = r6 * r0 + r5; // s20 mad
|
||||
r3 = r3 - r1; // s21 sub
|
||||
r6 = r6 * r0; // s22 mul
|
||||
r2 = r2 + r0 + select(0x45c37cecu, 0x96e8f127u, ((sel >> 1u) & 1u) != 0u); // s23 add
|
||||
r6 = r6 - r4; // s24 sub
|
||||
r7 = r3 * r4 + r7; // s25 mad
|
||||
r3 = rotl_imm(r3, 9u); // s26 rotl
|
||||
r2 = r2 - r1; // s27 sub
|
||||
r6 = mulhi(r6, r0); // s28 mulhi
|
||||
r2 = r2 ^ simd_shuffle_xor(r4, (ushort)2); // s29 shfl
|
||||
r1 = r1 ^ simd_shuffle_xor(r6, (ushort)1); // s30 shfl
|
||||
r1 = r1 + r6 + select(0x37985632u, 0xb1cdb2abu, ((sel >> 1u) & 1u) != 0u); // s31 add
|
||||
r0 = rotr_var(r0, r1); // s32 rotr
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // s33 shfl
|
||||
r0 = r0 ^ simd_shuffle_xor(r3, (ushort)4); // s34 shfl
|
||||
r7 = r7 * r0; // s35 mul
|
||||
r3 = r3 + r1 + select(0x804e777eu, 0x856e0180u, ((sel >> 0u) & 1u) != 0u); // s36 add
|
||||
r1 = r1 ^ r6; // s37 xor
|
||||
r2 = r2 * r7; // s38 mul
|
||||
r6 = r6 + r5 + select(0x0b06c8a7u, 0xe9bd0cc4u, ((sel >> 2u) & 1u) != 0u); // s39 add
|
||||
r5 = r5 * r7; // s40 mul
|
||||
r2 = mulhi(r2, r3); // s41 mulhi
|
||||
r2 = r2 ^ r0; // s42 xor
|
||||
r0 = rotl_imm(r0, 4u); // s43 rotl
|
||||
r4 = mulhi(r4, r3); // s44 mulhi
|
||||
r6 = r6 + r7 + select(0xee822e17u, 0xcdcb63f6u, ((sel >> 12u) & 1u) != 0u); // s45 add
|
||||
r3 = r2 * r0 + r3; // s46 mad
|
||||
r4 = r4 ^ simd_shuffle_xor(r3, (ushort)8); // s47 shfl
|
||||
r6 = mulhi(r6, r5); // s48 mulhi
|
||||
r0 = r0 - r2; // s49 sub
|
||||
r3 = r3 - r5; // s50 sub
|
||||
r1 = rotr_var(r1, r4); // s51 rotr
|
||||
r6 = r7 * r7 + r6; // s52 mad
|
||||
r5 = r5 ^ r3; // s53 xor
|
||||
r1 = r1 - r0; // s54 sub
|
||||
r5 = r5 - r6; // s55 sub
|
||||
r3 = r3 + r2 + select(0x3dfad1b6u, 0xd4758987u, ((sel >> 10u) & 1u) != 0u); // s56 add
|
||||
r4 = r4 + r1 + select(0xfca75bc2u, 0x0602d6beu, ((sel >> 0u) & 1u) != 0u); // s57 add
|
||||
r5 = mulhi(r5, r6); // s58 mulhi
|
||||
r2 = r2 ^ simd_shuffle_xor(r1, (ushort)2); // s59 shfl
|
||||
r2 = rotl_imm(r2, 9u); // s60 rotl
|
||||
r4 = r4 | r6; // s61 or
|
||||
r6 = rotr_var(r6, r4); // s62 rotr
|
||||
r2 = r2 * r4; // s63 mul
|
||||
r0 = r0 ^ r5; // s64 xor
|
||||
r2 = r2 ^ r7; // s65 xor
|
||||
r2 = r2 + r1 + select(0xd71c02ffu, 0x8596687au, ((sel >> 6u) & 1u) != 0u); // s66 add
|
||||
r1 = rotl_imm(r1, 21u); // s67 rotl
|
||||
r3 = r3 * r2; // s68 mul
|
||||
r7 = r7 ^ simd_shuffle_xor(r5, (ushort)2); // s69 shfl
|
||||
r3 = r3 * r2; // s70 mul
|
||||
r0 = r2 * r5 + r0; // s71 mad
|
||||
r6 = r6 + r7 + select(0x3c6fe15du, 0x08ac5733u, ((sel >> 0u) & 1u) != 0u); // s72 add
|
||||
r3 = r3 ^ simd_shuffle_xor(r2, (ushort)8); // s73 shfl
|
||||
r3 = r3 * r6; // s74 mul
|
||||
r6 = r6 ^ r7; // s75 xor
|
||||
r3 = r3 | r0; // s76 or
|
||||
r2 = r2 + r4 + select(0x295d5faeu, 0xc100b495u, ((sel >> 1u) & 1u) != 0u); // s77 add
|
||||
r3 = mulhi(r3, r7); // s78 mulhi
|
||||
r4 = r4 | r1; // s79 or
|
||||
r4 = rotr_var(r4, r3); // s80 rotr
|
||||
r4 = r4 + r3 + select(0x274a9221u, 0x5cc59530u, ((sel >> 15u) & 1u) != 0u); // s81 add
|
||||
r7 = r7 + r3 + select(0xc8651f8eu, 0x141479ecu, ((sel >> 5u) & 1u) != 0u); // s82 add
|
||||
r3 = rotr_var(r3, r6); // s83 rotr
|
||||
r2 = r4 * r7 + r2; // s84 mad
|
||||
r2 = r2 ^ simd_shuffle_xor(r5, (ushort)16); // s85 shfl
|
||||
r3 = r3 ^ r2; // s86 xor
|
||||
r5 = r5 ^ simd_shuffle_xor(r7, (ushort)2); // s87 shfl
|
||||
r0 = r0 ^ r4; // s88 xor
|
||||
r3 = r3 + r2 + select(0x53f915c2u, 0x883c0c92u, ((sel >> 18u) & 1u) != 0u); // s89 add
|
||||
r7 = rotr_var(r7, r5); // s90 rotr
|
||||
r3 = r3 + r1 + select(0xe2be00a5u, 0x3cc5bd20u, ((sel >> 31u) & 1u) != 0u); // s91 add
|
||||
r7 = r7 ^ simd_shuffle_xor(r2, (ushort)1); // s92 shfl
|
||||
r6 = rotr_var(r6, r2); // s93 rotr
|
||||
r7 = rotl_imm(r7, 14u); // s94 rotl
|
||||
r4 = r5 * r5 + r4; // s95 mad
|
||||
r2 = r4 * r5 + r2; // s96 mad
|
||||
r1 = r1 ^ r0; // s97 xor
|
||||
r5 = r5 * r4; // s98 mul
|
||||
r2 = r2 - r0; // s99 sub
|
||||
r7 = rotl_imm(r7, 30u); // s100 rotl
|
||||
r5 = r5 + r6 + select(0x423fd9c9u, 0xbfd646cbu, ((sel >> 17u) & 1u) != 0u); // s101 add
|
||||
r7 = r7 + r6 + select(0x8d3c011du, 0x19b74a43u, ((sel >> 11u) & 1u) != 0u); // s102 add
|
||||
r5 = rotl_imm(r5, 6u); // s103 rotl
|
||||
r0 = r0 * r4; // s104 mul
|
||||
r0 = r0 | r5; // s105 or
|
||||
r0 = r0 + r1 + select(0x8ce14721u, 0x7dcb7e18u, ((sel >> 15u) & 1u) != 0u); // s106 add
|
||||
r0 = rotl_imm(r0, 15u); // s107 rotl
|
||||
r4 = r4 - r2; // s108 sub
|
||||
r2 = mulhi(r2, r0); // s109 mulhi
|
||||
r1 = mulhi(r1, r0); // s110 mulhi
|
||||
r0 = r0 + r2 + select(0x2d9fc6b9u, 0x3c179ad8u, ((sel >> 28u) & 1u) != 0u); // s111 add
|
||||
r2 = mulhi(r2, r0); // s112 mulhi
|
||||
r6 = r6 - r3; // s113 sub
|
||||
r7 = r6 * r3 + r7; // s114 mad
|
||||
r5 = rotr_var(r5, r1); // s115 rotr
|
||||
r6 = rotr_var(r6, r4); // s116 rotr
|
||||
r2 = r2 + r1 + select(0x502138e2u, 0x47359729u, ((sel >> 22u) & 1u) != 0u); // s117 add
|
||||
r3 = r2 * r0 + r3; // s118 mad
|
||||
r1 = r1 * r3; // s119 mul
|
||||
r2 = r0 * r4 + r2; // s120 mad
|
||||
r0 = r0 * r3; // s121 mul
|
||||
r2 = mulhi(r2, r0); // s122 mulhi
|
||||
r5 = r5 * r6; // s123 mul
|
||||
r4 = r2 * r4 + r4; // s124 mad
|
||||
r4 = r4 ^ simd_shuffle_xor(r2, (ushort)2); // s125 shfl
|
||||
r2 = r2 ^ r0; // s126 xor
|
||||
r1 = rotl_imm(r1, 7u); // s127 rotl
|
||||
r7 = r7 ^ simd_shuffle_xor(r2, (ushort)4); // s128 shfl
|
||||
r6 = rotl_imm(r6, 17u); // s129 rotl
|
||||
r2 = r2 + r5 + select(0x33dff776u, 0x6c57e4e7u, ((sel >> 15u) & 1u) != 0u); // s130 add
|
||||
r0 = r0 | r3; // s131 or
|
||||
r5 = rotr_var(r5, r0); // s132 rotr
|
||||
r2 = r2 + r3 + select(0x5db25b34u, 0xe8212c0cu, ((sel >> 30u) & 1u) != 0u); // s133 add
|
||||
r4 = r4 ^ r3; // s134 xor
|
||||
r6 = r6 ^ r1; // s135 xor
|
||||
r1 = r1 - r7; // s136 sub
|
||||
r2 = r6 * r2 + r2; // s137 mad
|
||||
r0 = mulhi(r0, r5); // s138 mulhi
|
||||
r2 = r2 + r7 + select(0x218d4090u, 0xd44ff710u, ((sel >> 10u) & 1u) != 0u); // s139 add
|
||||
r1 = rotr_var(r1, r4); // s140 rotr
|
||||
r3 = r3 ^ simd_shuffle_xor(r6, (ushort)1); // s141 shfl
|
||||
r7 = r6 * r2 + r7; // s142 mad
|
||||
r2 = r2 * r3; // s143 mul
|
||||
r7 = r7 + r4 + select(0xf4b1a8deu, 0xb98942fau, ((sel >> 29u) & 1u) != 0u); // s144 add
|
||||
r7 = rotl_imm(r7, 15u); // s145 rotl
|
||||
r7 = r7 ^ r5; // s146 xor
|
||||
r4 = r7 * r4 + r4; // s147 mad
|
||||
r6 = rotr_var(r6, r5); // s148 rotr
|
||||
r1 = r2 * r4 + r1; // s149 mad
|
||||
r1 = r1 + r7 + select(0x2bef10f2u, 0x0d48ba42u, ((sel >> 17u) & 1u) != 0u); // s150 add
|
||||
r5 = r5 ^ r4; // s151 xor
|
||||
r7 = r7 + r0 + select(0xdc5cc080u, 0xc98dea9cu, ((sel >> 30u) & 1u) != 0u); // s152 add
|
||||
r4 = r4 * r6; // s153 mul
|
||||
r0 = r0 * r2; // s154 mul
|
||||
r6 = r6 ^ r5; // s155 xor
|
||||
r4 = rotr_var(r4, r2); // s156 rotr
|
||||
r1 = rotl_imm(r1, 11u); // s157 rotl
|
||||
r5 = r5 + r0 + select(0x18197438u, 0x6c752dcbu, ((sel >> 11u) & 1u) != 0u); // s158 add
|
||||
r4 = r1 * r5 + r4; // s159 mad
|
||||
r4 = rotl_imm(r4, 26u); // s160 rotl
|
||||
r3 = r0 * r7 + r3; // s161 mad
|
||||
r3 = rotr_var(r3, r1); // s162 rotr
|
||||
r4 = r4 + r0 + select(0xf1c46574u, 0x8e481727u, ((sel >> 3u) & 1u) != 0u); // s163 add
|
||||
r0 = mulhi(r0, r4); // s164 mulhi
|
||||
r2 = r2 + r6 + select(0xac578137u, 0x550ab406u, ((sel >> 20u) & 1u) != 0u); // s165 add
|
||||
r0 = rotl_imm(r0, 13u); // s166 rotl
|
||||
r3 = r3 + r1 + select(0xa33e6706u, 0xaebb5966u, ((sel >> 14u) & 1u) != 0u); // s167 add
|
||||
r3 = r3 | r5; // s168 or
|
||||
r6 = rotr_var(r6, r2); // s169 rotr
|
||||
r4 = r4 ^ r6; // s170 xor
|
||||
r6 = r6 - r1; // s171 sub
|
||||
r7 = rotl_imm(r7, 22u); // s172 rotl
|
||||
r5 = rotl_imm(r5, 15u); // s173 rotl
|
||||
r7 = r7 ^ simd_shuffle_xor(r0, (ushort)8); // s174 shfl
|
||||
r0 = r0 ^ r5; // s175 xor
|
||||
r7 = rotl_imm(r7, 6u); // s176 rotl
|
||||
r7 = r7 - r0; // s177 sub
|
||||
r3 = rotl_imm(r3, 30u); // s178 rotl
|
||||
r7 = r6 * r1 + r7; // s179 mad
|
||||
r6 = rotl_imm(r6, 9u); // s180 rotl
|
||||
r2 = r2 ^ r4; // s181 xor
|
||||
r2 = r2 ^ r7; // s182 xor
|
||||
r7 = r7 ^ r2; // s183 xor
|
||||
r1 = r1 + r2 + select(0xd94d55acu, 0x5bb7550fu, ((sel >> 21u) & 1u) != 0u); // s184 add
|
||||
r3 = r3 | r5; // s185 or
|
||||
r6 = mulhi(r6, r3); // s186 mulhi
|
||||
r4 = r4 | r0; // s187 or
|
||||
r7 = r7 ^ simd_shuffle_xor(r1, (ushort)2); // s188 shfl
|
||||
r6 = r6 + r5 + select(0x89e747feu, 0x2a354e2du, ((sel >> 6u) & 1u) != 0u); // s189 add
|
||||
r1 = r1 ^ r4; // s190 xor
|
||||
r7 = mulhi(r7, r4); // s191 mulhi
|
||||
r2 = rotl_imm(r2, 26u); // s192 rotl
|
||||
r5 = rotl_imm(r5, 8u); // s193 rotl
|
||||
r4 = r4 ^ simd_shuffle_xor(r5, (ushort)16); // s194 shfl
|
||||
r4 = r4 ^ r5; // s195 xor
|
||||
r1 = r1 * r3; // s196 mul
|
||||
r5 = r5 + r2 + select(0xea03e8e7u, 0x10cfdc71u, ((sel >> 7u) & 1u) != 0u); // s197 add
|
||||
r7 = r7 + r5 + select(0xa7ee0102u, 0x66e148d3u, ((sel >> 15u) & 1u) != 0u); // s198 add
|
||||
r5 = r5 - r2; // s199 sub
|
||||
r5 = r5 ^ simd_shuffle_xor(r1, (ushort)2); // s200 shfl
|
||||
r7 = r7 ^ simd_shuffle_xor(r1, (ushort)8); // s201 shfl
|
||||
r4 = rotr_var(r4, r5); // s202 rotr
|
||||
r7 = r7 ^ r5; // s203 xor
|
||||
r7 = r7 ^ r0; // s204 xor
|
||||
r6 = r6 + r2 + select(0x8379a4deu, 0x8558b619u, ((sel >> 10u) & 1u) != 0u); // s205 add
|
||||
r4 = rotl_imm(r4, 13u); // s206 rotl
|
||||
r1 = mulhi(r1, r3); // s207 mulhi
|
||||
r1 = r4 * r4 + r1; // s208 mad
|
||||
r2 = r2 ^ simd_shuffle_xor(r0, (ushort)4); // s209 shfl
|
||||
r7 = r7 + r0 + select(0xaa8cb14eu, 0xf4049c4cu, ((sel >> 11u) & 1u) != 0u); // s210 add
|
||||
r7 = r7 ^ simd_shuffle_xor(r1, (ushort)16); // s211 shfl
|
||||
r1 = r1 ^ r0; // s212 xor
|
||||
r7 = rotl_imm(r7, 3u); // s213 rotl
|
||||
r4 = rotr_var(r4, r7); // s214 rotr
|
||||
r3 = r3 ^ simd_shuffle_xor(r2, (ushort)16); // s215 shfl
|
||||
r5 = r5 | r1; // s216 or
|
||||
r1 = r1 - r2; // s217 sub
|
||||
r6 = r6 - r5; // s218 sub
|
||||
r6 = rotl_imm(r6, 4u); // s219 rotl
|
||||
r2 = mulhi(r2, r0); // s220 mulhi
|
||||
r2 = r2 | r0; // s221 or
|
||||
r5 = r5 ^ simd_shuffle_xor(r2, (ushort)8); // s222 shfl
|
||||
r5 = mulhi(r5, r6); // s223 mulhi
|
||||
r0 = r0 - r6; // s224 sub
|
||||
r7 = rotl_imm(r7, 23u); // s225 rotl
|
||||
r4 = r4 | r2; // s226 or
|
||||
r2 = r2 * r4; // s227 mul
|
||||
r3 = rotl_imm(r3, 12u); // s228 rotl
|
||||
r0 = rotr_var(r0, r4); // s229 rotr
|
||||
r0 = r0 + r6 + select(0x1d176220u, 0x2a7fecb2u, ((sel >> 16u) & 1u) != 0u); // s230 add
|
||||
r1 = r1 ^ simd_shuffle_xor(r2, (ushort)4); // s231 shfl
|
||||
r2 = r2 * r1; // s232 mul
|
||||
r7 = r0 * r1 + r7; // s233 mad
|
||||
r5 = rotl_imm(r5, 22u); // s234 rotl
|
||||
r4 = rotr_var(r4, r6); // s235 rotr
|
||||
r0 = r5 * r1 + r0; // s236 mad
|
||||
r6 = r6 ^ r4; // s237 xor
|
||||
r4 = r4 ^ r6; // s238 xor
|
||||
r6 = rotl_imm(r6, 18u); // s239 rotl
|
||||
r4 = r4 + r7 + select(0x17dafb4du, 0xadce39f3u, ((sel >> 25u) & 1u) != 0u); // s240 add
|
||||
r0 = r0 - r7; // s241 sub
|
||||
r1 = rotr_var(r1, r6); // s242 rotr
|
||||
r3 = r3 + r0 + select(0xf2f77d26u, 0x0e7033b6u, ((sel >> 29u) & 1u) != 0u); // s243 add
|
||||
r2 = r7 * r4 + r2; // s244 mad
|
||||
r4 = r0 * r6 + r4; // s245 mad
|
||||
r5 = r5 * r7; // s246 mul
|
||||
r3 = r3 + r5 + select(0xfed2da4eu, 0x7b2ff6b7u, ((sel >> 31u) & 1u) != 0u); // s247 add
|
||||
r0 = r0 + r7 + select(0x03b2891cu, 0xb5fad7b1u, ((sel >> 0u) & 1u) != 0u); // s248 add
|
||||
r5 = mulhi(r5, r4); // s249 mulhi
|
||||
r5 = r5 + r2 + select(0x042cd6e3u, 0xa7e71c2fu, ((sel >> 11u) & 1u) != 0u); // s250 add
|
||||
r6 = r6 ^ simd_shuffle_xor(r1, (ushort)1); // s251 shfl
|
||||
r6 = r6 ^ r1; // s252 xor
|
||||
r2 = r2 * r5; // s253 mul
|
||||
r0 = r0 + r6 + select(0x784a302bu, 0xb83d78deu, ((sel >> 16u) & 1u) != 0u); // s254 add
|
||||
r2 = r2 - r4; // s255 sub
|
||||
}
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
370
proto-cuda/packs-ca3-v5/v5-genesis/program_bound.metal
Normal file
370
proto-cuda/packs-ca3-v5/v5-genesis/program_bound.metal
Normal file
|
|
@ -0,0 +1,370 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
|
||||
#define MASK 0x0fffffffu
|
||||
constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u };
|
||||
|
||||
inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
|
||||
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW.
|
||||
kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
|
||||
device ulong* out [[buffer(1)]],
|
||||
constant uint& baseNonce [[buffer(2)]],
|
||||
constant uint* initw [[buffer(3)]],
|
||||
uint gid [[thread_position_in_grid]]) {
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint x = nonce ^ initw[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ initw[1]; }
|
||||
{ uint x = nonce ^ initw[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ initw[2]; }
|
||||
{ uint x = nonce ^ initw[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ initw[3]; }
|
||||
{ uint x = nonce ^ initw[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ initw[4]; }
|
||||
{ uint x = nonce ^ initw[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ initw[5]; }
|
||||
{ uint x = nonce ^ initw[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ initw[6]; }
|
||||
{ uint x = nonce ^ initw[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ initw[7]; }
|
||||
{ uint x = nonce ^ initw[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ initw[0]; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0
|
||||
r2 = r1 * r1 + r2; // 1
|
||||
r2 = r3 * r2 + r2; // 2
|
||||
r3 = r3 ^ r5; // 3
|
||||
r7 = r7 ^ dataset[r2 & MASK]; // 4
|
||||
r5 = r5 ^ dataset[r7 & MASK]; // 5
|
||||
r1 = r1 ^ simd_shuffle_xor(r4, (ushort)8); // 6
|
||||
r7 = r7 ^ simd_shuffle_xor(r3, (ushort)8); // 7
|
||||
r1 = mulhi(r1, r5); // 8
|
||||
r6 = rotr_var(r6, r3); // 9
|
||||
r3 = r3 | r4; // 10
|
||||
r4 = r4 ^ dataset[r6 & MASK]; // 11
|
||||
r0 = mulhi(r0, r4); // 12
|
||||
r5 = r5 + r1 + select(0xc7934706u, 0xd3177981u, ((sel >> 30u) & 1u) != 0u); // 13
|
||||
r0 = r0 ^ dataset[r5 & MASK]; // 14
|
||||
r2 = r2 - r4; // 15
|
||||
r2 = r2 ^ dataset[r7 & MASK]; // 16
|
||||
r7 = r7 ^ dataset[r4 & MASK]; // 17
|
||||
r7 = r7 ^ simd_shuffle_xor(r3, (ushort)4); // 18
|
||||
r5 = r5 * r0; // 19
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 20
|
||||
r2 = r2 ^ simd_shuffle_xor(r4, (ushort)16); // 21
|
||||
r6 = mulhi(r6, r2); // 22
|
||||
r6 = r6 ^ dataset[r3 & MASK]; // 23
|
||||
r5 = r5 * r0; // 24
|
||||
r5 = rotl_imm(r5, 19u); // 25
|
||||
r7 = r7 ^ simd_shuffle_xor(r6, (ushort)2); // 26
|
||||
r0 = r0 ^ r5; // 27
|
||||
r0 = r0 ^ r4; // 28
|
||||
r3 = r3 - r0; // 29
|
||||
r5 = r5 * r1; // 30
|
||||
r7 = r7 ^ dataset[r3 & MASK]; // 31
|
||||
r1 = r1 ^ dataset[r2 & MASK]; // 32
|
||||
r5 = r5 ^ r6; // 33
|
||||
r5 = r5 ^ dataset[r0 & MASK]; // 34
|
||||
r0 = mulhi(r0, r5); // 35
|
||||
r5 = r5 ^ simd_shuffle_xor(r2, (ushort)4); // 36
|
||||
r7 = r7 ^ dataset[r6 & MASK]; // 37
|
||||
r3 = r3 + r1 + select(0x75ba2fadu, 0x230c005cu, ((sel >> 27u) & 1u) != 0u); // 38
|
||||
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)4); // 39
|
||||
r2 = r2 ^ r5; // 40
|
||||
r3 = r6 * r3 + r3; // 41
|
||||
r6 = r6 - r7; // 42
|
||||
r7 = r7 ^ r0; // 43
|
||||
r1 = r1 ^ dataset[r7 & MASK]; // 44
|
||||
r2 = r2 * r3; // 45
|
||||
r1 = mulhi(r1, r5); // 46
|
||||
r4 = r4 - r3; // 47
|
||||
r2 = rotr_var(r2, r6); // 48
|
||||
r3 = r3 ^ dataset[r4 & MASK]; // 49
|
||||
r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 50
|
||||
r0 = r0 * r2; // 51
|
||||
r0 = r0 + r2 + select(0x4f92b968u, 0x699fd448u, ((sel >> 6u) & 1u) != 0u); // 52
|
||||
r1 = r1 + r0 + select(0x2bb965afu, 0x77b1520du, ((sel >> 12u) & 1u) != 0u); // 53
|
||||
r7 = rotl_imm(r7, 14u); // 54
|
||||
r3 = r3 + r7 + select(0x7b0fe07au, 0xa54c55a0u, ((sel >> 1u) & 1u) != 0u); // 55
|
||||
r6 = r6 ^ dataset[r7 & MASK]; // 56
|
||||
r1 = rotr_var(r1, r5); // 57
|
||||
r5 = r5 ^ dataset[r3 & MASK]; // 58
|
||||
r6 = r6 ^ dataset[r1 & MASK]; // 59
|
||||
r3 = r5 * r0 + r3; // 60
|
||||
r5 = r5 + r7 + select(0xaf9dd72du, 0xad7493e7u, ((sel >> 31u) & 1u) != 0u); // 61
|
||||
r4 = r4 + r6 + select(0x89841d87u, 0x1e07c3d9u, ((sel >> 27u) & 1u) != 0u); // 62
|
||||
r5 = rotl_imm(r5, 19u); // 63
|
||||
// latency-shadow block (Counter ASIC 3.0 item 8): 256 ALU instructions x 27 passes after instruction 63, no load
|
||||
for (uint sh = 0u; sh < 27u; ++sh) {
|
||||
r5 = r5 + r2 + select(0x92199f99u, 0x8bc12da9u, ((sel >> 18u) & 1u) != 0u); // s0 add
|
||||
r0 = r0 + r7 + select(0x8d72d3adu, 0x63079e5au, ((sel >> 26u) & 1u) != 0u); // s1 add
|
||||
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)2); // s2 shfl
|
||||
r4 = r4 - r2; // s3 sub
|
||||
r7 = r7 + r0 + select(0xb21b4babu, 0x5d4c7a60u, ((sel >> 31u) & 1u) != 0u); // s4 add
|
||||
r0 = rotl_imm(r0, 11u); // s5 rotl
|
||||
r0 = r0 + r1 + select(0x2fe0e98bu, 0xc88e2942u, ((sel >> 16u) & 1u) != 0u); // s6 add
|
||||
r7 = r7 ^ r1; // s7 xor
|
||||
r1 = r6 * r5 + r1; // s8 mad
|
||||
r6 = r6 ^ simd_shuffle_xor(r1, (ushort)4); // s9 shfl
|
||||
r1 = r2 * r2 + r1; // s10 mad
|
||||
r5 = r0 * r3 + r5; // s11 mad
|
||||
r2 = r2 ^ simd_shuffle_xor(r6, (ushort)4); // s12 shfl
|
||||
r1 = r1 + r5 + select(0xbdf8f9a5u, 0xbc48c63eu, ((sel >> 25u) & 1u) != 0u); // s13 add
|
||||
r1 = rotl_imm(r1, 29u); // s14 rotl
|
||||
r1 = r1 - r4; // s15 sub
|
||||
r7 = r7 | r1; // s16 or
|
||||
r2 = r2 ^ simd_shuffle_xor(r4, (ushort)8); // s17 shfl
|
||||
r7 = r7 ^ r4; // s18 xor
|
||||
r6 = r6 * r1; // s19 mul
|
||||
r5 = r6 * r0 + r5; // s20 mad
|
||||
r3 = r3 - r1; // s21 sub
|
||||
r6 = r6 * r0; // s22 mul
|
||||
r2 = r2 + r0 + select(0x45c37cecu, 0x96e8f127u, ((sel >> 1u) & 1u) != 0u); // s23 add
|
||||
r6 = r6 - r4; // s24 sub
|
||||
r7 = r3 * r4 + r7; // s25 mad
|
||||
r3 = rotl_imm(r3, 9u); // s26 rotl
|
||||
r2 = r2 - r1; // s27 sub
|
||||
r6 = mulhi(r6, r0); // s28 mulhi
|
||||
r2 = r2 ^ simd_shuffle_xor(r4, (ushort)2); // s29 shfl
|
||||
r1 = r1 ^ simd_shuffle_xor(r6, (ushort)1); // s30 shfl
|
||||
r1 = r1 + r6 + select(0x37985632u, 0xb1cdb2abu, ((sel >> 1u) & 1u) != 0u); // s31 add
|
||||
r0 = rotr_var(r0, r1); // s32 rotr
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // s33 shfl
|
||||
r0 = r0 ^ simd_shuffle_xor(r3, (ushort)4); // s34 shfl
|
||||
r7 = r7 * r0; // s35 mul
|
||||
r3 = r3 + r1 + select(0x804e777eu, 0x856e0180u, ((sel >> 0u) & 1u) != 0u); // s36 add
|
||||
r1 = r1 ^ r6; // s37 xor
|
||||
r2 = r2 * r7; // s38 mul
|
||||
r6 = r6 + r5 + select(0x0b06c8a7u, 0xe9bd0cc4u, ((sel >> 2u) & 1u) != 0u); // s39 add
|
||||
r5 = r5 * r7; // s40 mul
|
||||
r2 = mulhi(r2, r3); // s41 mulhi
|
||||
r2 = r2 ^ r0; // s42 xor
|
||||
r0 = rotl_imm(r0, 4u); // s43 rotl
|
||||
r4 = mulhi(r4, r3); // s44 mulhi
|
||||
r6 = r6 + r7 + select(0xee822e17u, 0xcdcb63f6u, ((sel >> 12u) & 1u) != 0u); // s45 add
|
||||
r3 = r2 * r0 + r3; // s46 mad
|
||||
r4 = r4 ^ simd_shuffle_xor(r3, (ushort)8); // s47 shfl
|
||||
r6 = mulhi(r6, r5); // s48 mulhi
|
||||
r0 = r0 - r2; // s49 sub
|
||||
r3 = r3 - r5; // s50 sub
|
||||
r1 = rotr_var(r1, r4); // s51 rotr
|
||||
r6 = r7 * r7 + r6; // s52 mad
|
||||
r5 = r5 ^ r3; // s53 xor
|
||||
r1 = r1 - r0; // s54 sub
|
||||
r5 = r5 - r6; // s55 sub
|
||||
r3 = r3 + r2 + select(0x3dfad1b6u, 0xd4758987u, ((sel >> 10u) & 1u) != 0u); // s56 add
|
||||
r4 = r4 + r1 + select(0xfca75bc2u, 0x0602d6beu, ((sel >> 0u) & 1u) != 0u); // s57 add
|
||||
r5 = mulhi(r5, r6); // s58 mulhi
|
||||
r2 = r2 ^ simd_shuffle_xor(r1, (ushort)2); // s59 shfl
|
||||
r2 = rotl_imm(r2, 9u); // s60 rotl
|
||||
r4 = r4 | r6; // s61 or
|
||||
r6 = rotr_var(r6, r4); // s62 rotr
|
||||
r2 = r2 * r4; // s63 mul
|
||||
r0 = r0 ^ r5; // s64 xor
|
||||
r2 = r2 ^ r7; // s65 xor
|
||||
r2 = r2 + r1 + select(0xd71c02ffu, 0x8596687au, ((sel >> 6u) & 1u) != 0u); // s66 add
|
||||
r1 = rotl_imm(r1, 21u); // s67 rotl
|
||||
r3 = r3 * r2; // s68 mul
|
||||
r7 = r7 ^ simd_shuffle_xor(r5, (ushort)2); // s69 shfl
|
||||
r3 = r3 * r2; // s70 mul
|
||||
r0 = r2 * r5 + r0; // s71 mad
|
||||
r6 = r6 + r7 + select(0x3c6fe15du, 0x08ac5733u, ((sel >> 0u) & 1u) != 0u); // s72 add
|
||||
r3 = r3 ^ simd_shuffle_xor(r2, (ushort)8); // s73 shfl
|
||||
r3 = r3 * r6; // s74 mul
|
||||
r6 = r6 ^ r7; // s75 xor
|
||||
r3 = r3 | r0; // s76 or
|
||||
r2 = r2 + r4 + select(0x295d5faeu, 0xc100b495u, ((sel >> 1u) & 1u) != 0u); // s77 add
|
||||
r3 = mulhi(r3, r7); // s78 mulhi
|
||||
r4 = r4 | r1; // s79 or
|
||||
r4 = rotr_var(r4, r3); // s80 rotr
|
||||
r4 = r4 + r3 + select(0x274a9221u, 0x5cc59530u, ((sel >> 15u) & 1u) != 0u); // s81 add
|
||||
r7 = r7 + r3 + select(0xc8651f8eu, 0x141479ecu, ((sel >> 5u) & 1u) != 0u); // s82 add
|
||||
r3 = rotr_var(r3, r6); // s83 rotr
|
||||
r2 = r4 * r7 + r2; // s84 mad
|
||||
r2 = r2 ^ simd_shuffle_xor(r5, (ushort)16); // s85 shfl
|
||||
r3 = r3 ^ r2; // s86 xor
|
||||
r5 = r5 ^ simd_shuffle_xor(r7, (ushort)2); // s87 shfl
|
||||
r0 = r0 ^ r4; // s88 xor
|
||||
r3 = r3 + r2 + select(0x53f915c2u, 0x883c0c92u, ((sel >> 18u) & 1u) != 0u); // s89 add
|
||||
r7 = rotr_var(r7, r5); // s90 rotr
|
||||
r3 = r3 + r1 + select(0xe2be00a5u, 0x3cc5bd20u, ((sel >> 31u) & 1u) != 0u); // s91 add
|
||||
r7 = r7 ^ simd_shuffle_xor(r2, (ushort)1); // s92 shfl
|
||||
r6 = rotr_var(r6, r2); // s93 rotr
|
||||
r7 = rotl_imm(r7, 14u); // s94 rotl
|
||||
r4 = r5 * r5 + r4; // s95 mad
|
||||
r2 = r4 * r5 + r2; // s96 mad
|
||||
r1 = r1 ^ r0; // s97 xor
|
||||
r5 = r5 * r4; // s98 mul
|
||||
r2 = r2 - r0; // s99 sub
|
||||
r7 = rotl_imm(r7, 30u); // s100 rotl
|
||||
r5 = r5 + r6 + select(0x423fd9c9u, 0xbfd646cbu, ((sel >> 17u) & 1u) != 0u); // s101 add
|
||||
r7 = r7 + r6 + select(0x8d3c011du, 0x19b74a43u, ((sel >> 11u) & 1u) != 0u); // s102 add
|
||||
r5 = rotl_imm(r5, 6u); // s103 rotl
|
||||
r0 = r0 * r4; // s104 mul
|
||||
r0 = r0 | r5; // s105 or
|
||||
r0 = r0 + r1 + select(0x8ce14721u, 0x7dcb7e18u, ((sel >> 15u) & 1u) != 0u); // s106 add
|
||||
r0 = rotl_imm(r0, 15u); // s107 rotl
|
||||
r4 = r4 - r2; // s108 sub
|
||||
r2 = mulhi(r2, r0); // s109 mulhi
|
||||
r1 = mulhi(r1, r0); // s110 mulhi
|
||||
r0 = r0 + r2 + select(0x2d9fc6b9u, 0x3c179ad8u, ((sel >> 28u) & 1u) != 0u); // s111 add
|
||||
r2 = mulhi(r2, r0); // s112 mulhi
|
||||
r6 = r6 - r3; // s113 sub
|
||||
r7 = r6 * r3 + r7; // s114 mad
|
||||
r5 = rotr_var(r5, r1); // s115 rotr
|
||||
r6 = rotr_var(r6, r4); // s116 rotr
|
||||
r2 = r2 + r1 + select(0x502138e2u, 0x47359729u, ((sel >> 22u) & 1u) != 0u); // s117 add
|
||||
r3 = r2 * r0 + r3; // s118 mad
|
||||
r1 = r1 * r3; // s119 mul
|
||||
r2 = r0 * r4 + r2; // s120 mad
|
||||
r0 = r0 * r3; // s121 mul
|
||||
r2 = mulhi(r2, r0); // s122 mulhi
|
||||
r5 = r5 * r6; // s123 mul
|
||||
r4 = r2 * r4 + r4; // s124 mad
|
||||
r4 = r4 ^ simd_shuffle_xor(r2, (ushort)2); // s125 shfl
|
||||
r2 = r2 ^ r0; // s126 xor
|
||||
r1 = rotl_imm(r1, 7u); // s127 rotl
|
||||
r7 = r7 ^ simd_shuffle_xor(r2, (ushort)4); // s128 shfl
|
||||
r6 = rotl_imm(r6, 17u); // s129 rotl
|
||||
r2 = r2 + r5 + select(0x33dff776u, 0x6c57e4e7u, ((sel >> 15u) & 1u) != 0u); // s130 add
|
||||
r0 = r0 | r3; // s131 or
|
||||
r5 = rotr_var(r5, r0); // s132 rotr
|
||||
r2 = r2 + r3 + select(0x5db25b34u, 0xe8212c0cu, ((sel >> 30u) & 1u) != 0u); // s133 add
|
||||
r4 = r4 ^ r3; // s134 xor
|
||||
r6 = r6 ^ r1; // s135 xor
|
||||
r1 = r1 - r7; // s136 sub
|
||||
r2 = r6 * r2 + r2; // s137 mad
|
||||
r0 = mulhi(r0, r5); // s138 mulhi
|
||||
r2 = r2 + r7 + select(0x218d4090u, 0xd44ff710u, ((sel >> 10u) & 1u) != 0u); // s139 add
|
||||
r1 = rotr_var(r1, r4); // s140 rotr
|
||||
r3 = r3 ^ simd_shuffle_xor(r6, (ushort)1); // s141 shfl
|
||||
r7 = r6 * r2 + r7; // s142 mad
|
||||
r2 = r2 * r3; // s143 mul
|
||||
r7 = r7 + r4 + select(0xf4b1a8deu, 0xb98942fau, ((sel >> 29u) & 1u) != 0u); // s144 add
|
||||
r7 = rotl_imm(r7, 15u); // s145 rotl
|
||||
r7 = r7 ^ r5; // s146 xor
|
||||
r4 = r7 * r4 + r4; // s147 mad
|
||||
r6 = rotr_var(r6, r5); // s148 rotr
|
||||
r1 = r2 * r4 + r1; // s149 mad
|
||||
r1 = r1 + r7 + select(0x2bef10f2u, 0x0d48ba42u, ((sel >> 17u) & 1u) != 0u); // s150 add
|
||||
r5 = r5 ^ r4; // s151 xor
|
||||
r7 = r7 + r0 + select(0xdc5cc080u, 0xc98dea9cu, ((sel >> 30u) & 1u) != 0u); // s152 add
|
||||
r4 = r4 * r6; // s153 mul
|
||||
r0 = r0 * r2; // s154 mul
|
||||
r6 = r6 ^ r5; // s155 xor
|
||||
r4 = rotr_var(r4, r2); // s156 rotr
|
||||
r1 = rotl_imm(r1, 11u); // s157 rotl
|
||||
r5 = r5 + r0 + select(0x18197438u, 0x6c752dcbu, ((sel >> 11u) & 1u) != 0u); // s158 add
|
||||
r4 = r1 * r5 + r4; // s159 mad
|
||||
r4 = rotl_imm(r4, 26u); // s160 rotl
|
||||
r3 = r0 * r7 + r3; // s161 mad
|
||||
r3 = rotr_var(r3, r1); // s162 rotr
|
||||
r4 = r4 + r0 + select(0xf1c46574u, 0x8e481727u, ((sel >> 3u) & 1u) != 0u); // s163 add
|
||||
r0 = mulhi(r0, r4); // s164 mulhi
|
||||
r2 = r2 + r6 + select(0xac578137u, 0x550ab406u, ((sel >> 20u) & 1u) != 0u); // s165 add
|
||||
r0 = rotl_imm(r0, 13u); // s166 rotl
|
||||
r3 = r3 + r1 + select(0xa33e6706u, 0xaebb5966u, ((sel >> 14u) & 1u) != 0u); // s167 add
|
||||
r3 = r3 | r5; // s168 or
|
||||
r6 = rotr_var(r6, r2); // s169 rotr
|
||||
r4 = r4 ^ r6; // s170 xor
|
||||
r6 = r6 - r1; // s171 sub
|
||||
r7 = rotl_imm(r7, 22u); // s172 rotl
|
||||
r5 = rotl_imm(r5, 15u); // s173 rotl
|
||||
r7 = r7 ^ simd_shuffle_xor(r0, (ushort)8); // s174 shfl
|
||||
r0 = r0 ^ r5; // s175 xor
|
||||
r7 = rotl_imm(r7, 6u); // s176 rotl
|
||||
r7 = r7 - r0; // s177 sub
|
||||
r3 = rotl_imm(r3, 30u); // s178 rotl
|
||||
r7 = r6 * r1 + r7; // s179 mad
|
||||
r6 = rotl_imm(r6, 9u); // s180 rotl
|
||||
r2 = r2 ^ r4; // s181 xor
|
||||
r2 = r2 ^ r7; // s182 xor
|
||||
r7 = r7 ^ r2; // s183 xor
|
||||
r1 = r1 + r2 + select(0xd94d55acu, 0x5bb7550fu, ((sel >> 21u) & 1u) != 0u); // s184 add
|
||||
r3 = r3 | r5; // s185 or
|
||||
r6 = mulhi(r6, r3); // s186 mulhi
|
||||
r4 = r4 | r0; // s187 or
|
||||
r7 = r7 ^ simd_shuffle_xor(r1, (ushort)2); // s188 shfl
|
||||
r6 = r6 + r5 + select(0x89e747feu, 0x2a354e2du, ((sel >> 6u) & 1u) != 0u); // s189 add
|
||||
r1 = r1 ^ r4; // s190 xor
|
||||
r7 = mulhi(r7, r4); // s191 mulhi
|
||||
r2 = rotl_imm(r2, 26u); // s192 rotl
|
||||
r5 = rotl_imm(r5, 8u); // s193 rotl
|
||||
r4 = r4 ^ simd_shuffle_xor(r5, (ushort)16); // s194 shfl
|
||||
r4 = r4 ^ r5; // s195 xor
|
||||
r1 = r1 * r3; // s196 mul
|
||||
r5 = r5 + r2 + select(0xea03e8e7u, 0x10cfdc71u, ((sel >> 7u) & 1u) != 0u); // s197 add
|
||||
r7 = r7 + r5 + select(0xa7ee0102u, 0x66e148d3u, ((sel >> 15u) & 1u) != 0u); // s198 add
|
||||
r5 = r5 - r2; // s199 sub
|
||||
r5 = r5 ^ simd_shuffle_xor(r1, (ushort)2); // s200 shfl
|
||||
r7 = r7 ^ simd_shuffle_xor(r1, (ushort)8); // s201 shfl
|
||||
r4 = rotr_var(r4, r5); // s202 rotr
|
||||
r7 = r7 ^ r5; // s203 xor
|
||||
r7 = r7 ^ r0; // s204 xor
|
||||
r6 = r6 + r2 + select(0x8379a4deu, 0x8558b619u, ((sel >> 10u) & 1u) != 0u); // s205 add
|
||||
r4 = rotl_imm(r4, 13u); // s206 rotl
|
||||
r1 = mulhi(r1, r3); // s207 mulhi
|
||||
r1 = r4 * r4 + r1; // s208 mad
|
||||
r2 = r2 ^ simd_shuffle_xor(r0, (ushort)4); // s209 shfl
|
||||
r7 = r7 + r0 + select(0xaa8cb14eu, 0xf4049c4cu, ((sel >> 11u) & 1u) != 0u); // s210 add
|
||||
r7 = r7 ^ simd_shuffle_xor(r1, (ushort)16); // s211 shfl
|
||||
r1 = r1 ^ r0; // s212 xor
|
||||
r7 = rotl_imm(r7, 3u); // s213 rotl
|
||||
r4 = rotr_var(r4, r7); // s214 rotr
|
||||
r3 = r3 ^ simd_shuffle_xor(r2, (ushort)16); // s215 shfl
|
||||
r5 = r5 | r1; // s216 or
|
||||
r1 = r1 - r2; // s217 sub
|
||||
r6 = r6 - r5; // s218 sub
|
||||
r6 = rotl_imm(r6, 4u); // s219 rotl
|
||||
r2 = mulhi(r2, r0); // s220 mulhi
|
||||
r2 = r2 | r0; // s221 or
|
||||
r5 = r5 ^ simd_shuffle_xor(r2, (ushort)8); // s222 shfl
|
||||
r5 = mulhi(r5, r6); // s223 mulhi
|
||||
r0 = r0 - r6; // s224 sub
|
||||
r7 = rotl_imm(r7, 23u); // s225 rotl
|
||||
r4 = r4 | r2; // s226 or
|
||||
r2 = r2 * r4; // s227 mul
|
||||
r3 = rotl_imm(r3, 12u); // s228 rotl
|
||||
r0 = rotr_var(r0, r4); // s229 rotr
|
||||
r0 = r0 + r6 + select(0x1d176220u, 0x2a7fecb2u, ((sel >> 16u) & 1u) != 0u); // s230 add
|
||||
r1 = r1 ^ simd_shuffle_xor(r2, (ushort)4); // s231 shfl
|
||||
r2 = r2 * r1; // s232 mul
|
||||
r7 = r0 * r1 + r7; // s233 mad
|
||||
r5 = rotl_imm(r5, 22u); // s234 rotl
|
||||
r4 = rotr_var(r4, r6); // s235 rotr
|
||||
r0 = r5 * r1 + r0; // s236 mad
|
||||
r6 = r6 ^ r4; // s237 xor
|
||||
r4 = r4 ^ r6; // s238 xor
|
||||
r6 = rotl_imm(r6, 18u); // s239 rotl
|
||||
r4 = r4 + r7 + select(0x17dafb4du, 0xadce39f3u, ((sel >> 25u) & 1u) != 0u); // s240 add
|
||||
r0 = r0 - r7; // s241 sub
|
||||
r1 = rotr_var(r1, r6); // s242 rotr
|
||||
r3 = r3 + r0 + select(0xf2f77d26u, 0x0e7033b6u, ((sel >> 29u) & 1u) != 0u); // s243 add
|
||||
r2 = r7 * r4 + r2; // s244 mad
|
||||
r4 = r0 * r6 + r4; // s245 mad
|
||||
r5 = r5 * r7; // s246 mul
|
||||
r3 = r3 + r5 + select(0xfed2da4eu, 0x7b2ff6b7u, ((sel >> 31u) & 1u) != 0u); // s247 add
|
||||
r0 = r0 + r7 + select(0x03b2891cu, 0xb5fad7b1u, ((sel >> 0u) & 1u) != 0u); // s248 add
|
||||
r5 = mulhi(r5, r4); // s249 mulhi
|
||||
r5 = r5 + r2 + select(0x042cd6e3u, 0xa7e71c2fu, ((sel >> 11u) & 1u) != 0u); // s250 add
|
||||
r6 = r6 ^ simd_shuffle_xor(r1, (ushort)1); // s251 shfl
|
||||
r6 = r6 ^ r1; // s252 xor
|
||||
r2 = r2 * r5; // s253 mul
|
||||
r0 = r0 + r6 + select(0x784a302bu, 0xb83d78deu, ((sel >> 16u) & 1u) != 0u); // s254 add
|
||||
r2 = r2 - r4; // s255 sub
|
||||
}
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
BIN
proto-cuda/packs-ca3-v5/v5-genesis/state.igsd1
Normal file
BIN
proto-cuda/packs-ca3-v5/v5-genesis/state.igsd1
Normal file
Binary file not shown.
57
proto-cuda/packs-ca3-v5/v5-genesis/vectors.h
Normal file
57
proto-cuda/packs-ca3-v5/v5-genesis/vectors.h
Normal file
|
|
@ -0,0 +1,57 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Expected outputs: igneum-pow (Rust) CPU interpreter, generator v5, memory-hard dataset
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
|
||||
#define IGNEUM_VEC_WARPS 3
|
||||
static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u };
|
||||
static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {
|
||||
{ // base nonce 0
|
||||
0x61b73fdc4b19aa6eull, 0x610aa89caabd9033ull, 0x2098b7208e1a87a2ull, 0xadab17772714d58aull, 0x122ffea36de09b58ull, 0x37ef2b162861daf7ull, 0x2fafc41dccce3551ull, 0x896f2f5b273d8d22ull,
|
||||
0x9eeaf06bf26dc93aull, 0x912bc36017365292ull, 0xaee3489661bf7408ull, 0x746a2d45731a5b82ull, 0x37e102af2acb343aull, 0xc55d3e0cd361c1dcull, 0xb406caab7c29e79eull, 0x836aac23d9e1e071ull,
|
||||
0x3546dde84ff5d903ull, 0xf6ea2ee17788aacfull, 0x421ea20c767e9bc8ull, 0x4f6d341a8ed74b31ull, 0xc3caca19146fce10ull, 0xed94a3809c545272ull, 0x4c81410daf155777ull, 0x7e9c21bb3fcae920ull,
|
||||
0x177df4095c5a3186ull, 0xa0b272d801631069ull, 0xe1caaa749196d41dull, 0x46587eb06c695377ull, 0x2da5268e1a91b4b8ull, 0x65d1458d5f6c4b22ull, 0xf5eba1ff3c23abecull, 0x975164a63bd857aaull
|
||||
},
|
||||
{ // base nonce 4096
|
||||
0x8acb1dab413356ceull, 0xe1496796ba02e646ull, 0x3ef41217af95845eull, 0x98734db55629c78cull, 0x2a3772782cb06469ull, 0xac322419bf616aacull, 0xb31df4cd19f1ca85ull, 0xd99cf1c5ff71a59aull,
|
||||
0x5d9150df6882a437ull, 0x8a840e6b19a67039ull, 0x958258a0794f5a7full, 0x7920ac856c5ccf03ull, 0x3e2d32bba1854ee8ull, 0x624b7ee6c7188baaull, 0xa6561af2cbaada13ull, 0x149fc2be416c9d05ull,
|
||||
0x64d49fb97adc18b4ull, 0xcf1392ed093384d1ull, 0xf9b3006b778820b2ull, 0x996cbcefdb19a8feull, 0x3cdad4075fe603baull, 0xeacff90cb39702ecull, 0x76f8fcc773ffd064ull, 0x47d7a476038c62d7ull,
|
||||
0x364abcb1cebfd3dcull, 0x4a693b6e773abb8dull, 0x89fa1d025bbbef40ull, 0x09b6d8fd62661e9aull, 0xb25a12fc86d1672bull, 0x709698b8f3610c8full, 0xea90332c099f4b12ull, 0x182efd5917a3ac88ull
|
||||
},
|
||||
{ // base nonce 1000000
|
||||
0xfff0c5f72bc7f65bull, 0xceed1f32eaab246bull, 0xaeb2bc4ae4d9a349ull, 0x3a99b0b3ea04f8f2ull, 0x6787bcd03abb4a32ull, 0xb869e7541d7b9e0dull, 0x41cdb2fac5e770ebull, 0x8c57c331da34a520ull,
|
||||
0x7ca4eea1312b3303ull, 0x0e8da8c8bf2baba0ull, 0x31936e024742611eull, 0x482d90d85ac75f85ull, 0xfef10b7265a9c3b1ull, 0x666b8515840eea5aull, 0x38e56615f60ddb1dull, 0xf5847616b5a6ba3dull,
|
||||
0x9c1d61f35848f26aull, 0x9cb3a3de2ab6b4f4ull, 0x4cb2d12cc9e57a66ull, 0xb73a5582760087e6ull, 0x2c93f2a481804d57ull, 0xf3061af8c1cbbef7ull, 0xb2bd9a0c7a42f393ull, 0xf863114deaad0538ull,
|
||||
0x4f37fdf6937fabefull, 0x52663000df31a556ull, 0xe0e721a0f5a85910ull, 0x6d0db0745408d49aull, 0x99d908bdfd76f8b4ull, 0xb0aa805e18e4f13aull, 0x2a313940928c4c1dull, 0x6c969be6b1b760e2ull
|
||||
}
|
||||
};
|
||||
|
||||
// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (268435455).
|
||||
static const uint32_t IGNEUM_DS_HEAD[16] = {
|
||||
0x3fdb9cc7u, 0x20a68a16u, 0xf543d4a5u, 0x12b21f10u, 0x1580e7f0u, 0x5a6f4182u, 0xb164b095u, 0x33ac8b62u,
|
||||
0x29296437u, 0xaada1879u, 0xa49620d3u, 0x761db725u, 0x24a81921u, 0x45c70b8bu, 0x3a674109u, 0x62647f74u
|
||||
};
|
||||
static const uint32_t IGNEUM_DS_LAST_INDEX = 268435455u;
|
||||
static const uint32_t IGNEUM_DS_LAST = 0x769c356du;
|
||||
// 64 sampled dataset words (index, value) computed on the Mac.
|
||||
#define IGNEUM_DS_SAMPLES 64
|
||||
static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = {
|
||||
59471966u, 217795994u, 208353206u, 42483309u, 172547758u, 148076330u, 183853158u, 214389424u, 267488061u, 169781097u, 184093494u, 153880993u, 84977930u, 46426879u, 3093825u, 225364072u, 44593546u, 260713159u, 168250303u, 52384140u, 223401610u, 45554030u, 95410555u, 175039924u, 79171087u, 267580473u, 24168642u, 37981670u, 171551130u, 195559979u, 204611762u, 140997658u, 138925853u, 86637313u, 20736778u, 219665210u, 160430336u, 264654675u, 8013395u, 228945585u, 213884386u, 104419827u, 44185464u, 142737231u, 99284897u, 132475900u, 61861762u, 132056166u, 262388043u, 91878046u, 117353561u, 124768597u, 71352993u, 190698941u, 46055428u, 55281366u, 165145231u, 106810753u, 171985651u, 232085256u, 159510492u, 40072060u, 209107596u, 39023794u
|
||||
};
|
||||
static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = {
|
||||
0xdcafdb01u, 0xc091b066u, 0x98318ab1u, 0x73662aa4u, 0xf46b46b0u, 0x91ab95d4u, 0x3a5659c2u, 0x6e81c60au, 0x3d39ce88u, 0x1f6a2b6bu, 0x017b89d2u, 0x2e1b0d2fu, 0xd0d3fdcdu, 0x65478f05u, 0x09c627c5u, 0x6a12a65fu, 0x673dfe78u, 0x5e3cb389u, 0x9b6c3ab5u, 0x7db5867eu, 0x9e4e3844u, 0x0fb8d2a7u, 0x94da7c76u, 0x9047aae4u, 0x48bc05f4u, 0x8deb15bcu, 0x9b630120u, 0x7d2ec996u, 0x9162e70bu, 0x66303b2du, 0x7b61b5ecu, 0x2c72a23du, 0x6212ac42u, 0xa4212c34u, 0x7a37c02eu, 0xeb8b8117u, 0xb05dca47u, 0x06a6da71u, 0xa0b72498u, 0xd5917506u, 0xff1a7fb9u, 0x9809a9e4u, 0x729a7cf0u, 0xa14a997eu, 0x1f4b63e5u, 0xbdc389a4u, 0xfcec4807u, 0x49f01ffeu, 0xaa064d33u, 0xf752f881u, 0x485b414du, 0x60280cbcu, 0x14bfd7ccu, 0x646f720cu, 0x18e3d162u, 0x31b6781au, 0xc240fbd7u, 0x79b9d07eu, 0x2bcb0484u, 0xf8264c4cu, 0xe7ceb1e2u, 0x9ca84419u, 0x44770a37u, 0x9521c2dbu
|
||||
};
|
||||
// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^26 words.
|
||||
static const uint32_t IGNEUM_CACHE_HEAD[16] = {
|
||||
0x355a86d2u, 0x7957db1cu, 0xd21772afu, 0x6fc1e09bu, 0xd55ce61du, 0x6e6a278bu, 0xd3f543ceu, 0x223d8e82u,
|
||||
0x143ab337u, 0x2e9f05bdu, 0x2eb389bfu, 0x0c6e449eu, 0x5cfa4222u, 0xba6560feu, 0x8e3e1aa4u, 0xdbcc1d53u
|
||||
};
|
||||
static const uint32_t IGNEUM_CACHE_LAST[16] = {
|
||||
0x41190d91u, 0xbd277957u, 0x22ddbb49u, 0x6986f207u, 0xdf69a4d6u, 0x26401a3au, 0x818230fbu, 0xc417122du,
|
||||
0x3597b211u, 0xb553ce55u, 0xcf39cc0du, 0x3b7fc43au, 0x3fd43b00u, 0x67e1c80eu, 0xffa7ea7du, 0xca2960abu
|
||||
};
|
||||
static const uint64_t IGNEUM_CACHE_FNV64 = 0x48c4f5bf24166b2eull;
|
||||
36
proto-cuda/packs-ca3-v5/v5-genesis/vectors.json
Normal file
36
proto-cuda/packs-ca3-v5/v5-genesis/vectors.json
Normal file
|
|
@ -0,0 +1,36 @@
|
|||
{
|
||||
"seed": "igneum-genesis",
|
||||
"day": "2026-10-03",
|
||||
"dataset_mode": "memory-hard",
|
||||
"dataset_log2_words": 28,
|
||||
"mask": "0x0fffffff",
|
||||
"lanes": 32,
|
||||
"source": "igneum-pow (Rust) CPU interpreter, generator v5, memory-hard dataset",
|
||||
"warps": [
|
||||
{"base_nonce": 0, "expected": [
|
||||
"0x61b73fdc4b19aa6e", "0x610aa89caabd9033", "0x2098b7208e1a87a2", "0xadab17772714d58a", "0x122ffea36de09b58", "0x37ef2b162861daf7", "0x2fafc41dccce3551", "0x896f2f5b273d8d22",
|
||||
"0x9eeaf06bf26dc93a", "0x912bc36017365292", "0xaee3489661bf7408", "0x746a2d45731a5b82", "0x37e102af2acb343a", "0xc55d3e0cd361c1dc", "0xb406caab7c29e79e", "0x836aac23d9e1e071",
|
||||
"0x3546dde84ff5d903", "0xf6ea2ee17788aacf", "0x421ea20c767e9bc8", "0x4f6d341a8ed74b31", "0xc3caca19146fce10", "0xed94a3809c545272", "0x4c81410daf155777", "0x7e9c21bb3fcae920",
|
||||
"0x177df4095c5a3186", "0xa0b272d801631069", "0xe1caaa749196d41d", "0x46587eb06c695377", "0x2da5268e1a91b4b8", "0x65d1458d5f6c4b22", "0xf5eba1ff3c23abec", "0x975164a63bd857aa"
|
||||
]},
|
||||
{"base_nonce": 4096, "expected": [
|
||||
"0x8acb1dab413356ce", "0xe1496796ba02e646", "0x3ef41217af95845e", "0x98734db55629c78c", "0x2a3772782cb06469", "0xac322419bf616aac", "0xb31df4cd19f1ca85", "0xd99cf1c5ff71a59a",
|
||||
"0x5d9150df6882a437", "0x8a840e6b19a67039", "0x958258a0794f5a7f", "0x7920ac856c5ccf03", "0x3e2d32bba1854ee8", "0x624b7ee6c7188baa", "0xa6561af2cbaada13", "0x149fc2be416c9d05",
|
||||
"0x64d49fb97adc18b4", "0xcf1392ed093384d1", "0xf9b3006b778820b2", "0x996cbcefdb19a8fe", "0x3cdad4075fe603ba", "0xeacff90cb39702ec", "0x76f8fcc773ffd064", "0x47d7a476038c62d7",
|
||||
"0x364abcb1cebfd3dc", "0x4a693b6e773abb8d", "0x89fa1d025bbbef40", "0x09b6d8fd62661e9a", "0xb25a12fc86d1672b", "0x709698b8f3610c8f", "0xea90332c099f4b12", "0x182efd5917a3ac88"
|
||||
]},
|
||||
{"base_nonce": 1000000, "expected": [
|
||||
"0xfff0c5f72bc7f65b", "0xceed1f32eaab246b", "0xaeb2bc4ae4d9a349", "0x3a99b0b3ea04f8f2", "0x6787bcd03abb4a32", "0xb869e7541d7b9e0d", "0x41cdb2fac5e770eb", "0x8c57c331da34a520",
|
||||
"0x7ca4eea1312b3303", "0x0e8da8c8bf2baba0", "0x31936e024742611e", "0x482d90d85ac75f85", "0xfef10b7265a9c3b1", "0x666b8515840eea5a", "0x38e56615f60ddb1d", "0xf5847616b5a6ba3d",
|
||||
"0x9c1d61f35848f26a", "0x9cb3a3de2ab6b4f4", "0x4cb2d12cc9e57a66", "0xb73a5582760087e6", "0x2c93f2a481804d57", "0xf3061af8c1cbbef7", "0xb2bd9a0c7a42f393", "0xf863114deaad0538",
|
||||
"0x4f37fdf6937fabef", "0x52663000df31a556", "0xe0e721a0f5a85910", "0x6d0db0745408d49a", "0x99d908bdfd76f8b4", "0xb0aa805e18e4f13a", "0x2a313940928c4c1d", "0x6c969be6b1b760e2"
|
||||
]}
|
||||
],
|
||||
"dataset_head": ["0x3fdb9cc7", "0x20a68a16", "0xf543d4a5", "0x12b21f10", "0x1580e7f0", "0x5a6f4182", "0xb164b095", "0x33ac8b62", "0x29296437", "0xaada1879", "0xa49620d3", "0x761db725", "0x24a81921", "0x45c70b8b", "0x3a674109", "0x62647f74"],
|
||||
"dataset_last_index": 268435455,
|
||||
"dataset_last": "0x769c356d",
|
||||
"dataset_samples": [{"index": 59471966, "value": "0xdcafdb01"}, {"index": 217795994, "value": "0xc091b066"}, {"index": 208353206, "value": "0x98318ab1"}, {"index": 42483309, "value": "0x73662aa4"}, {"index": 172547758, "value": "0xf46b46b0"}, {"index": 148076330, "value": "0x91ab95d4"}, {"index": 183853158, "value": "0x3a5659c2"}, {"index": 214389424, "value": "0x6e81c60a"}, {"index": 267488061, "value": "0x3d39ce88"}, {"index": 169781097, "value": "0x1f6a2b6b"}, {"index": 184093494, "value": "0x017b89d2"}, {"index": 153880993, "value": "0x2e1b0d2f"}, {"index": 84977930, "value": "0xd0d3fdcd"}, {"index": 46426879, "value": "0x65478f05"}, {"index": 3093825, "value": "0x09c627c5"}, {"index": 225364072, "value": "0x6a12a65f"}, {"index": 44593546, "value": "0x673dfe78"}, {"index": 260713159, "value": "0x5e3cb389"}, {"index": 168250303, "value": "0x9b6c3ab5"}, {"index": 52384140, "value": "0x7db5867e"}, {"index": 223401610, "value": "0x9e4e3844"}, {"index": 45554030, "value": "0x0fb8d2a7"}, {"index": 95410555, "value": "0x94da7c76"}, {"index": 175039924, "value": "0x9047aae4"}, {"index": 79171087, "value": "0x48bc05f4"}, {"index": 267580473, "value": "0x8deb15bc"}, {"index": 24168642, "value": "0x9b630120"}, {"index": 37981670, "value": "0x7d2ec996"}, {"index": 171551130, "value": "0x9162e70b"}, {"index": 195559979, "value": "0x66303b2d"}, {"index": 204611762, "value": "0x7b61b5ec"}, {"index": 140997658, "value": "0x2c72a23d"}, {"index": 138925853, "value": "0x6212ac42"}, {"index": 86637313, "value": "0xa4212c34"}, {"index": 20736778, "value": "0x7a37c02e"}, {"index": 219665210, "value": "0xeb8b8117"}, {"index": 160430336, "value": "0xb05dca47"}, {"index": 264654675, "value": "0x06a6da71"}, {"index": 8013395, "value": "0xa0b72498"}, {"index": 228945585, "value": "0xd5917506"}, {"index": 213884386, "value": "0xff1a7fb9"}, {"index": 104419827, "value": "0x9809a9e4"}, {"index": 44185464, "value": "0x729a7cf0"}, {"index": 142737231, "value": "0xa14a997e"}, {"index": 99284897, "value": "0x1f4b63e5"}, {"index": 132475900, "value": "0xbdc389a4"}, {"index": 61861762, "value": "0xfcec4807"}, {"index": 132056166, "value": "0x49f01ffe"}, {"index": 262388043, "value": "0xaa064d33"}, {"index": 91878046, "value": "0xf752f881"}, {"index": 117353561, "value": "0x485b414d"}, {"index": 124768597, "value": "0x60280cbc"}, {"index": 71352993, "value": "0x14bfd7cc"}, {"index": 190698941, "value": "0x646f720c"}, {"index": 46055428, "value": "0x18e3d162"}, {"index": 55281366, "value": "0x31b6781a"}, {"index": 165145231, "value": "0xc240fbd7"}, {"index": 106810753, "value": "0x79b9d07e"}, {"index": 171985651, "value": "0x2bcb0484"}, {"index": 232085256, "value": "0xf8264c4c"}, {"index": 159510492, "value": "0xe7ceb1e2"}, {"index": 40072060, "value": "0x9ca84419"}, {"index": 209107596, "value": "0x44770a37"}, {"index": 39023794, "value": "0x9521c2db"}],
|
||||
"cache_head": ["0x355a86d2", "0x7957db1c", "0xd21772af", "0x6fc1e09b", "0xd55ce61d", "0x6e6a278b", "0xd3f543ce", "0x223d8e82", "0x143ab337", "0x2e9f05bd", "0x2eb389bf", "0x0c6e449e", "0x5cfa4222", "0xba6560fe", "0x8e3e1aa4", "0xdbcc1d53"],
|
||||
"cache_last_line": ["0x41190d91", "0xbd277957", "0x22ddbb49", "0x6986f207", "0xdf69a4d6", "0x26401a3a", "0x818230fb", "0xc417122d", "0x3597b211", "0xb553ce55", "0xcf39cc0d", "0x3b7fc43a", "0x3fd43b00", "0x67e1c80e", "0xffa7ea7d", "0xca2960ab"],
|
||||
"cache_fnv1a64": "0x48c4f5bf24166b2e"
|
||||
}
|
||||
|
|
@ -2992,9 +2992,9 @@ final class ServeDataset {
|
|||
/// Counter ASIC 2.0 and 3.0: the classes this worker runs from a prepared pack only (never from the Swift version 2
|
||||
/// generator): class v3 (generator 3) and class v4 (generator 4, class v3 plus the latency-shadow block, which is in
|
||||
/// the pack's own program_bound.metal). Each carries an era seed.
|
||||
func isPackClass(_ cls: String) -> Bool { cls == "v3" || cls == "v4" }
|
||||
func isPackClass(_ cls: String) -> Bool { cls == "v3" || cls == "v4" || cls == "v5" } // class v5 (7 October 2026): class v4 over the state leaves, from a pack
|
||||
/// The class name of a pack's IGNEUM_GENERATOR (packfile.h's rule).
|
||||
func packClassOf(generator: UInt32) -> String { generator == 4 ? "v4" : generator == 3 ? "v3" : "v2" }
|
||||
func packClassOf(generator: UInt32) -> String { generator == 5 ? "v5" : generator == 4 ? "v4" : generator == 3 ? "v3" : "v2" }
|
||||
|
||||
// The resident programs and datasets, shared by the job loop (main thread) and the prepare queue (background).
|
||||
final class ServeStore {
|
||||
|
|
@ -3085,13 +3085,17 @@ func servePackProgram(_ gpu: GPU, _ store: ServeStore, seedHex: String, seed: [U
|
|||
}
|
||||
func refuse(_ why: String) -> NSError { NSError(domain: "pack", code: 2, userInfo: [NSLocalizedDescriptionKey: "pack \(dir): \(why)"]) }
|
||||
let generator = defineU32("IGNEUM_GENERATOR") ?? 1
|
||||
guard generator == 2 || generator == 3 || generator == 4 else { throw refuse("program pack generator \(generator) is not a generator version this worker runs (2, 3 or 4)") }
|
||||
guard generator == 2 || generator == 3 || generator == 4 || generator == 5 else { throw refuse("program pack generator \(generator) is not a generator version this worker runs (2, 3, 4 or 5)") }
|
||||
let packClass = packClassOf(generator: generator)
|
||||
if let named = defineStr("IGNEUM_PROGRAM_CLASS"), named != packClass { throw refuse("program pack IGNEUM_PROGRAM_CLASS \"\(named)\" does not match IGNEUM_GENERATOR \(generator)") }
|
||||
// Counter ASIC 3.0 (6 October 2026): the shadow block marks class v4 (packfile.h's rule): a generator 3 pack with
|
||||
// IGNEUM_SHADOW_INSTRS is a v4 program stamped v3 (the v3 control's program id) and a generator 4 pack without it is no v4 pack
|
||||
let shadow = defineU32("IGNEUM_SHADOW_INSTRS") ?? 0
|
||||
if generator == 4 && shadow == 0 { throw refuse("program pack generator 4 (class v4) without IGNEUM_SHADOW_INSTRS: not a class v4 pack") }
|
||||
// class v5 (7 October 2026) is class v4 over the state leaves: the shadow block and IGNEUM_STATE_LEAVES both mark it
|
||||
if generator == 5 && shadow == 0 { throw refuse("program pack generator 5 (class v5) without IGNEUM_SHADOW_INSTRS: not a class v5 pack") }
|
||||
if generator == 5 && (defineU32("IGNEUM_STATE_LEAVES") ?? 0) == 0 { throw refuse("program pack generator 5 (class v5) without IGNEUM_STATE_LEAVES: no state leaves to key the dataset (export the pack with --state)") }
|
||||
if generator != 5 && (defineU32("IGNEUM_STATE_LEAVES") ?? 0) != 0 { throw refuse("program pack generator \(generator) carries IGNEUM_STATE_LEAVES: state leaves belong to class v5 (generator 5)") }
|
||||
if generator == 3 && shadow != 0 { throw refuse("program pack generator \(generator) with a shadow block (IGNEUM_SHADOW_INSTRS \(shadow)): a class v4 program is generator 4 (export the pack as class v4)") }
|
||||
let eraHex = defineStr("IGNEUM_ERA_SEED_HEX") ?? ""
|
||||
if wantClass != "" && wantClass != packClass { throw refuse("program class mismatch: this pack is class \(packClass), the line names class \(wantClass) (export the pack again)") }
|
||||
|
|
@ -3159,6 +3163,29 @@ func servePackDataset(_ gpu: GPU, _ store: ServeStore, dayHex: String, dir: Stri
|
|||
let fillPipe = try gpu.device.makeComputePipelineState(function: fillFn)
|
||||
let buildPipe = try gpu.device.makeComputePipelineState(function: buildFn)
|
||||
let words = 1 << datasetLog2
|
||||
// Class v5 (docs/design/class-v5-stored-state.md, 7 October 2026): the window's state leaves, leaves.bin beside program.h
|
||||
// (IGNEUM_STATE_LEAVES leaves of 16 little-endian words), checked against the pack's count and FNV-1a 64
|
||||
// (IGNEUM_STATE_LEAVES_FNV64) and bound as buffer 2 of igneum_build with the count in buffer 3 (packbench.swift's shape,
|
||||
// the kernel text the Rust emitter wrote). A v5 pack without its leaves builds nothing (the known-failed case); the buffer
|
||||
// is dropped after the build, so device memory while hashing is the class v4 worker's.
|
||||
func defineU64(_ name: String) -> UInt64? {
|
||||
guard let re = try? NSRegularExpression(pattern: "#define \(name) 0x([0-9a-fA-F]+)ull"), let m = re.firstMatch(in: programH, range: NSRange(programH.startIndex..., in: programH)) else { return nil }
|
||||
return UInt64(programH[Range(m.range(at: 1), in: programH)!], radix: 16)
|
||||
}
|
||||
let stateLeaves = defineU32("IGNEUM_STATE_LEAVES") ?? 0
|
||||
if (packClass == "v5") != (stateLeaves > 0) { throw refuse(packClass == "v5" ? "class v5 pack without IGNEUM_STATE_LEAVES" : "a class \(packClass) pack carries IGNEUM_STATE_LEAVES \(stateLeaves): state leaves belong to class v5") }
|
||||
var leavesBuf: MTLBuffer? = nil
|
||||
var nLeaves: UInt32 = 0
|
||||
if stateLeaves > 0 {
|
||||
let file = defineStr("IGNEUM_STATE_LEAVES_FILE") ?? "leaves.bin"
|
||||
guard let data = FileManager.default.contents(atPath: dir + "/" + file) else { throw refuse("class v5 pack without its leaves file \(file) (the state leaves key every item: nothing can be built without them)") }
|
||||
if data.count != Int(stateLeaves) * 64 { throw refuse("\(file) is \(data.count) bytes, the pack says \(stateLeaves) leaves of 64") }
|
||||
guard let want = defineU64("IGNEUM_STATE_LEAVES_FNV64") else { throw refuse("class v5 pack without IGNEUM_STATE_LEAVES_FNV64: the leaves cannot be checked") }
|
||||
let got = fnv1a64Bytes([UInt8](data))
|
||||
if got != want { throw refuse("\(file) FNV-1a 64 \(h64(got)) is not the pack's IGNEUM_STATE_LEAVES_FNV64 \(h64(want)) (the leaves are of another state root; export the pack again)") }
|
||||
guard let b = gpu.device.makeBuffer(bytes: (data as NSData).bytes, length: data.count, options: .storageModeShared) else { throw refuse("cannot allocate the \(data.count) byte leaf buffer") }
|
||||
leavesBuf = b; nLeaves = stateLeaves
|
||||
}
|
||||
guard let cache = gpu.device.makeBuffer(length: (1 << cacheLog2) * 4, options: .storageModePrivate) else { throw refuse("cannot allocate the 2^\(cacheLog2) word cache") }
|
||||
guard let dataset = gpu.device.makeBuffer(length: words * 4, options: .storageModePrivate) else { throw refuse("cannot allocate the 2^\(datasetLog2) word dataset") }
|
||||
func run(_ body: (MTLComputeCommandEncoder) -> Void) throws -> Double {
|
||||
|
|
@ -3174,8 +3201,29 @@ func servePackDataset(_ gpu: GPU, _ store: ServeStore, dayHex: String, dir: Stri
|
|||
}
|
||||
let buildMs = try run { enc in
|
||||
enc.setComputePipelineState(buildPipe); enc.setBuffer(cache, offset: 0, index: 0); enc.setBuffer(dataset, offset: 0, index: 1)
|
||||
if let lb = leavesBuf { enc.setBuffer(lb, offset: 0, index: 2); var n = nLeaves; enc.setBytes(&n, length: 4, index: 3) }
|
||||
enc.dispatchThreadgroups(MTLSize(width: words / 16 / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
|
||||
}
|
||||
leavesBuf = nil
|
||||
// The build's self-test when the pack carries vectors.json (igneum-pow export writes it): the dataset's head 16 words and
|
||||
// its last word against the CPU reference, as the one-click workers' pf_selftest does; a mismatch refuses the day (the
|
||||
// miner's CPU re-check would catch every found nonce later, but a wrong dataset is wrong on every hash, so it stops here)
|
||||
if let vdata = FileManager.default.contents(atPath: dir + "/vectors.json"), let vj = (try? JSONSerialization.jsonObject(with: vdata)) as? [String: Any],
|
||||
let headHex = vj["dataset_head"] as? [String], headHex.count == 16, let lastIdx = (vj["dataset_last_index"] as? NSNumber)?.uint64Value, let lastHex = vj["dataset_last"] as? String {
|
||||
func hex32(_ x: String) -> UInt32? { UInt32(x.hasPrefix("0x") ? String(x.dropFirst(2)) : x, radix: 16) }
|
||||
let wantHead = headHex.compactMap(hex32)
|
||||
if wantHead.count == 16, let wantLast = hex32(lastHex), lastIdx < UInt64(words) {
|
||||
let copy = gpu.device.makeBuffer(length: 64 + 4, options: .storageModeShared)!
|
||||
let cb = gpu.queue.makeCommandBuffer()!; let bl = cb.makeBlitCommandEncoder()!
|
||||
bl.copy(from: dataset, sourceOffset: 0, to: copy, destinationOffset: 0, size: 64)
|
||||
bl.copy(from: dataset, sourceOffset: Int(lastIdx) * 4, to: copy, destinationOffset: 64, size: 4)
|
||||
bl.endEncoding(); cb.commit(); cb.waitUntilCompleted()
|
||||
let p = copy.contents().bindMemory(to: UInt32.self, capacity: 17)
|
||||
let headOk = (0..<16).allSatisfy { p[$0] == wantHead[$0] }
|
||||
let lastOk = p[16] == wantLast
|
||||
if !headOk || !lastOk { throw refuse("dataset self-test FAIL against vectors.json (head \(headOk ? "ok" : "BAD"), word [\(lastIdx)] \(lastOk ? "ok" : "BAD")\(stateLeaves > 0 ? "; class v5 with \(stateLeaves) state leaves" : ""))") }
|
||||
}
|
||||
}
|
||||
let sd = ServeDataset(dayHex: dayHex, packClass: packClass, eraHex: eraHex, buffer: dataset, cacheFillGPUms: fillMs, buildGPUms: buildMs)
|
||||
store.add(sd)
|
||||
return (sd, ms(t0, nowNs()))
|
||||
|
|
@ -3206,8 +3254,8 @@ func runServe(_ opts: Options) -> Never {
|
|||
}
|
||||
if wantClass != "" && wantClass != "v2" && !isPackClass(wantClass) {
|
||||
let id = f.count > 1 ? f[1] : "0"
|
||||
if f[0] == "prepare" { emit("prepare-failed \(id) \(f.count > 2 ? f[2] : "0") program class \(wantClass) is not one this worker runs (v2, v3 or v4)") }
|
||||
else { emit("error \(id) program class \(wantClass) is not one this worker runs (v2, v3 or v4)") }
|
||||
if f[0] == "prepare" { emit("prepare-failed \(id) \(f.count > 2 ? f[2] : "0") program class \(wantClass) is not one this worker runs (v2, v3, v4 or v5)") }
|
||||
else { emit("error \(id) program class \(wantClass) is not one this worker runs (v2, v3, v4 or v5)") }
|
||||
continue
|
||||
}
|
||||
if f[0] == "prepare" {
|
||||
|
|
|
|||
|
|
@ -51,6 +51,12 @@ func defineStr(_ name: String) -> String? {
|
|||
guard let re = try? NSRegularExpression(pattern: pat), let m = re.firstMatch(in: programH, range: NSRange(programH.startIndex..., in: programH)) else { return nil }
|
||||
return String(programH[Range(m.range(at: 1), in: programH)!])
|
||||
}
|
||||
/// An unquoted 64-bit define (`0x...ull`), the pack's FNV lines.
|
||||
func defineU64(_ name: String) -> UInt64? {
|
||||
let pat = "#define \(name) 0x([0-9a-fA-F]+)ull"
|
||||
guard let re = try? NSRegularExpression(pattern: pat), let m = re.firstMatch(in: programH, range: NSRange(programH.startIndex..., in: programH)) else { return nil }
|
||||
return UInt64(programH[Range(m.range(at: 1), in: programH)!], radix: 16)
|
||||
}
|
||||
let datasetLog2 = Int(defineU32("IGNEUM_DATASET_LOG2") ?? 28)
|
||||
let datasetMode = defineU32("IGNEUM_DATASET_MODE") ?? 1
|
||||
if datasetMode != 1 { fail("packbench runs memory-hard packs only") }
|
||||
|
|
@ -126,9 +132,30 @@ let (cacheWall, cacheGpu) = run { enc in
|
|||
enc.setComputePipelineState(fillPipe); enc.setBuffer(cache, offset: 0, index: 0)
|
||||
enc.dispatchThreadgroups(MTLSize(width: cacheSegments / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
|
||||
}
|
||||
func fnv1a64(_ p: UnsafeRawPointer, _ n: Int) -> UInt64 {
|
||||
var h: UInt64 = 0xcbf29ce484222325
|
||||
let b = p.bindMemory(to: UInt8.self, capacity: n)
|
||||
for i in 0..<n { h ^= UInt64(b[i]); h = h &* 0x100000001b3 }
|
||||
return h
|
||||
}
|
||||
let items = words / 16
|
||||
// class v5 (docs/design/class-v5-stored-state.md): a pack with IGNEUM_STATE_LEAVES carries leaves.bin (16 words per leaf), bound
|
||||
// as buffer 2 of igneum_build with the count in buffer 3; the pack's FNV of the leaves is checked first
|
||||
var leavesBuf: MTLBuffer? = nil
|
||||
var nLeaves: UInt32 = 0
|
||||
if let n = defineU32("IGNEUM_STATE_LEAVES") {
|
||||
let path = opts.pack + "/" + (defineStr("IGNEUM_STATE_LEAVES_FILE") ?? "leaves.bin")
|
||||
guard let data = FileManager.default.contents(atPath: path) else { fail("class v5 pack without its leaves file \(path)") }
|
||||
if data.count != Int(n) * 64 { fail("leaves.bin is \(data.count) bytes, the pack says \(n) leaves of 64") }
|
||||
let fnv = data.withUnsafeBytes { fnv1a64($0.baseAddress!, data.count) }
|
||||
guard let want = defineU64("IGNEUM_STATE_LEAVES_FNV64") else { fail("class v5 pack without IGNEUM_STATE_LEAVES_FNV64") }
|
||||
if want != fnv { fail("leaves.bin FNV \(String(fnv, radix: 16)) is not the pack's \(String(want, radix: 16))") }
|
||||
guard let b = device.makeBuffer(bytes: (data as NSData).bytes, length: data.count, options: .storageModeShared) else { fail("leaves alloc") }
|
||||
leavesBuf = b; nLeaves = n
|
||||
}
|
||||
let (buildWall, buildGpu) = run { enc in
|
||||
enc.setComputePipelineState(buildPipe); enc.setBuffer(cache, offset: 0, index: 0); enc.setBuffer(dataset, offset: 0, index: 1)
|
||||
if let lb = leavesBuf { enc.setBuffer(lb, offset: 0, index: 2); var n = nLeaves; enc.setBytes(&n, length: 4, index: 3) }
|
||||
enc.dispatchThreadgroups(MTLSize(width: items / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
|
||||
}
|
||||
// cache fingerprint and dataset head/last through a blit to shared memory
|
||||
|
|
@ -138,12 +165,6 @@ func blit(_ src: MTLBuffer, _ offset: Int, _ n: Int) -> MTLBuffer {
|
|||
b.copy(from: src, sourceOffset: offset, to: dst, destinationOffset: 0, size: n); b.endEncoding(); cb.commit(); cb.waitUntilCompleted()
|
||||
return dst
|
||||
}
|
||||
func fnv1a64(_ p: UnsafeRawPointer, _ n: Int) -> UInt64 {
|
||||
var h: UInt64 = 0xcbf29ce484222325
|
||||
let b = p.bindMemory(to: UInt8.self, capacity: n)
|
||||
for i in 0..<n { h ^= UInt64(b[i]); h = h &* 0x100000001b3 }
|
||||
return h
|
||||
}
|
||||
let cacheCopy = blit(cache, 0, cacheWords * 4)
|
||||
let cacheFnv = fnv1a64(cacheCopy.contents(), cacheWords * 4)
|
||||
let cacheOk = cacheFnv == cacheFnvWant
|
||||
|
|
|
|||
247
proto-newpow/class-v5/bench.cu
Normal file
247
proto-newpow/class-v5/bench.cu
Normal file
|
|
@ -0,0 +1,247 @@
|
|||
// Class v5 bench (docs/design/class-v5-stored-state.md section 7). TEST HARNESS ONLY: no pool, no network, no wallet.
|
||||
//
|
||||
// One pack directory in (program.h, vectors.h, memhard.h, kernel.cu, and leaves.bin for a class v5 pack), the shape of
|
||||
// proto-newpow/state-dataset/bench.cu: cache fill on the GPU (twice, CUDA events), the dataset build (twice, the
|
||||
// second pass reported; a class v5 pack's build takes the leaves of leaves.bin, IGNEUM_STATE_LEAVES x 16 words), the
|
||||
// pack's dataset self-test (head, last, 64 samples) and its three vector warps, a warm-up batch of 2^24 at base
|
||||
// nonce 0 (fingerprinted: FNV-1a 64 over the little-endian u64 outputs), N timed batches, then a power window where
|
||||
// nvidia-smi samples at 1 Hz while the hash kernel runs back to back. The same binary shape for the class v4 control
|
||||
// pack and the class v5 pack, so the two rows differ in the pack alone.
|
||||
//
|
||||
// bench [--batches 10] [--batch-log2 24] [--power-seconds 20] [--device 0] [--leaves <leaves.bin>]
|
||||
//
|
||||
// Build (in the pack directory): nvcc -O3 -std=c++17 -arch=sm_89 -Xcompiler -pthread -I. -o bench ../bench.cu kernel.cu
|
||||
|
||||
#include <cuda_runtime.h>
|
||||
#include <cstdint>
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
#include <chrono>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
#include <thread>
|
||||
#include <mutex>
|
||||
|
||||
#include "program.h"
|
||||
#include "vectors.h"
|
||||
#include "memhard.h"
|
||||
|
||||
#define CUDA_CHECK(call) do { cudaError_t err_ = (call); if (err_ != cudaSuccess) { \
|
||||
std::fprintf(stderr, "CUDA error: %s (%d)\n at %s:%d\n in %s\n", cudaGetErrorString(err_), (int)err_, __FILE__, __LINE__, #call); \
|
||||
std::exit(2); } } while (0)
|
||||
|
||||
static uint64_t fnv1a64(const void* p, size_t n) {
|
||||
const uint8_t* b = (const uint8_t*)p;
|
||||
uint64_t h = 0xcbf29ce484222325ull;
|
||||
for (size_t i = 0; i < n; ++i) { h ^= b[i]; h *= 0x100000001b3ull; }
|
||||
return h;
|
||||
}
|
||||
static float eventMs(cudaEvent_t a, cudaEvent_t b) { float ms = 0.f; CUDA_CHECK(cudaEventElapsedTime(&ms, a, b)); return ms; }
|
||||
|
||||
struct Opts { int batches = 10; int batchLog2 = 24; int powerSeconds = 20; int device = 0; std::string leaves = "leaves.bin"; };
|
||||
static Opts parse(int argc, char** argv) {
|
||||
Opts o;
|
||||
for (int i = 1; i < argc; ++i) {
|
||||
std::string a = argv[i];
|
||||
auto need = [&](int n) { if (i + n >= argc) { std::printf("%s needs %d argument(s)\n", a.c_str(), n); std::exit(2); } };
|
||||
if (a == "--batches") { need(1); o.batches = std::atoi(argv[++i]); }
|
||||
else if (a == "--batch-log2") { need(1); o.batchLog2 = std::atoi(argv[++i]); }
|
||||
else if (a == "--power-seconds") { need(1); o.powerSeconds = std::atoi(argv[++i]); }
|
||||
else if (a == "--device") { need(1); o.device = std::atoi(argv[++i]); }
|
||||
else if (a == "--leaves") { need(1); o.leaves = argv[++i]; }
|
||||
else { std::printf("unknown argument %s\n", argv[i]); std::exit(2); }
|
||||
}
|
||||
return o;
|
||||
}
|
||||
|
||||
// nvidia-smi sampler: one line per second on its own thread until `timeout` ends the process.
|
||||
struct Sampler {
|
||||
std::vector<double> power, sm, mem;
|
||||
std::mutex m;
|
||||
std::thread t;
|
||||
void start(int device, int seconds) {
|
||||
t = std::thread([this, device, seconds]() {
|
||||
char cmd[256];
|
||||
std::snprintf(cmd, sizeof(cmd), "timeout %d nvidia-smi -i %d --query-gpu=power.draw,clocks.sm,clocks.mem --format=csv,noheader,nounits -l 1 2>/dev/null", seconds + 2, device);
|
||||
FILE* f = popen(cmd, "r");
|
||||
if (!f) return;
|
||||
char line[256];
|
||||
while (std::fgets(line, sizeof(line), f)) {
|
||||
double p = 0, s = 0, mm = 0;
|
||||
if (std::sscanf(line, "%lf, %lf, %lf", &p, &s, &mm) == 3) { std::lock_guard<std::mutex> g(m); power.push_back(p); sm.push_back(s); mem.push_back(mm); }
|
||||
}
|
||||
pclose(f);
|
||||
});
|
||||
}
|
||||
void join() { if (t.joinable()) t.join(); }
|
||||
};
|
||||
|
||||
int main(int argc, char** argv) {
|
||||
Opts o = parse(argc, argv);
|
||||
#ifdef IGNEUM_STATE_LEAVES
|
||||
const bool v5 = true;
|
||||
#else
|
||||
const bool v5 = false;
|
||||
#endif
|
||||
std::printf("class-v5 bench pack \"%s\" class %s generator %d (test harness: no pool, no network, no wallet)\n", IGNEUM_SEED_STRING, v5 ? "v5" : "control", (int)IGNEUM_GENERATOR);
|
||||
CUDA_CHECK(cudaSetDevice(o.device));
|
||||
cudaDeviceProp prop;
|
||||
std::memset(&prop, 0, sizeof(prop));
|
||||
CUDA_CHECK(cudaGetDeviceProperties(&prop, o.device));
|
||||
int drv = 0, rt = 0;
|
||||
CUDA_CHECK(cudaDriverGetVersion(&drv));
|
||||
CUDA_CHECK(cudaRuntimeGetVersion(&rt));
|
||||
std::printf("GPU: %s (%d SMs, cc %d.%d, %.0f MiB), CUDA driver %d.%d runtime %d.%d\n", prop.name, prop.multiProcessorCount, prop.major, prop.minor,
|
||||
(double)prop.totalGlobalMem / 1048576.0, drv / 1000, (drv % 100) / 10, rt / 1000, (rt % 100) / 10);
|
||||
int regs = 0, bps = 0;
|
||||
CUDA_CHECK(igneum_hash_info(®s, &bps, 1u));
|
||||
std::printf("hash kernel: %d registers/thread, %d resident blocks/SM at 1 warp/block\n", regs, bps);
|
||||
size_t free0 = 0, total = 0;
|
||||
CUDA_CHECK(cudaMemGetInfo(&free0, &total));
|
||||
std::printf("device memory at start: %.0f MiB used of %.0f MiB\n", (double)(total - free0) / 1048576.0, (double)total / 1048576.0);
|
||||
|
||||
cudaEvent_t e0, e1;
|
||||
CUDA_CHECK(cudaEventCreate(&e0));
|
||||
CUDA_CHECK(cudaEventCreate(&e1));
|
||||
|
||||
// ---- cache
|
||||
const uint32_t cacheWords = 1u << IGNEUM_CACHE_LOG2_WORDS;
|
||||
uint32_t* dCache = nullptr;
|
||||
CUDA_CHECK(cudaMalloc(&dCache, (size_t)cacheWords * 4u));
|
||||
float cacheFill[2] = {0, 0};
|
||||
for (int pass = 0; pass < 2; ++pass) {
|
||||
CUDA_CHECK(cudaEventRecord(e0));
|
||||
CUDA_CHECK(igneum_launch_cache_fill(dCache, IGNEUM_CACHE_SEGMENTS));
|
||||
CUDA_CHECK(cudaEventRecord(e1));
|
||||
CUDA_CHECK(cudaEventSynchronize(e1));
|
||||
cacheFill[pass] = eventMs(e0, e1);
|
||||
}
|
||||
std::printf("cache fill (GPU): %.2f ms first, %.2f ms second\n", cacheFill[0], cacheFill[1]);
|
||||
{
|
||||
std::vector<uint32_t> head(16), last(16);
|
||||
CUDA_CHECK(cudaMemcpy(head.data(), dCache, 64, cudaMemcpyDeviceToHost));
|
||||
CUDA_CHECK(cudaMemcpy(last.data(), dCache + (cacheWords - 16), 64, cudaMemcpyDeviceToHost));
|
||||
bool ok = std::memcmp(head.data(), IGNEUM_CACHE_HEAD, 64) == 0 && std::memcmp(last.data(), IGNEUM_CACHE_LAST, 64) == 0;
|
||||
std::printf("cache head and last 16 words against the pack: %s\n", ok ? "PASS" : "FAIL");
|
||||
}
|
||||
|
||||
// ---- leaves (class v5)
|
||||
uint32_t* dLeaves = nullptr;
|
||||
uint32_t nLeaves = 0;
|
||||
#ifdef IGNEUM_STATE_LEAVES
|
||||
{
|
||||
FILE* f = std::fopen(o.leaves.c_str(), "rb");
|
||||
if (!f) { std::printf("FAIL: cannot open %s (the pack's leaves.bin)\n", o.leaves.c_str()); return 2; }
|
||||
std::fseek(f, 0, SEEK_END);
|
||||
long n = std::ftell(f);
|
||||
std::fseek(f, 0, SEEK_SET);
|
||||
std::vector<uint32_t> h((size_t)n / 4);
|
||||
if (std::fread(h.data(), 1, (size_t)n, f) != (size_t)n) { std::printf("FAIL: short read of %s\n", o.leaves.c_str()); return 2; }
|
||||
std::fclose(f);
|
||||
nLeaves = (uint32_t)(h.size() / 16);
|
||||
uint64_t fnv = fnv1a64(h.data(), h.size() * 4);
|
||||
std::printf("leaves: %u x 64 B from %s (%ld bytes), FNV-1a 64 %016llx against the pack's %016llx: %s; state root %s, chain block %s (%llu)\n",
|
||||
nLeaves, o.leaves.c_str(), n, (unsigned long long)fnv, (unsigned long long)IGNEUM_STATE_LEAVES_FNV64,
|
||||
fnv == IGNEUM_STATE_LEAVES_FNV64 && nLeaves == IGNEUM_STATE_LEAVES ? "PASS" : "FAIL", IGNEUM_STATE_ROOT_HEX, IGNEUM_STATE_BLOCK_HEX, (unsigned long long)IGNEUM_STATE_BLOCK_NUMBER);
|
||||
CUDA_CHECK(cudaMalloc(&dLeaves, h.size() * 4));
|
||||
CUDA_CHECK(cudaMemcpy(dLeaves, h.data(), h.size() * 4, cudaMemcpyHostToDevice));
|
||||
}
|
||||
#endif
|
||||
|
||||
// ---- dataset build, twice
|
||||
const uint32_t words = 1u << IGNEUM_DATASET_LOG2;
|
||||
const uint32_t nItems = words >> 4;
|
||||
uint32_t* dDs = nullptr;
|
||||
CUDA_CHECK(cudaMalloc(&dDs, (size_t)words * 4u));
|
||||
float build[2] = {0, 0};
|
||||
for (int pass = 0; pass < 2; ++pass) {
|
||||
CUDA_CHECK(cudaEventRecord(e0));
|
||||
#ifdef IGNEUM_STATE_LEAVES
|
||||
CUDA_CHECK(igneum_launch_build(dDs, dCache, dLeaves, nLeaves, nItems));
|
||||
#else
|
||||
CUDA_CHECK(igneum_launch_build(dDs, dCache, nItems));
|
||||
#endif
|
||||
CUDA_CHECK(cudaEventRecord(e1));
|
||||
CUDA_CHECK(cudaEventSynchronize(e1));
|
||||
build[pass] = eventMs(e0, e1);
|
||||
}
|
||||
std::printf("dataset build (GPU): %.2f ms first, %.2f ms second (%u items, %u MiB)\n", build[0], build[1], nItems, words >> 18);
|
||||
size_t free1 = 0;
|
||||
CUDA_CHECK(cudaMemGetInfo(&free1, &total));
|
||||
std::printf("device memory after the build: %.0f MiB used\n", (double)(total - free1) / 1048576.0);
|
||||
{
|
||||
std::vector<uint32_t> head(16);
|
||||
uint32_t last = 0;
|
||||
CUDA_CHECK(cudaMemcpy(head.data(), dDs, 64, cudaMemcpyDeviceToHost));
|
||||
CUDA_CHECK(cudaMemcpy(&last, dDs + IGNEUM_DS_LAST_INDEX, 4, cudaMemcpyDeviceToHost));
|
||||
int sampleOk = 0;
|
||||
for (int i = 0; i < IGNEUM_DS_SAMPLES; ++i) {
|
||||
uint32_t v = 0;
|
||||
CUDA_CHECK(cudaMemcpy(&v, dDs + IGNEUM_DS_SAMPLE_INDEX[i], 4, cudaMemcpyDeviceToHost));
|
||||
if (v == IGNEUM_DS_SAMPLE_VALUE[i]) ++sampleOk;
|
||||
}
|
||||
std::printf("dataset self-test: head %s, last %s, samples %d of %d\n", std::memcmp(head.data(), IGNEUM_DS_HEAD, 64) == 0 ? "PASS" : "FAIL", last == IGNEUM_DS_LAST ? "PASS" : "FAIL", sampleOk, (int)IGNEUM_DS_SAMPLES);
|
||||
}
|
||||
#ifdef IGNEUM_STATE_LEAVES
|
||||
if (dLeaves) { CUDA_CHECK(cudaFree(dLeaves)); dLeaves = nullptr; }
|
||||
#endif
|
||||
|
||||
// ---- the three vector warps
|
||||
const uint32_t nonces = 1u << o.batchLog2;
|
||||
uint64_t* dOut = nullptr;
|
||||
CUDA_CHECK(cudaMalloc(&dOut, (size_t)nonces * 8u));
|
||||
std::vector<uint64_t> hOut(nonces);
|
||||
int vecOk = 0;
|
||||
for (int w = 0; w < IGNEUM_VEC_WARPS; ++w) {
|
||||
CUDA_CHECK(igneum_launch_hash(dDs, dOut, IGNEUM_VEC_BASE[w], IGNEUM_MASK, 32u, 1u));
|
||||
CUDA_CHECK(cudaDeviceSynchronize());
|
||||
CUDA_CHECK(cudaMemcpy(hOut.data(), dOut, 32 * 8, cudaMemcpyDeviceToHost));
|
||||
if (std::memcmp(hOut.data(), IGNEUM_VEC_OUT[w], 32 * 8) == 0) ++vecOk;
|
||||
}
|
||||
std::printf("vector warps against the pack: %d of %d PASS\n", vecOk, (int)IGNEUM_VEC_WARPS);
|
||||
|
||||
// ---- warm-up and fingerprint, then timed batches
|
||||
const uint32_t blockWarps = 1u;
|
||||
CUDA_CHECK(igneum_launch_hash(dDs, dOut, 0u, IGNEUM_MASK, nonces, blockWarps));
|
||||
CUDA_CHECK(cudaDeviceSynchronize());
|
||||
CUDA_CHECK(cudaMemcpy(hOut.data(), dOut, (size_t)nonces * 8u, cudaMemcpyDeviceToHost));
|
||||
std::printf("fingerprint of 2^%d outputs at base 0: %016llx (lane 0 %016llx)\n", o.batchLog2, (unsigned long long)fnv1a64(hOut.data(), (size_t)nonces * 8u), (unsigned long long)hOut[0]);
|
||||
double totalMs = 0;
|
||||
for (int b = 0; b < o.batches; ++b) {
|
||||
CUDA_CHECK(cudaEventRecord(e0));
|
||||
CUDA_CHECK(igneum_launch_hash(dDs, dOut, (uint32_t)(b + 1) * nonces, IGNEUM_MASK, nonces, blockWarps));
|
||||
CUDA_CHECK(cudaEventRecord(e1));
|
||||
CUDA_CHECK(cudaEventSynchronize(e1));
|
||||
totalMs += eventMs(e0, e1);
|
||||
}
|
||||
double rate = (double)o.batches * nonces / (totalMs / 1000.0) / 1e6;
|
||||
std::printf("hash rate: %.3f MH/s over %d batches of 2^%d (GPU event time %.1f ms)\n", rate, o.batches, o.batchLog2, totalMs);
|
||||
|
||||
// ---- power window
|
||||
if (o.powerSeconds > 0) {
|
||||
Sampler s;
|
||||
s.start(o.device, o.powerSeconds);
|
||||
auto t0 = std::chrono::steady_clock::now();
|
||||
uint64_t done = 0;
|
||||
uint32_t base = 0;
|
||||
while (std::chrono::duration<double>(std::chrono::steady_clock::now() - t0).count() < o.powerSeconds) {
|
||||
CUDA_CHECK(igneum_launch_hash(dDs, dOut, base, IGNEUM_MASK, nonces, blockWarps));
|
||||
CUDA_CHECK(cudaDeviceSynchronize());
|
||||
base += nonces;
|
||||
done += nonces;
|
||||
}
|
||||
double secs = std::chrono::duration<double>(std::chrono::steady_clock::now() - t0).count();
|
||||
s.join();
|
||||
std::lock_guard<std::mutex> g(s.m);
|
||||
double p = 0, sm = 0, mem = 0;
|
||||
int n = 0;
|
||||
for (size_t i = 10; i < s.power.size(); ++i) { p += s.power[i]; sm += s.sm[i]; mem += s.mem[i]; ++n; }
|
||||
if (n > 0) { p /= n; sm /= n; mem /= n; }
|
||||
double rateW = done / secs / 1e6;
|
||||
std::printf("power window: %.1f s, %.3f MH/s sustained, %.1f W mean after the first 10 s (%d samples of %zu), SM %.0f MHz, mem %.0f MHz, %.3f MH/s per W, %.2f microjoules per hash\n",
|
||||
secs, rateW, p, n, s.power.size(), sm, mem, p > 0 ? rateW / p : 0.0, p > 0 ? p / (rateW * 1e6) * 1e6 : 0.0);
|
||||
}
|
||||
std::printf("RESULT pack=%s class=%s build_ms=%.2f rate_mhs=%.3f vectors=%d/%d\n", IGNEUM_SEED_STRING, v5 ? "v5" : "control", build[1], rate, vecOk, (int)IGNEUM_VEC_WARPS);
|
||||
return vecOk == IGNEUM_VEC_WARPS ? 0 : 1;
|
||||
}
|
||||
18
proto-newpow/class-v5/run.sh
Executable file
18
proto-newpow/class-v5/run.sh
Executable file
|
|
@ -0,0 +1,18 @@
|
|||
#!/usr/bin/env bash
|
||||
# Class v5 against class v4 on one card (docs/design/class-v5-stored-state.md section 7). Run in proto-newpow/class-v5
|
||||
# on the pod with the two exported packs beside it: packs/v4-genesis (control) and packs/v5-genesis (leaves.bin inside).
|
||||
# bash run.sh [sm_89] > run.log 2>&1
|
||||
set -u
|
||||
ARCH=${1:-sm_89}
|
||||
export PATH=/usr/local/cuda/bin:$PATH
|
||||
nvidia-smi --query-gpu=name,driver_version,power.limit --format=csv,noheader
|
||||
for pack in v4-genesis v5-genesis; do
|
||||
( cd packs/$pack && nvcc -O3 -std=c++17 -arch=$ARCH -Xcompiler -pthread -I. -o bench ../../bench.cu kernel.cu 2>&1 | grep -v "^$" ; echo "nvcc $pack rc ${PIPESTATUS[0]}" )
|
||||
done
|
||||
for pass in 1 2; do
|
||||
for pack in v4-genesis v5-genesis; do
|
||||
echo "=== pass $pass $pack $(date -u +%H:%M:%SZ)"
|
||||
( cd packs/$pack && ./bench --batches 10 --power-seconds 20 )
|
||||
done
|
||||
done
|
||||
echo "=== done $(date -u +%H:%M:%SZ)"
|
||||
|
|
@ -11,6 +11,10 @@
|
|||
//
|
||||
// Build: see README.md (macOS -framework OpenCL, Linux -lOpenCL, Windows cl.exe + OpenCL.lib), or build.sh / build.bat.
|
||||
|
||||
#if !defined(_WIN32) && !defined(__APPLE__) && !defined(_POSIX_C_SOURCE)
|
||||
#define _POSIX_C_SOURCE 200809L /* clock_gettime and strtok_r under -std=c99 on Linux (glibc hides them without it; the kit build of 7 October 2026) */
|
||||
#define _DEFAULT_SOURCE
|
||||
#endif
|
||||
#define _CRT_SECURE_NO_WARNINGS
|
||||
#define CL_TARGET_OPENCL_VERSION 120
|
||||
#define CL_USE_DEPRECATED_OPENCL_1_2_APIS
|
||||
|
|
@ -184,10 +188,19 @@ static uint64_t fnv1a64(const void* p, size_t n) {
|
|||
}
|
||||
static const uint32_t CACHE_WORDS_HOST = 1u << IGNEUM_CACHE_LOG2_WORDS;
|
||||
static uint32_t* hCache = NULL;
|
||||
#ifdef IGNEUM_STATE_LEAVES
|
||||
/* class v5 compiled-in pack: the host copy of the pack's leaves (leaves.bin beside IGNEUM_KERNEL_PATH, loaded and checked by
|
||||
* compiledInLeavesBuffer before the build), so the host reference derives the same words the device does */
|
||||
static uint32_t* hLeaves = NULL;
|
||||
#endif
|
||||
/* dataset[w] through the pack's own mh_word (memhard.h), which carries the pack's item-to-word layout (era layout,
|
||||
* 5 October 2026; the former w >> 4 / w & 15 here failed the random points of every interleaved pack). */
|
||||
static uint32_t host_ds_word(uint32_t w) {
|
||||
#ifdef IGNEUM_STATE_LEAVES
|
||||
return mh_word(hCache, hLeaves, IGNEUM_STATE_LEAVES, w); /* class v5: every item keyed by leaf(t) = leaves[t mod n] */
|
||||
#else
|
||||
return mh_word(hCache, w);
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
|
||||
|
|
@ -505,6 +518,13 @@ typedef struct {
|
|||
int groupSize; // work-group size the program was built for (IGNEUM_GROUP = 32 x group-warps)
|
||||
} Device;
|
||||
|
||||
/* class v5 build arguments and leaf buffers (defined after the serve pair below) */
|
||||
static cl_int setBuildArgs(cl_kernel k, cl_mem ds, cl_mem cache, cl_mem leaves, cl_uint nLeaves, cl_uint nItems);
|
||||
static int packLeavesBuffer(Device* dv, const char* packDir, cl_mem* leaves, cl_uint* nLeaves, char* err, size_t errCap);
|
||||
#ifdef IGNEUM_STATE_LEAVES
|
||||
static int compiledInLeavesBuffer(Device* dv, cl_mem* leaves, cl_uint* nLeaves, char* err, size_t errCap);
|
||||
#endif
|
||||
|
||||
static const char* exchangeName(int m) { return m == 1 ? "sub_group_shuffle_xor (cl_khr_subgroup_shuffle)" : m == 2 ? "intel_sub_group_shuffle_xor (cl_intel_subgroups)" : "local-memory exchange with barrier"; }
|
||||
|
||||
// Returns 0 on success, 1 on build failure (log printed).
|
||||
|
|
@ -819,10 +839,19 @@ static SizeResult runSize(Device* dv, const DeviceInfo* di, const Options* o, in
|
|||
#if IGNEUM_DATASET_MODE == 1
|
||||
cl_uint nItems = r.words / 16u;
|
||||
size_t local = kernelMaxLocal(dv, dv->kBuild, di, 256);
|
||||
CL_CHECK(clSetKernelArg(dv->kBuild, 0, sizeof(cl_mem), &dDs));
|
||||
CL_CHECK(clSetKernelArg(dv->kBuild, 1, sizeof(cl_mem), &gCache));
|
||||
CL_CHECK(clSetKernelArg(dv->kBuild, 2, sizeof(cl_uint), &nItems));
|
||||
cl_mem leaves = NULL;
|
||||
cl_uint nLeaves = 0;
|
||||
#ifdef IGNEUM_STATE_LEAVES
|
||||
{ /* class v5 compiled-in pack: its leaves from the pack directory beside the kernel */
|
||||
char lerr[800];
|
||||
if (!compiledInLeavesBuffer(dv, &leaves, &nLeaves, lerr, sizeof(lerr))) { printf("FAIL: %s\n", lerr); exit(2); }
|
||||
if (pass == 0) printf("class v5: %u state leaves uploaded for the build (FNV-1a 64 %016llx checked against the pack)\n", (unsigned)nLeaves, (unsigned long long)IGNEUM_STATE_LEAVES_FNV64);
|
||||
}
|
||||
#endif
|
||||
CL_CHECK(setBuildArgs(dv->kBuild, dDs, gCache, leaves, nLeaves, nItems));
|
||||
ev = launch1D(dv, dv->kBuild, nItems, local);
|
||||
CL_CHECK(clFinish(dv->q));
|
||||
if (leaves) { clReleaseMemObject(leaves); ++gMemReleased; }
|
||||
#else
|
||||
cl_uint n = r.words, d0 = IGNEUM_DAY0, d1 = IGNEUM_DAY1;
|
||||
size_t local = kernelMaxLocal(dv, dv->kFill, di, 256);
|
||||
|
|
@ -1088,6 +1117,7 @@ typedef struct {
|
|||
int checked;
|
||||
char programClass[8]; /* Counter ASIC 2.0: the pack's class ("v2" or "v3") and era seed, from packfile.h */
|
||||
char eraHex[65];
|
||||
cl_uint stateLeaves; /* class v5: the state leaves the dataset was built from (uploaded for the build, released after); 0 otherwise */
|
||||
} ServePair;
|
||||
|
||||
static int hexEq(const char* a, const char* b) {
|
||||
|
|
@ -1132,6 +1162,65 @@ static void releasePair(ServePair* p) {
|
|||
free(p);
|
||||
}
|
||||
|
||||
/* Class v5 (docs/design/class-v5-stored-state.md, 7 October 2026): igneum_build takes the window's state leaves. The
|
||||
* emitted kernel of a class v5 pack is igneum_build(ds, cache, leaves, nLeaves, nItems); every other class's is
|
||||
* igneum_build(ds, cache, nItems). One place sets the arguments for both shapes. */
|
||||
static cl_int setBuildArgs(cl_kernel k, cl_mem ds, cl_mem cache, cl_mem leaves, cl_uint nLeaves, cl_uint nItems) {
|
||||
cl_int e = clSetKernelArg(k, 0, sizeof(cl_mem), &ds);
|
||||
if (e == CL_SUCCESS) e = clSetKernelArg(k, 1, sizeof(cl_mem), &cache);
|
||||
if (leaves) {
|
||||
if (e == CL_SUCCESS) e = clSetKernelArg(k, 2, sizeof(cl_mem), &leaves);
|
||||
if (e == CL_SUCCESS) e = clSetKernelArg(k, 3, sizeof(cl_uint), &nLeaves);
|
||||
if (e == CL_SUCCESS) e = clSetKernelArg(k, 4, sizeof(cl_uint), &nItems);
|
||||
} else if (e == CL_SUCCESS) e = clSetKernelArg(k, 2, sizeof(cl_uint), &nItems);
|
||||
return e;
|
||||
}
|
||||
|
||||
/* The leaves of a pack directory as a device buffer (NULL with *nLeaves 0 for a pack of another class): read and checked
|
||||
* against the pack's count and FNV-1a 64 by pf_load_leaves before the upload. The caller releases the buffer after the
|
||||
* build (the leaves are not needed while hashing). Returns 1, or 0 with err set. */
|
||||
static int packLeavesBuffer(Device* dv, const char* packDir, cl_mem* leaves, cl_uint* nLeaves, char* err, size_t errCap) {
|
||||
PfPack pk;
|
||||
char perr[700];
|
||||
uint32_t* h = NULL;
|
||||
size_t bytes = 0;
|
||||
cl_int e = 0;
|
||||
*leaves = NULL; *nLeaves = 0;
|
||||
if (!packDir || !packDir[0]) return 1;
|
||||
if (!pf_load(packDir, &pk, perr, sizeof(perr))) { snprintf(err, errCap, "pack %s: %s", packDir, perr); return 0; }
|
||||
if (!pf_load_leaves(packDir, &pk, &h, &bytes, perr, sizeof(perr))) { snprintf(err, errCap, "pack %s: %s", packDir, perr); return 0; }
|
||||
if (!h) return 1;
|
||||
*leaves = clCreateBuffer(dv->ctx, CL_MEM_READ_ONLY | CL_MEM_COPY_HOST_PTR, bytes, h, &e);
|
||||
free(h);
|
||||
if (e != CL_SUCCESS) { *leaves = NULL; snprintf(err, errCap, "clCreateBuffer state leaves (%s)", clErrName(e)); return 0; }
|
||||
++gMemCreated;
|
||||
*nLeaves = pk.stateLeaves;
|
||||
return 1;
|
||||
}
|
||||
|
||||
#ifdef IGNEUM_STATE_LEAVES
|
||||
/* The compiled-in pack is a class v5 pack: its leaves live beside IGNEUM_KERNEL_PATH (the directory build.sh compiled
|
||||
* against). The plain bench and the placeholder serve path build from them; without the file they build nothing. */
|
||||
static int compiledInLeavesBuffer(Device* dv, cl_mem* leaves, cl_uint* nLeaves, char* err, size_t errCap) {
|
||||
static char dir[1200];
|
||||
const char* k = IGNEUM_KERNEL_PATH;
|
||||
const char* slash = strrchr(k, '/');
|
||||
#ifdef _WIN32
|
||||
const char* bslash = strrchr(k, '\\');
|
||||
if (bslash && (!slash || bslash > slash)) slash = bslash;
|
||||
#endif
|
||||
if (slash) { size_t n = (size_t)(slash - k); if (n >= sizeof(dir)) n = sizeof(dir) - 1; memcpy(dir, k, n); dir[n] = 0; } else strcpy(dir, ".");
|
||||
if (!hLeaves) {
|
||||
/* the host copy for host_ds_word (the 64 random points of the dataset self-test), checked by pf_load_leaves */
|
||||
PfPack pk; char perr[700]; size_t bytes = 0;
|
||||
if (!pf_load(dir, &pk, perr, sizeof(perr))) { snprintf(err, errCap, "pack %s: %s", dir, perr); return 0; }
|
||||
if (!pf_load_leaves(dir, &pk, &hLeaves, &bytes, perr, sizeof(perr))) { snprintf(err, errCap, "pack %s: %s", dir, perr); return 0; }
|
||||
if (!hLeaves || pk.stateLeaves != IGNEUM_STATE_LEAVES) { snprintf(err, errCap, "pack %s: program.h on disk says %u state leaves, this exe was built for %u", dir, (unsigned)pk.stateLeaves, (unsigned)IGNEUM_STATE_LEAVES); return 0; }
|
||||
}
|
||||
return packLeavesBuffer(dv, dir, leaves, nLeaves, err, errCap);
|
||||
}
|
||||
#endif
|
||||
|
||||
/* The prepare request and its result, handed between the main loop and the prepare thread. */
|
||||
typedef struct {
|
||||
Device* dv;
|
||||
|
|
@ -1254,12 +1343,19 @@ static int pairBuffers(Device* dv, const DeviceInfo* di, cl_command_queue q, Ser
|
|||
++gMemCreated;
|
||||
local = kernelMaxLocal(dv, p->kBuild, di, 256);
|
||||
g = ((nItems + local - 1) / local) * local;
|
||||
e = clSetKernelArg(p->kBuild, 0, sizeof(cl_mem), &p->ds);
|
||||
if (e == CL_SUCCESS) e = clSetKernelArg(p->kBuild, 1, sizeof(cl_mem), &p->cache);
|
||||
if (e == CL_SUCCESS) e = clSetKernelArg(p->kBuild, 2, sizeof(cl_uint), &nItems);
|
||||
if (e == CL_SUCCESS) e = clEnqueueNDRangeKernel(q, p->kBuild, 1, NULL, &g, &local, 0, NULL, NULL);
|
||||
if (e == CL_SUCCESS) e = clFinish(q);
|
||||
if (e != CL_SUCCESS) { snprintf(err, errCap, "dataset build (%s)", clErrName(e)); return 0; }
|
||||
{
|
||||
/* class v5: the pack's state leaves, uploaded for the build (checked by pf_load_leaves) and released after it;
|
||||
* a v5 pack without them returns 0 here and the pair is never served (the known-failed case) */
|
||||
cl_mem leaves = NULL;
|
||||
cl_uint nLeaves = 0;
|
||||
if (!packLeavesBuffer(dv, packDir, &leaves, &nLeaves, err, errCap)) return 0;
|
||||
e = setBuildArgs(p->kBuild, p->ds, p->cache, leaves, nLeaves, nItems);
|
||||
if (e == CL_SUCCESS) e = clEnqueueNDRangeKernel(q, p->kBuild, 1, NULL, &g, &local, 0, NULL, NULL);
|
||||
if (e == CL_SUCCESS) e = clFinish(q);
|
||||
if (leaves) { clReleaseMemObject(leaves); ++gMemReleased; }
|
||||
if (e != CL_SUCCESS) { snprintf(err, errCap, "dataset build (%s%s)", clErrName(e), nLeaves ? ", class v5 with the state leaves" : ""); return 0; }
|
||||
p->stateLeaves = nLeaves;
|
||||
}
|
||||
p->datasetMs = wallMs() - tb;
|
||||
if (p->kHotFill && p->hotWords) {
|
||||
/* hot-table experiment: the epoch's table from the pack's own fill kernel (one work-item per segment) */
|
||||
|
|
@ -1395,7 +1491,8 @@ static int runBenchPack(Device* dv, const DeviceInfo* di, const Options* o) {
|
|||
memcpy(cur->sw, gPack.seedw, 32); memcpy(cur->kw, gPack.keyw, 32);
|
||||
if (!pairHotKernel(cur, o->packDir, perr, sizeof(perr))) { printf("FAIL: pack %s: %s\n", o->packDir, perr); return 1; }
|
||||
if (!pairBuffers(dv, di, dv->q, cur, words, gServeCacheWords, gServeSegments, o->packDir, perr, sizeof(perr))) { printf("FAIL: pack %s: %s\n", o->packDir, perr); return 1; }
|
||||
printf("pack %s: cache %.0f dataset %.0f hot %.0f check %.0f ms (%.0f ms in all); %s\n", o->packDir, cur->cacheMs, cur->datasetMs, cur->hotMs, cur->checkMs, wallMs() - t0, cur->check);
|
||||
printf("pack %s: cache %.0f dataset %.0f hot %.0f check %.0f ms (%.0f ms in all); class %s%s; %s\n", o->packDir, cur->cacheMs, cur->datasetMs, cur->hotMs, cur->checkMs, wallMs() - t0,
|
||||
gPack.programClass, cur->stateLeaves ? " (class v5: the state leaves uploaded for the build and released)" : "", cur->check);
|
||||
printKernelInfo(di, cur->kHashBound, "igneum_hash_bound", (int)groupSize, "");
|
||||
dOut = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, (size_t)nonces * sizeof(uint64_t), NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer out");
|
||||
dInit = clCreateBuffer(dv->ctx, CL_MEM_READ_ONLY | CL_MEM_COPY_HOST_PTR, 32, cur->sw, &err); CL_CHECK_ERR(err, "clCreateBuffer init words");
|
||||
|
|
@ -1478,17 +1575,26 @@ static int runServe(Device* dv, const DeviceInfo* di, const Options* o) {
|
|||
strncpy(cur->programClass, gPack.programClass, sizeof(cur->programClass) - 1); strncpy(cur->eraHex, gPack.eraHex, sizeof(cur->eraHex) - 1);
|
||||
if (!pairHotKernel(cur, o->packDir, perr, sizeof(perr))) { printf("error 0 pack %s: %s\n", o->packDir, perr); fflush(stdout); return 1; }
|
||||
if (!pairBuffers(dv, di, dv->q, cur, words, gServeCacheWords, gServeSegments, o->packDir, perr, sizeof(perr))) { printf("error 0 pack %s: %s\n", o->packDir, perr); fflush(stdout); return 1; }
|
||||
printf("info first pack %s: cache %.0f dataset %.0f check %.0f ms (%.0f ms in all); %s\n", o->packDir, cur->cacheMs, cur->datasetMs, cur->checkMs, wallMs() - t0, cur->check);
|
||||
printf("info first pack %s: cache %.0f dataset %.0f check %.0f ms (%.0f ms in all); class %s%s; %s\n", o->packDir, cur->cacheMs, cur->datasetMs, cur->checkMs, wallMs() - t0,
|
||||
cur->programClass[0] ? cur->programClass : "v2", cur->stateLeaves ? " (class v5: the state leaves uploaded for the build and released)" : "", cur->check);
|
||||
} else {
|
||||
if (!setupCache(dv, di)) { printf("error 0 cache check failed (device cache differs from the host cache or the pack's FNV)\n"); fflush(stdout); return 1; }
|
||||
memcpy(cur->sw, SEEDW, 32); memcpy(cur->kw, KEYW, 32);
|
||||
cur->cache = gCache; gCache = NULL;
|
||||
cur->ds = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, (size_t)words * 4u, NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer dataset");
|
||||
CL_CHECK(clSetKernelArg(cur->kBuild, 0, sizeof(cl_mem), &cur->ds));
|
||||
CL_CHECK(clSetKernelArg(cur->kBuild, 1, sizeof(cl_mem), &cur->cache));
|
||||
CL_CHECK(clSetKernelArg(cur->kBuild, 2, sizeof(cl_uint), &nItems));
|
||||
countRelease(launch1D(dv, cur->kBuild, nItems, kernelMaxLocal(dv, cur->kBuild, di, 256)));
|
||||
CL_CHECK(clFinish(dv->q));
|
||||
{
|
||||
cl_mem leaves = NULL;
|
||||
cl_uint nLeaves = 0;
|
||||
#ifdef IGNEUM_STATE_LEAVES
|
||||
char lerr[800];
|
||||
if (!compiledInLeavesBuffer(dv, &leaves, &nLeaves, lerr, sizeof(lerr))) { printf("error 0 %s\n", lerr); fflush(stdout); return 1; }
|
||||
#endif
|
||||
CL_CHECK(setBuildArgs(cur->kBuild, cur->ds, cur->cache, leaves, nLeaves, nItems));
|
||||
countRelease(launch1D(dv, cur->kBuild, nItems, kernelMaxLocal(dv, cur->kBuild, di, 256)));
|
||||
CL_CHECK(clFinish(dv->q));
|
||||
if (leaves) { clReleaseMemObject(leaves); ++gMemReleased; }
|
||||
cur->stateLeaves = nLeaves;
|
||||
}
|
||||
}
|
||||
dOut = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, (size_t)batch * sizeof(uint64_t), NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer out");
|
||||
dInit = clCreateBuffer(dv->ctx, CL_MEM_READ_ONLY, 32, NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer init words");
|
||||
|
|
|
|||
|
|
@ -4,13 +4,13 @@
|
|||
<meta charset="utf-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1, viewport-fit=cover">
|
||||
<title>Igneum ledger: every criticism, answered</title>
|
||||
<meta name="description" content="Every criticism Igneum expects, in the critic's words, with what was done, the status and the date. 183 entries. Nothing deleted, nothing softened.">
|
||||
<meta name="description" content="Every criticism Igneum expects, in the critic's words, with what was done, the status and the date. 184 entries. Nothing deleted, nothing softened.">
|
||||
<link rel="canonical" href="https://igneum.network/ledger">
|
||||
<meta name="theme-color" content="#0C0C0E">
|
||||
<meta property="og:type" content="website">
|
||||
<meta property="og:site_name" content="Igneum">
|
||||
<meta property="og:title" content="Igneum ledger: every criticism, answered">
|
||||
<meta property="og:description" content="183 criticisms in the critic's words, with what was done, the status and the date.">
|
||||
<meta property="og:description" content="184 criticisms in the critic's words, with what was done, the status and the date.">
|
||||
<meta property="og:url" content="https://igneum.network/ledger">
|
||||
<meta property="og:image" content="https://igneum.network/og-small.png?v=3">
|
||||
<meta property="og:image:width" content="256">
|
||||
|
|
@ -18,7 +18,7 @@
|
|||
<meta property="og:image:alt" content="Igneum. Mined by GPUs. Proven by fire.">
|
||||
<meta name="twitter:card" content="summary">
|
||||
<meta name="twitter:title" content="Igneum ledger: every criticism, answered">
|
||||
<meta name="twitter:description" content="183 criticisms in the critic's words, with what was done, the status and the date.">
|
||||
<meta name="twitter:description" content="184 criticisms in the critic's words, with what was done, the status and the date.">
|
||||
<meta name="twitter:image" content="https://igneum.network/og-small.png?v=3">
|
||||
<meta name="twitter:image:alt" content="Igneum. Mined by GPUs. Proven by fire.">
|
||||
<link rel="icon" href="/favicon.ico" sizes="48x48">
|
||||
|
|
@ -212,9 +212,9 @@ blockquote{margin:10px 0;padding:10px 14px;border-left:3px solid var(--line-2);c
|
|||
<section class="page-hero">
|
||||
<div class="container">
|
||||
<div class="breadcrumb"><a href="/">Igneum</a><span>/</span><span>The ledger</span></div>
|
||||
<div class="page-heading"><div><div class="eyebrow"><span class="line"></span>Ledger · 183 entries · regenerated from the repository</div>
|
||||
<div class="page-heading"><div><div class="eyebrow"><span class="line"></span>Ledger · 184 entries · regenerated from the repository</div>
|
||||
<h1>Every criticism, answered<br><span class="accent">or conceded.</span></h1>
|
||||
<p class="lead">This is every criticism the project expects, in the critic's words, with what was done about it and the date. 183 entries since 3 October 2026. Entries are never deleted; a status that changes keeps its history on the line. Where the critic was right the entry says Conceded. Where nothing has been done it says Open and names what settles it. The founder mined through the GPU years. Ethereum's move to proof of stake in September 2022 ended that income and the miners' place in that chain. This is one person building, with AI systems doing the engineering, the coin he wanted to exist for miners: GPU-mined, the miners are the provers, no founder allocation, every cost stated. Help is welcome and a team is wanted: cryptographers, node engineers, miners who will test. This ledger is the application form: pick an open row and write to <a href="mailto:hello@igneum.network">hello@igneum.network</a> with its id.</p></div></div>
|
||||
<p class="lead">This is every criticism the project expects, in the critic's words, with what was done about it and the date. 184 entries since 3 October 2026. Entries are never deleted; a status that changes keeps its history on the line. Where the critic was right the entry says Conceded. Where nothing has been done it says Open and names what settles it. The founder mined through the GPU years. Ethereum's move to proof of stake in September 2022 ended that income and the miners' place in that chain. This is one person building, with AI systems doing the engineering, the coin he wanted to exist for miners: GPU-mined, the miners are the provers, no founder allocation, every cost stated. Help is welcome and a team is wanted: cryptographers, node engineers, miners who will test. This ledger is the application form: pick an open row and write to <a href="mailto:hello@igneum.network">hello@igneum.network</a> with its id.</p></div></div>
|
||||
</div>
|
||||
</section>
|
||||
<section class="section compact"><div class="container">
|
||||
|
|
@ -227,8 +227,8 @@ blockquote{margin:10px 0;padding:10px 14px;border-left:3px solid var(--line-2);c
|
|||
<tr><td class="num">27</td><td><button type="button" class="chip" data-filter="Closed by rule or decided">Closed by rule or decided</button></td><td>A consensus rule or a decision by the owner answers it, dated</td></tr>
|
||||
<tr><td class="num">13</td><td><button type="button" class="chip" data-filter="Answered with evidence">Answered with evidence</button></td><td>A measurement or a simulation exists and is named</td></tr>
|
||||
<tr><td class="num">13</td><td><button type="button" class="chip" data-filter="Answered by design">Answered by design</button></td><td>A design rule answers it; no measurement is possible yet</td></tr>
|
||||
<tr><td class="num">4</td><td><button type="button" class="chip" data-filter="Other">Other</button></td><td>A status outside the six above, read the line</td></tr>
|
||||
<tr><td class="num">183</td><td><button type="button" class="chip" data-filter="">All</button></td><td>Every entry. The sections: <a href="#m">Mining and chips</a>, <a href="#f">Finality and attacks</a>, <a href="#p">Proving and the zkEVM</a>, <a href="#e">Economics and the coin</a>, <a href="#g">Governance and the founders</a>, <a href="#c">Comparisons</a>, <a href="#l">Legal and regulatory</a>, <a href="#x">Launch and operations</a>, <a href="#d">Builders</a></td></tr>
|
||||
<tr><td class="num">5</td><td><button type="button" class="chip" data-filter="Other">Other</button></td><td>A status outside the six above, read the line</td></tr>
|
||||
<tr><td class="num">184</td><td><button type="button" class="chip" data-filter="">All</button></td><td>Every entry. The sections: <a href="#m">Mining and chips</a>, <a href="#f">Finality and attacks</a>, <a href="#p">Proving and the zkEVM</a>, <a href="#e">Economics and the coin</a>, <a href="#g">Governance and the founders</a>, <a href="#c">Comparisons</a>, <a href="#l">Legal and regulatory</a>, <a href="#x">Launch and operations</a>, <a href="#d">Builders</a></td></tr>
|
||||
</tbody></table></div>
|
||||
<div class="toolbar"><input type="search" id="q" placeholder="Search the ledger" aria-label="Search the ledger"><span id="shown"></span></div>
|
||||
<h2 id="m">Mining and chips</h2>
|
||||
|
|
@ -328,6 +328,12 @@ blockquote{margin:10px 0;padding:10px 14px;border-left:3px solid var(--line-2);c
|
|||
<div class="status"><span class="badge b-other">Relabelled</span> <span class="did">7 October 2026, morning, X36): the X9 in this row and in the litepaper paragraph is the chip Bitmain announced and withdrew before launch, its core claimed and never measured; the ladder's arithmetic against that core is unchanged.</span></div>
|
||||
<details><summary>The answer as first written</summary><p>Correct on both counts, and the second was the sharper one. The X9 (Bitmain, about 1 MH/s at 2,472 W, approximate, github.com/monero-project/monero/issues/10270) is a shipped 3x per-joule edge over a desktop CPU on the best-known latency-bound random-program design, seven years after launch; it makes the k = 0.3 column of the chip model a product class rather than an attacker's claim, and the public headline is now the range 2.1x (k = 1) to 3.9x (k = 0.33) over the RTX 5090 at class v4, with the ladder taking the X9 bracket to about 2.8x by rung 2 and the Apple tier's to about 1.1x by rung 3. The ladder does not close the gap; it is the chain's only automatic answer, it moves at the pace of the cards that pay for it, and the honest card's watts remain the lever that moves every row (algorithm lane proposal 7). What the ladder gives up by design: a chip holding over 10 percent of weight can stall it, and the status quo it stalls is a rung the cards already run.</p></details>
|
||||
</article>
|
||||
<article class="entry" id="M35" data-bucket="Other">
|
||||
<div class="head"><span class="id">M35</span><h3>"Proof of stored state" proves nothing a pool cannot ship, and "proof of following" is a 32 ms rebuild per hour</h3><span class="date">7 October 2026</span></div>
|
||||
<blockquote>Class v5 keys the dataset by the chain's state so that every hash proves the miner holds the chain. The state is 6 KB. A pool ships it with the template. The hourly refresh is a 32 ms rebuild that a farm's one node does once and broadcasts. Nothing on the chain can tell a card that derived the leaves from a card that received them, so the class proves nothing about who holds what, and it adds consensus-critical serialisation code for the privilege.</blockquote>
|
||||
<div class="status"><span class="badge b-other">Implemented behind a switch</span> <span class="did">7 October 2026, <code>a repository file</code>, branch <code>class-v5</code>, fork branch <code>class-v5-node</code> from the 0.3.18 node, behind <code>program_class_v5_activation_daa</code>, never until set): the dataset of every epoch is built from the canonical state stream after the epoch's seed block (<code>a repository file</code>: accounts, non-zero slots, code chunks, one serialisation that rebuilds to its root), hashed into leaves <code>D[i] = Blake2b-512('igneum-sd1/' || root || i || record)</code> and folded into every item (<code>leaf(t) = D[t mod n]</code>, so a hasher without the state, with another root, with a stream one record short or with the previous epoch's leaves is wrong on 64 of 64 items and 32 of 32 lanes: the known-failed case, <code>igneum-pow/src/state.rs</code>); the object byte 5 and the 95 percent seven-window tally beside class v4's (<code>consensus/core/src/igneum.rs</code> <code>program_class_for_epoch_signalled_v5</code>, one step per epoch); a node without the state refuses the header with a retryable error and serves no template (<code>kaspa-pow</code> <code>DayStateUnavailable</code>, the known-failed case <code>class_v5_refuses_without_state_and_refreshes_the_leaves_per_epoch</code>); the miner fetches the stream from its node's exec RPC (<code>igneum_getPowStateLeaves</code>); the fast-time gate <code>a repository file</code> with a stateless node and a stale miner.</span></div>
|
||||
<details><summary>The answer as first written</summary><p>The first sentence is right and the page says it first: anything the lottery derives is derived from a seed and the state, so the only bytes a central node cannot compress away are the state's, 5,952 bytes at today's devnet state (93 records), and the chain cannot distinguish a card that derived the leaves from one that received them, exactly as it cannot distinguish a pool member from a solo miner today. The numbers are on the page: the leaves pass 45 MB per member per hour, the line where a WAN pool at 1 Gbit/s serving 10,000 members can no longer ship them inside the window, at about 700,000 state records; a LAN farm at 100 Gbit/s is never bounded below the 2 GiB sample cap. What the class does force, per machine and per hour: holding the current state or its leaves, a rebuild from it (32 ms on a 4090, measured on 6 October), and knowledge of the chain's reference block inside the ten-minute lead; a machine cut off from the chain for an hour stops producing valid blocks at the next refresh, where under a daily rule it kept mining until midnight. It also removes the recompute chip (the f = 0 row of chip-model-v3) as a category and moves nothing against the dataset-storing chip, which the page and the litepaper both say. The cost is measured and small (hash rate and watts unchanged on the 4090, build +1.4 ms resident, verifier +0.11 to 0.21 ms per unit); the risk is the serialisation, which is why the stream must rebuild to its root on the capturing node before it is served and why the gate crosses a day boundary with a non-trivial state before Devnet 2.</p></details>
|
||||
</article>
|
||||
<h2 id="f">Finality and attacks</h2>
|
||||
<article class="entry" id="F1" data-bucket="Fixed or built">
|
||||
<div class="head"><span class="id">F1</span><h3>Finality is attackable for the first month</h3><span class="date">5 October 2026</span></div>
|
||||
|
|
|
|||
|
|
@ -434,6 +434,7 @@ body.all .pager{display:none}
|
|||
</table></div>
|
||||
<p>Three ideas carry the chip resistance. <strong>The hash rewrites itself.</strong> A new program every hour, drawn from the chain. Its memory pattern changes with it. The rules change on a schedule fixed at launch. No release, no vote. These are automatic schedule changes: they defeat a chip wired for one datapath and they need no human fork. Against a chip that stores the dataset every drawn parameter is firmware, and what meets that chip is the latency-shadow work (class v4) and the price per joule (the Horizon lane analysis, 6 October 2026, section 5.4; ledger M32). <strong>It waits on memory, not maths.</strong> Every hash is a chain of random reads into a table too big for a chip to carry. The wait is the same physics for everyone. <strong>Miners hold the switch.</strong> Spare defences are written into the rules, switched off. A miner signal turns one on, at the class-change threshold: miners signal three things at three thresholds, 60 percent of blue blocks over two weeks for a parameter genesis leaves open, 90 percent for an upgrade (new code), and 95 percent with a floor height for a class change. No fork.</p>
|
||||
<p><strong>The work that waits can grow.</strong> Class v4 adds a block of latency-shadow arithmetic to every hash, about 100,000 integer operations that run while the memory reads are in flight, so a chip that stores the whole dataset still has to pay for a core. That size sits on a ladder fixed at genesis, six rungs from about 100,000 to about 1,000,000 operations, and it moves one rung at a time only when 90 percent of blue blocks in each of seven consecutive days ask for it; it can never move two rungs inside a week and never past a rung the reference verifier cannot check under 10 ms with its sibling thread busy (measured on the build server, 6 October 2026: the first three rungs pass at 8.8, 8.9 and 9.2 ms, the fourth misses by 0.08 ms on a loaded box and stays out until a quiet re-measurement, the two doublings are out at 12.4 and 15.0 ms). What it buys, on the measured cards: against a dataset-storing chip whose core costs what an RTX 5090's does per operation, the chip's per-joule edge falls from 2.1x at the first rung to 1.3x at the third; against a core as good as the one Bitmain claimed for its withdrawn Antminer X9 (about 3x per joule over a desktop CPU, never measured), from 3.9x to 2.8x. What it costs, per rung, is measured too: the Apple tier gives up 3 points of rate at the first step and 6 more at the second, the RTX 5090 nothing until the second; so the miners who pay for a step are the ones who take it (<a href="/ledger#M34">ledger M34</a>).</p>
|
||||
<p><strong>The dataset is the chain.</strong> Class v5 builds each hour's dataset from the chain's own execution state at that hour's reference block: every account, every storage slot and every byte of contract code, hashed under the state root and folded into every item. A card that does not hold the state cannot build the dataset, and a card that builds it from stale state is wrong on every hash from the next refresh on, so a miner must keep following the chain to keep mining. The hash itself does not change, and neither does its speed: measured on an RTX 4090 on 7 October 2026, 63.06 against 63.07 million hashes a second, the hourly rebuild 0.2 ms longer at the devnet's state, a node's check 0.2 to 0.3 ms longer per block. What it buys is a floor under what mining means: a chip that recomputes items instead of storing them must now hold the state too, and a pool miner who today needs nothing but the date needs the state every hour. What it does not buy is stated as plainly: a pool or a farm with one node can ship the state to its cards each hour, 6 KB today and at most 2 GB on a used chain, so the claim is that every mining operation holds and follows the chain, not every card; and against a chip that stores the whole dataset it changes nothing, which is why it sits beside the latency-shadow work, not in its place. It ships switched off, behind the 95 percent class signal with a floor height (<a href="/ledger#M35">ledger M35</a>).</p>
|
||||
<h3 id="chip-model">The chip model</h3>
|
||||
<p>We price the strongest chip we can design against an RTX 5090 and publish the arithmetic. Class v4 is live from the first block on the testnet and the mainnet (the ladder’s rung 0 at genesis), so the launch number is the class v4 row. The honest card: an RTX 5090 mines class v3 at 136 MH/s on 350 W in the bench and 290 W in the app (measured, 6 October 2026); an Apple M5 Max at 27 MH/s on 21 W (measured, 6 October 2026); an H100 SXM at 249 MH/s, 98 percent of its random-read ceiling like the 5090, 1.78x the 5090’s hash at 1.15x the tuned 5090’s hash per watt and a third of the hash per rented dollar (measured, 7 October 2026), so datacentre silicon does not change the chip question. The CPU verifier takes 2.33 ms per warp of 32 hashes on one M5 Max core under class v4 (measured, 6 October 2026), against a gate of 10 ms.</p>
|
||||
<div class="tbl"><table><thead><tr><th>The chip and the class</th><th>Edge over an RTX 5090 per joule</th><th>Label and date</th></tr></thead><tbody>
|
||||
|
|
|
|||
51
tools/class-v5/fleet-cuda-v5-bench.sh
Executable file
51
tools/class-v5/fleet-cuda-v5-bench.sh
Executable file
|
|
@ -0,0 +1,51 @@
|
|||
#!/usr/bin/env bash
|
||||
# Class v5 kit bench on one fleet NVIDIA card (7 October 2026, docs/design/class-v5-stored-state.md section 13, "the kit:
|
||||
# CUDA"). Run ON the pod by the fleet lane, in a directory holding the unzipped kit (packs-ca3-v5-<stamp>.zip from
|
||||
# igneum-build-1:/srv/artefacts/packs/): the kit's Linux NVRTC worker (bin/linux/igneum-worker-cuda, THIS tree's
|
||||
# proto-cuda/nvrtc/worker.cpp with the class v5 leaf upload) runs --check and --bench on the first class v5 pack,
|
||||
# packs/v5-dn3-epoch0 (program id e5a4ac5978462156, 11 leaves under state root 7e37a9fb..., cache FNV 7334fa46e5d972eb);
|
||||
# then src/bench.cu (the nvcc harness of the 4090 rows) on the same pack. Expected on every line: the self-test PASS and
|
||||
# the fingerprint of the 2^24 outputs at base nonce 0 equal to the Metal reading 82b19cbde8557ea5 (M5 Max, 18:07:09Z).
|
||||
# One card, about a minute. Prints RESULT lines; exit 0 only when both fingerprints match.
|
||||
# bash tools/fleet-cuda-v5-bench.sh [device index, default 0] [sm arch for nvcc, default from nvidia-smi]
|
||||
set -u
|
||||
DEV="${1:-0}"
|
||||
EXPECTED=82b19cbde8557ea5
|
||||
KIT="$(cd "$(dirname "$0")/.." && pwd)"
|
||||
export PATH=/usr/local/cuda/bin:$PATH
|
||||
echo "RESULT start $(date -u +%Y-%m-%dT%H:%M:%SZ) host=$(hostname) kit=$KIT expected_fingerprint=$EXPECTED"
|
||||
nvidia-smi -i "$DEV" --query-gpu=name,driver_version,power.limit,memory.total --format=csv,noheader | sed 's/^/RESULT card /'
|
||||
ARCH="${2:-}"
|
||||
if [ -z "$ARCH" ]; then cc=$(nvidia-smi -i "$DEV" --query-gpu=compute_cap --format=csv,noheader | tr -d ' .'); ARCH="sm_$cc"; fi
|
||||
PACK="$KIT/packs/v5-dn3-epoch0"
|
||||
[ -f "$PACK/leaves.bin" ] || { echo "RESULT error leaves.bin missing in $PACK (a class v5 pack without its leaves builds nothing)"; exit 2; }
|
||||
echo "RESULT leaves $(wc -c < "$PACK/leaves.bin") bytes sha256 $(sha256sum "$PACK/leaves.bin" | cut -c1-64)"
|
||||
W="$KIT/bin/linux/igneum-worker-cuda"
|
||||
chmod +x "$W" 2>/dev/null
|
||||
echo "RESULT worker $W sha256 $(sha256sum "$W" | cut -c1-64)"
|
||||
ok=0
|
||||
echo "RESULT check start $(date -u +%H:%M:%SZ) cmd=igneum-worker-cuda --check --pack packs/v5-dn3-epoch0 --device $DEV"
|
||||
"$W" --check --pack "$PACK" --device "$DEV" 2>&1 | sed 's/^/RESULT check out /'
|
||||
echo "RESULT bench start $(date -u +%H:%M:%SZ) cmd=igneum-worker-cuda --bench --pack packs/v5-dn3-epoch0 --device $DEV --batch-log2 24 --batches 5"
|
||||
out=$("$W" --bench --pack "$PACK" --device "$DEV" --batch-log2 24 --batches 5 2>&1); rc=$?
|
||||
echo "$out" | grep -E '^(pack |RESULT|FAIL|warm-up)' | sed 's/^/RESULT bench out /'
|
||||
fp=$(echo "$out" | grep -o 'fingerprint=[0-9a-f]*' | tail -1 | cut -d= -f2)
|
||||
mhs=$(echo "$out" | grep -o 'mhs=[0-9.]*' | tail -1 | cut -d= -f2)
|
||||
chk=$(echo "$out" | grep -o 'check=[A-Za-z]*' | tail -1 | cut -d= -f2)
|
||||
match=no; [ "$fp" = "$EXPECTED" ] && [ "$chk" = PASS ] && { match=yes; ok=$((ok + 1)); }
|
||||
echo "RESULT worker-v5 fingerprint=$fp expected=$EXPECTED match=$match check=$chk mhs=$mhs exit=$rc $(date -u +%H:%M:%SZ)"
|
||||
# the nvcc harness (the 4090 rows of the design page, section 7): compiled in the pack directory against its kernel.cu
|
||||
if command -v nvcc > /dev/null; then
|
||||
( cd "$PACK" && nvcc -O3 -std=c++17 -arch="$ARCH" -Xcompiler -pthread -I. -o bench "$KIT/src/bench.cu" kernel.cu 2>&1 | grep -v '^$' | sed 's/^/RESULT nvcc /'; echo "RESULT nvcc rc ${PIPESTATUS[0]} arch=$ARCH" )
|
||||
if [ -x "$PACK/bench" ]; then
|
||||
out2=$(cd "$PACK" && CUDA_VISIBLE_DEVICES="$DEV" ./bench --batches 10 --power-seconds 0 2>&1); rc2=$?
|
||||
echo "$out2" | sed 's/^/RESULT bench.cu out /'
|
||||
fp2=$(echo "$out2" | grep -o 'fingerprint of 2^24 outputs at base 0: [0-9a-f]*' | grep -o '[0-9a-f]\{16\}$')
|
||||
m2=no; [ "$fp2" = "$EXPECTED" ] && echo "$out2" | grep -q 'vector warps against the pack: 3 of 3' && { m2=yes; ok=$((ok + 1)); }
|
||||
echo "RESULT bench.cu-v5 fingerprint=$fp2 expected=$EXPECTED match=$m2 exit=$rc2 $(date -u +%H:%M:%SZ)"
|
||||
fi
|
||||
else
|
||||
echo "RESULT nvcc absent: the bench.cu row is skipped (the worker row stands)"; ok=$((ok + 1))
|
||||
fi
|
||||
echo "RESULT end $(date -u +%Y-%m-%dT%H:%M:%SZ) rows_matched=$ok of 2"
|
||||
[ "$ok" = 2 ]
|
||||
32
tools/class-v5/harness-remote.sh
Executable file
32
tools/class-v5/harness-remote.sh
Executable file
|
|
@ -0,0 +1,32 @@
|
|||
#!/usr/bin/env bash
|
||||
# The class v5 fast-time gate on igneum-build-1 (docs/design/class-v5-stored-state.md section 10): runs one case of
|
||||
# infra/fast-time/class-v5-signal.mjs on the box from this worktree's mirror there (the fork's Linux binaries in
|
||||
# vendor/igneum-node-class-v5/target/release and igneum-pow's in igneum-pow/target/release, both built by
|
||||
# tools/build-remote.sh), under its own data dir, and copies the summary and log back under
|
||||
# docs/design/class-v5-harness/<case>.{json,log}. A functional run (counts, ids, locks), not a measurement: it takes
|
||||
# no build slot and no measure hold (the Mac rule's `run` kind).
|
||||
#
|
||||
# tools/class-v5/harness-remote.sh <case-name> -- <class-v5-signal.mjs arguments>
|
||||
# tools/class-v5/harness-remote.sh failed-case -- --signal 6,6,7 --expect flip
|
||||
# BASE=29830 tools/class-v5/harness-remote.sh flip-stale -- --signal 6,6,6 --expect flip --stale 2 (another port base per
|
||||
# case, so two cases never share ports; the harness kills the previous run's pids of its own data dir first)
|
||||
set -euo pipefail
|
||||
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
ROOT="$(cd "$HERE/../.." && pwd)"
|
||||
CASE="${1:?case name}"; shift
|
||||
[ "${1:-}" = "--" ] && shift
|
||||
WT="$(basename "$ROOT")"
|
||||
HOST=build@188.40.146.49
|
||||
KEY="$HOME/.ssh/igneum_ed25519"
|
||||
REMOTE_ROOT="/srv/builds/$WT"
|
||||
REMOTE_LOG="/srv/builds/_log/v5-class/harness"
|
||||
OUT="$ROOT/docs/design/class-v5-harness"
|
||||
mkdir -p "$OUT"
|
||||
ARGS="$*"
|
||||
BASE="${BASE:-29760}"
|
||||
echo "harness-remote: case $CASE on $HOST: node $REMOTE_ROOT/infra/fast-time/class-v5-signal.mjs $ARGS"
|
||||
# the harness file travels by rsync (the mirror carries HEAD; an uncommitted change rides along, re-stamped: the copied-sources rule)
|
||||
rsync -a -e "ssh -i $KEY" "$ROOT/infra/fast-time/class-v5-signal.mjs" "$ROOT/infra/fast-time/override-60x.json" "$HOST:$REMOTE_ROOT/infra/fast-time/"
|
||||
ssh -i "$KEY" "$HOST" "touch $REMOTE_ROOT/infra/fast-time/class-v5-signal.mjs $REMOTE_ROOT/infra/fast-time/override-60x.json; mkdir -p $REMOTE_LOG; cd $REMOTE_ROOT && IGNEUM_V5_BASE=$BASE IGNEUM_V5_TMP=/tmp/igneum-fast-time-v5s-$CASE IGNEUM_V5_BIN=$REMOTE_ROOT/vendor/igneum-node-class-v5/target/release IGNEUM_POW=$REMOTE_ROOT/igneum-pow/target/release/igneum-pow node infra/fast-time/class-v5-signal.mjs $ARGS > $REMOTE_LOG/$CASE.log 2>&1; rc=\$?; cp /tmp/igneum-fast-time-v5s-$CASE/summary.json $REMOTE_LOG/$CASE.json 2>/dev/null || true; echo \"harness-remote: rc \$rc\"; exit \$rc" || true
|
||||
rsync -a -e "ssh -i $KEY" "$HOST:$REMOTE_LOG/$CASE.log" "$HOST:$REMOTE_LOG/$CASE.json" "$OUT/" 2>/dev/null || true
|
||||
grep -E "SUMMARY|FAILED CHECK|PROGRAM ID|CLASS SWITCH" "$OUT/$CASE.log" || tail -20 "$OUT/$CASE.log"
|
||||
86
tools/class-v5/kits-on-box.sh
Executable file
86
tools/class-v5/kits-on-box.sh
Executable file
|
|
@ -0,0 +1,86 @@
|
|||
#!/usr/bin/env bash
|
||||
# The class v5 kit build as it runs ON igneum-build-1 (called by tools/class-v5/kits-remote.sh through the build-server
|
||||
# library's remote runner, from the worktree's mirror checkout, so this file travels with the commit). A script file by
|
||||
# path, never an inline string: main's rule of 7 October 2026, 21:33 BST (no rm, find -delete or truncation inside a
|
||||
# `bash -c` string; tools/ci/inline-rm-check.sh). What it does, in order: the pack loader test with the class v5 cases,
|
||||
# the Linux and Windows builds of the two one-click workers, the CPU emulation's --check on the class v5 pack, then the
|
||||
# kit zip at /srv/artefacts/packs/packs-ca3-v5-<stamp>.zip with SHA256SUMS. Logs land in /srv/builds/_log/v5-class/kits-<stamp>.
|
||||
# bash tools/class-v5/kits-on-box.sh <stamp> <skip-emu 0|1> (cwd = the worktree root on the box)
|
||||
set -u
|
||||
STAMP="${1:?stamp}"; SKIP_EMU="${2:-0}"
|
||||
ZIP="/srv/artefacts/packs/packs-ca3-v5-$STAMP.zip"
|
||||
LOGDIR="/srv/builds/_log/v5-class/kits-$STAMP"
|
||||
T=$(mktemp -d /tmp/v5-kits.XXXXXX); S=$T/stage; mkdir -p "$S/bin/linux" "$S/bin/windows" "$S/src" "$S/packs" "$S/tools" "$LOGDIR"
|
||||
rc=0
|
||||
step() { echo "STEP $1 $(date -u +%H:%M:%SZ)"; }
|
||||
fail() { echo "FAIL $1"; rc=1; }
|
||||
finish() { cp "$T"/*.log "$LOGDIR"/ 2>/dev/null; echo "LOGS $LOGDIR"; rm -rf "$T"; }
|
||||
V5=proto-cuda/packs-ca3-v5/v5-dn3-epoch0
|
||||
step packfile-test
|
||||
cc -std=c99 -Wall -Wextra -Wno-unused-function -O1 -o "$T/packfile-test" proto-cuda/nvrtc/emu/packfile-test.c && "$T/packfile-test" proto-cuda/packs/igneum-devnet-v4-epoch0 "$V5" > "$T/packfile-test.log" 2>&1 || fail "packfile-test ($(tail -1 "$T/packfile-test.log"))"
|
||||
grep -E '^(FAIL| v5 pack)' "$T/packfile-test.log"; grep -c '^ok' "$T/packfile-test.log" | sed 's/^/packfile-test ok lines /'
|
||||
step linux-opencl-worker
|
||||
gcc -std=c99 -O2 -Wall -Wextra -Wno-stringop-truncation -Wno-format-truncation -DIGNEUM_CL_DYNAMIC -DCL_TARGET_OPENCL_VERSION=120 -I proto-cuda/packs/igneum-devnet-v4-epoch0 -DIGNEUM_KERNEL_PATH='"kernel_bound.cl"' -o "$S/bin/linux/igneum-worker-opencl" proto-opencl/host.c -ldl -lpthread 2> "$T/cl-linux.log" || { fail "linux opencl worker"; head -20 "$T/cl-linux.log"; }
|
||||
step linux-cuda-worker
|
||||
g++ -std=c++17 -O2 -Wall -Wextra -I proto-cuda/nvrtc -I /usr/local/cuda/include -o "$S/bin/linux/igneum-worker-cuda" proto-cuda/nvrtc/worker.cpp -ldl -lpthread 2> "$T/cuda-linux.log" || { fail "linux cuda worker"; head -20 "$T/cuda-linux.log"; }
|
||||
step windows-cuda-worker
|
||||
x86_64-w64-mingw32-g++ -std=c++17 -O2 -Wall -Wextra -static -I proto-cuda/nvrtc -I /usr/local/cuda/include -o "$S/bin/windows/igneum-worker-cuda.exe" proto-cuda/nvrtc/worker.cpp 2> "$T/cuda-win.log" && x86_64-w64-mingw32-strip "$S/bin/windows/igneum-worker-cuda.exe" || { fail "windows cuda worker"; head -20 "$T/cuda-win.log"; }
|
||||
step windows-opencl-worker
|
||||
# the Khronos CL headers alone (never -I /usr/include: that puts glibc's stdint.h ahead of mingw's)
|
||||
mkdir -p "$T/inc" && cp -r /usr/include/CL "$T/inc/"
|
||||
x86_64-w64-mingw32-gcc -std=c99 -O2 -Wall -Wextra -Wno-stringop-truncation -Wno-format-truncation -static -DIGNEUM_CL_DYNAMIC -DCL_TARGET_OPENCL_VERSION=120 -I "$T/inc" -I proto-cuda/packs/igneum-devnet-v4-epoch0 -DIGNEUM_KERNEL_PATH='"kernel_bound.cl"' -o "$S/bin/windows/igneum-worker-opencl.exe" proto-opencl/host.c 2> "$T/cl-win.log" && x86_64-w64-mingw32-strip "$S/bin/windows/igneum-worker-opencl.exe" || { fail "windows opencl worker"; head -20 "$T/cl-win.log"; }
|
||||
if [ "$SKIP_EMU" = 0 ]; then
|
||||
step emu-check-v5
|
||||
# the CPU emulation (proto-cuda/nvrtc/emu/test.sh's build) with pack A = the class v5 pack and pack B = the v4 control
|
||||
E=$T/emu; mkdir -p "$E"
|
||||
emu_kernel() { { echo '#include <cuda_runtime.h>'; echo '#include <cstdint>'; echo "namespace $2 {"; sed -E 's/([A-Za-z_0-9]+)<<<([^,]+), ([^>]+)>>>\(/emu_launch(\1, \2, \3, /' "$1/$3.cu"; echo "}"; } > "$E/$3_$2.cpp"; g++ -std=c++17 -O2 -w -I proto-cuda/emu -I "$1" -c "$E/$3_$2.cpp" -o "$E/$3_$2.o"; }
|
||||
emu_kernel "$V5" emu_pack_a kernel && emu_kernel "$V5" emu_pack_a kernel_bound && emu_kernel proto-cuda/packs-ca3-v5/v4-genesis emu_pack_b kernel && emu_kernel proto-cuda/packs-ca3-v5/v4-genesis emu_pack_b kernel_bound \
|
||||
&& g++ -std=c++17 -O2 -Wall -Wextra -DIGNEUM_EMU -I proto-cuda/nvrtc -I /usr/local/cuda/include -c proto-cuda/nvrtc/worker.cpp -o "$E/worker.o" \
|
||||
&& g++ -std=c++17 -O2 -Wall -Wextra -DIGNEUM_EMU -DIGNEUM_EMU_TWO_PACKS -I proto-cuda/nvrtc -I proto-cuda/emu -I /usr/local/cuda/include -c proto-cuda/nvrtc/emu/emu_backend.cpp -o "$E/emu_backend.o" \
|
||||
&& g++ -std=c++17 -O2 -w -I proto-cuda/emu -c proto-cuda/emu/shim.cpp -o "$E/shim.o" \
|
||||
&& g++ -o "$E/igneum-worker-cuda-emu" "$E"/*.o -pthread 2> "$T/emu-build.log" || { fail "emu build"; head -30 "$T/emu-build.log"; }
|
||||
if [ -x "$E/igneum-worker-cuda-emu" ]; then
|
||||
( IGNEUM_EMU_PACK="$PWD/$V5" IGNEUM_EMU_PACK2="$PWD/proto-cuda/packs-ca3-v5/v4-genesis" timeout 1500 nice -n 10 "$E/igneum-worker-cuda-emu" --check --pack "$V5" > "$T/emu-check.log" 2>&1 ) || fail "emu --check on the class v5 pack (rc $?)"
|
||||
grep -E '^check PASS|self-test|FAIL|error' "$T/emu-check.log" | cut -c1-400
|
||||
fi
|
||||
fi
|
||||
if [ "$rc" != 0 ]; then echo "KIT not written: a step failed (rc $rc)"; finish; exit 1; fi
|
||||
step stage
|
||||
for p in v4-genesis v5-genesis v5-dn3-epoch0; do mkdir -p "$S/packs/$p"; cp proto-cuda/packs-ca3-v5/$p/* "$S/packs/$p/"; done
|
||||
cp proto-cuda/nvrtc/worker.cpp proto-cuda/nvrtc/packfile.h proto-cuda/nvrtc/cuda_api.h proto-opencl/host.c proto-opencl/cl_dynamic.h proto-newpow/class-v5/bench.cu proto-newpow/class-v5/run.sh proto-metal/packbench.swift "$S/src/"
|
||||
cp tools/class-v5/fleet-cuda-v5-bench.sh tools/class-v5/pc1-amd-v5-bench.ps1 "$S/tools/"
|
||||
cat > "$S/README.txt" <<'R'
|
||||
packs-ca3-v5: the class v5 kit (Igneum, 7 October 2026; docs/design/class-v5-stored-state.md)
|
||||
packs/v5-dn3-epoch0 the first class v5 pack: Devnet 3 epoch 0, program id e5a4ac5978462156, 11 state leaves (leaves.bin, 704 B) under
|
||||
state root 7e37a9fb19b154d32daf5bf30a50d339a75029fbc9eec9ea20e95439dba5a311, cache FNV-1a 64 7334fa46e5d972eb
|
||||
expected fingerprint of the 2^24 outputs at base nonce 0: 82b19cbde8557ea5 (Metal, M5 Max, 18:07:09Z) on every platform
|
||||
packs/v5-genesis the string-seed class v5 pack over the devnet's 93 leaves; packs/v4-genesis the class v4 control (sub-version 3)
|
||||
bin/linux igneum-worker-cuda (NVRTC, libcuda + libnvrtc.so.12 at run time), igneum-worker-opencl (libOpenCL.so.1 at run time)
|
||||
bin/windows igneum-worker-cuda.exe (nvcuda.dll + nvrtc64 at run time), igneum-worker-opencl.exe (OpenCL.dll at run time)
|
||||
Every worker reads a class v5 pack's leaves.bin, checks it against the pack's IGNEUM_STATE_LEAVES_FNV64, uploads it for igneum_build and
|
||||
frees it after the build; a pack without its leaves builds nothing. Self-test: cache head, last line and FNV; dataset head, last word and
|
||||
64 samples; 96 vector lanes. Bench: igneum-worker-cuda --bench --pack packs/v5-dn3-epoch0 --batch-log2 24 (fleet-cuda-v5-bench.sh);
|
||||
igneum-worker-opencl --bench-pack --pack packs/v5-dn3-epoch0 --batch-log2 24 --device <n> (pc1-amd-v5-bench.ps1); src/bench.cu with nvcc.
|
||||
Intel: not measured tonight (7 October 2026); no Arc B580 sits on PC 1 or PC 2 and the only Arc path needs a driver click, which no PC job
|
||||
may raise; the card's holder and a click-free driver path are owed. The OpenCL worker takes the Intel device by index the same way.
|
||||
R
|
||||
( cd "$S" && find . -type f ! -name SHA256SUMS | sort | xargs sha256sum > SHA256SUMS )
|
||||
mkdir -p /srv/artefacts/packs
|
||||
python3 - "$S" "$ZIP" <<'PY'
|
||||
import os, sys, zipfile
|
||||
stage, out = sys.argv[1], sys.argv[2]
|
||||
with zipfile.ZipFile(out, 'w', zipfile.ZIP_DEFLATED) as z:
|
||||
for dp, dn, fn in os.walk(stage):
|
||||
dn.sort()
|
||||
for f in sorted(fn):
|
||||
p = os.path.join(dp, f); rel = os.path.relpath(p, stage)
|
||||
zi = zipfile.ZipInfo(rel, date_time=(2026, 10, 7, 0, 0, 0)); zi.compress_type = zipfile.ZIP_DEFLATED
|
||||
zi.external_attr = (0o755 if rel.startswith('bin/') or rel.endswith('.sh') else 0o644) << 16
|
||||
with open(p, 'rb') as fh: z.writestr(zi, fh.read())
|
||||
PY
|
||||
sha256sum "$ZIP" | cut -c1-64 > "$ZIP.sha256"
|
||||
echo "KIT $ZIP bytes $(stat -c %s "$ZIP") files $(python3 -c "import zipfile,sys; print(len(zipfile.ZipFile(sys.argv[1]).namelist()))" "$ZIP")"
|
||||
echo "KIT sha256 $(cat "$ZIP.sha256")"
|
||||
for b in "$S"/bin/linux/* "$S"/bin/windows/*; do echo "BIN $(basename "$(dirname "$b")")/$(basename "$b") $(stat -c %s "$b") bytes sha256 $(sha256sum "$b" | cut -c1-64)"; done
|
||||
finish
|
||||
exit 0
|
||||
46
tools/class-v5/kits-remote.sh
Executable file
46
tools/class-v5/kits-remote.sh
Executable file
|
|
@ -0,0 +1,46 @@
|
|||
#!/usr/bin/env bash
|
||||
# The class v5 kits for every platform, built and tested on igneum-build-1 (docs/design/class-v5-stored-state.md section 13,
|
||||
# the kit rows; 7 October 2026). From THIS worktree's commit (HEAD through the bare mirror, as tools/build-remote.sh moves
|
||||
# sources; nothing is built on the Mac): the pack loader test with the class v5 cases (proto-cuda/nvrtc/emu/packfile-test.c),
|
||||
# the CPU emulation of the NVRTC worker running --check on the class v5 pack (the real kernel text on host threads, the
|
||||
# dataset built from the pack's leaves and checked against the pack's cache FNV, dataset head, last word, 64 samples and
|
||||
# three vector warps: the "exactly as igneum-pow's verifier" line without a GPU), the Linux and Windows builds of the two
|
||||
# one-click workers (proto-cuda/nvrtc/worker.cpp through NVRTC, proto-opencl/host.c for AMD, Intel and Apple), then one kit
|
||||
# zip at /srv/artefacts/packs/packs-ca3-v5-<stamp>.zip holding the three packs (v4-genesis, v5-genesis, v5-dn3-epoch0), the
|
||||
# binaries, the sources, the bench scripts for the fleet and PC 1, and SHA256SUMS. Prints the zip's sha256 last.
|
||||
#
|
||||
# IGNEUM_AGENT=v5-kits tools/class-v5/kits-remote.sh [--box 1] [--skip-emu]
|
||||
#
|
||||
# Takes one of the box's build slots through infra/build-server/remote-run.sh (bs_remote_run), box 1 by default (the
|
||||
# build-server lane's word of 7 October 2026, 19:4x BST: the kit builds on build-1 explicitly). The work itself is
|
||||
# tools/class-v5/kits-on-box.sh, a script file run by path on the box (the inline-rm rule of 21:33 BST); when the
|
||||
# build-server lane's lease tool is in place, the same file goes through `/srv/builds/_bin/lease pool <threads> -- bash
|
||||
# tools/class-v5/kits-on-box.sh ...` with the label "v5 kit".
|
||||
set -euo pipefail
|
||||
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
BS_TOOL=class-v5-kits
|
||||
# shellcheck source=../../infra/build-server/lib.sh
|
||||
. "$HERE/../../infra/build-server/lib.sh"
|
||||
BOX=1; SKIP_EMU=0
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in --box) BOX="$2"; shift 2 ;; --skip-emu) SKIP_EMU=1; shift ;; *) echo "unknown argument $1" >&2; exit 2 ;; esac
|
||||
done
|
||||
export IGNEUM_AGENT="${IGNEUM_AGENT:-v5-kits}"
|
||||
bs_host "$BOX"
|
||||
pushd "$HERE/../../igneum-pow" > /dev/null; bs_context; popd > /dev/null
|
||||
WT="$BS_REMOTE_WT"
|
||||
STAMP=$(date -u +%Y%m%dT%H%M%SZ)
|
||||
ZIP="/srv/artefacts/packs/packs-ca3-v5-$STAMP.zip"
|
||||
bs_toolchain_check
|
||||
bs_sync_sources
|
||||
bs_log "sources at $WT (commit $BS_SHA on $BS_BRANCH); building the class v5 kits on box $BOX"
|
||||
# the body runs from a script FILE at the worktree's mirror checkout (tools/class-v5/kits-on-box.sh, carried by the commit), never
|
||||
# an inline string: main's rule of 7 October 2026, 21:33 BST (no rm, find -delete or truncation inside a bash -c string)
|
||||
CMD="cd '$WT' && bash tools/class-v5/kits-on-box.sh '$STAMP' '$SKIP_EMU'"
|
||||
set +e
|
||||
BR_KIND=build BR_COMMAND="class v5 kits (packfile-test, emu --check v5, Linux and Windows workers, kit zip)" BR_TARGET="x86_64-linux+windows" BR_ARTEFACTS="" \
|
||||
bs_remote_run "$WT" "$BS_WT class-v5 kits" "$CMD" 2>&1 | tee "/tmp/v5-kits-$STAMP.log" | grep -E '^(STEP|FAIL|KIT|BIN|LOGS|check PASS|self-test|packfile-test|emu:| v5|build-remote: RESULT)'
|
||||
rc=${PIPESTATUS[0]} # the remote runner's status, read before anything else runs (the first run read `true`'s and called a failed build built)
|
||||
set -e
|
||||
bs_wt_unlock
|
||||
[ "$rc" = 0 ] && bs_log "class v5 kits built: $ZIP" || bs_die "class v5 kits FAILED (rc $rc); full log /tmp/v5-kits-$STAMP.log"
|
||||
80
tools/class-v5/pc1-amd-v5-bench.ps1
Normal file
80
tools/class-v5/pc1-amd-v5-bench.ps1
Normal file
|
|
@ -0,0 +1,80 @@
|
|||
# Class v5 kit bench on PC 1's RX 9070 XT (7 October 2026, docs/design/class-v5-stored-state.md section 13, "the kit:
|
||||
# OpenCL (AMD)"): the kit's igneum-worker-opencl.exe (THIS tree's proto-opencl/host.c with the class v5 leaf upload:
|
||||
# igneum_build(ds, cache, leaves, nLeaves, nItems), the leaves of leaves.bin checked against the pack's FNV-1a 64 first)
|
||||
# runs --bench-pack on the first class v5 pack, proto-cuda/packs-ca3-v5/v5-dn3-epoch0 (program id e5a4ac5978462156, 11
|
||||
# leaves under state root 7e37a9fb..., cache FNV 7334fa46e5d972eb), on the gfx1201 device. Expected: the self-test
|
||||
# PASS line (cache head, last line and FNV; dataset head, word [268435455] and 64 samples; 96 of 96 vector lanes) and the
|
||||
# fingerprint of the 2^24 outputs at base nonce 0 equal to the Metal reading, 82b19cbde8557ea5 (the M5 Max, 7 October
|
||||
# 2026, 18:07:09Z). The v4-genesis control pack runs after it so the run has a known class v4 row beside the v5 row.
|
||||
# Beside the miners (a correctness and fingerprint gate; the rate row is labelled loaded when the card mines): no card is
|
||||
# switched, nothing is posted to the installed app, nothing is built on the PC. Published by the Counter ASIC coordinator
|
||||
# only (the PC 1 queue is its); the kit is fetch job $kitId. Every result line starts with RESULT; SUMMARY {json} ends it.
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$jobName = 'pc1-amd-v5-bench'
|
||||
$kitId = $env:IGNEUM_V5_KIT_ID; if (-not $kitId) { $kitId = 'fetch-ca3-v5-kit-20261007' }
|
||||
$expected = '82b19cbde8557ea5'
|
||||
$started = Get-Date
|
||||
function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') }
|
||||
function Summary([string] $status, [hashtable] $extra) {
|
||||
$o = [ordered]@{ job = $jobName; status = $status; duration_s = [int]((Get-Date) - $started).TotalSeconds; finished_at = (Stamp) }
|
||||
foreach ($k in $extra.Keys) { $o[$k] = $extra[$k] }
|
||||
'SUMMARY ' + ($o | ConvertTo-Json -Compress -Depth 4)
|
||||
}
|
||||
"RESULT start $(Stamp) job=$jobName machine=$env:COMPUTERNAME app_version=$env:IGNEUM_APP_VERSION expected_fingerprint=$expected"
|
||||
$jobs = Split-Path $env:IGNEUM_JOB_DIR
|
||||
$kit = Join-Path $jobs $kitId
|
||||
if (-not (Test-Path $kit)) { "RESULT error kit missing at $kit (the fetch job $kitId runs first; republish it after any app update)"; Summary 'failed' @{ error = 'kit missing' }; exit 2 }
|
||||
$exe = Join-Path $kit 'bin\windows\igneum-worker-opencl.exe'
|
||||
$packs = Join-Path $kit 'packs'
|
||||
if (-not (Test-Path $exe)) { "RESULT error worker missing at $exe"; Summary 'failed' @{ error = 'worker missing' }; exit 2 }
|
||||
"RESULT worker kit $exe sha256 $((Get-FileHash -Algorithm SHA256 $exe).Hash.ToLower()) bytes $((Get-Item $exe).Length)"
|
||||
foreach ($pk in @('v5-dn3-epoch0', 'v4-genesis')) {
|
||||
$d = Join-Path $packs $pk
|
||||
if (-not (Test-Path (Join-Path $d 'kernel_bound.cl'))) { "RESULT error pack $pk missing at $d"; Summary 'failed' @{ error = "pack $pk missing" }; exit 2 }
|
||||
"RESULT pack $pk kernel_bound.cl sha256 $((Get-FileHash -Algorithm SHA256 (Join-Path $d 'kernel_bound.cl')).Hash.ToLower()) program.h sha256 $((Get-FileHash -Algorithm SHA256 (Join-Path $d 'program.h')).Hash.ToLower())"
|
||||
}
|
||||
$leaves = Join-Path $packs 'v5-dn3-epoch0\leaves.bin'
|
||||
if (-not (Test-Path $leaves)) { "RESULT error leaves.bin missing at $leaves (a class v5 pack without its leaves builds nothing)"; Summary 'failed' @{ error = 'leaves missing' }; exit 2 }
|
||||
"RESULT leaves $leaves bytes $((Get-Item $leaves).Length) sha256 $((Get-FileHash -Algorithm SHA256 $leaves).Hash.ToLower())"
|
||||
# the OpenCL device index of the 9070 XT (the installed worker's list when present: the app's own indices)
|
||||
$inst = @("$env:LOCALAPPDATA\Programs\Igneum Miner", "$env:ProgramFiles\Igneum Miner") | Where-Object { Test-Path (Join-Path $_ 'igneum-app.exe') } | Select-Object -First 1
|
||||
$listExe = $exe
|
||||
if ($inst -and (Test-Path (Join-Path $inst 'igneum-worker-opencl.exe'))) { $listExe = Join-Path $inst 'igneum-worker-opencl.exe' }
|
||||
$list = @(& $listExe --list 2>&1 | ForEach-Object { "$_" })
|
||||
$list | ForEach-Object { "RESULT list $_" }
|
||||
$dev = $null
|
||||
foreach ($l in $list) { if ($l -match '^\s*\[(\d+)\].*gfx1201' -and $l -notmatch 'dup') { $dev = [int]$Matches[1]; break } }
|
||||
if ($null -eq $dev) { "RESULT error no gfx1201 device in --list (the eGPU is off the bus: the AMD v5 row stays OWED)"; Summary 'failed' @{ error = 'no gfx1201' }; exit 2 }
|
||||
"RESULT device $dev gfx1201 (list from $listExe)"
|
||||
function Workers { @(Get-CimInstance Win32_Process -Filter "Name = 'igneum-worker-opencl.exe' OR Name = 'igneum-worker-cuda.exe'" -ErrorAction SilentlyContinue | ForEach-Object { "$($_.Name):$($_.ProcessId):[$($_.CommandLine -replace '\s+', ' ')]" }) }
|
||||
$w = @(Workers)
|
||||
$loaded = ($w | Where-Object { $_ -match "igneum-worker-opencl.*--device\s+$dev(\s|$)" }).Count -gt 0
|
||||
"RESULT workers_before $(Stamp) $($w -join ' ')"
|
||||
$state = if ($loaded) { 'loaded' } else { 'quiet' }
|
||||
"RESULT context card_state=$state (a fingerprint gate: the load changes the rate row, never the bytes)"
|
||||
$rows = @{}
|
||||
$fpOk = $false
|
||||
foreach ($pk in @('v5-dn3-epoch0', 'v4-genesis')) {
|
||||
$d = Join-Path $packs $pk
|
||||
$t0 = Get-Date
|
||||
"RESULT bench $pk start $(Stamp) cmd=igneum-worker-opencl.exe --bench-pack --pack $d --device $dev --batch-log2 24 --batches 5"
|
||||
$out = @(& $exe --bench-pack --pack $d --device $dev --batch-log2 24 --batches 5 2>&1 | ForEach-Object { "$_" })
|
||||
$code = $LASTEXITCODE
|
||||
$secs = [int]((Get-Date) - $t0).TotalSeconds
|
||||
foreach ($l in $out) { if ($l -match '^(pack |class v5|RESULT |FAIL|error|warm-up)') { "RESULT bench $pk out $l" } }
|
||||
$res = $out | Where-Object { $_ -match '^RESULT ' } | Select-Object -Last 1
|
||||
$fp = ''; $mhs = ''; $check = ''
|
||||
if ($res -match 'fingerprint=([0-9a-f]{16})') { $fp = $Matches[1] }
|
||||
if ($res -match 'mhs=([0-9.]+)') { $mhs = $Matches[1] }
|
||||
if ($res -match 'check=(\w+)') { $check = $Matches[1] }
|
||||
$rows[$pk] = [ordered]@{ exit = $code; seconds = $secs; fingerprint = $fp; mhs = $mhs; check = $check; card_state = $state }
|
||||
if ($pk -eq 'v5-dn3-epoch0') {
|
||||
$fpOk = ($fp -eq $expected -and $check -eq 'PASS')
|
||||
"RESULT v5 fingerprint=$fp expected=$expected match=$fpOk check=$check mhs=$mhs card_state=$state exit=$code seconds=$secs $(Stamp)"
|
||||
} else {
|
||||
"RESULT v4-control fingerprint=$fp check=$check mhs=$mhs card_state=$state exit=$code seconds=$secs $(Stamp)"
|
||||
}
|
||||
}
|
||||
"RESULT workers_after $(Stamp) $((@(Workers)) -join ' ')"
|
||||
Summary $(if ($fpOk) { 'done' } else { 'failed' }) @{ expected = $expected; v5 = $rows['v5-dn3-epoch0']; v4_control = $rows['v4-genesis']; device = $dev; kit = $kitId }
|
||||
exit $(if ($fpOk) { 0 } else { 1 })
|
||||
58
tools/class-v5/verify-bench-remote.sh
Executable file
58
tools/class-v5/verify-bench-remote.sh
Executable file
|
|
@ -0,0 +1,58 @@
|
|||
#!/usr/bin/env bash
|
||||
# The class v5 verifier cost against class v4 on igneum-build-1 (docs/design/class-v5-stored-state.md section 7), the
|
||||
# latency ladder's method (tools/ladder/verify-bench-remote.sh): the cold verify of one 32-lane warp on the reference
|
||||
# core (core 40, nice 19), alone and with its SMT sibling (core 88) running the same bench, plus the average of 50
|
||||
# warps; class v4 at rung 0 against class v5 over the devnet's state stream (node 1's exec snapshot, 93 records).
|
||||
#
|
||||
# tools/class-v5/verify-bench-remote.sh [--pow <path on the box>] [--state <IGSD1 path on the box>] [--out <dir on the Mac>]
|
||||
#
|
||||
# Runs ON the box under its measure hold (infra/build-server/remote-run.sh BR_MEASURE=1), as the ladder's does.
|
||||
set -euo pipefail
|
||||
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
BS_TOOL=class-v5-verify-bench
|
||||
# shellcheck source=../../infra/build-server/lib.sh
|
||||
. "$HERE/../../infra/build-server/lib.sh"
|
||||
POW=""; STATE="/srv/builds/_log/v5-class/node1-state.igsd1"; OUT=""
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--pow) POW="$2"; shift 2 ;; --state) STATE="$2"; shift 2 ;; --out) OUT="$2"; shift 2 ;;
|
||||
*) echo "unknown argument $1" >&2; exit 2 ;;
|
||||
esac
|
||||
done
|
||||
bs_host
|
||||
pushd "$HERE/../../igneum-pow" > /dev/null; bs_context; popd > /dev/null
|
||||
WT_ROOT="$BS_WT_ROOT"
|
||||
WT_NAME="$BS_WT"
|
||||
POW="${POW:-/srv/builds/$WT_NAME/igneum-pow/target/release/igneum-pow}"
|
||||
OUT="${OUT:-$WT_ROOT/docs/design/class-v5-bench}"
|
||||
STAMP=$(date -u +%Y%m%dT%H%M%SZ)
|
||||
REMOTE_OUT="/srv/builds/_log/v5-class/bench/$STAMP"
|
||||
mkdir -p "$OUT"
|
||||
read -r -d '' CMD <<EOF || true
|
||||
set -u; mkdir -p '$REMOTE_OUT'; cd '$REMOTE_OUT'
|
||||
echo "host \$(hostname) load \$(cut -d' ' -f1-3 /proc/loadavg) freq40 \$(cat /sys/devices/system/cpu/cpu40/cpufreq/scaling_cur_freq 2>/dev/null || echo ?) kHz siblings \$(cat /sys/devices/system/cpu/cpu40/topology/thread_siblings_list) pow \$(sha256sum '$POW' | cut -c1-16) state \$(sha256sum '$STATE' | cut -c1-16)" > meta.txt
|
||||
for cls in v4 v5; do
|
||||
extra=""; [ "\$cls" = v5 ] && extra="--state '$STATE'"
|
||||
eval nice -n 19 taskset -c 40 '$POW' bench --seed igneum-genesis --day 2026-10-03 --program-class \$cls \$extra --warps 50 > alone-\$cls.txt 2>&1
|
||||
eval nice -n 19 taskset -c 88 '$POW' bench --seed igneum-genesis --day 2026-10-03 --program-class \$cls \$extra --warps 4000 > sibling-\$cls.txt 2>&1 &
|
||||
sib=\$!
|
||||
sleep 1.5
|
||||
eval nice -n 19 taskset -c 40 '$POW' bench --seed igneum-genesis --day 2026-10-03 --program-class \$cls \$extra --warps 50 > loaded-\$cls.txt 2>&1
|
||||
kill \$sib 2>/dev/null; wait \$sib 2>/dev/null
|
||||
echo "class \$cls done \$(date -u +%H:%M:%SZ)"
|
||||
done
|
||||
echo "load after \$(cut -d' ' -f1-3 /proc/loadavg)" >> meta.txt
|
||||
EOF
|
||||
bs_log "measuring class v4 and v5 verifier on $BS_HOST under the measure hold (pow $POW, state $STATE); results to $OUT"
|
||||
BR_MEASURE=1 BR_KIND=measure BR_COMMAND="igneum-pow bench class v4 and v5 (core 40 alone, then with core 88 loaded)" BR_TARGET=x86_64-unknown-linux-gnu \
|
||||
bs_remote_run "$(dirname "$POW")" "class v5 verifier bench" "$CMD"
|
||||
bs_rsync -a "$BS_HOST:$REMOTE_OUT/" "$OUT/$STAMP/"
|
||||
for cls in v4 v5; do
|
||||
for kind in alone loaded; do
|
||||
f="$OUT/$STAMP/$kind-$cls.txt"
|
||||
cold=$(grep -o "single cold run[^0-9]*[0-9.]* ms" "$f" | head -1 || true)
|
||||
avg=$(grep -i "CPU verify" "$f" | head -1 || true)
|
||||
echo "$cls $kind: $cold | $avg"
|
||||
done
|
||||
done
|
||||
cat "$OUT/$STAMP/meta.txt"
|
||||
81
tools/ladder/verify-bench-remote.sh
Executable file
81
tools/ladder/verify-bench-remote.sh
Executable file
|
|
@ -0,0 +1,81 @@
|
|||
#!/usr/bin/env bash
|
||||
# The latency ladder's verifier bound, measured on igneum-build-1 (docs/design/latency-ladder.md section 4): for every rung
|
||||
# of the ladder, the cold verify of one 32-lane warp of class v4 at that rung on the reference core (core 40, 3.8 GHz under
|
||||
# schedutil, nice 19), alone and with its SMT sibling (core 88) running the same bench, plus the average of 50 warps. The
|
||||
# figure a rung's admissibility reads is the cold run with the sibling loaded; the gate is 10 ms.
|
||||
#
|
||||
# tools/ladder/verify-bench-remote.sh [--rungs "27 35 53 88 173 267"] [--pow <path on the box>] [--out <dir on the Mac>]
|
||||
#
|
||||
# Runs ON the box under its measure hold (infra/build-server/remote-run.sh BR_MEASURE=1: waits for every running build, blocks
|
||||
# new ones and the capacity layer until it ends; one JSONL line of kind measure in /srv/builds/_log/builds.jsonl). The binary is
|
||||
# the box's own build of igneum-pow from this worktree (tools/build-remote.sh from igneum-pow/ puts it at
|
||||
# /srv/builds/<worktree>/igneum-pow/target/release/igneum-pow); the results come back as one text file per run plus a table.
|
||||
# A number taken beside another build is not a number (CLAUDE.md), which is what the hold is for.
|
||||
set -euo pipefail
|
||||
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
BS_TOOL=ladder-verify-bench
|
||||
# shellcheck source=../../infra/build-server/lib.sh
|
||||
. "$HERE/../../infra/build-server/lib.sh"
|
||||
RUNGS="27 35 53 88 173 267"; POW=""; OUT=""
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--rungs) RUNGS="$2"; shift 2 ;; --pow) POW="$2"; shift 2 ;; --out) OUT="$2"; shift 2 ;;
|
||||
*) echo "unknown argument $1" >&2; exit 2 ;;
|
||||
esac
|
||||
done
|
||||
bs_host
|
||||
# the context of the igneum-pow crate (BS_WT, BS_WT_ROOT, branch and sha for the JSONL line); the binary is already on the box
|
||||
( cd "$HERE/../../igneum-pow" ) || bs_die "no igneum-pow beside tools/"
|
||||
pushd "$HERE/../../igneum-pow" > /dev/null; bs_context; popd > /dev/null
|
||||
WT_ROOT="$BS_WT_ROOT"
|
||||
WT_NAME="$BS_WT"
|
||||
POW="${POW:-/srv/builds/$WT_NAME/igneum-pow/target/release/igneum-pow}"
|
||||
OUT="${OUT:-$WT_ROOT/docs/design/latency-ladder-bench}"
|
||||
STAMP=$(date -u +%Y%m%dT%H%M%SZ)
|
||||
REMOTE_OUT="/srv/builds/_log/ladder-bench/$STAMP"
|
||||
mkdir -p "$OUT"
|
||||
# the command remote-run.sh evals under the measure hold: per rung, the quiet-core run, then the two siblings at once
|
||||
read -r -d '' CMD <<EOF || true
|
||||
set -u; mkdir -p '$REMOTE_OUT'; cd '$REMOTE_OUT'
|
||||
echo "host \$(hostname) load \$(cut -d' ' -f1-3 /proc/loadavg) freq40 \$(cat /sys/devices/system/cpu/cpu40/cpufreq/scaling_cur_freq 2>/dev/null || echo ?) kHz siblings \$(cat /sys/devices/system/cpu/cpu40/topology/thread_siblings_list) pow \$(sha256sum '$POW' | cut -c1-16)" > meta.txt
|
||||
for r in $RUNGS; do
|
||||
nice -n 19 taskset -c 40 '$POW' bench --seed igneum-genesis --day 2026-10-03 --class mx8+sh256x\$r --warps 50 > alone-\$r.txt 2>&1
|
||||
# the sibling's load must outlast the measured run (the first run of this script, 22:06Z: a 50-warp sibling finished during
|
||||
# the measured run's own cache fill, so the "loaded" warps ran alone; 4,000 warps is 20 to 40 s, killed when the run ends)
|
||||
nice -n 19 taskset -c 88 '$POW' bench --seed igneum-genesis --day 2026-10-03 --class mx8+sh256x\$r --warps 4000 > sibling-\$r.txt 2>&1 &
|
||||
sib=\$!
|
||||
sleep 1.5
|
||||
nice -n 19 taskset -c 40 '$POW' bench --seed igneum-genesis --day 2026-10-03 --class mx8+sh256x\$r --warps 50 > loaded-\$r.txt 2>&1
|
||||
kill \$sib 2>/dev/null; wait \$sib 2>/dev/null
|
||||
echo "rung reps=\$r done \$(date -u +%H:%M:%SZ)"
|
||||
done
|
||||
echo "load after \$(cut -d' ' -f1-3 /proc/loadavg)" >> meta.txt
|
||||
EOF
|
||||
bs_log "measuring rungs $RUNGS on $BS_HOST under the measure hold (pow $POW); results to $OUT"
|
||||
BR_MEASURE=1 BR_KIND=measure BR_COMMAND="igneum-pow bench per ladder rung (core 40 alone, then with core 88 loaded)" BR_TARGET=x86_64-unknown-linux-gnu \
|
||||
bs_remote_run "$(dirname "$POW")" "latency ladder verifier bench rungs $RUNGS" "$CMD"
|
||||
bs_rsync -a "$BS_HOST:$REMOTE_OUT/" "$OUT/$STAMP/"
|
||||
# the table: cold = the "warp base 0: single cold run" line, avg = the "CPU verify" line
|
||||
python3 - "$OUT/$STAMP" $RUNGS <<'PY'
|
||||
import re, sys, os
|
||||
d = sys.argv[1]; rungs = sys.argv[2:]
|
||||
def read(name):
|
||||
try: t = open(os.path.join(d, name)).read()
|
||||
except FileNotFoundError: return (None, None, None)
|
||||
cold = re.search(r"warp base 0: single cold run ([\d.]+) ms", t)
|
||||
avg = re.search(r"CPU verify: ([\d.]+) ms per 32-lane warp", t)
|
||||
shadow = re.search(r"(\d+) shadow instructions per hash", t)
|
||||
return (float(cold.group(1)) if cold else None, float(avg.group(1)) if avg else None, int(shadow.group(1)) if shadow else None)
|
||||
out = [open(os.path.join(d, "meta.txt")).read().strip()]
|
||||
out.append("| Rung | reps | Shadow instrs per hash | Counted ops (approx) | Cold, core alone (ms) | Avg of 50, alone (ms) | Cold, sibling loaded (ms) | Avg of 50, sibling loaded (ms) | Under 10 ms loaded |")
|
||||
out.append("|---|---|---|---|---|---|---|---|---|")
|
||||
for i, r in enumerate(rungs):
|
||||
a = read(f"alone-{r}.txt"); l = read(f"loaded-{r}.txt")
|
||||
ops = 930 + (8 * 256 * int(r) * 183 + 99) // 100
|
||||
f = lambda x: "?" if x is None else f"{x:.2f}"
|
||||
ok = "?" if l[0] is None else ("yes" if l[0] < 10.0 else "NO")
|
||||
out.append(f"| {i} | {r} | {a[2] or '?'} | {ops:,} | {f(a[0])} | {f(a[1])} | {f(l[0])} | {f(l[1])} | {ok} |")
|
||||
print("\n".join(out))
|
||||
open(os.path.join(d, "table.md"), "w").write("\n".join(out) + "\n")
|
||||
PY
|
||||
bs_log "done; raw lines in $OUT/$STAMP"
|
||||
Loading…
Reference in a new issue