From cb3bc0efe7a0a787e19748e18225ba4b7ac9dd9d Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 21:43:02 +0000 Subject: [PATCH] hot table (Counter ASIC 2.0 layer 5), measured and not adopted, on the ca2-v3 composed class (squash of tag ca2-cache-history-2026-10-05) LoadClass::hot (Option) beside mix, slots, scratch, mixer_mult, growth and era; V3_CLASS = { era: None, hot: None, ..LoadClass::MX4 } (hot stays None: the hot code is behind the flag, measured and not adopted). Op::Hot, the hot slots drawn after the scratch slots, no width roll (v2_loads allows the added form's extra slots), the id suffix hot/[added], the hot parsing inside parse_loads, the era branch first in name(). HotTable under seed_words("igneum-hot/" || epoch seed) with the cache chain and tag umHT, read at H[mulhi(src, HOT_WORDS)]; DatasetSource::hot attached by new_class_day and from_seed_bytes_class; the acceptance stand-in dataset_elem(idx, S[2], S[3]); the three emitters (hot argument after the init words, ht_segment and igneum_hot_fill beside the layout-aware cores); packfile.h hot fields beside class, era, attempt and mixerMult; OpenCL host, Metal packbench and NVRTC worker fill H on the device and self-test it. Eight packs under proto-cuda/packs-ca2-hot (replaced hot32k4 hot64k4 hot96k4 hot64k2 hot64k8, added hot32k4a hot64k4a hot96k4a) re-exported on the merged crate: vectors unchanged, program.h and program.json carry the mixer fields. docs/plans/hot-table.md (design, spec text, per-tier budget, chip model, Mac and PC measurements, the decision: layer 5 out of v3, the 3.0 note); bench-log entry and addenda with job ids and worker sha256s; the two PC playbooks. Checks on this commit: cargo test --release 53 + 19 pass (the pinned v2, mx4, era, readwidth and hot packs); the pinned packs under proto-cuda/packs, packs-ca2-mixer, packs-ca2-era and packs-readwidth untouched; Metal packbench and Apple OpenCL --bench-pack on all eight hot packs (run lock, 2^20 at base 0): 96/96 lanes, hot table head, last line and FNV PASS, one fingerprint per pack on both harnesses, equal to the fingerprints before the rebase (hot32k4 679e5e83378d3790, hot64k4 d4c9e456b039fdef, hot96k4 7c98eceffee9fd73, hot64k2 f43b10a95879b8e5, hot64k8 c11309d743be9392, hot32k4a afb700b2d997c847, hot64k4a ba214baa9c1a9e85, hot96k4a 29e1916aed6deff5). No era or mixer behaviour changed: every resolution kept the ca2-v3 side and appended the hot branch. Co-Authored-By: Claude Fable 5.1 --- docs/bench-log.md | 74 ++++ docs/plans/hot-table.md | 266 ++++++++++++ igneum-pow/src/accept.rs | 67 ++- igneum-pow/src/emit.rs | 248 +++++++++-- igneum-pow/src/generator.rs | 264 +++++++++++- igneum-pow/src/main.rs | 7 + igneum-pow/src/memhard.rs | 131 +++++- igneum-pow/src/verify.rs | 71 +++- igneum-pow/tests/packs.rs | 203 ++++++++- proto-cuda/nvrtc/packfile.h | 35 +- proto-cuda/nvrtc/worker.cpp | 54 ++- proto-cuda/packs-ca2-hot/hot32k4/kernel.cl | 305 +++++++++++++ proto-cuda/packs-ca2-hot/hot32k4/kernel.cu | 179 ++++++++ .../packs-ca2-hot/hot32k4/kernel_bound.cl | 399 ++++++++++++++++++ .../packs-ca2-hot/hot32k4/kernel_bound.cu | 125 ++++++ proto-cuda/packs-ca2-hot/hot32k4/memhard.h | 129 ++++++ .../packs-ca2-hot/hot32k4/memhard.metal | 131 ++++++ proto-cuda/packs-ca2-hot/hot32k4/program.h | 70 +++ proto-cuda/packs-ca2-hot/hot32k4/program.json | 129 ++++++ .../packs-ca2-hot/hot32k4/program.metal | 112 +++++ .../packs-ca2-hot/hot32k4/program_bound.metal | 114 +++++ proto-cuda/packs-ca2-hot/hot32k4/vectors.h | 67 +++ proto-cuda/packs-ca2-hot/hot32k4/vectors.json | 39 ++ proto-cuda/packs-ca2-hot/hot32k4a/kernel.cl | 305 +++++++++++++ proto-cuda/packs-ca2-hot/hot32k4a/kernel.cu | 179 ++++++++ .../packs-ca2-hot/hot32k4a/kernel_bound.cl | 399 ++++++++++++++++++ .../packs-ca2-hot/hot32k4a/kernel_bound.cu | 125 ++++++ proto-cuda/packs-ca2-hot/hot32k4a/memhard.h | 129 ++++++ .../packs-ca2-hot/hot32k4a/memhard.metal | 131 ++++++ proto-cuda/packs-ca2-hot/hot32k4a/program.h | 70 +++ .../packs-ca2-hot/hot32k4a/program.json | 129 ++++++ .../packs-ca2-hot/hot32k4a/program.metal | 112 +++++ .../hot32k4a/program_bound.metal | 114 +++++ proto-cuda/packs-ca2-hot/hot32k4a/vectors.h | 67 +++ .../packs-ca2-hot/hot32k4a/vectors.json | 39 ++ proto-cuda/packs-ca2-hot/hot64k2/kernel.cl | 305 +++++++++++++ proto-cuda/packs-ca2-hot/hot64k2/kernel.cu | 179 ++++++++ .../packs-ca2-hot/hot64k2/kernel_bound.cl | 399 ++++++++++++++++++ .../packs-ca2-hot/hot64k2/kernel_bound.cu | 125 ++++++ proto-cuda/packs-ca2-hot/hot64k2/memhard.h | 129 ++++++ .../packs-ca2-hot/hot64k2/memhard.metal | 131 ++++++ proto-cuda/packs-ca2-hot/hot64k2/program.h | 70 +++ proto-cuda/packs-ca2-hot/hot64k2/program.json | 129 ++++++ .../packs-ca2-hot/hot64k2/program.metal | 112 +++++ .../packs-ca2-hot/hot64k2/program_bound.metal | 114 +++++ proto-cuda/packs-ca2-hot/hot64k2/vectors.h | 67 +++ proto-cuda/packs-ca2-hot/hot64k2/vectors.json | 39 ++ proto-cuda/packs-ca2-hot/hot64k4/kernel.cl | 305 +++++++++++++ proto-cuda/packs-ca2-hot/hot64k4/kernel.cu | 179 ++++++++ .../packs-ca2-hot/hot64k4/kernel_bound.cl | 399 ++++++++++++++++++ .../packs-ca2-hot/hot64k4/kernel_bound.cu | 125 ++++++ proto-cuda/packs-ca2-hot/hot64k4/memhard.h | 129 ++++++ .../packs-ca2-hot/hot64k4/memhard.metal | 131 ++++++ proto-cuda/packs-ca2-hot/hot64k4/program.h | 70 +++ proto-cuda/packs-ca2-hot/hot64k4/program.json | 129 ++++++ .../packs-ca2-hot/hot64k4/program.metal | 112 +++++ .../packs-ca2-hot/hot64k4/program_bound.metal | 114 +++++ proto-cuda/packs-ca2-hot/hot64k4/vectors.h | 67 +++ proto-cuda/packs-ca2-hot/hot64k4/vectors.json | 39 ++ proto-cuda/packs-ca2-hot/hot64k4a/kernel.cl | 305 +++++++++++++ proto-cuda/packs-ca2-hot/hot64k4a/kernel.cu | 179 ++++++++ .../packs-ca2-hot/hot64k4a/kernel_bound.cl | 399 ++++++++++++++++++ .../packs-ca2-hot/hot64k4a/kernel_bound.cu | 125 ++++++ proto-cuda/packs-ca2-hot/hot64k4a/memhard.h | 129 ++++++ .../packs-ca2-hot/hot64k4a/memhard.metal | 131 ++++++ proto-cuda/packs-ca2-hot/hot64k4a/program.h | 70 +++ .../packs-ca2-hot/hot64k4a/program.json | 129 ++++++ .../packs-ca2-hot/hot64k4a/program.metal | 112 +++++ .../hot64k4a/program_bound.metal | 114 +++++ proto-cuda/packs-ca2-hot/hot64k4a/vectors.h | 67 +++ .../packs-ca2-hot/hot64k4a/vectors.json | 39 ++ proto-cuda/packs-ca2-hot/hot64k8/kernel.cl | 305 +++++++++++++ proto-cuda/packs-ca2-hot/hot64k8/kernel.cu | 179 ++++++++ .../packs-ca2-hot/hot64k8/kernel_bound.cl | 399 ++++++++++++++++++ .../packs-ca2-hot/hot64k8/kernel_bound.cu | 125 ++++++ proto-cuda/packs-ca2-hot/hot64k8/memhard.h | 129 ++++++ .../packs-ca2-hot/hot64k8/memhard.metal | 131 ++++++ proto-cuda/packs-ca2-hot/hot64k8/program.h | 70 +++ proto-cuda/packs-ca2-hot/hot64k8/program.json | 129 ++++++ .../packs-ca2-hot/hot64k8/program.metal | 112 +++++ .../packs-ca2-hot/hot64k8/program_bound.metal | 114 +++++ proto-cuda/packs-ca2-hot/hot64k8/vectors.h | 67 +++ proto-cuda/packs-ca2-hot/hot64k8/vectors.json | 39 ++ proto-cuda/packs-ca2-hot/hot96k4/kernel.cl | 305 +++++++++++++ proto-cuda/packs-ca2-hot/hot96k4/kernel.cu | 179 ++++++++ .../packs-ca2-hot/hot96k4/kernel_bound.cl | 399 ++++++++++++++++++ .../packs-ca2-hot/hot96k4/kernel_bound.cu | 125 ++++++ proto-cuda/packs-ca2-hot/hot96k4/memhard.h | 129 ++++++ .../packs-ca2-hot/hot96k4/memhard.metal | 131 ++++++ proto-cuda/packs-ca2-hot/hot96k4/program.h | 70 +++ proto-cuda/packs-ca2-hot/hot96k4/program.json | 129 ++++++ .../packs-ca2-hot/hot96k4/program.metal | 112 +++++ .../packs-ca2-hot/hot96k4/program_bound.metal | 114 +++++ proto-cuda/packs-ca2-hot/hot96k4/vectors.h | 67 +++ proto-cuda/packs-ca2-hot/hot96k4/vectors.json | 39 ++ proto-cuda/packs-ca2-hot/hot96k4a/kernel.cl | 305 +++++++++++++ proto-cuda/packs-ca2-hot/hot96k4a/kernel.cu | 179 ++++++++ .../packs-ca2-hot/hot96k4a/kernel_bound.cl | 399 ++++++++++++++++++ .../packs-ca2-hot/hot96k4a/kernel_bound.cu | 125 ++++++ proto-cuda/packs-ca2-hot/hot96k4a/memhard.h | 129 ++++++ .../packs-ca2-hot/hot96k4a/memhard.metal | 131 ++++++ proto-cuda/packs-ca2-hot/hot96k4a/program.h | 70 +++ .../packs-ca2-hot/hot96k4a/program.json | 129 ++++++ .../packs-ca2-hot/hot96k4a/program.metal | 112 +++++ .../hot96k4a/program_bound.metal | 114 +++++ proto-cuda/packs-ca2-hot/hot96k4a/vectors.h | 67 +++ .../packs-ca2-hot/hot96k4a/vectors.json | 39 ++ proto-metal/packbench.swift | 42 +- proto-opencl/host.c | 71 +++- relay/playbooks/ca2-hot-5090-bench.ps1 | 66 +++ relay/playbooks/ca2-hot-9070-bench.ps1 | 85 ++++ 111 files changed, 16006 insertions(+), 70 deletions(-) create mode 100644 docs/plans/hot-table.md create mode 100644 proto-cuda/packs-ca2-hot/hot32k4/kernel.cl create mode 100644 proto-cuda/packs-ca2-hot/hot32k4/kernel.cu create mode 100644 proto-cuda/packs-ca2-hot/hot32k4/kernel_bound.cl create mode 100644 proto-cuda/packs-ca2-hot/hot32k4/kernel_bound.cu create mode 100644 proto-cuda/packs-ca2-hot/hot32k4/memhard.h create mode 100644 proto-cuda/packs-ca2-hot/hot32k4/memhard.metal create mode 100644 proto-cuda/packs-ca2-hot/hot32k4/program.h create mode 100644 proto-cuda/packs-ca2-hot/hot32k4/program.json create mode 100644 proto-cuda/packs-ca2-hot/hot32k4/program.metal create mode 100644 proto-cuda/packs-ca2-hot/hot32k4/program_bound.metal create mode 100644 proto-cuda/packs-ca2-hot/hot32k4/vectors.h create mode 100644 proto-cuda/packs-ca2-hot/hot32k4/vectors.json create mode 100644 proto-cuda/packs-ca2-hot/hot32k4a/kernel.cl create mode 100644 proto-cuda/packs-ca2-hot/hot32k4a/kernel.cu create mode 100644 proto-cuda/packs-ca2-hot/hot32k4a/kernel_bound.cl create mode 100644 proto-cuda/packs-ca2-hot/hot32k4a/kernel_bound.cu create mode 100644 proto-cuda/packs-ca2-hot/hot32k4a/memhard.h create mode 100644 proto-cuda/packs-ca2-hot/hot32k4a/memhard.metal create mode 100644 proto-cuda/packs-ca2-hot/hot32k4a/program.h create mode 100644 proto-cuda/packs-ca2-hot/hot32k4a/program.json create mode 100644 proto-cuda/packs-ca2-hot/hot32k4a/program.metal create mode 100644 proto-cuda/packs-ca2-hot/hot32k4a/program_bound.metal create mode 100644 proto-cuda/packs-ca2-hot/hot32k4a/vectors.h create mode 100644 proto-cuda/packs-ca2-hot/hot32k4a/vectors.json create mode 100644 proto-cuda/packs-ca2-hot/hot64k2/kernel.cl create mode 100644 proto-cuda/packs-ca2-hot/hot64k2/kernel.cu create mode 100644 proto-cuda/packs-ca2-hot/hot64k2/kernel_bound.cl create mode 100644 proto-cuda/packs-ca2-hot/hot64k2/kernel_bound.cu create mode 100644 proto-cuda/packs-ca2-hot/hot64k2/memhard.h create mode 100644 proto-cuda/packs-ca2-hot/hot64k2/memhard.metal create mode 100644 proto-cuda/packs-ca2-hot/hot64k2/program.h create mode 100644 proto-cuda/packs-ca2-hot/hot64k2/program.json create mode 100644 proto-cuda/packs-ca2-hot/hot64k2/program.metal create mode 100644 proto-cuda/packs-ca2-hot/hot64k2/program_bound.metal create mode 100644 proto-cuda/packs-ca2-hot/hot64k2/vectors.h create mode 100644 proto-cuda/packs-ca2-hot/hot64k2/vectors.json create mode 100644 proto-cuda/packs-ca2-hot/hot64k4/kernel.cl create mode 100644 proto-cuda/packs-ca2-hot/hot64k4/kernel.cu create mode 100644 proto-cuda/packs-ca2-hot/hot64k4/kernel_bound.cl create mode 100644 proto-cuda/packs-ca2-hot/hot64k4/kernel_bound.cu create mode 100644 proto-cuda/packs-ca2-hot/hot64k4/memhard.h create mode 100644 proto-cuda/packs-ca2-hot/hot64k4/memhard.metal create mode 100644 proto-cuda/packs-ca2-hot/hot64k4/program.h create mode 100644 proto-cuda/packs-ca2-hot/hot64k4/program.json create mode 100644 proto-cuda/packs-ca2-hot/hot64k4/program.metal create mode 100644 proto-cuda/packs-ca2-hot/hot64k4/program_bound.metal create mode 100644 proto-cuda/packs-ca2-hot/hot64k4/vectors.h create mode 100644 proto-cuda/packs-ca2-hot/hot64k4/vectors.json create mode 100644 proto-cuda/packs-ca2-hot/hot64k4a/kernel.cl create mode 100644 proto-cuda/packs-ca2-hot/hot64k4a/kernel.cu create mode 100644 proto-cuda/packs-ca2-hot/hot64k4a/kernel_bound.cl create mode 100644 proto-cuda/packs-ca2-hot/hot64k4a/kernel_bound.cu create mode 100644 proto-cuda/packs-ca2-hot/hot64k4a/memhard.h create mode 100644 proto-cuda/packs-ca2-hot/hot64k4a/memhard.metal create mode 100644 proto-cuda/packs-ca2-hot/hot64k4a/program.h create mode 100644 proto-cuda/packs-ca2-hot/hot64k4a/program.json create mode 100644 proto-cuda/packs-ca2-hot/hot64k4a/program.metal create mode 100644 proto-cuda/packs-ca2-hot/hot64k4a/program_bound.metal create mode 100644 proto-cuda/packs-ca2-hot/hot64k4a/vectors.h create mode 100644 proto-cuda/packs-ca2-hot/hot64k4a/vectors.json create mode 100644 proto-cuda/packs-ca2-hot/hot64k8/kernel.cl create mode 100644 proto-cuda/packs-ca2-hot/hot64k8/kernel.cu create mode 100644 proto-cuda/packs-ca2-hot/hot64k8/kernel_bound.cl create mode 100644 proto-cuda/packs-ca2-hot/hot64k8/kernel_bound.cu create mode 100644 proto-cuda/packs-ca2-hot/hot64k8/memhard.h create mode 100644 proto-cuda/packs-ca2-hot/hot64k8/memhard.metal create mode 100644 proto-cuda/packs-ca2-hot/hot64k8/program.h create mode 100644 proto-cuda/packs-ca2-hot/hot64k8/program.json create mode 100644 proto-cuda/packs-ca2-hot/hot64k8/program.metal create mode 100644 proto-cuda/packs-ca2-hot/hot64k8/program_bound.metal create mode 100644 proto-cuda/packs-ca2-hot/hot64k8/vectors.h create mode 100644 proto-cuda/packs-ca2-hot/hot64k8/vectors.json create mode 100644 proto-cuda/packs-ca2-hot/hot96k4/kernel.cl create mode 100644 proto-cuda/packs-ca2-hot/hot96k4/kernel.cu create mode 100644 proto-cuda/packs-ca2-hot/hot96k4/kernel_bound.cl create mode 100644 proto-cuda/packs-ca2-hot/hot96k4/kernel_bound.cu create mode 100644 proto-cuda/packs-ca2-hot/hot96k4/memhard.h create mode 100644 proto-cuda/packs-ca2-hot/hot96k4/memhard.metal create mode 100644 proto-cuda/packs-ca2-hot/hot96k4/program.h create mode 100644 proto-cuda/packs-ca2-hot/hot96k4/program.json create mode 100644 proto-cuda/packs-ca2-hot/hot96k4/program.metal create mode 100644 proto-cuda/packs-ca2-hot/hot96k4/program_bound.metal create mode 100644 proto-cuda/packs-ca2-hot/hot96k4/vectors.h create mode 100644 proto-cuda/packs-ca2-hot/hot96k4/vectors.json create mode 100644 proto-cuda/packs-ca2-hot/hot96k4a/kernel.cl create mode 100644 proto-cuda/packs-ca2-hot/hot96k4a/kernel.cu create mode 100644 proto-cuda/packs-ca2-hot/hot96k4a/kernel_bound.cl create mode 100644 proto-cuda/packs-ca2-hot/hot96k4a/kernel_bound.cu create mode 100644 proto-cuda/packs-ca2-hot/hot96k4a/memhard.h create mode 100644 proto-cuda/packs-ca2-hot/hot96k4a/memhard.metal create mode 100644 proto-cuda/packs-ca2-hot/hot96k4a/program.h create mode 100644 proto-cuda/packs-ca2-hot/hot96k4a/program.json create mode 100644 proto-cuda/packs-ca2-hot/hot96k4a/program.metal create mode 100644 proto-cuda/packs-ca2-hot/hot96k4a/program_bound.metal create mode 100644 proto-cuda/packs-ca2-hot/hot96k4a/vectors.h create mode 100644 proto-cuda/packs-ca2-hot/hot96k4a/vectors.json create mode 100644 relay/playbooks/ca2-hot-5090-bench.ps1 create mode 100644 relay/playbooks/ca2-hot-9070-bench.ps1 diff --git a/docs/bench-log.md b/docs/bench-log.md index d9bb12a0c..8a82a73b2 100644 --- a/docs/bench-log.md +++ b/docs/bench-log.md @@ -1571,3 +1571,77 @@ The 9070 XT rows are 2,048 persistent warps (4,096 within 1 percent), arena 64 M **Readings.** (1) Same count, wider: the vendor gap does not move at 16 B (7.8x) because on the 9070 XT a 4-byte read already costs a 64-byte line and on the 5090 a 16-byte read costs one 32-byte sector, the same as 4 bytes: the memory systems do identical work, only the fold's input grows. At 64 B the gap closes to 4.1x, entirely by the 5090 losing half its rate (its share falls to 0.58 and its DRAM traffic reaches 589 GB/s, 37 percent of the stream: bandwidth, not latency, bounds it), while the 9070 XT and the M5 Max do not move. (2) Fewer, wider (w64x4): 3.7x, but every card runs 4x faster because the dependent chain is 32 loads long instead of 128; the 5090 sits at a 0.56 share (bandwidth), so a chip with more bandwidth per dollar than a GPU gains, which is the Ethash shape the design avoids. (3) The mix: the hour-to-hour spread is 7 to 22 percent of the median per card (the 5090 the widest, because its 64-byte loads are the expensive ones and their count per program runs 2 to 8 of 16); the programs with many 64-byte loads (mixA-3, mixA-5, mixB-2) are the slow hours on the 5090 and the fast ones nowhere. (4) The scratch: on the 5090 every RMW share costs 12 to 48 percent against the persistent control, the 32 KiB arena less than the 128 KiB one (the smaller arena, 64 MiB over 2,048 warps, sits inside the 96 MB L2); on the M5 Max the 32 KiB rows are FASTER than the control (+12 and +74 percent at 25 and 50 percent), because the arena (128 MiB over 4,096 warps) lives in the chip's caches and a scratch op is cheaper than a dataset read, so replacing dataset loads raises the rate: the scratch at these sizes is not memory work on Apple and is partly cached on NVIDIA. The chip row for these variants comes from the ca2-soundness branch; what this entry gives is the GPU cost and the share. (5) Latency-bound shares: v2 0.87 to 1.01 on the three cards, w16 0.84 to 1.03, w64 0.58 (5090) and 0.78 (9070 XT); the Mac's shares above 1 are an Apple OpenCL probe under load against a Metal rate. Jobs: `run-readwidth-5090-20261005` and `run-readwidth-9070-20261005` (probes; the packs refused for their string seeds, fixed in a9e002c), `run-readwidth-5090-20261005c`, `run-readwidth-9070-20261005c` (benches), `run-readwidth-9070-scratch-20261005d` (the scratch packs after the `__local` fix d0018cf, AMD's compiler requires the exchange buffer at the kernel's outermost scope); read back with `node tools/jobs.mjs --all`. Mac commands and logs: `docs/plans/read-width.md` section 3. The worker exes for the jobs: `proto-cuda/nvrtc/build-windows.sh` on this branch (mingw), sha256 of the CUDA one `6f46336f...defe1`. +## 5 October 2026 (night), the hot table on the M5 Max: a second table sized to GPU cache beside the 1 GiB dataset (Counter ASIC 2.0 layer 5) + +Branch `ca2-cache` (on readwidth 1ea7a52), `docs/plans/hot-table.md`. Apple M5 Max, measure lock held, the Mac's load average 14 to 27 throughout (other agents' CPU work; the lock serialises builds and measurements, not every process), so the ratios inside one session are the result and the absolute rates are not quiet numbers. Packs `proto-cuda/packs-ca2-hot/hot{32,64,96}k4`, `hot64k2`, `hot64k8` from `igneum-pow export --seed igneum-genesis --day 2026-10-03 --class hotk` (the version 2 genesis program with k of its 16 loads redirected to an S MiB table H keyed by `seed_words("igneum-hot/" || seed bytes)`, read at `H[mulhi(src, words)]`). + +**Probe** (`proto-opencl/igneum-bench-cl-igneum-genesis-mh --memprobe --probe-mib S`, Apple OpenCL, wall time, best of 3, 256 dependent steps per lane, work-group 256; the ceiling row is 4,194,304 lanes): + +| MiB | chase at 4,096 lanes | ns per dependent load | chase ceiling, G loads/s | indep x8 ceiling | stream | +|---|---|---|---|---|---| +| 32 | 3.51 G/s | 1,168 | 21.7 | 21.8 | 138.7 GB/s | +| 64 | 3.63 | 1,129 | 12.8 | 13.0 | 199.1 | +| 96 | 3.30 | 1,242 | 12.3 | 12.7 | 242.6 | +| 1024 | 2.22 | 1,844 | 3.50 | 3.50 | 521.5 | + +**Hash rate and bit-exactness** (Metal `proto-metal/packbench --pack --batches 5 --batch-log2 24`, GPU time; Apple OpenCL `--bench-pack --pack --batches 5 --batch-log2 24`, wall; both fill H on the device from the pack's `igneum_hot_fill` and check it; fingerprint = FNV-1a 64 over 2^24 outputs at base 0): + +| Pack | Metal Mhash/s | Apple OpenCL Mhash/s | fingerprint (equal on both) | vectors | hot table head, last line, FNV | g against v2 (Metal) | probe-predicted g | ideal g | +|---|---|---|---|---|---|---|---|---| +| igneum-genesis-mh (v2) | 27.68 | 27.61 | 25f96e7dce90bd4e | 96/96 both | none | 1 | 1 | 1 | +| hot32k4 | 33.90 | 33.92 | d2e6cf3b61d0b9fe | 96/96 both | PASS both | 1.22 | 1.27 | 1.33 | +| hot64k4 | 30.93 | 30.89 | e4c5263ac650cc0d | 96/96 both | PASS both | 1.12 | 1.22 | 1.33 | +| hot96k4 | 29.06 | 28.97 | 5d63439b6e394521 | 96/96 both | PASS both | 1.05 | 1.22 | 1.33 | +| hot64k2 | 27.67 | 27.27 | 352633bdbbb0d2b6 | 96/96 both | PASS both | 1.00 | 1.10 | 1.14 | +| hot64k8 | 47.42 | 46.73 | da54630d7dfaaf85 | 96/96 both | PASS both | 1.71 | 1.57 | 2.0 | + +Hot table fill, Metal GPU time: 0.07 ms (32 MiB), 0.15 (64), 0.22 (96). Hot table FNV-1a 64 of the genesis epoch: c1767ba3ef02719f (32 MiB), 77ca4b9527104530 (64), 79bcf436c4e5bc47 (96); cache 48c4f5bf24166b2e unchanged. + +**CPU verifier** (`igneum-pow bench --seed igneum-genesis --day 2026-10-03 --class --warps 50`, one core, release): + +| Class | hot fill, one core | items per warp | ms per warp | +|---|---|---|---| +| v2 | none | 4,096 | 0.626 | +| hot32k4 | 24.0 ms | 3,072 | 0.489 | +| hot64k4 | 46.4 ms | 3,072 | 0.504 | +| hot96k4 | 73.0 ms | 3,072 | 0.488 | +| hot64k2 | 45.5 ms | 3,584 | 0.560 | +| hot64k8 | 47.7 ms | 2,048 | 0.344 | + +Reading: bit-exact across Metal, Apple OpenCL and the Rust reference on every hot pack, hot table included. On this card the 32 MiB table delivers 92% of the probe's predicted gain with the dataset streaming beside it, 64 MiB about half, 96 MiB a quarter; k = 8 at 64 MiB gives 1.71x against an ideal 2.0x. The verifier gets cheaper with k (a hot load is one table read, a dataset load is an item derivation) and pays 24 to 73 ms per epoch for the fill. Chip model with these g in the plan, section 6.4. The RTX 5090 and RX 9070 XT rows are a prepared PC job (`relay/playbooks/ca2-hot-{5090,9070}-bench.ps1`, zip `~/Desktop/igneum-ca2-hot.zip`), not run. + +Crate: `cargo test --release` 52 pass (39 unit, 13 pack tests: the four pinned v2 packs byte-identical, the five hot packs pinned with their load-form count: exactly 16 - k masked dataset loads and k hot loads per hash kernel). + +**Addendum, the added form** (coordinator's form of 5 October 2026: 16 + k load slots, the k hot ones drawn among them, the 16 dataset loads and the 4,096-item verifier bound unchanged; packs `hot32k4a`, `hot64k4a`, `hot96k4a`; second Mac session 21:03 to 21:19 UTC, load average 7 to 14; same harnesses and commands, branch `ca2-cache` on ca2-v3 464d6e1, the hosts rebuilt on the merged packfile.h): + +| Pack | Metal Mhash/s | Apple OpenCL Mhash/s | fingerprint (equal on both) | vectors | hot table | g against v2 (Metal, v2 27.63 in this session) | probe-predicted g | CPU verify ms/warp (v2 0.602) | hot fill, one core | +|---|---|---|---|---|---|---|---|---|---| +| hot32k4a | 25.76 | 25.72 | 8a3414735db4523c | 96/96 both | PASS both | 0.93 | 0.96 | 0.631 | 21.7 ms | +| hot64k4a | 23.92 | 23.87 | 45668f34105f6307 | 96/96 both | PASS both | 0.87 | 0.94 | 0.609 | 43.3 ms | +| hot96k4a | 22.92 | 22.88 | af763997dfee4c82 | 96/96 both | PASS both | 0.83 | 0.93 | 0.614 | 64.9 ms | + +Reading: the added form costs this card 7, 13 and 17% of its rate at 32, 64 and 96 MiB for four extra loads per iteration, more than the probe predicts as the table grows; the verifier is unchanged (4,096 items, plus 32 table reads) and pays the fill per epoch. Chip arithmetic in the plan, section 6.4. All eight packs load and self-test through the rebuilt OpenCL host (the Windows exe's host.c) on the Mac. + +**Addendum, the PCs** (5 October 2026, 21:29 to 21:35 UTC, PC 1 ae432dc7, app 0.3.9 before and after; fetch `fetch-ca2-hot-20261005` (zip sha256 bd49faa1c9d48024f49c615481faff5c68a4c09f0889dbaf009c208674d67b3f), jobs `run-ca2-hot-5090-20261005` (126 s) and `run-ca2-hot-9070-20261005` (247 s), both exit 0, the card under test switched off in the app through `api/cards` and restored; workers `igneum-worker-cuda.exe` sha256 956c4ab34f42cbcd1d2c1c6fb1a58fd9b3a8c70166df771cafcd0296ca6a27d4 and `igneum-worker-opencl.exe` sha256 32d3d34390aad70485c3524424c354223387137d383b5c5daf01f40073c12703, built from ca2-cache 196db96 on ca2-v3's merged packfile.h d2cd6e1; read back with `node tools/jobs.mjs --all`): + +Probe (`--memprobe --probe-mib S`, dependent 4 B chase ceiling at 4,194,304 lanes, G loads/s; ns per dependent load at 4,096 lanes in brackets): + +| Card | 32 MiB | 64 | 96 | 1024 | stream at 1024 MiB | +|---|---|---|---|---|---| +| RTX 5090 (CUDA, wall) | 112.6 (320) | 112.6 (340) | 112.6 (336) | 17.6 (610) | 1,563 GB/s | +| RX 9070 XT (OpenCL, event) | 9.88 (396) | 9.47 (457) | 8.18 (454) | 2.43 (1,579) | 633 GB/s | + +Rates (5 dispatches of 2^24 after a warm-up; 5090 `--bench --block-warps 1`, 9070 XT `--bench-pack --device 1` work-group 256; every row check=PASS with the Mac's fingerprint; v2 references from the readwidth entry, same night, same workers: 136.1 and 18.15 MH/s): + +| Pack | 5090 MH/s | g | 9070 XT MH/s | g | ideal g | +|---|---|---|---|---|---| +| hot32k4 | 146.6 | 1.08 | 19.79 | 1.09 | 1.33 | +| hot64k4 | 140.8 | 1.03 | 18.73 | 1.03 | 1.33 | +| hot96k4 | 138.5 | 1.02 | 18.33 | 1.01 | 1.33 | +| hot64k2 | 137.5 | 1.01 | 18.17 | 1.00 | 1.14 | +| hot64k8 | 163.6 | 1.20 | 22.32 | 1.23 | 2.0 | +| hot32k4a | 118.7 | 0.87 | 15.27 | 0.84 | 1 | +| hot64k4a | 115.4 | 0.85 | 14.62 | 0.81 | 1 | +| hot96k4a | 114.4 | 0.84 | 14.56 | 0.80 | 1 | + +Reading: the probe promises a full hit rate on the 5090 (every S inside the 96 MiB L2 at one ceiling, 6.4x DRAM) and the hash gets 2 to 8% at k = 4 and 20% at k = 8; the 9070 XT the same shape. The dataset's random lines evict the table from the shared cache on every card. The added form costs 13 to 20% of the rate. Recommendation in `docs/plans/hot-table.md` section 6.4: do not adopt layer 5 in either form on these measurements. diff --git a/docs/plans/hot-table.md b/docs/plans/hot-table.md new file mode 100644 index 000000000..426ca7aa3 --- /dev/null +++ b/docs/plans/hot-table.md @@ -0,0 +1,266 @@ +# Hot table: a second table sized to GPU cache, read beside the 1 GiB dataset + +Counter ASIC 2.0, layer 5 (`docs/plans/counter-asic-2.md`). Experiment branch `ca2-cache`, 5 October 2026 (night), on top of the read-width branch (`readwidth` b970dda: `LoadClass`, `verify::fold_words`, the scratch variant, `proto-metal/packbench.swift`, `proto-opencl/host.c --bench-pack`). Nothing here is the lottery hash: every hot class sits behind the generator flag and the default v2 path is byte-identical (the pinned packs `igneum-genesis-mh` and `igneum-devnet-v4-epoch0` are diffed by `igneum-pow/tests/packs.rs`). + +## 1. The idea + +The honest hash is 128 dependent random 4-byte reads over 1 GiB (spec 01 section 1.5). A chip that wants a gain on that must beat a GPU at DRAM random reads, which is the same physics for both (the plan's last section). What a chip can do that a GPU cannot is choose its memory: a chip can mirror read-only data into SRAM and serve it at SRAM latency, if the data fits. + +The hot table turns that around. A second table `H` of `S` MiB (32, 64 or 96 in this experiment) is derived from the epoch seed and read by `k` of the 16 load slots (2, 4 or 8). `S` is chosen to fit the caches of the cards that mine: 96 MiB L2 on the RTX 5090, 64 MB Infinity Cache plus 8 MB L2 on the RX 9070 XT, the system level cache on the M5 Max (figures approximate, from memory; the probe of section 6 measures what each card does at each size). A GPU gets the hot loads as cache hits for free. A chip must carry `S` MiB of SRAM for the same hits, beside the DRAM path it still needs for the other `16 - k` slots, or serve `H` from DRAM and fall behind the GPU by the hot share. Either way the chip pays for something the GPU already has. + +What this does not change: the cold loads (the 1 GiB dataset) stay a dependent chain at DRAM latency, so the latency-bound property holds for them; the hot loads are interleaved in the same chain (every load's address is a fresh register of the same iteration, G2), so a hot hit shortens the chain by one DRAM latency and nothing else. + +## 2. The specification text (proposed; prototype values, to be fixed at gate 1) + +### 2.1 Hot key and fill + +For an epoch whose program seed bytes are `e` (spec 01 section 1.12: the UTF-8 of a seed string in the packs, the 32-byte VDF output on the chain; the bytes before the attempt suffix of 1.4.6, so every attempt of one epoch shares one table): + +``` +KH = seed_words_from_bytes("igneum-hot/" || e) 8 words +``` + +`H` has `N_H = S x 2^18` words (`S` MiB) in `N_seg = S x 256` segments of 64 chained lines of 16 words, filled exactly as the cache of section 1.8.3 with `KH` in place of `K` and the tag `("Igne", "umHT") = (0x49676e65, 0x756d4854)` in place of `("Igne", "umMH")`: + +``` +prev = 0^16 +for j in 0..63: + in = prev XOR (sigma[0..3] || KH[0..7] || s || j || tag[0..1]) + line = B(in) ChaCha12 with feed-forward, section 1.8.2 + H[segment s, line j] = line + prev = line +``` + +One GPU thread per segment, as the cache fill. The fill is a once-per-epoch cost: `S / 256` of the 256 MiB cache fill (section 1.8.3 table: 0.6 to 2.1 ms on the M5 Max GPU, 0.67 ms on the RTX 5090, 175 to 190 ms on one CPU core for 256 MiB), so under 1 ms on a GPU and 22 to 71 ms on one core, measured below. + +### 2.2 The hot load + +A hot load slot reads `H` instead of the dataset with the same fold (width 1: a plain XOR): + +``` +hot: dst = dst XOR H[mulhi(src, N_H)] +``` + +`mulhi(a, b)` is the high 32 bits of the 64-bit product (the `mulhi` family of section 1.4.1, bit-exact on Metal, CUDA and OpenCL). The index lies in `[0, N_H)` for any `N_H`, which is what lets `S = 96` exist: 96 MiB is not a power of two, so `src AND MASK` cannot address it. This is the multiply-shift range reduction section 1.13.3 proposes for the growing dataset, so the hot table is also its first measured use. For `S = 32` and `64` the mapping takes the top 23 or 24 bits of `src` where the dataset load takes the low 28; a fresh source is uniform, so neither choice costs uniformity (section 6, hot-load uniformity test). + +The emitted text has one form per dialect, checkable by text search as the mask check of section 1.14 item 2: `hot[mulhi(rN, HOT_WORDS)]` (Metal), `hot[__umulhi(rN, HOT_WORDS)]` (CUDA), `hot[mul_hi(rN, HOT_WORDS)]` (OpenCL), with `HOT_WORDS` a literal of the pack. A hot pack's hash kernels carry exactly `16 - k` masked dataset loads and exactly `k` hot loads (`igneum-pow/tests/packs.rs`). + +### 2.3 Which slots are hot + +The 16 load slots are drawn first by the partial Fisher-Yates of section 1.4.3, which emits them in a uniformly random order. The first `k` slots in that draw order are the hot slots. Drawn, not fixed, because a fixed pattern (every fourth load, say) would let a pipeline schedule its SRAM reads statically for every hour; drawn costs no extra draw, so a `hot(S, k)` program takes the version 2 stream exactly and is the version 2 program with `k` of its loads redirected (the width roll of the read-width classes is not taken: `LoadClass::takes_width_roll`). A hot slot keeps every rule of a load: the fresh-source draw (G2), the injecting-write count (acceptance (b)), the lane-constant test and the distinct-address count (acceptance (c)); its address is tagged apart from dataset addresses in the count so a hot word and a dataset word at the same index are two addresses. + +The class composes with the read-width fields of `LoadClass` (width mix, scratch `k` and `kb`): the hot slots are taken from the drawn slots after the scratch slots, so a class may carry a width mix, a scratch and a hot table at once. The experiment packs below use the version 2 widths and no scratch. + +Two forms (coordinator's decision, 5 October 2026, after the first Mac measurement): + +| Form | Load slots drawn | Dataset loads per hash | Hot loads per hash | Items per warp (verifier bound) | Class name | What a chip pays | +|---|---|---|---|---|---|---| +| replaced (`hot(S, k)`) | 16, the first `k` in draw order hot | `(16 - k) x 8` | `8k` | `(16 - k) x 256` | `hot64k4` | the on-die-cache recompute chip of `docs/analysis/scratch-soundness.md` 3.4 (the 256 MiB cache in SRAM, items derived on the fly, 333 MH/s at 50 T op/s, 2.4x the 5090) GAINS: a hot load replaces an item derivation (1,170 ops) with an SRAM read, so at `k = 4` its rate rises 1.33x against the GPU's measured 1.05 to 1.22x | +| added (`hot(S, k, added)`) | `16 + k`, the first `k` in draw order hot | `16 x 8 = 128` | `8k` | 4,096, unchanged | `hot64k4a` | `S` MiB of SRAM and `k` reads per iteration for nothing: the 16 item derivations stay; the GPU pays `k` cache hits | + +The added form is the one that taxes the named chip; the replaced form stays as the measured record (section 6). In the added form the slot draw is a partial Fisher-Yates of `16 + k` slots over 1..63, so the program stream differs from version 2 (another slot count), and the width roll is still not taken (`takes_width_roll`: the slot count is 16 plus the class's own added hot slots). `loads_per_hash` is `128 + 8k`; the acceptance rule's distinct-address bound scales with it as for the read-width classes. + +Program id: `FNV-1a 64 over "igneum-program-rw/" || generator || seed words || attempt || mix || load_slots || "hot/" || S || k [|| "added"]` (the read-width id with the hot fields appended), so no hot pack can be mistaken for a version 2 one, for another hot class or for the other form. + +### 2.4 Acceptance (section 1.4.6) + +The dynamic test stays a pure function of the program. A hot load reads `dataset_elem(mulhi(src, N_H), SW[2], SW[3])` (the six-operation closed form keyed by seed words 2 and 3, where the dataset stand-in is keyed by words 0 and 1), so the two stand-ins are two tables without any cache or day. Every other test is unchanged; the distinct-address bound is the read-width rule's (dataset and hot loads counted, scratch read-modify-writes not). + +### 2.5 The verifier (section 1.11) + +A verifier holds, per day, the mixer parameters and the 256 MiB cache, and, per epoch, `H` (`S` MiB, filled on one core in the times of section 6). A hot load is one table read per lane; a dataset load is the item derivation of section 1.11 as before. The verifier never holds the dataset. Per epoch the verifier's memory is `256 MiB + S MiB`. + +## 3. Memory budget (the project lead's cap, 5 October 2026: the working set on a card stays under 6 GB on an 8 GB card) + +Decided by the coordinator on 5 October 2026 after the card-lifetime review (`docs/analysis/card-lifetime-2026-10-05.md`, branch `card-lifetime` 1fecfe2, merged into `ca2-coord`): the GPU frees the 256 MiB cache after the daily dataset build (the hash never reads it; the rebuild costs 0.67 ms of fill and 13.4 ms of build on the 5090, bench-log 3 October 2026), so the cache is not resident and the working set is dataset + hot table + scratch (0 in v3) + buffers. The table below is that review's per-tier table with the hot table at its largest (96 MiB), buffers 128 MiB (the harnesses' 2^24-nonce output), resident warps = SMs x 48 on NVIDIA Ampere and later (SM counts approximate, from memory; the GTX 1650 is Turing at 32 warps per SM), 2,048 launched warps on Apple (the Metal harness). The scratch columns are the read-width variant's 32 and 128 KiB per resident warp, kept for the record; v3 carries no scratch. + +| Tier | Card assumed (SMs, approximate) | Resident warps | Scratch at 32 KiB | Scratch at 128 KiB | Hot table | Dataset at genesis | Buffers | Total, no scratch | Total at 32 KiB | Total at 128 KiB | Years the dataset leaves under mapping (b), cache freed | +|---|---|---|---|---|---|---|---|---|---|---|---| +| 4 GB | GTX 1650 (14 SMs x 32) | 448 | 14 MiB | 56 MiB | 96 MiB | 2,048 MiB | 128 MiB | 2,272 MiB | 2,286 MiB | 2,328 MiB | to year 4 (the 4 GiB step) | +| 8 GB | RTX 3050 (20) | 960 | 30 | 120 | 96 | 2,048 | 128 | 2,272 | 2,302 | 2,392 | to year 12 (the 8 GiB step) | +| 12 GB | RTX 3060 (28) | 1,344 | 42 | 168 | 96 | 2,048 | 128 | 2,272 | 2,314 | 2,440 | to year 28 (the 16 GiB step; year 12 if the cache were resident) | +| 16 GB | RTX 5060 Ti (36) | 1,728 | 54 | 216 | 96 | 2,048 | 128 | 2,272 | 2,326 | 2,488 | to year 28 | +| 24 GB | RTX 4090 (128) | 6,144 | 192 | 768 | 96 | 2,048 | 128 | 2,272 | 2,464 | 3,040 | to year 60 (the 32 GiB step) | +| 32 GB | RTX 5090 (170) | 8,160 | 255 | 1,020 | 96 | 2,048 | 128 | 2,272 | 2,527 | 3,292 | to year 60 | +| Apple 8 to 64 GB | M-series, 2,048 launched | 2,048 | 64 | 256 | 96 | 2,048 | 128 | 2,272 | 2,336 | 2,528 | 8 GB to year 4, 16 GB to year 12, 32 GB to year 28, 64 GB to year 60 (50% of unified memory usable, the review's assumption) | + +The prototype packs here use the 1 GiB dataset of spec 1.5 (1,024 MiB less in every total). Every total is under 6 GB with the hot table at its largest, so `S` is not what the cap binds: the dataset's growth is, and the hot table takes 96 MiB of the room at every tier (about 2 months of the 0.5 GiB-a-year schedule). The verifier's memory is `256 MiB + S MiB` (section 2.5); the GPU's is the table above. + +## 4. The chip model with the SRAM it would need + +Figures: SRAM area per bit from `docs/analysis/m16-recompute-attacker-2026-10-05.md` section 3 (256 MiB in about 100 to 300 mm^2 at a current node: the low end from a 0.02 um^2 bit cell with array overhead, the high end from wafer-scale parts at about 1 MB per mm^2; approximate, from memory) and from `docs/plans/counter-asic-2.md` (256 MB in about 45 mm^2 at a leading node, approximate). That is 0.18, 0.39 and 1.17 mm^2 per MiB. Die areas: a 750 mm^2 GPU-class die (the M16 model's equal-silicon comparison) and a 100 mm^2 memory-chip die (an assumption for a latency-bound chip whose die holds memory controllers and little else; labelled as such). + +| S (MiB) | SRAM at 0.18 mm^2/MiB | at 0.39 | at 1.17 | Share of a 100 mm^2 die (0.39) | Share of a 750 mm^2 die (0.39) | +|---|---|---|---|---|---| +| 32 | 6 mm^2 | 12 mm^2 | 37 mm^2 | 11% | 1.6% | +| 64 | 12 mm^2 | 25 mm^2 | 75 mm^2 | 20% | 3.2% | +| 96 | 17 mm^2 | 37 mm^2 | 112 mm^2 | 27% | 4.8% | + +The gain arithmetic. Let `G0` be a chip's gain over a GPU on the version 2 hash (the plan's public claim: under 2x; the plan's last section says the bound is DRAM latency, the same physics on both). Let `g` be the GPU's own measured speed-up from the hot class over version 2 on the same card (section 6: `hash rate hot(S, k) / hash rate v2`). The ideal `g` with every hot load a hit at zero cost is `16 / (16 - k)`: 1.14x at `k = 2`, 1.33x at `k = 4`, 2.0x at `k = 8`; the measured `g` says what share of that a real cache delivers while the 1 GiB dataset streams through the same cache. + +| Chip | Hot loads served from | Gain after the hot table | Arithmetic | +|---|---|---|---| +| A, no SRAM for H | DRAM | `G0 / g` | the chip's rate is what it was; the honest GPU gained `g` | +| B, S MiB of SRAM for H | SRAM | `G0 x A_die / (A_die + A_S)` per unit of silicon | the chip regains `g` and pays `A_S` on top of its die | +| B on a 100 mm^2 die, S = 64, 0.39 mm^2/MiB | SRAM | `0.80 x G0` | 100 / 125 | +| B on a 750 mm^2 die, S = 96, 0.39 mm^2/MiB | SRAM | `0.95 x G0` | 750 / 787 | + +What the big-table loads still cost the chip: `(16 - k) x 8` dependent DRAM reads per hash at the card's loaded latency (the 9070 XT entry's probe: 451 ns on the 5090 at 4,096 lanes, 276 ns unloaded on the 9070 XT, 1,949 ns on the M5 Max at 4,096 lanes; section 6 repeats the probe here). At `k = 4` that is 96 reads per hash; to match one RTX 5090 at its honest 229 Mhash/s (the M16 model's reference) a chip must keep `229 M x 96 x 300 ns = about 6,600` DRAM reads in flight at a 300 ns latency, and 11,000 at 500 ns, whatever its arithmetic. That queue depth is a memory-controller property, which is the latency-bound argument of the plan restated for the cold share. + +What the hot table does to the recompute attacker of M16: nothing good. `H` is a plain ChaCha12 chain, so a word of it can be recomputed from `KH` at `j + 1` block evaluations (32.5 on average, about 40,000 integer operations per hot load, approximate, against 1,170 per dataset item), which is 35x the cost of recomputing a dataset word. A chip stores `H` or reads it from DRAM; it does not recompute it. + +Reading, before the measurements: the hot table costs a chip `A_S` of die area or `g` of rate. Which one binds depends on the measured `g`, which is why the measurement comes first. If a GPU's cache delivers most of the ideal `g` with the dataset streaming beside it, `k = 8` at `S = 64` doubles the honest rate and halves chip A. If the cache delivers little, the hot table is a cost to the verifier (`S` MiB per epoch) with no gain, and the layer is dropped. + +## 5. Implementation (branch `ca2-cache`) + +| Where | What | +|---|---| +| `igneum-pow/src/generator.rs` | `LoadClass { hot: Option }` beside `mix`, `load_slots`, `scratch`, `scratch_kb`; `HotClass { mb, k }`; names `hot32k4`, `hot64k2`; `Op::Hot`; the first `k` drawn load slots after the scratch slots are hot; no width roll for a hot class with version 2 widths (`takes_width_roll`); the program id carries `hot/S/k` | +| `igneum-pow/src/memhard.rs` | `HotTable` (key, S, words), `hot_key(seed_bytes)`, `hot_words(mb)`, `hot_segments(mb)`, `hot_index(src, words)`, the tagged segment fill shared with the cache, `HOT_TAG` | +| `igneum-pow/src/verify.rs` | `DatasetSource::hot: Option`; `Op::Hot` in `step`; `Epoch::new_class` and `from_seed_bytes_class` fill `H` from the program's seed bytes when the class has a hot table | +| `igneum-pow/src/accept.rs` | `Op::Hot` with the closed-form stand-in of 2.4 | +| `igneum-pow/src/emit.rs` | the hot load statement in the three dialects; `hot` as the buffer after the init words (Metal buffer 3, or 4 when bound; CUDA and OpenCL argument after `mask`, or after the init words when bound) and before the scratch triple; `HOT_WORDS` literal; `ht_cache_segment` and `igneum_hot_fill` kernels in memhard.h, kernel.cu, kernel.cl and memhard.metal; `program.h` `IGNEUM_HOT_MB`, `IGNEUM_HOT_WORDS`, `IGNEUM_HOT_SEGMENTS`, `IGNEUM_HOT_SLOTS`, `IGNEUM_HOT_KEY_INIT`; `vectors.h` and `vectors.json` the hot head, last line and FNV-1a 64 | +| `igneum-pow/src/main.rs` | `--class hotk` on every command (`bench` reports the fill time and the verifier ms per warp) | +| `igneum-pow/tests/packs.rs` | the five hot packs pinned (program, vectors, every emitted file, the load-form count: `16 - k` masked loads and `k` hot loads) | +| `proto-cuda/packs-ca2-hot/` | replaced: `hot32k4`, `hot64k4`, `hot96k4`, `hot64k2`, `hot64k8`; added: `hot32k4a`, `hot64k4a`, `hot96k4a`; all from seed `igneum-genesis`, day `2026-10-03` | +| `proto-cuda/nvrtc/packfile.h` | `hotMb`, `hotWords`, `hotSegments`, `hotSlots`; the hot self-test values; `pf_selftest` checks them when the pack carries them | +| `proto-opencl/host.c` | `--bench-pack` and the self-test allocate and fill `H` from the pack's `igneum_hot_fill`, check its head, last line and FNV-1a 64, pass it as the argument after the init words | +| `proto-metal/packbench.swift` | the same on Metal from `memhard.metal` | +| `proto-cuda/nvrtc/worker.cpp` | `--check` and `--bench` (new: the base kernel timed over `--batches` dispatches, one RESULT line) allocate, fill and self-test `H`; the launch passes it | + +## 6. Measurements + +Every row names the machine, the harness and the command; the bench-log entry of 5 October 2026 ("the hot table on the M5 Max") carries the raw lines. The Mac rows were taken on 5 October 2026, 20:19 to 20:21 UTC, under the measure lock, with the Mac's load average at 14 to 27 from other agents' CPU work (the lock serialises builds and measurements, not every process), so they are ordered, repeatable to within a few percent against each other, and not the Mac's quiet numbers. The PC rows wait for the coordinator's go. + +### 6.1 Random-read probe at the hot sizes + +`igneum-bench-cl-igneum-genesis-mh --memprobe --probe-mib S` (`proto-opencl/host.c`; dependent random 4-byte loads, best of 3, 256 steps per lane, work-group 256; the ceiling is the 4,194,304-lane row; Apple OpenCL reports wall time). + +| Card | 32 MiB ceiling | 64 MiB | 96 MiB | 1024 MiB | Ratio 32 / 1024 | 64 / 1024 | 96 / 1024 | ns per dependent load at 4,096 lanes (32, 64, 96, 1024 MiB) | +|---|---|---|---|---|---|---|---|---| +| M5 Max (Apple OpenCL) | 21.7 G loads/s | 12.8 | 12.3 | 3.50 | 6.2 | 3.7 | 3.5 | 1,168; 1,129; 1,242; 1,844 | +| RTX 5090 (CUDA worker `--memprobe`, PC 1, job run-ca2-hot-5090-20261005, card off in the app) | 112.6 | 112.6 | 112.6 | 17.6 | 6.4 | 6.4 | 6.4 | 320; 340; 336; 610 | +| RX 9070 XT (OpenCL worker `--memprobe --device 1`, PC 1, job run-ca2-hot-9070-20261005, card off in the app) | 9.88 (10.8 at 262,144 lanes) | 9.47 | 8.18 | 2.43 | 4.1 | 3.9 | 3.4 | 396; 457; 454; 1,579 | + +Reading, RTX 5090: all three sizes sit inside the 96 MiB L2 at one ceiling (112.6 G loads/s, 6.4x the DRAM figure), so the probe alone promises a full hit rate for every S. RX 9070 XT: 32 and 64 MiB inside the Infinity Cache at 9.5 to 10.8 G loads/s (3.9 to 4.1x), 96 MiB at 8.2 (3.4x), as the 9070 XT entry's 64 MiB row said. Streams: 5090 1,563 GB/s at 1024 MiB, 9070 XT 633 GB/s (both at their rated figures, the cards were not parked). + +Reading, M5 Max: the step from 32 to 64 MiB halves the ceiling (21.7 to 12.8 G loads/s) and 96 MiB sits with 64, so 32 MiB is inside a cache level that 64 MiB is not (the M5 Max's system level cache size is not published; approximate reading: the 32 MiB table fits, the two larger ones mostly do not and run at a 3.5 to 3.7x advantage over DRAM from whatever hits they get). The coalesced stream grows with the buffer (139, 199, 243, 522 GB/s) because the small buffers are read once from cold. + +Predicted `g` from the probe alone, with a hot load costing `1 / ceiling_S` and a cold load `1 / ceiling_1024`: `g = 1 / ((16 - k) / 16 + (k / 16) x ceiling_1024 / ceiling_S)`. + +### 6.2 Bit-exactness and hash rate per hot pack + +Metal: `proto-metal/packbench --pack --batches 5 --batch-log2 24` (GPU time). Apple OpenCL: `igneum-bench-cl-igneum-genesis-mh --bench-pack --pack --batches 5 --batch-log2 24` (wall time). Both harnesses fill the hot table on the device from the pack's `igneum_hot_fill` and check its head, last line and FNV-1a 64 against vectors.json or vectors.h; vectors are the 96 lanes of the Rust reference; the fingerprint is FNV-1a 64 over the 2^24 outputs at base nonce 0. + +| Pack | Vectors (Metal, OpenCL) | Hot table FNV (Metal, OpenCL) | Fingerprint 2^24 (both harnesses equal) | Metal Mhash/s | Apple OpenCL Mhash/s | g against v2 (Metal) | Predicted g from the probe | Ideal g | +|---|---|---|---|---|---|---|---|---| +| v2 (igneum-genesis-mh) | 3/3, 96/96 | none | 25f96e7dce90bd4e | 27.68 | 27.61 | 1 | 1 | 1 | +| hot32k4 | 3/3, 96/96 | PASS, PASS | d2e6cf3b61d0b9fe | 33.90 | 33.92 | 1.22 | 1.27 | 1.33 | +| hot64k4 | 3/3, 96/96 | PASS, PASS | e4c5263ac650cc0d | 30.93 | 30.89 | 1.12 | 1.22 | 1.33 | +| hot96k4 | 3/3, 96/96 | PASS, PASS | 5d63439b6e394521 | 29.06 | 28.97 | 1.05 | 1.22 | 1.33 | +| hot64k2 | 3/3, 96/96 | PASS, PASS | 352633bdbbb0d2b6 | 27.67 | 27.27 | 1.00 | 1.10 | 1.14 | +| hot64k8 | 3/3, 96/96 | PASS, PASS | da54630d7dfaaf85 | 47.42 | 46.73 | 1.71 | 1.57 | 2.0 | +| hot32k4a (added) | 3/3, 96/96 | PASS, PASS | 8a3414735db4523c | 25.76 | 25.72 | 0.93 | 0.96 | 1 | +| hot64k4a (added) | 3/3, 96/96 | PASS, PASS | 45668f34105f6307 | 23.92 | 23.87 | 0.87 | 0.94 | 1 | +| hot96k4a (added) | 3/3, 96/96 | PASS, PASS | af763997dfee4c82 | 22.92 | 22.88 | 0.83 | 0.93 | 1 | + +For the added form the ideal `g` is 1 (the 16 dataset loads stay) and the probe predicts `g = 1 / (1 + (k / 16) x ceiling_1024 / ceiling_S)`: 0.96 at 32 MiB, 0.94 at 64 and 96 MiB for `k = 4` on the M5 Max; what matters is how far below 1 the GPU lands (its cost of the layer) against the chip's `S` MiB of SRAM and `k` reads. + +The PCs (5 October 2026, 21:29 to 21:35 UTC, PC 1 ae432dc7 on app 0.3.9 before and after, the card under test switched off in the app through `api/cards` and restored; CUDA worker `--bench --batches 5 --batch-log2 24 --block-warps 1` on the RTX 5090, wall time around the stream sync; OpenCL worker `--bench-pack --batches 5 --batch-log2 24 --device 1` on the RX 9070 XT, device event time, work-group 256; the v2 references are the readwidth entry's same-night, same-worker numbers: 136.1 and 18.15 MH/s). Every pack bit-exact with the Mac's fingerprint, hot table head, last line and FNV PASS, 96/96 lanes, on both cards. + +| Pack | RTX 5090 MH/s | g (v2 136.1) | probe-predicted g (5090) | RX 9070 XT MH/s | g (v2 18.15) | probe-predicted g (9070 XT) | ideal g | +|---|---|---|---|---|---|---|---| +| hot32k4 | 146.6 | 1.08 | 1.27 | 19.79 | 1.09 | 1.23 | 1.33 | +| hot64k4 | 140.8 | 1.03 | 1.27 | 18.73 | 1.03 | 1.23 | 1.33 | +| hot96k4 | 138.5 | 1.02 | 1.27 | 18.33 | 1.01 | 1.21 | 1.33 | +| hot64k2 | 137.5 | 1.01 | 1.13 | 18.17 | 1.00 | 1.11 | 1.14 | +| hot64k8 | 163.6 | 1.20 | 1.73 | 22.32 | 1.23 | 1.59 | 2.0 | +| hot32k4a (added) | 118.7 | 0.87 | 0.96 | 15.27 | 0.84 | 0.94 | 1 | +| hot64k4a (added) | 115.4 | 0.85 | 0.96 | 14.62 | 0.81 | 0.94 | 1 | +| hot96k4a (added) | 114.4 | 0.84 | 0.96 | 14.56 | 0.80 | 0.93 | 1 | + +The 5090 at `--block-warps 8` (6 blocks per SM, 8,160 resident warps) is within 0.7% of every row above; the 9070 XT at work-group 32 within 0.5%. Hot fill on the 5090: 1.2 to 2.5 ms (wall, driver API); on the 9070 XT 1.6 to 5.3 ms. + +Reading, the PCs. The probe promised a full hit rate for every S on the 5090 (all three tables inside the 96 MiB L2 at one ceiling) and 3.4 to 4.1x on the 9070 XT, and the hash delivered a fraction of it: 1.02 to 1.08x at `k = 4` against the probe's 1.27x and the ideal 1.33x on the 5090, 1.01 to 1.09x on the 9070 XT, 1.20 to 1.23x at `k = 8` against 1.73 and 2.0x. The table that stands alone in the probe does not stand up with the 1 GiB dataset streaming through the same cache: the dataset's random lines evict it (the 5090's L2 and the 9070 XT's Infinity Cache are shared by every load; neither card partitions them). The added form costs the 5090 13 to 16% and the 9070 XT 16 to 20% of its rate for four extra loads per iteration, three to five times the probe's 4 to 7%. The three cards agree on the shape; the 5090's larger cache buys it nothing over the Mac at 32 MiB (1.08 against 1.22). + +Hot table fill on the M5 Max GPU (Metal, GPU time): 0.07 ms at 32 MiB, 0.15 ms at 64 MiB, 0.22 ms at 96 MiB. + +Reading, M5 Max. Bit-exactness holds: Metal and Apple OpenCL give one fingerprint per pack and every vector lane against the Rust reference, hot table included. On the rate, the hot loads are worth 92% of the probe's prediction at 32 MiB (1.22 against 1.27), 50% at 64 MiB (1.12 against 1.22 is 0.12 of 0.22) and 23% at 96 MiB; at `k = 8` the measured 1.71 is above the prediction (1.57), which says the cold loads also go faster when half of them are gone (fewer dependent DRAM reads in the chain per hash). `hot64k2` gained nothing on this Mac at load (27.67 against 27.68). So on Apple silicon, with the 1 GiB dataset streaming through the same cache, the table that fits (32 MiB) delivers most of its ideal and the larger ones lose most of theirs. The 5090 and 9070 XT, whose caches are the plan's targets, are the PC job. + +### 6.3 CPU verifier, one core, avg of 50 warps + +`igneum-pow bench --seed igneum-genesis --day 2026-10-03 --class --warps 50` (release build, one M5 Max core, the Mac at load as above; the v2 row from the same session, the 0.604 ms of the brief was an earlier quiet run). + +| Class | Hot fill, one core | Items derived per warp | ms per warp | Against v2 | +|---|---|---|---|---| +| v2 | none | 4,096 | 0.626 | 1 | +| hot32k4 | 24.0 ms | 3,072 | 0.489 | 0.78 | +| hot64k4 | 46.4 ms | 3,072 | 0.504 | 0.81 | +| hot96k4 | 73.0 ms | 3,072 | 0.488 | 0.78 | +| hot64k2 | 45.5 ms | 3,584 | 0.560 | 0.89 | +| hot64k8 | 47.7 ms | 2,048 | 0.344 | 0.55 | +| hot32k4a (added) | 21.7 ms | 4,096 | 0.631 | 1.05 (v2 in the same session 0.602) | +| hot64k4a (added) | 43.3 ms | 4,096 | 0.609 | 1.01 | +| hot96k4a (added) | 64.9 ms | 4,096 | 0.614 | 1.02 | + +Reading: the verifier gets cheaper with `k`, because a hot load is one table read where a dataset load is an item derivation (9 mixers and 8 cache reads); the per-epoch cost is the fill, 24 to 73 ms on one core, against a 3,600 s epoch and the 20-minute seed lead. The 10 ms gate (spec 1.16) is unaffected. + +### 6.4 The chip model with the measured g (M5 Max; the PC rows will replace it) + +Chip A (no SRAM for `H`): gain after = `G0 / g`. Chip B (`H` in SRAM): `G0 x A_die / (A_die + A_S)` at 0.39 mm^2 per MiB, 100 mm^2 die. + +| Class | g (M5 Max) | Chip A, gain after as a share of G0 | Chip B, share of G0 at 100 mm^2 | Chip B at 750 mm^2 | +|---|---|---|---|---| +| hot32k4 | 1.22 | 0.82 | 0.89 | 0.98 | +| hot64k4 | 1.12 | 0.89 | 0.80 | 0.97 | +| hot96k4 | 1.05 | 0.95 | 0.73 | 0.95 | +| hot64k8 | 1.71 | 0.58 | 0.80 | 0.97 | + +The added form (second Mac session, 21:03 to 21:19 UTC, load average 7 to 14; v2 in that session 27.63 Metal, 27.59 OpenCL): the GPU keeps 0.93 / 0.87 / 0.83 of its rate at 32 / 64 / 96 MiB (`k = 4`), against the probe's 0.96 / 0.94 / 0.93, so the hot hits cost this card more than the probe says as the table grows, and 96 MiB is not resident. The named chip's position under the added form, same assumptions: + +| Chip | Hot loads served from | Rate against its own version 2 rate | Gain after, as a share of G0 (S = 32 / 64 / 96) | Arithmetic | +|---|---|---|---|---| +| A, no SRAM for H | DRAM | at most 16 / 20 = 0.80 (20 dependent DRAM loads per iteration in place of 16) | 0.86 / 0.92 / 0.96 | `0.80 / g` | +| B, S MiB of SRAM for H, 100 mm^2 die, 0.39 mm^2/MiB | SRAM (the k reads near free) | 1.00 | 0.96 / 0.92 / 0.88 | `(1 / g) x A_die / (A_die + A_S)` | +| B on a 750 mm^2 die | SRAM | 1.00 | 1.06 / 1.12 / 1.15 | the SRAM is 1.6 to 4.8% of the die, the GPU's loss is larger | + +With the PCs' `g` (added form, `k = 4`): RTX 5090 0.87 / 0.85 / 0.84, RX 9070 XT 0.84 / 0.81 / 0.80 at 32 / 64 / 96 MiB. + +| Chip, added form, S = 32 / 64 / 96 MiB | Gain after as a share of G0, 5090 g | 9070 XT g | +|---|---|---| +| A, H from DRAM (rate 0.80 of its own) | 0.92 / 0.94 / 0.95 | 0.95 / 0.99 / 1.00 | +| B, S MiB of SRAM, 100 mm^2 die, 0.39 mm^2/MiB | 1.03 / 0.94 / 0.87 | 1.06 / 0.99 / 0.91 | +| B, 750 mm^2 die | 1.13 / 1.14 / 1.17 | 1.17 / 1.20 / 1.22 | + +Reading: on the cards the plan named, the added form taxes no chip. A chip that serves H from its DRAM loses at most 8% of its gain on the 5090's figures and nothing on the 9070 XT's; a chip with the SRAM comes out ahead on every row except the smallest die at 64 and 96 MiB. The replaced form helps the recompute chip outright (section 2.3). The hot hits are not free on a GPU while the 1 GiB dataset streams through the same cache, and the layer's premise (section 1) needed them to be. + +**Recommendation (owner: the project lead, gate 1):** do not adopt layer 5 in either form on these measurements. What could change it: a GPU-side way to keep H resident (cache partitioning or persisting-access controls exist on NVIDIA, approximate, from memory, and are a driver setting, not a consensus rule), or a dataset access pattern that bypasses the cache; both are outside the hash and were not measured. The experiment stays behind the flag with its packs and vectors as the record. + +Reading, replaced form: on the M5 Max figures the chip's cheaper way out of `hot64k8` is the SRAM (0.80 of `G0` at a 100 mm^2 die) rather than serving `H` from DRAM (0.58). The layer's value is then the SRAM area, which is small on a large die (0.97 at 750 mm^2). The strongest configuration on this card is the one whose table fits the cache and whose `k` is large; whether 64 MiB fits the 5090's L2 and the 9070 XT's Infinity Cache with the dataset streaming beside it is the PC measurement. + +## 7. Decision and what would make it pay (the Counter ASIC 3.0 note) + +Decided by the coordinator on 5 October 2026 after the PC rows: layer 5 is out of v3. The rule was a GPU cost under 3% (`g` above 0.97) for a chip cost worth having; measured `g` 0.84 to 0.87 on the RTX 5090 and 0.80 to 0.84 on the RX 9070 XT for the added form, and the replaced form helps the recompute chip. No re-run. + +What would make a hot table pay, for a later round: + +| Condition | What the measurements say | What would have to be shown | +|---|---|---| +| A table that stays resident beside a streaming 1 GiB | The probe's ceiling at every S inside the cache (5090: 112.6 G loads/s at 32, 64 and 96 MiB) and the hash's 1.02 to 1.08x say the dataset's random lines evict it; 32 MiB on the M5 Max kept 92% of its probe gain, the 5090 kept 29% | A size small enough to survive: the probe cannot say which (it has no competing stream); a sweep of `S` down from 32 MiB (16, 8, 4, 2) with the hash itself, on the 5090 first, would. A 2 to 4 MiB table costs a chip under 2 mm^2 of SRAM, so the layer would then tax nothing; the point of the layer was a table a chip cannot afford, and a table a GPU keeps is one a chip affords | +| The `k` and `S` the probe says | `g` moved with `k` (1.20x at `k = 8`, 1.02 to 1.08 at `k = 4`, 1.00 at `k = 2`) and barely with `S`; the probe predicted 1.27 to 1.73 | Only a resident table makes `k` worth raising; at `k = 8` with a resident table the replaced form reaches 2x (ideal) and the added form costs 0 by the probe. Without residency, `k` buys rate on the replaced form (which helps the chip) and costs rate on the added form | +| A different access shape | Both forms read H at one random word per hot load, the dataset's own pattern, and share the cache with 128 dataset reads per hash | A shape that touches the table in cache-line units and few lines per hash (one 64-byte line per iteration, say) would need a smaller resident set per hash and could be measured with the read-width emitters (`wide_load_stmt`) pointed at H. Or a hot table whose lines are read in a fixed order per epoch (a stream, not a random read), which a GPU prefetches and a chip must still hold or fetch; not designed here | +| Cache partitioning on the GPU | NVIDIA exposes persisting L2 access controls to CUDA programs (approximate, from memory); AMD and Apple do not expose an equivalent to OpenCL or Metal | A consensus rule cannot depend on a driver feature of one vendor; it could be a miner-side optimisation if the layer were in, which it is not | + +What the round leaves in place: the code behind the flag (`LoadClass::hot`, both forms, the hosts filling H on the device from the epoch seed, the eight packs and their vectors), the probe at the hot sizes on three cards, and the measured rule that a read-only table a GPU cache could hold is not one it does hold while the dataset streams. + +## 8. What is unverified + +Listed here until measured, and carried into the bench-log entry. + +1. Measured on all three cards (6.2): no card keeps `H` resident enough to deliver the probe's hit rate while the dataset streams. Not measured: whether a driver-side cache partition (NVIDIA's persisting L2 access controls, approximate, from memory) would; it is not something a consensus rule can rely on. +2. The PC rows are one run each (jobs run-ca2-hot-5090-20261005 and run-ca2-hot-9070-20261005, 5 October 2026); the two launch shapes per card agree within 1%, and no re-run was taken (PC 1 was handed on). +3. The resident warp counts of section 3 are approximate; the occupancy the hosts reach is printed by each harness (`kernel:` lines) and should replace them. +4. The SRAM area figures are approximate (section 4 cites their sources); no chip was priced. +5. `S = 96` uses the multiply-shift mapping; the spec's dataset still uses `AND MASK`. If the layer goes in, gate 1 decides whether the dataset mapping follows (section 1.13.3) or `S` stays a power of two. +6. The hot table on the chain needs the epoch seed about 1 ms (GPU) or up to 71 ms (CPU) before the epoch starts; the 20-minute lead of section 4.3 covers it. Not exercised on a node. +7. The Mac numbers were taken at load average 14 to 27 (other agents' CPU work); the ratios between packs of one session are the result, the absolute Mhash/s are not quiet numbers. diff --git a/igneum-pow/src/accept.rs b/igneum-pow/src/accept.rs index 7a5352ee2..198871a28 100644 --- a/igneum-pow/src/accept.rs +++ b/igneum-pow/src/accept.rs @@ -11,11 +11,13 @@ //! | (c) dynamic | the program is interpreted for [`ACCEPT_UNITS`] (64) units of 32 lanes at base nonces drawn from SplitMix64 seeded with `FNV-1a-64("igneum-accept/" \|\| seed words as little-endian bytes)`, each `low32(next()) AND NOT 31`, with init words equal to the seed words and the closed-form dataset `dataset_elem(idx, S[0], S[1])` at [`ACCEPT_DATASET_LOG2`] (2^28 words) in place of the memory-hard dataset. Over the 2,048 evaluations: no register has a bit equal in every final value; no load site (iteration, instruction) reads one address in all 32 lanes of any unit; fewer than [`MAX_SATURATED`] (164, 1 percent of 16,384) final register values are 0 or 2^32 - 1; every output bit's ones count is within [`BIAS_TOLERANCE`] (136, 6 sigma) of 1,024; the distinct masked addresses read by one lane in one evaluation, summed over the 2,048 evaluations, exceed [`MIN_DISTINCT_SUM`] (245,760, a mean above 120 of the 128 loads) | //! //! The dynamic test uses the closed form so that it is a pure function of the program (no cache, no day) and -//! costs about a millisecond on one core. The census (section 7.3) checked on 100,000 programs that the +//! costs about a millisecond on one core. A hot-table load (`docs/plans/hot-table.md`) reads the closed form keyed by +//! seed words 2 and 3 at its multiply-shift index, a second pure table beside the dataset stand-in (words 0 and 1). The census (section 7.3) checked on 100,000 programs that the //! closed-form verdict agrees with the memory-hard one on all but 39 threshold-edge cases. use crate::generator::{Instr, Op, Program, INSTR_COUNT, ITERATIONS, LANES}; use crate::seed::{fnv1a64, SplitMix64}; +use crate::memhard::hot_index; use crate::verify::{dataset_elem, fold_words, load_index, splitmix32, ScratchModel}; /// Units (32-lane warps) the dynamic test interprets. @@ -178,6 +180,8 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut let seed = &p.seed; let mask: u32 = (1u32 << ACCEPT_DATASET_LOG2) - 1; let (d0, d1) = (seed[0], seed[1]); + let (h0, h1) = (seed[2], seed[3]); + let hot_words = p.hot_words(); let loads = p.loads_per_hash(); let mut r = [[0u32; LANES]; 8]; for lane in 0..LANES { @@ -305,6 +309,21 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut } nload += 1; } + Op::Hot => { + // Hot table: the stand-in is dataset_elem keyed by seed words 2 and 3; the address is tagged with + // bit 30 so a hot word and a dataset word at one index count as two addresses. + for lane in 0..LANES { + idx[lane] = hot_index(r[a][lane], hot_words); + } + if idx.iter().all(|&x| x == idx[0]) { + return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 }); + } + for lane in 0..LANES { + r[d][lane] ^= dataset_elem(idx[lane], h0, h1); + lane_addrs[lane * loads + nload] = 0x4000_0000 | idx[lane]; + } + nload += 1; + } Op::WLoad => { let b = (r[a][0] & mask) & !31; for lane in 0..LANES { @@ -461,6 +480,52 @@ mod tests { } } + /// Hot-table experiment: the hot classes pass the rule at about the version 2 rate, and the hot addresses are + /// uniform over the table (16 buckets of the index over 64 units x 32 lanes x 32 hot loads). + #[test] + fn hot_classes_pass_and_hot_loads_are_uniform() { + for name in ["hot32k4", "hot64k4", "hot96k4", "hot64k2", "hot64k8", "scr4k32+hot64k4", "hot32k4a", "hot64k4a", "hot96k4a"] { + let c = LoadClass::parse(name).unwrap(); + let p = generate_class("igneum-genesis", c); + assert!(check(&p).is_ok(), "{name}"); + let mut rejected = 0; + for i in 0..60u32 { + let s = format!("igneum-hot-accept/{i}"); + let q = candidate_class(&s, s.as_bytes(), 0, c); + if check(&q).is_err() { + rejected += 1; + } + } + assert!(rejected < 15, "{name}: {rejected} of 60 rejected"); + } + let p = generate_class("igneum-genesis", LoadClass::hot(96, 4)); + let words = p.hot_words(); + let loads = p.loads_per_hash(); + let mut acc = Acc { and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 }; + let mut la = vec![0u32; LANES * loads]; + let mut buckets = [0u64; 16]; + let mut hot_count = 0u64; + for (u, &b) in accept_base_nonces(&p.seed).iter().enumerate() { + run_unit(&p, u, b, &mut acc, &mut la).unwrap(); + for &a in &la { + if a & 0xC000_0000 == 0x4000_0000 { + let idx = a & 0x3FFF_FFFF; + assert!(idx < words); + buckets[(idx as u64 * 16 / words as u64) as usize] += 1; + hot_count += 1; + } + } + } + assert_eq!(hot_count, 64 * 32 * 32, "32 hot loads per hash over 2,048 hashes"); + let mean = hot_count as f64 / 16.0; + for (i, &b) in buckets.iter().enumerate() { + assert!((b as f64 - mean).abs() < 0.15 * mean, "bucket {i}: {b} against a mean of {mean}"); + } + // the dataset distinct count still holds for the dataset loads alone + let r = check(&p).unwrap(); + assert!(r.distinct_mean() > 120.0); + } + #[test] fn base_nonces_are_aligned_and_seed_dependent() { let a = accept_base_nonces(&[1, 2, 3, 4, 5, 6, 7, 8]); diff --git a/igneum-pow/src/emit.rs b/igneum-pow/src/emit.rs index ef9410eff..7cb1ceed1 100644 --- a/igneum-pow/src/emit.rs +++ b/igneum-pow/src/emit.rs @@ -11,8 +11,8 @@ use crate::generator::{EraParams, Instr, Op, Program, ProgramClass, GENERATOR_VERSION, INSTR_COUNT, ITERATIONS, LOAD_SLOTS}; use crate::memhard::{ - Layout, MixParams, Shape, CACHE_LINES_PER_SEGMENT, CACHE_SEGMENT_LOG2_LINES, CACHE_TAG, CHACHA_ROUNDS, CHACHA_SIGMA, - ITEM_ROUNDS, + hot_key, hot_segments, hot_words, Layout, MixParams, Shape, CACHE_LINES_PER_SEGMENT, CACHE_SEGMENT_LOG2_LINES, CACHE_TAG, + CHACHA_ROUNDS, CHACHA_SIGMA, HOT_TAG, ITEM_ROUNDS, }; use crate::seed::SplitMix64; use crate::verify::{window, DatasetMode, DatasetSource, Epoch, DEFAULT_DATASET_LOG2, FOLD_MUL, FOLD_ROT}; @@ -306,6 +306,87 @@ fn scratch_header_lines(p: &Program) -> String { s } +/// Hot-table experiment (`docs/plans/hot-table.md`): the `HOT_WORDS` literal of a hot pack's hash kernels (empty +/// for every other class, so the pinned packs do not change). +fn hot_define(p: &Program) -> String { + match p.class.hot { + Some(h) => format!("// Hot table ({} MiB, {} of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS {} +", h.mb, h.k, hex(hot_words(h.mb as u32))), + None => String::new(), + } +} + +/// The hot load as one statement per dialect: the high 32 bits of `src x HOT_WORDS` index the table. +fn hot_stmt(dialect: CoreDialect, d: &str, a: &str) -> String { + match dialect { + CoreDialect::Metal => format!("{d} = {d} ^ hot[mulhi({a}, HOT_WORDS)];"), + CoreDialect::Cuda => format!("{d} = {d} ^ hot[__umulhi({a}, HOT_WORDS)];"), + CoreDialect::OpenCl => format!("{d} = {d} ^ hot[mul_hi({a}, HOT_WORDS)];"), + } +} + +/// The hot table's fill core: `ht_segment(hot, seg)`, the cache chain of `mh_cache_segment` under the hot key and +/// the hot tag (`mh_chacha_block` and `MH_SEGMENT_LINES` must be in scope: the memory-hard core comes first). +fn emit_hot_core(p: &Program, dialect: CoreDialect) -> String { + let Some(h) = p.class.hot else { return String::new() }; + let (u, fn_, wptr) = match dialect { + CoreDialect::Metal => ("uint", "inline", "device uint*"), + CoreDialect::Cuda => ("uint32_t", "IGNEUM_HD", "uint32_t*"), + CoreDialect::OpenCl => ("uint", "static inline", "__global uint*"), + }; + let k = hot_key(&p.seed_bytes); + let mut s = String::with_capacity(1500); + s.push_str(&format!( + "// Hot table (docs/plans/hot-table.md): {} MiB = {} segments of {} chained ChaCha{} lines under the hot key KH = seed_words(\"igneum-hot/\" || epoch seed bytes), tag \"Igne\" \"umHT\". The cache chain with another key and tag.\n", + h.mb, + hot_segments(h.mb as u32), + CACHE_LINES_PER_SEGMENT, + CHACHA_ROUNDS + )); + s.push_str(&format!("{fn_} void ht_segment({wptr} hot, {u} seg) {{\n")); + s.push_str(&format!(" {u} prev[16]; {u} x[16]; {u} y[16];\n")); + s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) prev[i] = 0u;\n")); + s.push_str(&format!(" for ({u} j = 0u; j < MH_SEGMENT_LINES; ++j) {{\n")); + s.push_str(&format!( + " x[0] = {} ^ prev[0]; x[1] = {} ^ prev[1]; x[2] = {} ^ prev[2]; x[3] = {} ^ prev[3];\n", + hex(CHACHA_SIGMA[0]), + hex(CHACHA_SIGMA[1]), + hex(CHACHA_SIGMA[2]), + hex(CHACHA_SIGMA[3]) + )); + for i in 0..8 { + s.push_str(&format!(" x[{}] = {} ^ prev[{}];\n", 4 + i, hex(k[i]), 4 + i)); + } + s.push_str(&format!( + " x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = {} ^ prev[14]; x[15] = {} ^ prev[15];\n", + hex(HOT_TAG[0]), + hex(HOT_TAG[1]) + )); + s.push_str(" mh_chacha_block(x, y);\n"); + s.push_str(&format!(" {wptr} line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u);\n")); + s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) {{ line[i] = y[i]; prev[i] = y[i]; }}\n")); + s.push_str(" }\n"); + s.push_str("}\n"); + s +} + +/// The hot lines of program.h. +fn hot_header_lines(p: &Program) -> String { + let Some(h) = p.class.hot else { return String::new() }; + let mut s = String::new(); + s.push_str("// Hot table (hot-table experiment, 5 October 2026, docs/plans/hot-table.md): NOT the lottery hash. IGNEUM_HOT_SLOTS of the\n"); + s.push_str("// 16 load slots read H[mulhi(src, IGNEUM_HOT_WORDS)], H = IGNEUM_HOT_MB MiB of chained ChaCha12 lines (the cache chain) under\n"); + s.push_str("// KH = seed_words(\"igneum-hot/\" || epoch seed bytes) and the tag \"Igne\" \"umHT\"; filled by igneum_hot_fill once per epoch.\n"); + s.push_str(&format!("#define IGNEUM_HOT_MB {}\n", h.mb)); + s.push_str(&format!("#define IGNEUM_HOT_WORDS {}\n", hex(hot_words(h.mb as u32)))); + s.push_str(&format!("#define IGNEUM_HOT_SEGMENTS {}u\n", hot_segments(h.mb as u32))); + s.push_str(&format!("#define IGNEUM_HOT_SLOTS {} // hot loads per program ({} per hash), {} the dataset loads ({} of them)\n", h.k, p.hot_loads_per_hash(), if h.added { "added beside" } else { "replacing" }, p.class.dataset_slots())); + s.push_str(&format!("#define IGNEUM_HOT_ADDED {}\n", h.added as u8)); + s.push_str(&format!("#define IGNEUM_HOT_KEY_INIT {{ {} }}\n", join_hex(&hot_key(&p.seed_bytes)))); + s +} + pub fn hex(v: u32) -> String { format!("0x{v:08x}u") } @@ -538,6 +619,21 @@ fn build_store(layout: Layout, dialect: CoreDialect, ds: &str, t: &str) -> Strin } } +/// [`metal_memhard`] plus, for a hot pack, the hot table's `ht_segment` and `igneum_hot_fill` kernel (one thread per +/// segment, `IGNEUM_HOT_SEGMENTS` threads). Byte-identical to [`metal_memhard`] for every other class. +pub fn metal_memhard_for(p: &Program, mp: &MixParams) -> String { + let mut s = metal_memhard_layout(mp, p.class.layout()); + if p.has_hot() { + s.push('\n'); + s.push_str(&emit_hot_core(p, CoreDialect::Metal)); + s.push_str("// Hot table: one thread per segment (IGNEUM_HOT_SEGMENTS threads).\n"); + s.push_str("kernel void igneum_hot_fill(device uint* hot [[buffer(0)]], uint gid [[thread_position_in_grid]]) {\n"); + s.push_str(" ht_segment(hot, gid);\n"); + s.push_str("}\n"); + } + s +} + /// Metal library with the cache fill and dataset build kernels for one day key (`memhardMSL`, memhard.metal). pub fn metal_memhard(mp: &MixParams) -> String { metal_memhard_layout(mp, Layout::LINEAR) @@ -589,6 +685,7 @@ fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound: s.push_str("using namespace metal;\n"); s.push('\n'); s.push_str(&format!("#define MASK {}\n", hex(mask))); + s.push_str(&hot_define(p)); s.push_str(&format!("constant uint SEEDW[8] = {{ {} }};\n", join_hex(&p.seed))); s.push('\n'); s.push_str("inline uint splitmix32(uint x) {\n"); @@ -624,8 +721,11 @@ fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound: if bound { s.push_str(" constant uint* initw [[buffer(3)]],\n"); } + if p.has_hot() { + s.push_str(&format!(" device const uint* hot [[buffer({})]],\n", if bound { 4 } else { 3 })); + } if p.has_scratch() { - let b = if bound { 4 } else { 3 }; + let b = (if bound { 4 } else { 3 }) + p.has_hot() as usize; s.push_str(&format!(" device uint* scratch [[buffer({b})]],\n")); s.push_str(&format!(" constant uint& groups [[buffer({})]],\n", b + 1)); s.push_str(&format!(" constant uint& salt [[buffer({})]],\n", b + 2)); @@ -695,6 +795,7 @@ fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound: Op::Load => format!("{d} = {d} ^ {};", fetch(word_index(&a, false, ins))), Op::WLoad => format!("{d} = {d} ^ {};", fetch(word_index(&a, true, ins))), Op::Scratch => scratch_stmt(CoreDialect::Metal, &d, &a, p.class.scratch_slot_mask()), + Op::Hot => hot_stmt(CoreDialect::Metal, &d, &a), }; s.push_str(&format!(" {line} // {k}\n")); } @@ -763,6 +864,7 @@ fn cuda_instr_lines(p: &Program, dataset_log2: u32) -> String { Op::Load => format!("{d} = {d} ^ ds[{}];", load_index_expr(CoreDialect::Cuda, era.as_ref(), ins, &a, dataset_log2)), Op::WLoad => format!("{d} = {d} ^ ds[(__shfl_sync(0xffffffffu, {a}, 0) & wmask) + lane];"), Op::Scratch => scratch_stmt(CoreDialect::Cuda, &d, &a, p.class.scratch_slot_mask()), + Op::Hot => hot_stmt(CoreDialect::Cuda, &d, &a), }; s.push_str(&format!(" {line} // {k} {}\n", ins.op.name())); } @@ -791,6 +893,7 @@ pub fn cuda_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u3 s.push_str("#include \"memhard.h\"\n"); } s.push('\n'); + s.push_str(&hot_define(p)); s.push_str("__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {\n"); s.push_str(" x ^= x >> 16; x *= 0x7feb352du;\n"); s.push_str(" x ^= x >> 15; x *= 0x846ca68bu;\n"); @@ -831,14 +934,25 @@ pub fn cuda_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u3 s.push_str(&build_store(layout, CoreDialect::Cuda, "ds", "t")); s.push_str(" }\n"); s.push_str("}\n"); + if p.has_hot() { + s.push_str("// Hot table (ht_segment is in memhard.h): one thread per segment.\n"); + s.push_str("__global__ void igneum_hot_fill(uint32_t* hot, uint32_t nSegments) {\n"); + s.push_str(" uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;\n"); + s.push_str(" if (seg < nSegments) ht_segment(hot, seg);\n"); + s.push_str("}\n"); + } s.push('\n'); } + if memhard.is_none() && p.has_hot() { + panic!("a hot-table pack needs the memory-hard dataset (the hot fill shares its ChaCha core)"); + } s.push_str("// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every\n"); s.push_str("// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a\n"); s.push_str("// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.\n"); s.push_str(&scratch_prelude(p, CoreDialect::Cuda)); let scratch_args = if p.has_scratch() { ", uint32_t* scratch, uint32_t groups, uint32_t salt" } else { "" }; - s.push_str(&format!("__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask{scratch_args}) {{\n")); + let hot_args = if p.has_hot() { ", const uint32_t* hot" } else { "" }; + s.push_str(&format!("__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask{hot_args}{scratch_args}) {{\n")); if p.has_scratch() { s.push_str(&persistent_prologue(CoreDialect::Cuda, p.class.scratch_words_per_lane())); } else { @@ -890,26 +1004,37 @@ pub fn cuda_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u3 s.push_str(" return cudaGetLastError();\n"); s.push_str("}\n"); s.push('\n'); + if p.has_hot() { + s.push_str("cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments) {\n"); + s.push_str(" if (nSegments == 0u) return cudaErrorInvalidValue;\n"); + s.push_str(" uint32_t block = 256u;\n"); + s.push_str(" uint32_t grid = (nSegments + block - 1u) / block;\n"); + s.push_str(" igneum_hot_fill<<>>(hot, nSegments);\n"); + s.push_str(" return cudaGetLastError();\n"); + s.push_str("}\n"); + s.push('\n'); + } } + let (hot_decl, hot_pass) = if p.has_hot() { (" const uint32_t* hot,", " hot,") } else { ("", "") }; if p.has_scratch() { s.push_str("// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it).\n"); - s.push_str("cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,\n"); + s.push_str(&format!("cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,{hot_decl}\n")); s.push_str(" uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) {\n"); s.push_str(" if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue;\n"); s.push_str(" uint32_t block = 32u * blockWarps;\n"); s.push_str(" if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue;\n"); - s.push_str(" igneum_hash<<>>(ds, out, baseNonce, mask, scratch, nonces / 32u, salt);\n"); + s.push_str(&format!(" igneum_hash<<>>(ds, out, baseNonce, mask,{hot_pass} scratch, nonces / 32u, salt);\n")); s.push_str(" return cudaGetLastError();\n"); s.push_str("}\n"); } else { - s.push_str( - "cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,\n", - ); + s.push_str(&format!( + "cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,{hot_decl}\n", + )); s.push_str(" uint32_t nonces, uint32_t blockWarps) {\n"); s.push_str(" if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;\n"); s.push_str(" uint32_t block = 32u * blockWarps;\n"); s.push_str(" if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;\n"); - s.push_str(" igneum_hash<<>>(ds, out, baseNonce, mask);\n"); + s.push_str(&format!(" igneum_hash<<>>(ds, out, baseNonce, mask{});\n", if p.has_hot() { ", hot" } else { "" })); s.push_str(" return cudaGetLastError();\n"); s.push_str("}\n"); } @@ -952,6 +1077,7 @@ pub fn cuda_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_lo s.push('\n'); s.push_str("struct IgneumInitWords { uint32_t w[8]; };\n"); s.push('\n'); + s.push_str(&hot_define(p)); s.push_str("__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {\n"); s.push_str(" x ^= x >> 16; x *= 0x7feb352du;\n"); s.push_str(" x ^= x >> 15; x *= 0x846ca68bu;\n"); @@ -964,7 +1090,8 @@ pub fn cuda_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_lo let _ = memhard; // the bound kernel reads the stored dataset in both constructions s.push_str(&scratch_prelude(p, CoreDialect::Cuda)); let scratch_args = if p.has_scratch() { ", uint32_t* scratch, uint32_t groups, uint32_t salt" } else { "" }; - s.push_str(&format!("__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw{scratch_args}) {{\n")); + let hot_args = if p.has_hot() { ", const uint32_t* hot" } else { "" }; + s.push_str(&format!("__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw{hot_args}{scratch_args}) {{\n")); if p.has_scratch() { s.push_str(&persistent_prologue(CoreDialect::Cuda, p.class.scratch_words_per_lane())); } else { @@ -993,25 +1120,26 @@ pub fn cuda_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_lo } s.push_str("}\n"); s.push('\n'); + let (hot_decl, hot_pass) = if p.has_hot() { (" const uint32_t* hot,", ", hot") } else { ("", "") }; if p.has_scratch() { s.push_str("// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it).\n"); s.push_str("cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,\n"); - s.push_str(" IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) {\n"); + s.push_str(&format!(" IgneumInitWords iw,{hot_decl} uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) {{\n")); s.push_str(" if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue;\n"); s.push_str(" uint32_t block = 32u * blockWarps;\n"); s.push_str(" if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue;\n"); - s.push_str(" igneum_hash_bound<<>>(ds, out, baseNonce, mask, iw, scratch, nonces / 32u, salt);\n"); + s.push_str(&format!(" igneum_hash_bound<<>>(ds, out, baseNonce, mask, iw{hot_pass}, scratch, nonces / 32u, salt);\n")); s.push_str(" return cudaGetLastError();\n"); s.push_str("}\n"); } else { s.push_str( "cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,\n", ); - s.push_str(" IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps) {\n"); + s.push_str(&format!(" IgneumInitWords iw,{hot_decl} uint32_t nonces, uint32_t blockWarps) {{\n")); s.push_str(" if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;\n"); s.push_str(" uint32_t block = 32u * blockWarps;\n"); s.push_str(" if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;\n"); - s.push_str(" igneum_hash_bound<<>>(ds, out, baseNonce, mask, iw);\n"); + s.push_str(&format!(" igneum_hash_bound<<>>(ds, out, baseNonce, mask, iw{hot_pass});\n")); s.push_str(" return cudaGetLastError();\n"); s.push_str("}\n"); } @@ -1056,6 +1184,7 @@ fn opencl_instr_lines(p: &Program, dataset_log2: u32) -> String { Op::Load => format!("{d} = {d} ^ ds[{}];", load_index_expr(CoreDialect::OpenCl, era.as_ref(), ins, &a, dataset_log2)), Op::WLoad => format!("{{ uint t_; IGNEUM_BCAST0(t_, {a}); {d} = {d} ^ ds[(t_ & wmask) + lane]; }}"), Op::Scratch => scratch_stmt(CoreDialect::OpenCl, &d, &a, p.class.scratch_slot_mask()), + Op::Hot => hot_stmt(CoreDialect::OpenCl, &d, &a), }; s.push_str(&format!(" {line} // {k} {}\n", ins.op.name())); } @@ -1077,7 +1206,8 @@ pub fn opencl_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_ "// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash.\n", ); let scratch_args = if p.has_scratch() { ", __global uint* scratch, uint groups, uint salt" } else { "" }; - s.push_str(&format!("IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw{scratch_args}) {{\n")); + let hot_args = if p.has_hot() { ", __global const uint* hot" } else { "" }; + s.push_str(&format!("IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw{hot_args}{scratch_args}) {{\n")); let (setup, unit_loop) = persistent_prologue_parts(CoreDialect::OpenCl, p.class.scratch_words_per_lane()); if p.has_scratch() { s.push_str(&setup); @@ -1184,6 +1314,7 @@ pub fn opencl_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: s.push_str("#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }\n"); s.push_str("#endif\n"); s.push('\n'); + s.push_str(&hot_define(p)); s.push_str("static inline uint splitmix32(uint x) {\n"); s.push_str(" x ^= x >> 16; x *= 0x7feb352du;\n"); s.push_str(" x ^= x >> 15; x *= 0x846ca68bu;\n"); @@ -1215,7 +1346,17 @@ pub fn opencl_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: s.push_str(&build_store(layout, CoreDialect::OpenCl, "ds", "t")); s.push_str(" }\n"); s.push_str("}\n"); + if p.has_hot() { + s.push_str(&emit_hot_core(p, CoreDialect::OpenCl)); + s.push_str("// Hot table: one work-item per segment.\n"); + s.push_str("__kernel void igneum_hot_fill(__global uint* hot, uint nSegments) {\n"); + s.push_str(" uint seg = (uint)get_global_id(0);\n"); + s.push_str(" if (seg < nSegments) ht_segment(hot, seg);\n"); + s.push_str("}\n"); + } s.push('\n'); + } else if p.has_hot() { + panic!("a hot-table pack needs the memory-hard dataset (the hot fill shares its ChaCha core)"); } else { s.push_str("// dataset[i] = ds_elem(i, d0, d1) for i < n. Same closed form as the Metal igneum_fill kernel.\n"); s.push_str("__kernel void igneum_fill(__global uint* ds, uint n, uint d0, uint d1) {\n"); @@ -1229,7 +1370,8 @@ pub fn opencl_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: s.push_str("// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).\n"); s.push_str(&scratch_prelude(p, CoreDialect::OpenCl)); let scratch_args = if p.has_scratch() { ", __global uint* scratch, uint groups, uint salt" } else { "" }; - s.push_str(&format!("IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask{scratch_args}) {{\n")); + let hot_args = if p.has_hot() { ", __global const uint* hot" } else { "" }; + s.push_str(&format!("IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask{hot_args}{scratch_args}) {{\n")); let (setup, unit_loop) = persistent_prologue_parts(CoreDialect::OpenCl, p.class.scratch_words_per_lane()); if p.has_scratch() { s.push_str(&setup); @@ -1330,6 +1472,7 @@ pub fn program_header(p: &Program, day: &str, ds: &DatasetSource) -> String { s.push_str(&class_header_lines(p)); s.push_str(&scratch_header_lines(p)); s.push_str(&era_header_lines(p)); + s.push_str(&hot_header_lines(p)); s.push_str("// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)\n"); s.push_str(&format!("#define IGNEUM_DATASET_MODE {}\n", if memhard.is_some() { 1 } else { 0 })); s.push('\n'); @@ -1354,18 +1497,22 @@ pub fn program_header(p: &Program, day: &str, ds: &DatasetSource) -> String { s.push_str("// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().\n"); s.push_str("cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);\n"); s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);\n"); + if p.has_hot() { + s.push_str("cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments);\n"); + } } else { s.push_str("#ifndef IGNEUM_NO_CUDA\n"); s.push_str("// Defined in kernel.cu. Both launch on the default stream and return cudaGetLastError().\n"); s.push_str("cudaError_t igneum_launch_fill(uint32_t* ds, uint32_t nWords, uint32_t d0, uint32_t d1);\n"); } + let hot_decl = if p.has_hot() { " const uint32_t* hot," } else { "" }; if p.has_scratch() { - s.push_str("cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,\n"); + s.push_str(&format!("cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,{hot_decl}\n")); s.push_str(" uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt);\n"); } else { - s.push_str( - "cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,\n", - ); + s.push_str(&format!( + "cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,{hot_decl}\n", + )); s.push_str(" uint32_t nonces, uint32_t blockWarps);\n"); } s.push_str("cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);\n"); @@ -1392,6 +1539,10 @@ pub fn cuda_memhard_header(p: &Program, mp: &MixParams) -> String { s.push_str("#define IGNEUM_HD static inline\n"); s.push_str("#endif\n"); s.push_str(&emit_memhard_core_layout(mp, CoreDialect::Cuda, p.class.layout())); + if p.has_hot() { + s.push('\n'); + s.push_str(&emit_hot_core(p, CoreDialect::Cuda)); + } s } @@ -1413,6 +1564,11 @@ pub struct PackVectors { pub cache_fnv: u64, /// The cache is 2^cache_log2_words words (memory-hard only; 26 under version 2) pub cache_log2_words: u32, + /// Hot table (hot packs only): head line, last line, FNV-1a 64 over the whole table + pub has_hot: bool, + pub hot_head: Vec, + pub hot_last: Vec, + pub hot_fnv: u64, } /// The base nonces of the three vector warps every pack carries. @@ -1487,6 +1643,18 @@ pub fn vectors_header( s.push_str("};\n"); s.push_str(&format!("static const uint64_t IGNEUM_CACHE_FNV64 = {};\n", hex64(v.cache_fnv))); } + if v.has_hot { + s.push_str("// Hot table self-test (docs/plans/hot-table.md): hot[0..15], the last 16 words, and FNV-1a 64 over all IGNEUM_HOT_WORDS words.\n"); + s.push_str("static const uint32_t IGNEUM_HOT_HEAD[16] = {\n"); + s.push_str(&format!(" {},\n", join_hex(&v.hot_head[..8]))); + s.push_str(&format!(" {}\n", join_hex(&v.hot_head[8..16]))); + s.push_str("};\n"); + s.push_str("static const uint32_t IGNEUM_HOT_LAST[16] = {\n"); + s.push_str(&format!(" {},\n", join_hex(&v.hot_last[..8]))); + s.push_str(&format!(" {}\n", join_hex(&v.hot_last[8..16]))); + s.push_str("};\n"); + s.push_str(&format!("static const uint64_t IGNEUM_HOT_FNV64 = {};\n", hex64(v.hot_fnv))); + } s } @@ -1557,6 +1725,21 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String { s.push_str(" \"program_id_suffix\": \"'era/' || allowed[3] || width_words || stride_mul_le32 || stride_rot_le32 || interleave[4]\"\n"); s.push_str(" },\n"); } + if let Some(h) = p.class.hot { + let hk = hot_key(&p.seed_bytes); + s.push_str(&format!( + " \"hot_table\": {{\"mb\": {}, \"words\": {}, \"segments\": {}, \"slots\": {}, \"form\": {}, \"dataset_slots\": {}, \"hot_loads_per_hash\": {}, \"key\": [{}], \"key_derivation\": \"seed_words_from_bytes('igneum-hot/' || seed_bytes), the same for every attempt of the epoch\", \"tag\": [{}], \"chain\": \"the cache chain of dataset.cache with the hot key and tag: in_j = prev ^ (sigma || key || seg || j || tag); line_j = block(in_j); prev_0 = 0\", \"load\": \"dst = dst ^ hot[mulhi(src, words)] (the high 32 bits of the 64-bit product; the index lies in [0, words) for any size)\", \"slots_rule\": \"the first k load slots in the Fisher-Yates draw order after the scratch slots (a uniform k-subset); no extra draw, so with version 2 widths and no scratch the program is the version 2 program with k loads redirected\", \"acceptance_stand_in\": \"dataset_elem(mulhi(src, words), seed_words[2], seed_words[3])\", \"program_id\": \"the read-width id with 'hot/' || mb || k appended\", \"spec\": \"docs/plans/hot-table.md\"}},\n", + h.mb, + hot_words(h.mb as u32), + hot_segments(h.mb as u32), + h.k, + jstr(if h.added { "added: k load slots added beside the class's, the dataset loads unchanged" } else { "replaced: k of the class's load slots read the table" }), + p.class.dataset_slots(), + p.hot_loads_per_hash(), + join_jhex(&hk), + join_jhex(&HOT_TAG) + )); + } } s.push_str(&format!( " \"op_mix\": {{{}}},\n", @@ -1582,7 +1765,11 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String { " \"shfl\": \"dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp\",\n", ); s.push_str(" \"load\": \"dst = dst ^ dataset[src & dataset.mask]\",\n"); - s.push_str(" \"wload\": \"base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)\"\n"); + s.push_str(" \"wload\": \"base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)\""); + if p.has_hot() { + s.push_str(",\n \"hot\": \"dst = dst ^ hot[mulhi(src, hot_table.words)] (hot-table experiment)\""); + } + s.push('\n'); s.push_str(" },\n"); s.push_str(" \"dataset\": {\n"); s.push_str(&format!(" \"log2_words\": {dataset_log2},\n")); @@ -1723,7 +1910,13 @@ pub fn vectors_json( if memhard { s.push_str(&format!(",\n \"cache_head\": [{}],\n", join_jhex(&v.cache_head))); s.push_str(&format!(" \"cache_last_line\": [{}],\n", join_jhex(&v.cache_last))); - s.push_str(&format!(" \"cache_fnv1a64\": {}\n", jhex64(v.cache_fnv))); + s.push_str(&format!(" \"cache_fnv1a64\": {}", jhex64(v.cache_fnv))); + if v.has_hot { + s.push_str(&format!(",\n \"hot_head\": [{}],\n", join_jhex(&v.hot_head))); + s.push_str(&format!(" \"hot_last_line\": [{}],\n", join_jhex(&v.hot_last))); + s.push_str(&format!(" \"hot_fnv1a64\": {}", jhex64(v.hot_fnv))); + } + s.push('\n'); } else { s.push('\n'); } @@ -1773,6 +1966,13 @@ pub fn export_pack(epoch: &Epoch, day: &str, source: &str) -> Pack { v.cache_fnv = m.cache.fnv1a64(); v.cache_log2_words = m.shape().cache_log2_words; } + if let Some(h) = &ds.hot { + let w = h.words(); + v.has_hot = true; + v.hot_head = w[..16].to_vec(); + v.hot_last = w[w.len() - 16..].to_vec(); + v.hot_fnv = h.fnv1a64(); + } let is_mh = memhard.is_some(); let mut files = vec![ ("program.json".to_string(), program_json(p, day, ds)), @@ -1789,7 +1989,7 @@ pub fn export_pack(epoch: &Epoch, day: &str, source: &str) -> Pack { ]; if let Some(mp) = memhard { files.push(("memhard.h".to_string(), cuda_memhard_header(p, mp))); - files.push(("memhard.metal".to_string(), metal_memhard_layout(mp, p.class.layout()))); + files.push(("memhard.metal".to_string(), metal_memhard_for(p, mp))); } Pack { files, bases, outs, vectors: v } } diff --git a/igneum-pow/src/generator.rs b/igneum-pow/src/generator.rs index 7d5bbac42..0c22b7841 100644 --- a/igneum-pow/src/generator.rs +++ b/igneum-pow/src/generator.rs @@ -31,6 +31,12 @@ //! an era-fixed mix. The default class [`LoadClass::V2`] is the generator above, draw for draw and byte for byte; //! every other class takes one extra draw per instruction (the width roll), so its program stream differs from //! version 2 and its program id carries the class. +//! +//! Hot-table experiment (5 October 2026, Counter ASIC 2.0 layer 5, `docs/plans/hot-table.md`; NOT the lottery hash, +//! behind [`LoadClass::hot`]): `k` of the load slots read a second table `H` of `S` MiB derived from the epoch seed +//! ([`crate::memhard::HotTable`]) at `H[mulhi(src, words)]` with the plain one-word fold. The hot slots are the +//! first `k` drawn load slots after the scratch slots (a uniform `k`-subset, no extra draw), so a class with the +//! version 2 widths and no scratch takes the version 2 stream exactly ([`LoadClass::takes_width_roll`]). use crate::accept::{check, Reject}; use crate::seed::{fnv1a64, program_rng, seed_words_from_bytes, SplitMix64}; @@ -69,6 +75,9 @@ pub enum Op { /// Scratch read-modify-write (read-width experiment, variant 5, 5 October 2026): a 16-byte slot of the lane's /// own 32 KiB of the warp's 1 MiB scratch, read, folded into dst, rewritten. Never emitted by version 2. Scratch, + /// Hot-table load (hot-table experiment, 5 October 2026): `dst = dst XOR H[mulhi(src, HOT_WORDS)]`, one word + /// of the epoch's `S` MiB table. Never emitted by version 2. + Hot, } impl Op { @@ -88,6 +97,7 @@ impl Op { Op::Load => "load", Op::WLoad => "wload", Op::Scratch => "scratch", + Op::Hot => "hot", } } @@ -106,6 +116,7 @@ impl Op { "load" => Op::Load, "wload" => Op::WLoad, "scratch" => Op::Scratch, + "hot" => Op::Hot, _ => return None, }) } @@ -113,13 +124,13 @@ impl Op { /// An injecting op: bijective in `dst` and bringing another register (or the dataset) in. The acceptance /// rule's part (b) requires one such write per register. pub fn injects(self) -> bool { - matches!(self, Op::Add | Op::Sub | Op::Xor | Op::Mad | Op::Shfl | Op::Load | Op::WLoad | Op::Scratch) + matches!(self, Op::Add | Op::Sub | Op::Xor | Op::Mad | Op::Shfl | Op::Load | Op::WLoad | Op::Scratch | Op::Hot) } /// A memory operation: the fresh-source rule, the acceptance tests and the load count treat the scratch - /// read-modify-write as a load (it is one of the program's 128 memory operations). + /// read-modify-write and the hot-table load as loads (each is one of the program's 128 memory operations). pub fn is_load(self) -> bool { - matches!(self, Op::Load | Op::WLoad | Op::Scratch) + matches!(self, Op::Load | Op::WLoad | Op::Scratch | Op::Hot) } } @@ -202,6 +213,9 @@ pub struct LoadClass { /// Era layout (`docs/plans/era-layout.md`): `Some` turns on the strided, windowed load address and the /// interleaved dataset mapping with the parameters drawn from the era seed. `None` for every other class. pub era: Option, + /// Hot table (`docs/plans/hot-table.md`, measured 5 October 2026 and not adopted): `Some(HotClass { mb, k, added })` + /// turns `k` load slots into reads of an `mb` MiB epoch table. `None` for every other class, class v3 included. + pub hot: Option, } /// The parameters one era draws from its seed `E_n` (`docs/plans/era-layout.md` section 1.1, the proposed text of @@ -316,6 +330,18 @@ pub fn era_draw(era_bytes: &[u8], allowed: &[u8]) -> EraParams { EraParams { words, allowed: al, width_words, stride_mul, stride_rot, pos } } +/// The hot table of a class: `mb` MiB (32, 64 or 96 in the experiment) and `k` hot slots. Two forms: `replaced` +/// (`added = false`): `k` of the 16 load slots read the table, 16 - k dataset loads; `added` (`added = true`, +/// coordinator's form of 5 October 2026 against the on-die-cache recompute chip): the program has 16 + k load slots, +/// the `k` hot ones drawn among them, so the 16 dataset loads and the 4,096-item verifier bound are unchanged and the +/// hot loads are extra work (a cache hit on a GPU, SRAM and a read on a chip). +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub struct HotClass { + pub mb: u8, + pub k: u8, + pub added: bool, +} + /// Scratch geometry (variant 5): 16-byte slots, lane-major, 32 lanes per warp; `scratch_kb` KiB per warp gives /// `scratch_kb x 2` slots per lane (32 KiB: 64 slots, 128 KiB: 256 slots). pub const SCRATCH_SLOT_BYTES: usize = 16; @@ -339,13 +365,13 @@ impl LoadClass { impl LoadClass { /// Generator version 2 as adopted on 4 October 2026: 16 loads of one word. The lottery hash. pub const V2: LoadClass = - LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 1, growth: false, era: None }; + LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 1, growth: false, era: None, hot: None }; /// The construction decided for program class v3 on 5 October 2026 (Counter ASIC 2.0, `docs/plans/mixer-x4.md`): /// version 2 loads (16 slots of one word, no scratch, no width roll, so the program stream is version 2's), the /// mixer applied 4 times per round, and the cache growth rule. Name "mx4". pub const MX4: LoadClass = - LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 4, growth: true, era: None }; + LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 4, growth: true, era: None, hot: None }; /// The era class over `base` (`docs/plans/era-layout.md`): the parameters drawn by [`era_draw`]; when `allowed` /// has more than one width the drawn width becomes the class mix (every load that width), otherwise the base @@ -396,6 +422,50 @@ impl LoadClass { LoadClass { scratch: Some(k), scratch_kb: kb, ..LoadClass::V2 } } + /// Hot table, replaced form: version 2 widths, 16 load slots of which `k` read an `mb` MiB epoch table (`hot64k4`). + pub fn hot(mb: u8, k: u8) -> LoadClass { + LoadClass::V2.with_hot(mb, k) + } + + /// Hot table, added form: version 2 widths, 16 + `k` load slots of which `k` read the table (`hot64k4a`), the 16 + /// dataset loads unchanged. + pub fn hot_added(mb: u8, k: u8) -> LoadClass { + LoadClass::V2.with_hot_added(mb, k) + } + + /// This class with a hot table in the replaced form (composes with a width mix or a scratch: the hot slots are + /// drawn after the scratch slots, and `scratch + hot` must fit the slot count). + pub fn with_hot(mut self, mb: u8, k: u8) -> LoadClass { + assert!(mb >= 1, "a hot table needs at least 1 MiB"); + assert!(self.scratch_slots() + k as usize <= self.load_slots as usize, "scratch and hot slots exceed the load slots"); + self.hot = Some(HotClass { mb, k, added: false }); + self + } + + /// This class with a hot table in the added form: `k` load slots are added to the class's and read the table. + pub fn with_hot_added(mut self, mb: u8, k: u8) -> LoadClass { + assert!(mb >= 1, "a hot table needs at least 1 MiB"); + assert!((self.load_slots as usize) + (k as usize) < INSTR_COUNT, "added hot slots exceed the instruction count"); + self.load_slots += k; + self.hot = Some(HotClass { mb, k, added: true }); + self + } + + /// Hot slots per program (0 without a hot table). + pub fn hot_slots(&self) -> usize { + self.hot.map(|h| h.k as usize).unwrap_or(0) + } + + /// Hot slots that were added to the slot count (0 for the replaced form and without a hot table). + pub fn hot_added_slots(&self) -> usize { + self.hot.map(|h| if h.added { h.k as usize } else { 0 }).unwrap_or(0) + } + + /// Load slots that read the dataset: the slot count less the scratch and hot slots. + pub fn dataset_slots(&self) -> usize { + self.load_slots as usize - self.scratch_slots() - self.hot_slots() + } + /// This class with the mixer multiplier `m` (1, 2, 4, 8 or 16) and the cache growth rule on or off. pub fn with_mixer(self, mixer_mult: u8, growth: bool) -> LoadClass { assert!(mixer_mult >= 1 && mixer_mult <= 16 && mixer_mult.is_power_of_two(), "mixer multiplier must be 1, 2, 4, 8 or 16"); @@ -411,7 +481,7 @@ impl LoadClass { /// roll, so its program stream is the version 2 stream draw for draw (the mixer and the cache are properties of /// the dataset, not of the program). pub fn v2_loads(&self) -> bool { - self.mix == [100, 0, 0] && self.load_slots as usize == LOAD_SLOTS && self.scratch.is_none() + self.mix == [100, 0, 0] && self.load_slots as usize == LOAD_SLOTS + self.hot_added_slots() && self.scratch.is_none() } /// Whether every instruction takes the tenth draw (the width roll): every class whose loads are not version 2's. @@ -448,6 +518,36 @@ impl LoadClass { /// The load part of a class name (no mixer suffix). fn parse_loads(s: &str) -> Option { + // "+hotk[a]" composes a hot table with any class; "hotk[a]" alone is the version 2 base; + // the "a" suffix is the added form (k slots added to the class's), without it the replaced form + if let Some((base, hot)) = s.split_once("+hot") { + let (added, hot) = match hot.strip_suffix('a') { Some(h) => (true, h), None => (false, hot) }; + let (mb, k) = hot.split_once('k')?; + let (mb, k): (u8, u8) = (mb.parse().ok()?, k.parse().ok()?); + let c = LoadClass::parse_loads(base)?; + if mb == 0 || k == 0 { + return None; + } + if added { + if c.load_slots as usize + k as usize >= INSTR_COUNT { + return None; + } + return Some(c.with_hot_added(mb, k)); + } + if c.scratch_slots() + k as usize > c.load_slots as usize { + return None; + } + return Some(c.with_hot(mb, k)); + } + if let Some(rest) = s.strip_prefix("hot") { + let (added, rest) = match rest.strip_suffix('a') { Some(h) => (true, h), None => (false, rest) }; + let (mb, k) = rest.split_once('k')?; + let (mb, k): (u8, u8) = (mb.parse().ok()?, k.parse().ok()?); + if mb == 0 || k == 0 || k as usize > LOAD_SLOTS { + return None; + } + return Some(if added { LoadClass::hot_added(mb, k) } else { LoadClass::hot(mb, k) }); + } if let Some(rest) = s.strip_prefix("scr") { let (k, kb) = rest.split_once('k')?; let k: u8 = k.parse().ok()?; @@ -492,12 +592,23 @@ impl LoadClass { /// "v2", "w4", "w16", "w64", "w64x4", "mix50-35-15", "mix25-50-25x8", "scr4k32"; "mx4" for the v3 construction; /// any other mixer setting appends "m" and, with the growth rule, "g" ("v2m2", "w16m4g"). /// An era class is the base name with "-era" appended ("w4-era401998a5", "mx4-era..."). + /// A hot class appends "hotk[a]" ("hot64k4", "scr4k32+hot64k4a"; measured and not adopted). pub fn name(&self) -> String { if let Some(e) = self.era { let base = LoadClass { era: None, ..*self }; let base_name = if base.is_v2() { "w4".to_string() } else { base.name() }; return format!("{base_name}-era{}", e.label()); } + let base = LoadClass { hot: None, load_slots: self.load_slots - self.hot_added_slots() as u8, ..*self }.base_name(); + match self.hot { + None => base, + Some(h) if base == "v2" => format!("hot{}k{}{}", h.mb, h.k, if h.added { "a" } else { "" }), + Some(h) => format!("{base}+hot{}k{}{}", h.mb, h.k, if h.added { "a" } else { "" }), + } + } + + /// The name without the hot table. + fn base_name(&self) -> String { if self.is_v2() { return "v2".to_string(); } @@ -578,7 +689,7 @@ pub enum ProgramClass { /// placeholder of the seam (w16) is replaced here; nothing else in the seam names the class. /// Composed on 5 October 2026 (branch ca2-era): the era layout of `docs/plans/era-layout.md` is drawn inside this class by /// [`generate_from_seed_bytes_program_class`] (`LoadClass::era(V3_CLASS, era, &V3_ALLOWED)`); here `era` is `None`. -pub const V3_CLASS: LoadClass = LoadClass { era: None, ..LoadClass::MX4 }; +pub const V3_CLASS: LoadClass = LoadClass { era: None, hot: None, ..LoadClass::MX4 }; /// The width set class v3's era draw chooses from: 4 bytes only (the read-width decision of 5 October 2026; the /// draw is consumed, so widening the set at genesis keeps the derivation). @@ -667,6 +778,17 @@ impl Program { pub fn has_scratch(&self) -> bool { self.class.scratch.is_some() } + /// Hot-table loads per hash (one 4-byte word each). + pub fn hot_loads_per_hash(&self) -> usize { + self.instrs.iter().filter(|i| i.op == Op::Hot).count() * ITERATIONS + } + pub fn has_hot(&self) -> bool { + self.class.hot.is_some() + } + /// The hot table's words (0 without one). + pub fn hot_words(&self) -> u32 { + self.class.hot.map(|h| crate::memhard::hot_words(h.mb as u32)).unwrap_or(0) + } /// Width histogram of the loads, in words: (1, 4, 16) counts. pub fn width_counts(&self) -> [usize; 3] { let mut c = [0usize; 3]; @@ -677,9 +799,11 @@ impl Program { } c } - /// Distinct dataset items a 32-lane warp touches per hash: 32 per plain load, 2 per wide load. + /// Distinct dataset items a 32-lane warp touches per hash: 32 per plain load, 2 per wide load; scratch and + /// hot loads touch none (so the added form keeps 4,096). pub fn items_per_warp(&self) -> usize { - (self.loads_per_hash() - self.wide_loads_per_hash()) * 32 + self.wide_loads_per_hash() * 2 + (self.loads_per_hash() - self.wide_loads_per_hash() - self.scratch_ops_per_hash() - self.hot_loads_per_hash()) * 32 + + self.wide_loads_per_hash() * 2 } /// Op histogram, count descending then name ascending. pub fn histogram(&self) -> Vec<(&'static str, usize)> { @@ -760,6 +884,14 @@ pub fn program_id_class(generator: u32, seed: &[u32; 8], attempt: u32, class: &L b.extend_from_slice(b"era/"); b.extend_from_slice(&e.id_bytes()); } + if let Some(h) = class.hot { + b.extend_from_slice(b"hot/"); + b.push(h.mb); + b.push(h.k); + if h.added { + b.extend_from_slice(b"added"); + } + } fnv1a64(&b) } @@ -841,6 +973,11 @@ pub fn candidate_from_words_class( for &slot in &p[..class.scratch_slots()] { is_scratch[slot as usize] = true; } + // Hot table: the next k drawn load slots after the scratch slots are hot loads (again a uniform subset). + let mut is_hot = [false; INSTR_COUNT]; + for &slot in &p[class.scratch_slots()..class.scratch_slots() + class.hot_slots()] { + is_hot[slot as usize] = true; + } // (2) The instructions. `fresh[r]`: r was written by an earlier instruction and no load has read it since. let mut fresh = [false; 8]; let mut instrs = Vec::with_capacity(INSTR_COUNT); @@ -855,7 +992,13 @@ pub fn candidate_from_words_class( roll -= w; } if is_load[k] { - op = if is_scratch[k] { Op::Scratch } else { Op::Load }; + op = if is_scratch[k] { + Op::Scratch + } else if is_hot[k] { + Op::Hot + } else { + Op::Load + }; } let dst = rng.below(8); let src = if op.is_load() { @@ -1434,6 +1577,107 @@ mod tests { } } + /// Hot-table experiment: names and ids; a hot class with version 2 widths and no scratch takes the version 2 + /// stream, so its program is the version 2 program with k of the load slots turned into hot loads; the hot + /// table composes with a scratch class. + #[test] + fn hot_classes() { + assert_eq!(LoadClass::parse("hot64k4"), Some(LoadClass::hot(64, 4))); + assert_eq!(LoadClass::hot(64, 4).name(), "hot64k4"); + assert_eq!(LoadClass::parse("hot96k4").unwrap().name(), "hot96k4"); + assert_eq!(LoadClass::parse("hot0k4"), None); + assert_eq!(LoadClass::parse("hot64k17"), None); + assert_eq!(LoadClass::parse("hot64"), None); + // the added form: 16 + k slots, 16 dataset loads, no width roll, its own name and id + for (mb, k) in [(32u8, 4u8), (64, 4), (96, 4)] { + let c = LoadClass::parse(&format!("hot{mb}k{k}a")).unwrap(); + assert_eq!(c, LoadClass::hot_added(mb, k)); + assert_eq!(c.name(), format!("hot{mb}k{k}a")); + assert_eq!(c.load_slots as usize, 16 + k as usize); + assert_eq!(c.dataset_slots(), 16); + assert!(!c.takes_width_roll()); + assert_ne!(c, LoadClass::hot(mb, k)); + let p = candidate_class("igneum-genesis", b"igneum-genesis", 0, c); + assert_eq!(p.loads_per_hash(), 128 + 8 * k as usize); + assert_eq!(p.hot_loads_per_hash(), 8 * k as usize); + assert_eq!(p.bytes_per_hash(), 512); + assert_eq!(p.items_per_warp(), 4096); + assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Load).count(), 16); + assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Hot).count(), k as usize); + assert!(p.instrs.iter().all(|i| i.width == 1)); + assert_ne!(p.program_id(), candidate_class("igneum-genesis", b"igneum-genesis", 0, LoadClass::hot(mb, k)).program_id()); + } + assert_eq!(LoadClass::parse("scr4k32+hot64k4a").unwrap().name(), "scr4k32+hot64k4a"); + assert_eq!(LoadClass::parse("scr4k32+hot64k4a").unwrap().dataset_slots(), 12); + assert_eq!(LoadClass::parse("hot64k48a"), None); + assert!(!LoadClass::hot(64, 4).is_v2()); + assert!(!LoadClass::hot(64, 4).takes_width_roll()); + assert!(LoadClass::V2.takes_width_roll() == false); + assert!(LoadClass::scratch(4, 32).takes_width_roll()); + assert!(LoadClass::fixed(4, 16).takes_width_roll()); + let v2 = candidate("igneum-genesis", b"igneum-genesis", 0); + let mut ids = std::collections::HashSet::new(); + ids.insert(v2.program_id()); + for (mb, k) in [(32u8, 4u8), (64, 4), (96, 4), (64, 2), (64, 8)] { + let c = LoadClass::hot(mb, k); + let p = candidate_class("igneum-genesis", b"igneum-genesis", 0, c); + assert_eq!(p.class, c); + assert_eq!(p.loads_per_hash(), 128); + assert_eq!(p.hot_loads_per_hash(), 8 * k as usize); + assert_eq!(p.bytes_per_hash(), (16 - k as usize) * 8 * 4); + assert_eq!(p.hot_words(), crate::memhard::hot_words(mb as u32)); + assert!(ids.insert(p.program_id()), "hot{mb}k{k}: program id collides"); + // the version 2 program with k loads redirected: every other field identical, instruction by instruction + let mut hot = 0; + for (a, b) in p.instrs.iter().zip(v2.instrs.iter()) { + if a.op == Op::Hot { + hot += 1; + assert_eq!(b.op, Op::Load, "a hot slot is one of the version 2 load slots"); + assert_eq!((a.dst, a.src, a.src2, a.imm, a.imm2, a.rot, a.bit, a.mask, a.width), (b.dst, b.src, b.src2, b.imm, b.imm2, b.rot, b.bit, b.mask, b.width)); + } else { + assert_eq!(a, b); + } + } + assert_eq!(hot, k as usize); + } + // the same k at two sizes: the same instructions, different ids and tables + let a = candidate_class("igneum-genesis", b"igneum-genesis", 0, LoadClass::hot(32, 4)); + let b = candidate_class("igneum-genesis", b"igneum-genesis", 0, LoadClass::hot(64, 4)); + assert_eq!(a.instrs, b.instrs); + assert_ne!(a.program_id(), b.program_id()); + // composition with a scratch class: scratch slots first, then hot, the rest dataset loads + let c = LoadClass::parse("scr4k32+hot64k4").unwrap(); + assert_eq!(c, LoadClass::scratch(4, 32).with_hot(64, 4)); + assert_eq!(c.name(), "scr4k32+hot64k4"); + assert!(c.takes_width_roll()); + let p = candidate_class("igneum-genesis", b"igneum-genesis", 0, c); + assert_eq!(p.scratch_ops_per_hash(), 32); + assert_eq!(p.hot_loads_per_hash(), 32); + assert_eq!(p.loads_per_hash(), 128); + assert_eq!(p.bytes_per_hash(), 8 * 8 * 4); + assert!(ids.insert(p.program_id())); + assert_eq!(LoadClass::parse("scr12k32+hot8k8"), None, "scratch and hot slots exceed the 16"); + assert_eq!(LoadClass::parse("w16+hot64k4").unwrap().name(), "w16+hot64k4"); + // the hot slots are a uniform subset of the load slots over a population + let mut position_sum = 0usize; + let mut n = 0usize; + for i in 0..200u32 { + let s = format!("igneum-hot-slots/{i}"); + let p = candidate_class(&s, s.as_bytes(), 0, LoadClass::hot(64, 4)); + let loads: Vec = p.instrs.iter().enumerate().filter(|(_, x)| x.op.is_load()).map(|(i, _)| i).collect(); + assert_eq!(loads.len(), 16); + for (rank, &i) in loads.iter().enumerate() { + if p.instrs[i].op == Op::Hot { + position_sum += rank; + n += 1; + } + } + } + assert_eq!(n, 800); + let mean_rank = position_sum as f64 / n as f64; + assert!((mean_rank - 7.5).abs() < 0.6, "hot slots sit anywhere among the 16 loads: mean rank {mean_rank}"); + } + #[test] fn generate_returns_an_accepted_program() { let p = generate("igneum-genesis"); diff --git a/igneum-pow/src/main.rs b/igneum-pow/src/main.rs index 714ed3362..cf82ab766 100644 --- a/igneum-pow/src/main.rs +++ b/igneum-pow/src/main.rs @@ -252,6 +252,13 @@ fn bench(a: &Args, mode: DatasetMode) { ); drop(c); } + if let Some(h) = a.class.hot { + // the hot table of the epoch on its own first (one core), then the epoch (which fills it again) + let t0 = Instant::now(); + let t = igneum_pow::memhard::HotTable::for_seed_bytes(a.seed.as_bytes(), h.mb as u32); + let fill_ms = t0.elapsed().as_secs_f64() * 1e3; + println!("hot table: {} MiB filled in {fill_ms:.1} ms on one core ({} chains of 64 ChaCha12 blocks), FNV-1a 64 {:016x}", h.mb, igneum_pow::memhard::hot_segments(h.mb as u32), t.fnv1a64()); + } let t0 = Instant::now(); let (e, _) = epoch_of(a, mode); let build_ms = t0.elapsed().as_secs_f64() * 1e3; diff --git a/igneum-pow/src/memhard.rs b/igneum-pow/src/memhard.rs index dd3cdf925..5a0ecd8a5 100644 --- a/igneum-pow/src/memhard.rs +++ b/igneum-pow/src/memhard.rs @@ -29,6 +29,14 @@ pub const CHACHA_ROUNDS: usize = 12; pub const CHACHA_SIGMA: [u32; 4] = [0x61707865, 0x3320646e, 0x79622d32, 0x6b206574]; /// "Igne", "umMH". pub const CACHE_TAG: [u32; 2] = [0x49676e65, 0x756d4d48]; +/// "Igne", "umHT": the chain tag of the hot table (hot-table experiment, `docs/plans/hot-table.md`). +pub const HOT_TAG: [u32; 2] = [0x49676e65, 0x756d4854]; +/// Domain tag of the hot key: `KH = seed_words_from_bytes("igneum-hot/" || epoch seed bytes)`. +pub const HOT_KEY_TAG: &[u8] = b"igneum-hot/"; +/// Words per MiB of hot table. +pub const HOT_WORDS_PER_MIB: u32 = 1 << 18; +/// Segments (64 chained lines of 16 words, 4 KiB) per MiB of hot table. +pub const HOT_SEGMENTS_PER_MIB: u32 = 256; /// The shape of the item derivation and of the cache: the mixer multiplier and the cache size. #[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] @@ -306,6 +314,12 @@ impl Cache { /// One segment: 64 chained lines written at `cache[seg * 1024 ..]`. /// `in_j = prev XOR (sigma || K || seg || j || tag)`, `line_j = B(in_j)`, `prev_0 = 0`. pub fn fill_segment(words: &mut [u32], seg: usize, key: &[u32; 8]) { + Self::fill_segment_tagged(words, seg, key, &CACHE_TAG) + } + + /// [`Cache::fill_segment`] with an explicit chain tag: [`CACHE_TAG`] for the cache, [`HOT_TAG`] for the hot + /// table of `docs/plans/hot-table.md` (the same chain, another key and tag). + pub fn fill_segment_tagged(words: &mut [u32], seg: usize, key: &[u32; 8], tag: &[u32; 2]) { let base = (seg << CACHE_SEGMENT_LOG2_LINES) * 16; let seg_words = &mut words[base..base + CACHE_LINES_PER_SEGMENT * 16]; let mut prev = [0u32; 16]; @@ -315,8 +329,8 @@ impl Cache { x[4..12].copy_from_slice(key); x[12] = seg as u32; x[13] = j as u32; - x[14] = CACHE_TAG[0]; - x[15] = CACHE_TAG[1]; + x[14] = tag[0]; + x[15] = tag[1]; for i in 0..16 { x[i] ^= prev[i]; } @@ -376,6 +390,85 @@ impl Cache { } } +/// The hot key of an epoch: `seed_words_from_bytes("igneum-hot/" || seed_bytes)`, `seed_bytes` the program seed +/// bytes before any attempt suffix, so every attempt of one epoch shares one table. +pub fn hot_key(seed_bytes: &[u8]) -> [u32; 8] { + let mut b = Vec::with_capacity(HOT_KEY_TAG.len() + seed_bytes.len()); + b.extend_from_slice(HOT_KEY_TAG); + b.extend_from_slice(seed_bytes); + crate::seed::seed_words_from_bytes(&b) +} + +/// Words of a hot table of `mb` MiB. +pub fn hot_words(mb: u32) -> u32 { + mb * HOT_WORDS_PER_MIB +} + +/// Segments of a hot table of `mb` MiB. +pub fn hot_segments(mb: u32) -> u32 { + mb * HOT_SEGMENTS_PER_MIB +} + +/// The hot index of a source word: `mulhi(src, words)`, the high 32 bits of the 64-bit product, in `[0, words)` +/// for any table size (the multiply-shift range reduction of spec 01 section 1.13.3). +#[inline(always)] +pub fn hot_index(src: u32, words: u32) -> u32 { + ((src as u64 * words as u64) >> 32) as u32 +} + +/// The hot table `H` of one epoch (hot-table experiment): `mb` MiB of chained ChaCha12 lines under the hot key, +/// read by the hot load slots as `dst ^= H[hot_index(src, words)]`. The verifier holds it beside the cache. +pub struct HotTable { + pub key: [u32; 8], + pub mb: u32, + words: Vec, +} + +impl HotTable { + /// Fill `mb` MiB under `key` on the calling thread. + pub fn fill(key: [u32; 8], mb: u32) -> HotTable { + assert!(mb >= 1 && mb <= 4096, "hot table size in MiB out of range"); + let n = hot_words(mb) as usize; + let mut words = vec![0u32; n]; + for seg in 0..hot_segments(mb) as usize { + Cache::fill_segment_tagged(&mut words, seg, &key, &HOT_TAG); + } + HotTable { key, mb, words } + } + + /// The table of the epoch whose program seed bytes are `seed_bytes`. + pub fn for_seed_bytes(seed_bytes: &[u8], mb: u32) -> HotTable { + Self::fill(hot_key(seed_bytes), mb) + } + + #[inline(always)] + pub fn n_words(&self) -> u32 { + self.words.len() as u32 + } + + /// `H[i]`. + #[inline(always)] + pub fn at(&self, i: u32) -> u32 { + self.words[i as usize] + } + + /// `H[hot_index(src, words)]`: what a hot load reads for source word `src`. + #[inline(always)] + pub fn word(&self, src: u32) -> u32 { + self.words[hot_index(src, self.n_words()) as usize] + } + + #[inline(always)] + pub fn words(&self) -> &[u32] { + &self.words + } + + /// FNV-1a 64 over the table as little-endian bytes (what `vectors.h` carries as `IGNEUM_HOT_FNV64`). + pub fn fnv1a64(&self) -> u64 { + fnv1a64_words(&self.words) + } +} + /// Derive `ts.len()` items into `out`, all chains interleaved round by round so the cache-line misses of /// independent items overlap in the memory system (`deriveItems` in the Swift). Under multiplier `m` /// (`mp.shape.mixer_mult`) round `r` applies `M` with keys `round_key(r m + j)` for `j = 0 .. m - 1` before its @@ -577,6 +670,40 @@ mod tests { assert_eq!(y, z); } + /// Hot-table experiment: the genesis epoch's table (seed bytes "igneum-genesis") as the hot packs carry it + /// (`proto-cuda/packs-ca2-hot/hot32k4/vectors.json`: hot_head, hot_fnv1a64; the head is the same at every size, + /// a larger table is more segments). The index mapping stays inside the table for any size. + #[test] + fn hot_table_fill_vector_and_index() { + assert_ne!(HOT_TAG, CACHE_TAG); + let h = HotTable::for_seed_bytes(b"igneum-genesis", 32); + assert_eq!(h.n_words(), 1 << 23); + assert_eq!(hot_segments(32), 8192); + assert_eq!( + &h.words()[..16], + &[ + 0x8068cc73, 0x6036ebf9, 0xb604cd25, 0x8ffb840e, 0xc54074a2, 0x285c0695, 0x77512425, 0xc26a58a7, + 0x72c88757, 0xc10fca78, 0x513825dd, 0x30d6ccc8, 0x9a05e7cf, 0xb9533f50, 0x4bac3ba0, 0xa5c19528 + ] + ); + assert_eq!(h.fnv1a64(), 0xc1767ba3ef02719f, "hot32k4 pack, hot_fnv1a64"); + assert_eq!(h.key, hot_key(b"igneum-genesis")); + assert_ne!(h.key, day_key("2026-10-03")); + // a different seed, a different table; the same seed under the cache tag is not the hot table + assert_ne!(HotTable::for_seed_bytes(b"igneum-genesis\x01\x00\x00\x00", 1).words()[..16], h.words()[..16]); + let mut under_cache_tag = vec![0u32; 1024]; + Cache::fill_segment(&mut under_cache_tag, 0, &h.key); + assert_ne!(&under_cache_tag[..16], &h.words()[..16]); + for words in [hot_words(32), hot_words(64), hot_words(96)] { + assert_eq!(hot_index(0, words), 0); + assert!(hot_index(u32::MAX, words) < words); + assert_eq!(hot_index(u32::MAX, words), words - 1); + assert!(hot_index(0x8000_0000, words) == words / 2); + } + assert_eq!(hot_index(0x1234_5678, 1 << 24), 0x1234_5678 >> 8); + assert_eq!(h.word(0x8000_0000), h.at(1 << 22)); + } + #[test] fn first_cache_line_matches_pack() { // vectors.json cache_head for day 2026-10-03: segment 0, line 0, with prev = 0. diff --git a/igneum-pow/src/verify.rs b/igneum-pow/src/verify.rs index 940590c3f..ba7338b7c 100644 --- a/igneum-pow/src/verify.rs +++ b/igneum-pow/src/verify.rs @@ -2,7 +2,7 @@ //! calls. Dataset words come from the memory-hard cache (default) or from the closed form (old packs). use crate::generator::{generate, generate_class, EraParams, Instr, LoadClass, Op, Program, ProgramClass, ITERATIONS, LANES}; -use crate::memhard::{Layout, MemhardCpu, Shape}; +use crate::memhard::{hot_index, HotTable, Layout, MemhardCpu, Shape}; use crate::seed::day_key; /// The load address of an era program (`docs/plans/era-layout.md` section 1.3): `y = rotl(x * M, R)`, then the @@ -171,6 +171,10 @@ pub struct DatasetSource { /// in packs so any implementation can rebuild the key. Empty when the key was given directly. pub key_bytes: Vec, pub dataset: Dataset, + /// The hot table of the epoch (hot-table experiment, `docs/plans/hot-table.md`): `Some` when the program's + /// class has one; filled by [`Epoch::new_class`] and [`Epoch::from_seed_bytes_class`] from the program's seed + /// bytes. A hot load reads `hot[hot_index(src, words)]`. + pub hot: Option, } impl DatasetSource { @@ -199,7 +203,18 @@ impl DatasetSource { DatasetMode::ClosedForm => Dataset::ClosedForm { d0: key[0], d1: key[1] }, DatasetMode::MemoryHard => Dataset::MemoryHard(MemhardCpu::with_shape(key, shape)), }; - Self { log2_words, mask, key, key_bytes: Vec::new(), dataset } + Self { log2_words, mask, key, key_bytes: Vec::new(), dataset, hot: None } + } + + /// This source with the hot table of the epoch whose program seed bytes are `seed_bytes` (`mb` MiB). + pub fn with_hot(mut self, seed_bytes: &[u8], mb: u32) -> Self { + self.hot = Some(HotTable::for_seed_bytes(seed_bytes, mb)); + self + } + + /// The hot table of a program's class, filled from its seed bytes (none for a class without one). + pub fn attach_hot_for(&mut self, program: &Program) { + self.hot = program.class.hot.map(|h| HotTable::for_seed_bytes(&program.seed_bytes, h.mb as u32)); } /// The shape of the memory-hard construction ([`Shape::V2`] for the closed form, which has none). @@ -309,6 +324,10 @@ pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32, let mut val = [0u32; LANES]; let mut scratch = if program.has_scratch() { Some(ScratchModel::new(program.class.scratch_slots_per_lane())) } else { None }; let slot_mask = program.class.scratch_slot_mask(); + if program.has_hot() { + let h = ds.hot.as_ref().expect("a hot-table program needs the epoch's hot table on the dataset source"); + assert_eq!(h.n_words(), program.hot_words(), "the hot table's size is the class's"); + } for _ in 0..ITERATIONS { let sel = r[0]; for ins in &program.instrs { @@ -440,6 +459,14 @@ fn step( Op::Scratch => { // handled by the caller (interpret_warp_init), which owns the unit's scratch model } + Op::Hot => { + // Hot-table experiment: one word of the epoch table at the multiply-shift index, plain xor fold. + let h = ds.hot.as_ref().expect("a hot load needs the hot table"); + let n = h.n_words(); + for lane in 0..LANES { + r[d][lane] ^= h.at(hot_index(r[a][lane], n)); + } + } Op::WLoad => { // Lane 0's register, masked, aligned down to 32 words; lane l reads word base + l. let base = (r[a][0] & mask) & !31; @@ -491,7 +518,11 @@ impl Epoch { /// `memhard::cache_log2_words` for a class with the growth rule; the dataset size is the caller's). pub fn new_class_day(seed: &str, day: &str, mode: DatasetMode, dataset_log2: u32, class: LoadClass, days_since_genesis: u64) -> Self { let shape = Shape::for_class_day(&class, days_since_genesis); - Self { program: generate_class(seed, class), dataset: DatasetSource::new_shape(day, mode, dataset_log2, shape) } + let program = generate_class(seed, class); + let mut dataset = DatasetSource::new_shape(day, mode, dataset_log2, shape); + // hot-table experiment: a hot class fills its table from the seed bytes + dataset.attach_hot_for(&program); + Self { program, dataset } } /// `dataset[w]` as this epoch's program reads it: under the program's layout (era layout; linear for v2). @@ -531,6 +562,7 @@ impl Epoch { let dataset_log2 = if class.growth { crate::memhard::dataset_log2_words(genesis_dataset_log2, days_since_genesis) } else { genesis_dataset_log2 }; let mut dataset = DatasetSource::from_key_shape(key, DatasetMode::MemoryHard, dataset_log2, shape); dataset.key_bytes = day_bytes.to_vec(); + dataset.attach_hot_for(&program); Self { program, dataset } } @@ -751,6 +783,39 @@ mod tests { } } + /// Hot-table experiment: an epoch of a hot class carries the table, hashes deterministically and differs from + /// version 2; the reference interpreter agrees with a hand-stepped hot load; a hot program without its table is + /// refused. + #[test] + fn hot_epochs_hash() { + let e = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::hot(32, 4)); + let h = e.dataset.hot.as_ref().expect("the epoch fills the hot table"); + assert_eq!(h.n_words(), 1 << 23); + assert_eq!(h.key, crate::memhard::hot_key(b"igneum-genesis")); + let v2 = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::V2); + let a = e.hash_warp(0); + assert_eq!(a, e.hash_warp(0)); + assert_ne!(a, v2.hash_warp(0)); + assert_ne!(a[0], a[1]); + // the same program under a 64 MiB table reads other words + let e64 = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::hot(64, 4)); + assert_eq!(e64.program.instrs, e.program.instrs); + assert_ne!(e64.hash_warp(0), a); + // from seed bytes, the chain's shape, with a hot class + let genesis = crate::bind::unhex("edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07").unwrap(); + let ec = Epoch::from_seed_bytes_class(&genesis, &crate::bind::day_bytes(20_730), "devnet", LoadClass::hot(32, 2)); + assert_eq!(ec.dataset.hot.as_ref().unwrap().key, crate::memhard::hot_key(&genesis)); + assert_eq!(ec.hash_warp(0), ec.hash_warp(0)); + } + + #[test] + #[should_panic(expected = "needs the epoch's hot table")] + fn hot_program_without_a_table_is_refused() { + let p = generate_class("igneum-genesis", LoadClass::hot(32, 4)); + let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 20); + let _ = hash_warp(&p, 0, &ds); + } + #[test] fn wide_class_epochs_hash() { for name in ["w16", "w64x4", "50,35,15"] { diff --git a/igneum-pow/tests/packs.rs b/igneum-pow/tests/packs.rs index 54cf2692c..f03cf6436 100644 --- a/igneum-pow/tests/packs.rs +++ b/igneum-pow/tests/packs.rs @@ -12,10 +12,10 @@ use igneum_pow::accept; use igneum_pow::emit::{ - cuda_kernel, cuda_kernel_bound, cuda_memhard_header, export_pack, metal_memhard, metal_program, + cuda_kernel, cuda_kernel_bound, cuda_memhard_header, export_pack, metal_memhard, metal_memhard_for, metal_program, metal_program_bound, opencl_kernel, opencl_kernel_bound, program_header, program_json, LoadSource, }; -use igneum_pow::generator::{generate_from_seed_bytes, generate_from_seed_bytes_program_class, LoadClass, Op, ProgramClass, GENERATOR_VERSION, GENERATOR_VERSION_V3, LOAD_SLOTS, V3_CLASS}; +use igneum_pow::generator::{generate_from_seed_bytes, generate_from_seed_bytes_class, generate_from_seed_bytes_program_class, LoadClass, Op, ProgramClass, GENERATOR_VERSION, GENERATOR_VERSION_V3, LOAD_SLOTS, V3_CLASS}; use igneum_pow::memhard::{Shape, CACHE_WORDS}; use igneum_pow::verify::{DatasetMode, DatasetSource, Epoch}; use serde_json::Value; @@ -675,3 +675,202 @@ fn era_dataset_is_a_prefix_at_smaller_sizes() { } } } + +// --------------------------------------------------------------------------------------------------------- +// Hot-table experiment (5 October 2026, docs/plans/hot-table.md): the five packs under proto-cuda/packs-ca2-hot/ are +// pinned the same way (program, vectors, every emitted file byte for byte), plus the hot table's fingerprint and the +// one-form load check: exactly 16 - k masked dataset loads and exactly k hot loads in every hash kernel. +// --------------------------------------------------------------------------------------------------------- + +const HOT_PACKS: [&str; 8] = ["hot32k4", "hot64k4", "hot96k4", "hot64k2", "hot64k8", "hot32k4a", "hot64k4a", "hot96k4a"]; + +fn hot_packs_dir() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca2-hot") +} + +fn hread(pack: &str, file: &str) -> String { + let p = hot_packs_dir().join(pack).join(file); + std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display())) +} + +fn hjson(pack: &str, file: &str) -> Value { + serde_json::from_str(&hread(pack, file)).unwrap_or_else(|e| panic!("{pack}/{file}: {e}")) +} + +/// The epoch of a hot pack from program.json alone: the class from `load_class`, the program from the seed bytes, +/// the dataset from the day bytes, the hot table from the seed bytes (what `Epoch::from_seed_bytes_class` does). +fn hepoch(pack: &str) -> &'static Epoch { + static E: OnceLock> = OnceLock::new(); + let all = E.get_or_init(|| { + HOT_PACKS + .iter() + .map(|p| { + let j = hjson(p, "program.json"); + let seed = j["seed"].as_str().unwrap(); + let seed_bytes = unhex(&j["seed_bytes"]); + let day_bytes = unhex(&j["dataset"]["day_bytes"]); + assert_eq!(j["dataset_mode"].as_str().unwrap(), "memory-hard"); + let class = LoadClass::parse(j["load_class"].as_str().unwrap()).unwrap(); + assert_eq!(class.name(), *p, "the pack directory is the class name"); + let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32; + let program = generate_from_seed_bytes_class(seed, &seed_bytes, class); + let mut dataset = + DatasetSource::from_key(igneum_pow::seed::seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2); + dataset.key_bytes = day_bytes; + dataset.attach_hot_for(&program); + (p.to_string(), Epoch { program, dataset }) + }) + .collect() + }); + &all.iter().find(|(n, _)| n == pack).unwrap().1 +} + +fn hassert_same_text(pack: &str, file: &str, got: &str) { + let want = hread(pack, file); + if got != want { + let (gl, wl): (Vec<&str>, Vec<&str>) = (got.lines().collect(), want.lines().collect()); + for i in 0..gl.len().max(wl.len()) { + let g = gl.get(i).copied().unwrap_or(""); + let w = wl.get(i).copied().unwrap_or(""); + if g != w { + panic!("{pack}/{file} differs at line {}:\n pack: {w}\n rust: {g}", i + 1); + } + } + panic!("{pack}/{file} differs only in trailing bytes (len {} vs {})", got.len(), want.len()); + } +} + +#[test] +fn hot_packs_program_and_vectors() { + for pack in HOT_PACKS { + let e = hepoch(pack); + let p = &e.program; + let j = hjson(pack, "program.json"); + let h = p.class.hot.unwrap(); + assert_eq!(j["generator"].as_u64().unwrap() as u32, GENERATOR_VERSION); + assert_eq!(j["attempt"].as_u64().unwrap() as u32, p.attempt); + assert_eq!(hex64(&j["program_id"]), p.program_id(), "{pack}: program id"); + let dataset_slots = if h.added { 16 } else { 16 - h.k as usize }; + assert_eq!(j["loads_per_hash"].as_u64().unwrap() as usize, (dataset_slots + h.k as usize) * 8); + assert_eq!(j["hot_table"]["mb"].as_u64().unwrap(), h.mb as u64); + assert_eq!(j["hot_table"]["slots"].as_u64().unwrap(), h.k as u64); + assert_eq!(j["hot_table"]["dataset_slots"].as_u64().unwrap() as usize, dataset_slots); + assert_eq!(j["hot_table"]["words"].as_u64().unwrap() as u32, p.hot_words()); + assert_eq!(j["op_mix"]["hot"].as_u64().unwrap(), h.k as u64, "{pack}: k hot instructions"); + assert_eq!(j["op_mix"]["load"].as_u64().unwrap() as usize, dataset_slots); + assert_eq!(p.items_per_warp(), dataset_slots * 8 * 32); + assert!(accept::check(p).is_ok(), "{pack}: passes the acceptance rule"); + let v2 = &epoch("igneum-genesis-mh").program; + if !h.added { + // replaced form: the version 2 genesis program with k loads redirected (attempt 0 on both) + assert_eq!(p.attempt, v2.attempt); + for (a, b) in p.instrs.iter().zip(v2.instrs.iter()) { + if a.op == Op::Hot { + assert_eq!(b.op, Op::Load); + } else { + assert_eq!(a, b); + } + } + } else { + // added form: 16 + k load slots, so another slot draw and another program; 16 dataset loads stay + assert_ne!(p.instrs, v2.instrs); + assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Load).count(), 16); + assert!(p.instrs.iter().all(|i| i.width == 1)); + } + // the hot table: the pack's head, last line and fingerprint + let v = hjson(pack, "vectors.json"); + let t = e.dataset.hot.as_ref().unwrap(); + assert_eq!(t.n_words(), p.hot_words()); + let head: Vec = v["hot_head"].as_array().unwrap().iter().map(hex32).collect(); + assert_eq!(&t.words()[..16], &head[..]); + let last: Vec = v["hot_last_line"].as_array().unwrap().iter().map(hex32).collect(); + assert_eq!(&t.words()[t.words().len() - 16..], &last[..]); + assert_eq!(t.fnv1a64(), hex64(&v["hot_fnv1a64"]), "{pack}: hot_fnv1a64"); + assert_eq!(t.key, igneum_pow::memhard::hot_key(&p.seed_bytes)); + // the cache is the day's, unchanged by the class + assert_eq!(e.dataset.memhard().unwrap().cache.fnv1a64(), 0x48c4f5bf24166b2e); + // 96 vectors + let warps = v["warps"].as_array().unwrap(); + assert_eq!(warps.len(), 3); + for w in warps { + let base = w["base_nonce"].as_u64().unwrap() as u32; + let expected: Vec = w["expected"].as_array().unwrap().iter().map(hex64).collect(); + let got = e.hash_warp(base); + for lane in 0..32 { + assert_eq!(got[lane], expected[lane], "{pack}: base {base} lane {lane}"); + } + assert_eq!(e.hash(base + 5), expected[5]); + } + // the dataset words are the day's + let head: Vec = v["dataset_head"].as_array().unwrap().iter().map(hex32).collect(); + for (i, hd) in head.iter().enumerate() { + assert_eq!(e.dataset.word(i as u32), *hd); + } + } + // the same k at three sizes: identical programs, three fingerprints, three vector sets + let a = hepoch("hot32k4"); + let b = hepoch("hot64k4"); + let c = hepoch("hot96k4"); + assert_eq!(a.program.instrs, b.program.instrs); + assert_eq!(b.program.instrs, c.program.instrs); + assert_ne!(a.hash_warp(0), b.hash_warp(0)); + assert_ne!(b.hash_warp(0), c.hash_warp(0)); +} + +#[test] +fn hot_packs_emitted_sources_and_load_forms() { + for pack in HOT_PACKS { + let e = hepoch(pack); + let p = &e.program; + let k = p.class.hot.unwrap().k as usize; + let dataset_loads = p.class.dataset_slots(); + let day = hjson(pack, "program.json")["dataset"]["day"].as_str().unwrap().to_string(); + let mp = &e.dataset.memhard().unwrap().params; + hassert_same_text(pack, "kernel.cu", &cuda_kernel(p, Some(mp))); + hassert_same_text(pack, "kernel_bound.cu", &cuda_kernel_bound(p, Some(mp))); + hassert_same_text(pack, "program.metal", &metal_program(p, e.dataset.log2_words, LoadSource::Stored)); + hassert_same_text(pack, "program_bound.metal", &metal_program_bound(p, e.dataset.log2_words)); + hassert_same_text(pack, "kernel.cl", &opencl_kernel(p, Some(mp))); + hassert_same_text(pack, "kernel_bound.cl", &opencl_kernel_bound(p, Some(mp))); + hassert_same_text(pack, "program.h", &program_header(p, &day, &e.dataset)); + hassert_same_text(pack, "memhard.h", &cuda_memhard_header(p, mp)); + hassert_same_text(pack, "memhard.metal", &metal_memhard_for(p, mp)); + assert_ne!(metal_memhard_for(p, mp), metal_memhard(mp), "{pack}: the hot fill kernel is in memhard.metal"); + let got = program_json(p, &day, &e.dataset); + hassert_same_text(pack, "program.json", &got); + let _: Value = serde_json::from_str(&got).expect("program.json is valid JSON"); + let v = hjson(pack, "vectors.json"); + let out = export_pack(e, &day, v["source"].as_str().unwrap()); + let file = |name: &str| -> &str { &out.files.iter().find(|(n, _)| n == name).unwrap().1 }; + hassert_same_text(pack, "vectors.json", file("vectors.json")); + hassert_same_text(pack, "vectors.h", file("vectors.h")); + assert_eq!(out.files.len(), 12); + // One form per dialect, exactly 16 - k masked dataset loads and k hot loads in every hash kernel; the fill + // kernel is present once per source that builds the table. + for (file, load, masked, hot) in [ + ("kernel.cu", "ds[r", " & mask]", "hot[__umulhi(r"), + ("kernel_bound.cu", "ds[r", " & mask]", "hot[__umulhi(r"), + ("program.metal", "dataset[r", " & MASK]", "hot[mulhi(r"), + ("program_bound.metal", "dataset[r", " & MASK]", "hot[mulhi(r"), + ("kernel.cl", "ds[r", " & mask]", "hot[mul_hi(r"), + ] { + let text = hread(pack, file); + assert_eq!(text.matches(load).count(), dataset_loads, "{pack}/{file}: {dataset_loads} dataset loads"); + assert_eq!(text.matches(masked).count(), dataset_loads, "{pack}/{file}: masked loads"); + assert_eq!(text.matches(hot).count(), k, "{pack}/{file}: {k} hot loads"); + assert!(text.contains(&format!("#define HOT_WORDS 0x{:08x}u", p.hot_words())), "{pack}/{file}: HOT_WORDS literal"); + } + // kernel_bound.cl carries both kernels + let text = hread(pack, "kernel_bound.cl"); + assert_eq!(text.matches("hot[mul_hi(r").count(), 2 * k); + assert_eq!(text.matches("ds[r").count(), 2 * dataset_loads); + for file in ["kernel.cu", "kernel.cl", "kernel_bound.cl", "memhard.metal"] { + assert_eq!(hread(pack, file).matches("igneum_hot_fill(").count(), 1, "{pack}/{file}: one hot fill kernel"); + } + assert_eq!(hread(pack, "memhard.h").matches("void ht_segment(").count(), 1); + let ph = hread(pack, "program.h"); + assert!(ph.contains(&format!("#define IGNEUM_HOT_MB {}", p.class.hot.unwrap().mb))); + assert!(ph.contains(&format!("#define IGNEUM_HOT_SLOTS {k}"))); + assert!(ph.contains("igneum_launch_hot_fill(")); + } +} diff --git a/proto-cuda/nvrtc/packfile.h b/proto-cuda/nvrtc/packfile.h index 2a740998d..d4f2bc19c 100644 --- a/proto-cuda/nvrtc/packfile.h +++ b/proto-cuda/nvrtc/packfile.h @@ -34,6 +34,13 @@ typedef struct { // Counter ASIC 2.0 (5 October 2026): the mixer multiplier of the item derivation (IGNEUM_MIXER_MULT, 1 when absent: // version 2; 4 under class v3). The emitted memhard.h / kernel.cl carry it in their text; this is for the log lines. uint32_t mixerMult; + // hot-table experiment (5 October 2026, docs/plans/hot-table.md): hotMb 0 when the pack has no hot table; the + // table is filled on the device from the pack's igneum_hot_fill (never shipped), its self-test values from vectors.h + uint32_t hotMb, hotWords, hotSegments, hotSlots; + uint32_t hotKey[8]; + int haveHot; + uint32_t hotHead[16], hotLast[16]; + uint64_t hotFnv; // seeds.txt (or program.h): the seeds as the worker protocol carries them char epochHex[65]; char dayHex[PF_HEX_CAP]; @@ -300,6 +307,12 @@ static int pf_load(const char* dir, PfPack* pk, char* err, size_t cap) { pk->scratchWordsPerLane = 8192; pf_define_u32(prog, "IGNEUM_SCRATCH_WORDS_PER_LANE", &pk->scratchWordsPerLane); strcpy(pk->loadClass, "v2"); pf_define_str(prog, "IGNEUM_LOAD_CLASS", pk->loadClass, sizeof(pk->loadClass)); pk->mixerMult = 1; pf_define_u32(prog, "IGNEUM_MIXER_MULT", &pk->mixerMult); + pk->hotMb = 0; pf_define_u32(prog, "IGNEUM_HOT_MB", &pk->hotMb); + if (pk->hotMb) { + if (!pf_define_u32(prog, "IGNEUM_HOT_WORDS", &pk->hotWords) || !pf_define_u32(prog, "IGNEUM_HOT_SEGMENTS", &pk->hotSegments) || + !pf_define_u32(prog, "IGNEUM_HOT_SLOTS", &pk->hotSlots) || pf_define_words(prog, "IGNEUM_HOT_KEY_INIT", pk->hotKey, 8) != 8) { free(prog); return pf_fail(err, cap, "program.h has IGNEUM_HOT_MB but not IGNEUM_HOT_WORDS, IGNEUM_HOT_SEGMENTS, IGNEUM_HOT_SLOTS and IGNEUM_HOT_KEY_INIT"); } + if (pk->hotWords != pk->hotMb * 262144u || pk->hotSegments != pk->hotMb * 256u || pk->hotMb > 4096u) { free(prog); return pf_fail(err, cap, "program.h hot table sizes disagree (words must be MiB x 2^18, segments MiB x 256)"); } + } if (pf_define_words(prog, "IGNEUM_KEY_INIT", pk->keyw, 8) != 8) { free(prog); return pf_fail(err, cap, "program.h has no IGNEUM_KEY_INIT with 8 words"); } if (!pf_define_str(prog, "IGNEUM_SEED_STRING", pk->seedString, sizeof(pk->seedString))) strncpy(pk->seedString, "(no IGNEUM_SEED_STRING)", sizeof(pk->seedString) - 1); pf_define_str(prog, "IGNEUM_SEED_BYTES_HEX", ehex, sizeof(ehex)); @@ -364,6 +377,9 @@ static int pf_load(const char* dir, PfPack* pk, char* err, size_t cap) { pf_symbol_numbers(vec, "IGNEUM_CACHE_FNV64", &pk->cacheFnv, 1) == 1 && pk->vecWarps > 0) { uint32_t ns = 0; pk->haveVectors = 1; + if (pk->hotMb && pf_symbol_u32s(vec, "IGNEUM_HOT_HEAD", pk->hotHead, 16) == 16 && + pf_symbol_u32s(vec, "IGNEUM_HOT_LAST", pk->hotLast, 16) == 16 && + pf_symbol_numbers(vec, "IGNEUM_HOT_FNV64", &pk->hotFnv, 1) == 1) pk->haveHot = 1; if (pf_define_u32(vec, "IGNEUM_DS_SAMPLES", &ns) && ns > 0 && ns <= PF_MAX_SAMPLES) { int a = pf_symbol_u32s(vec, "IGNEUM_DS_SAMPLE_INDEX", pk->sampleIdx, (int)ns); int b = pf_symbol_u32s(vec, "IGNEUM_DS_SAMPLE_VALUE", pk->sampleVal, (int)ns); @@ -377,23 +393,32 @@ static int pf_load(const char* dir, PfPack* pk, char* err, size_t cap) { // The self-test verdict from values the host read back from the device. `vec` holds vecWarps x 32 outputs of the // bound kernel run with the pack's own seed words as init words (that is igneum_hash of kernel.cu). Writes one line. +// A hot-table pack (pk->hotMb) also hands the hot table's head, last line and FNV-1a 64 (NULL and 0 otherwise); a +// hot pack whose vectors.h carries no hot values is not checked on the table (the vectors cover it) and says so. static int pf_selftest(const PfPack* pk, const uint32_t* cacheHead, const uint32_t* cacheLast, uint64_t cacheFnv, const uint32_t* dsHead, uint32_t dsLast, const uint32_t* sampleVals, const uint64_t* vec, + const uint32_t* hotHead, const uint32_t* hotLast, uint64_t hotFnv, char* out, size_t cap) { int okCH = memcmp(cacheHead, pk->cacheHead, 64) == 0, okCL = memcmp(cacheLast, pk->cacheLast, 64) == 0; int okFnv = (cacheFnv == pk->cacheFnv); int okDH = memcmp(dsHead, pk->dsHead, 64) == 0, okDL = (dsLast == pk->dsLast); + int hotChecked = (pk->hotMb && pk->haveHot && hotHead && hotLast); + int okHot = !hotChecked || (memcmp(hotHead, pk->hotHead, 64) == 0 && memcmp(hotLast, pk->hotLast, 64) == 0 && hotFnv == pk->hotFnv); int badS = 0, badV = 0, i, w, l, firstBadWarp = -1, firstBadLane = -1; for (i = 0; i < pk->nSamples; ++i) if (sampleVals[i] != pk->sampleVal[i]) ++badS; for (w = 0; w < pk->vecWarps; ++w) for (l = 0; l < 32; ++l) if (vec[w * 32 + l] != pk->vecOut[w][l]) { if (firstBadWarp < 0) { firstBadWarp = w; firstBadLane = l; } ++badV; } - if (okCH && okCL && okFnv && okDH && okDL && badS == 0 && badV == 0) { - snprintf(out, cap, "self-test PASS (cache head, last line and FNV-1a 64 %016llx; dataset head, word [%u] and %d samples; %d of %d vector lanes)", - (unsigned long long)cacheFnv, pk->dsLastIndex, pk->nSamples, pk->vecWarps * 32, pk->vecWarps * 32); + if (okCH && okCL && okFnv && okDH && okDL && okHot && badS == 0 && badV == 0) { + if (pk->hotMb) + snprintf(out, cap, "self-test PASS (cache head, last line and FNV-1a 64 %016llx; dataset head, word [%u] and %d samples; hot table %u MiB %s; %d of %d vector lanes)", + (unsigned long long)cacheFnv, pk->dsLastIndex, pk->nSamples, pk->hotMb, hotChecked ? "head, last line and FNV-1a 64 ok" : "not in vectors.h (the vectors cover it)", pk->vecWarps * 32, pk->vecWarps * 32); + else + snprintf(out, cap, "self-test PASS (cache head, last line and FNV-1a 64 %016llx; dataset head, word [%u] and %d samples; %d of %d vector lanes)", + (unsigned long long)cacheFnv, pk->dsLastIndex, pk->nSamples, pk->vecWarps * 32, pk->vecWarps * 32); return 1; } - snprintf(out, cap, "self-test FAIL (cache head %s, cache last %s, cache FNV %016llx vs pack %016llx %s, dataset head %s, dataset last %s, samples %d bad of %d, vector lanes %d bad of %d%s)", + snprintf(out, cap, "self-test FAIL (cache head %s, cache last %s, cache FNV %016llx vs pack %016llx %s, dataset head %s, dataset last %s, samples %d bad of %d, hot table %s, vector lanes %d bad of %d%s)", okCH ? "ok" : "BAD", okCL ? "ok" : "BAD", (unsigned long long)cacheFnv, (unsigned long long)pk->cacheFnv, okFnv ? "ok" : "BAD", - okDH ? "ok" : "BAD", okDL ? "ok" : "BAD", badS, pk->nSamples, badV, pk->vecWarps * 32, + okDH ? "ok" : "BAD", okDL ? "ok" : "BAD", badS, pk->nSamples, hotChecked ? (okHot ? "ok" : "BAD") : "none", badV, pk->vecWarps * 32, firstBadWarp >= 0 ? " (first bad lane in the warp at base nonce" : ""); if (firstBadWarp >= 0) { size_t n = strlen(out); diff --git a/proto-cuda/nvrtc/worker.cpp b/proto-cuda/nvrtc/worker.cpp index 175b2bc0b..0439e4270 100644 --- a/proto-cuda/nvrtc/worker.cpp +++ b/proto-cuda/nvrtc/worker.cpp @@ -434,6 +434,12 @@ struct Pair { int warps = 0; // persistent warps launched (the arena holds this many) int residentWarps = 0; // the occupancy query's capacity: blocks/SM x warps/block x SMs size_t scratchBytes = 0; + // hot-table experiment (5 October 2026, docs/plans/hot-table.md): the epoch's hot table, filled on the device by the + // pack's igneum_hot_fill, the argument after the init words + uint32_t hotMb = 0, hotWords = 0, hotSegments = 0, hotSlots = 0; + CUfunction fHotFill = nullptr; + CUdeviceptr hot = 0; + double hotMs = 0; }; static bool pairIs(const Pair* p, const std::string& epochHex, const std::string& dayHex) { @@ -467,6 +473,7 @@ static void releasePair(Ctx& c, Pair* p) { if (p->ds) c.drv.memFree(p->ds); if (p->cache) c.drv.memFree(p->cache); if (p->scratch) c.drv.memFree(p->scratch); + if (p->hot) c.drv.memFree(p->hot); if (p->modBound) c.drv.moduleUnload(p->modBound); if (p->modKernel) c.drv.moduleUnload(p->modKernel); delete p; @@ -485,11 +492,13 @@ static bool launchHash(Ctx& c, Pair* p, CUdeviceptr out, uint32_t baseNonce, con while (warps > 1u && units % warps != 0u) warps >>= 1; if (block != 32u) { err = "a variant-5 pack runs one warp per block (--block-warps 1)"; return false; } uint32_t salt = c.salt; c.salt += units; - void* args[8] = { &p->ds, &out, &baseNonce, &mask, &a, &p->scratch, &units, &salt }; + // the hot table (when the pack has one) sits between the init words and the scratch triple + void* args[9] = { &p->ds, &out, &baseNonce, &mask, &a, &p->scratch, &units, &salt, nullptr }; + if (p->hot) { args[5] = &p->hot; args[6] = &p->scratch; args[7] = &units; args[8] = &salt; } DRV_CHECK(c, c.drv.launchKernel(p->fHashBound, warps, 1, 1, 32, 1, 1, 0, s, args, nullptr), "cuLaunchKernel igneum_hash_bound (persistent)"); return true; } - void* args[5] = { &p->ds, &out, &baseNonce, &mask, &a }; + void* args[6] = { &p->ds, &out, &baseNonce, &mask, &a, &p->hot }; DRV_CHECK(c, c.drv.launchKernel(p->fHashBound, nonces / block, 1, 1, block, 1, 1, 0, s, args, nullptr), "cuLaunchKernel igneum_hash_bound"); return true; } @@ -812,7 +821,9 @@ static Pair* buildPair(Ctx& c, const std::string& dir, CUstream s, std::string& p->datasetLog2 = pk.datasetLog2; p->words = 1u << pk.datasetLog2; p->cacheWords = 1u << pk.cacheLog2Words; p->cacheSegments = pk.cacheSegments; // Compile Compiled ck, cb; - if (!rtcCompile(c, kernelDev, "kernel.cu", programH, memhardH, { "igneum_cache_fill", "igneum_build" }, ck, err)) { releasePair(c, p); return nullptr; } + std::vector kernelNames = { "igneum_cache_fill", "igneum_build" }; + if (pk.hotMb) kernelNames.push_back("igneum_hot_fill"); + if (!rtcCompile(c, kernelDev, "kernel.cu", programH, memhardH, kernelNames, ck, err)) { releasePair(c, p); return nullptr; } if (!rtcCompile(c, boundDev, "kernel_bound.cu", programH, memhardH, { "igneum_hash_bound" }, cb, err)) { releasePair(c, p); return nullptr; } p->compileMs = ck.ms + cb.ms; // Load @@ -824,14 +835,16 @@ static Pair* buildPair(Ctx& c, const std::string& dir, CUstream s, std::string& if (c.drv.moduleGetFunction(&p->fCacheFill, p->modKernel, ck.lowered[0].c_str()) != CUDA_SUCCESS) { err = "igneum_cache_fill (" + ck.lowered[0] + ") not in the module"; releasePair(c, p); return nullptr; } if (c.drv.moduleGetFunction(&p->fBuild, p->modKernel, ck.lowered[1].c_str()) != CUDA_SUCCESS) { err = "igneum_build (" + ck.lowered[1] + ") not in the module"; releasePair(c, p); return nullptr; } if (c.drv.moduleGetFunction(&p->fHashBound, p->modBound, cb.lowered[0].c_str()) != CUDA_SUCCESS) { err = "igneum_hash_bound (" + cb.lowered[0] + ") not in the module"; releasePair(c, p); return nullptr; } + if (pk.hotMb && c.drv.moduleGetFunction(&p->fHotFill, p->modKernel, ck.lowered[2].c_str()) != CUDA_SUCCESS) { err = "igneum_hot_fill (" + ck.lowered[2] + ") not in the module"; releasePair(c, p); return nullptr; } c.drv.funcGetAttribute(&p->regs, CU_FUNC_ATTRIBUTE_NUM_REGS, p->fHashBound); c.drv.occupancy(&p->blocksPerSM, p->fHashBound, 32 * c.blockWarps, 0); } p->loadClass = pk.loadClass; p->loadsPerHash = pk.loadsPerHash; p->bytesPerHash = pk.bytesPerHash; p->scratchOps = pk.scratchOps; p->programClass = pk.programClass; p->eraHex = pk.eraHex; p->persistent = pk.persistent != 0; + p->hotMb = pk.hotMb; p->hotWords = pk.hotWords; p->hotSegments = pk.hotSegments; p->hotSlots = pk.hotSlots; p->residentWarps = p->blocksPerSM * c.blockWarps * c.sms; - size_t scratchBytes = 0; + size_t scratchBytes = 0, hotBytes = (size_t)pk.hotWords * 4u; if (p->persistent) { // Variant 5: one arena per launched warp. The launch is the resident capacity (the occupancy query), rounded // down to a power of two so it divides every batch, or --warps; the allocation cannot change the occupancy @@ -847,8 +860,8 @@ static Pair* buildPair(Ctx& c, const std::string& dir, CUstream s, std::string& size_t cacheBytes = (size_t)p->cacheWords * 4u, dsBytes = (size_t)p->words * 4u; { size_t freeB = 0, totalB = 0; - if (c.drv.memGetInfo(&freeB, &totalB) == CUDA_SUCCESS && freeB < cacheBytes + dsBytes + scratchBytes + (64u << 20)) { - err = fmt("%llu MiB free on the device, this pack needs %llu MiB (cache %llu + dataset %llu + scratch %llu)", (unsigned long long)(freeB >> 20), (unsigned long long)((cacheBytes + dsBytes + scratchBytes) >> 20), (unsigned long long)(cacheBytes >> 20), (unsigned long long)(dsBytes >> 20), (unsigned long long)(scratchBytes >> 20)); + if (c.drv.memGetInfo(&freeB, &totalB) == CUDA_SUCCESS && freeB < cacheBytes + dsBytes + scratchBytes + hotBytes + (64u << 20)) { + err = fmt("%llu MiB free on the device, this pack needs %llu MiB (cache %llu + dataset %llu + scratch %llu + hot %llu)", (unsigned long long)(freeB >> 20), (unsigned long long)((cacheBytes + dsBytes + scratchBytes + hotBytes) >> 20), (unsigned long long)(cacheBytes >> 20), (unsigned long long)(dsBytes >> 20), (unsigned long long)(scratchBytes >> 20), (unsigned long long)(hotBytes >> 20)); releasePair(c, p); return nullptr; } } @@ -883,6 +896,18 @@ static Pair* buildPair(Ctx& c, const std::string& dir, CUstream s, std::string& if (r != CUDA_SUCCESS) { err = "dataset build: " + c.err(r); releasePair(c, p); return nullptr; } } p->dsMs = wallMs() - t0; + // Hot table (hot-table experiment): filled from the epoch seed by the pack's own kernel, never shipped + if (pk.hotMb) { + t0 = wallMs(); + CUresult r = c.drv.memAlloc(&p->hot, hotBytes); + if (r != CUDA_SUCCESS) { err = "cuMemAlloc hot table: " + c.err(r); p->hot = 0; releasePair(c, p); return nullptr; } + uint32_t nSeg = pk.hotSegments, block = 256u, grid = (nSeg + block - 1u) / block; + void* args[2] = { &p->hot, &nSeg }; + r = c.drv.launchKernel(p->fHotFill, grid, 1, 1, block, 1, 1, 0, s, args, nullptr); + if (r == CUDA_SUCCESS) r = c.drv.streamSynchronize(s); + if (r != CUDA_SUCCESS) { err = "hot table fill: " + c.err(r); releasePair(c, p); return nullptr; } + p->hotMs = wallMs() - t0; + } // Self-test against vectors.h t0 = wallMs(); if (!pk.haveVectors) { @@ -911,8 +936,17 @@ static Pair* buildPair(Ctx& c, const std::string& dir, CUstream s, std::string& if ((r = c.drv.streamSynchronize(s)) != CUDA_SUCCESS || (r = c.drv.memcpyDtoH(&vec[(size_t)w * 32u], out, 32u * 8u)) != CUDA_SUCCESS) { err = "vector warp: " + c.err(r); c.drv.memFree(out); releasePair(c, p); return nullptr; } } c.drv.memFree(out); + uint32_t hotHead[16] = {0}, hotLast[16] = {0}; + uint64_t hotFnv = 0; + if (p->hot) { + std::vector hw(p->hotWords); + if ((r = c.drv.memcpyDtoH(hw.data(), p->hot, hotBytes)) != CUDA_SUCCESS) { err = "cuMemcpyDtoH hot table: " + c.err(r); releasePair(c, p); return nullptr; } + std::memcpy(hotHead, hw.data(), 64); + std::memcpy(hotLast, hw.data() + p->hotWords - 16u, 64); + hotFnv = pf_fnv1a64(hw.data(), hotBytes); + } char line[1024]; - p->checkPass = pf_selftest(&pk, cacheHead, cacheLast, fnv, dsHead, dsLast, samples.data(), vec.data(), line, sizeof(line)) != 0; + p->checkPass = pf_selftest(&pk, cacheHead, cacheLast, fnv, dsHead, dsLast, samples.data(), vec.data(), p->hot ? hotHead : nullptr, p->hot ? hotLast : nullptr, hotFnv, line, sizeof(line)) != 0; p->checked = true; p->check = line; } @@ -924,7 +958,7 @@ static Pair* buildPair(Ctx& c, const std::string& dir, CUstream s, std::string& } static std::string pairSummary(const Pair* p) { - return fmt("nvrtc %.0f cache %.0f dataset %.0f check %.0f race %.0f ms variant %s; %s", p->compileMs, p->cacheMs, p->dsMs, p->checkMs, p->raceMs, p->variant.c_str(), p->check.c_str()); + return fmt("nvrtc %.0f cache %.0f dataset %.0f hot %.0f check %.0f race %.0f ms variant %s; %s", p->compileMs, p->cacheMs, p->dsMs, p->hotMs, p->checkMs, p->raceMs, p->variant.c_str(), p->check.c_str()); } // --------------------------------------------------------------------------------------------- @@ -1276,8 +1310,8 @@ static int runBench(Ctx& c, const Options& o, Pair* p) { c.drv.memFree(dOut); std::string dev = c.name; for (char& ch : dev) if (ch == ' ') ch = '_'; std::printf("warm-up dispatch (base 0): %.2f ms; %d timed dispatches of %u nonces: mean %.2f ms\n", warm, o.batches, nonces, sum / o.batches); - std::printf("RESULT pack=%s class=%s device=%s arch=%s regs=%d blocks_per_sm=%d warps=%d resident=%d arena_mib=%llu nonces=%u batches=%d check=%s fingerprint=%016llx mhs=%.3f loads=%u bytes=%u scratch_ops=%u time=wall\n", - p->dir.c_str(), p->loadClass.c_str(), dev.c_str(), c.archOpt.c_str(), p->regs, p->blocksPerSM, p->warps, p->residentWarps, (unsigned long long)(p->scratchBytes >> 20), nonces, o.batches, + std::printf("RESULT pack=%s class=%s device=%s arch=%s regs=%d blocks_per_sm=%d warps=%d resident=%d arena_mib=%llu hot_mib=%u hot_slots=%u hot_fill_ms=%.2f nonces=%u batches=%d check=%s fingerprint=%016llx mhs=%.3f loads=%u bytes=%u scratch_ops=%u time=wall\n", + p->dir.c_str(), p->loadClass.c_str(), dev.c_str(), c.archOpt.c_str(), p->regs, p->blocksPerSM, p->warps, p->residentWarps, (unsigned long long)(p->scratchBytes >> 20), p->hotMb, p->hotSlots, p->hotMs, nonces, o.batches, p->checked ? (p->checkPass ? "PASS" : "FAIL") : "skipped", (unsigned long long)fp, (double)nonces * (double)o.batches / (sum / 1000.0) / 1e6, p->loadsPerHash, p->bytesPerHash, p->scratchOps * 8u); return 0; diff --git a/proto-cuda/packs-ca2-hot/hot32k4/kernel.cl b/proto-cuda/packs-ca2-hot/hot32k4/kernel.cl new file mode 100644 index 000000000..47a4fbfd2 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4/kernel.cl @@ -0,0 +1,305 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal). +// Built from source at runtime by proto-opencl/host.c, which passes these defines: +// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit) +// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default) +// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32 +// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition +// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units; +// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32. +#ifndef IGNEUM_GROUP +#define IGNEUM_GROUP 32 +#endif +#ifndef IGNEUM_EXCHANGE +#define IGNEUM_EXCHANGE 0 +#endif +#ifdef __OPENCL_VERSION__ +#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1))) +#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n] +#if IGNEUM_EXCHANGE == 1 +#ifdef cl_khr_subgroups +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif +#ifdef cl_khr_subgroup_shuffle +#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable +#endif +#elif IGNEUM_EXCHANGE == 2 +#pragma OPENCL EXTENSION cl_intel_subgroups : enable +#endif +#else +// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros. +#include "emu_opencl.h" +#endif + +#if IGNEUM_EXCHANGE == 1 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#elif IGNEUM_EXCHANGE == 2 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#else +// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per +// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane +// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's +// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier. +#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; } +#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; } +#endif + +// Hot table (32 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x00800000u +static inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32. +static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); } +// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x. +static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); } +static inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +static inline void mh_chacha_block(const uint* x, uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +static inline void mh_cache_segment(__global uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +static inline void mh_mixer(uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +static inline void mh_item(__global const uint* cache, uint t, uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + __global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item. +// The same constants as memhard.h in this pack (one emitter, three dialects). +__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) { + uint t = (uint)get_global_id(0); + if (t < nItems) { + uint s[16]; + mh_item(cache, t, s); + __global uint* d = ds + ((ulong)t * 16u); + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (docs/plans/hot-table.md): 32 MiB = 8192 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +static inline void ht_segment(__global uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one work-item per segment. +__kernel void igneum_hot_fill(__global uint* hot, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the +// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and +// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). +IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ ds[r2 & mask]; // 4 load + r5 = r5 ^ ds[r7 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl + r1 = mul_hi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = mul_hi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl + r5 = r5 * r0; // 19 mul + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl + r6 = mul_hi(r6, r2); // 22 mulhi + r6 = r6 ^ ds[r1 & mask]; // 23 load + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ hot[mul_hi(r0, HOT_WORDS)]; // 32 hot + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[mul_hi(r1, HOT_WORDS)]; // 34 hot + r0 = mul_hi(r0, r5); // 35 mulhi + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl + r7 = r7 ^ ds[r0 & mask]; // 37 load + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ ds[r7 & mask]; // 44 load + r2 = r2 * r3; // 45 mul + r1 = mul_hi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ hot[mul_hi(r2, HOT_WORDS)]; // 59 hot + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} + +#if IGNEUM_EXCHANGE != 0 +// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the +// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a +// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md. +IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) { + if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); } +} +#endif diff --git a/proto-cuda/packs-ca2-hot/hot32k4/kernel.cu b/proto-cuda/packs-ca2-hot/hot32k4/kernel.cu new file mode 100644 index 000000000..9d8898865 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4/kernel.cu @@ -0,0 +1,179 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal). +// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC. +#include +#include +#include "program.h" +#include "memhard.h" + +// Hot table (32 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x00800000u +__device__ __forceinline__ uint32_t splitmix32(uint32_t x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31. +__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } +// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x. +__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) { + uint32_t x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item. +// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host. +__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) { + uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x; + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) { + uint32_t t = blockIdx.x * blockDim.x + threadIdx.x; + if (t < nItems) { + uint32_t s[16]; + mh_item(cache, t, s); + uint32_t* d = ds + (size_t)t * 16u; + for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (ht_segment is in memhard.h): one thread per segment. +__global__ void igneum_hot_fill(uint32_t* hot, uint32_t nSegments) { + uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x; + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every +// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a +// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid. +__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot) { + uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x; + uint32_t nonce = baseNonce + gid; + uint32_t r0, r1, r2, r3, r4, r5, r6, r7; + { uint32_t x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint32_t x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint32_t x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint32_t x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint32_t x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint32_t x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint32_t x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint32_t x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint32_t it = 0u; it < 8u; ++it) { + uint32_t sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ ds[r2 & mask]; // 4 load + r5 = r5 ^ ds[r7 & mask]; // 5 load + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 6 shfl + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 8); // 7 shfl + r1 = __umulhi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = __umulhi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[__umulhi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 18 shfl + r5 = r5 * r0; // 19 mul + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 20 shfl + r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 16); // 21 shfl + r6 = __umulhi(r6, r2); // 22 mulhi + r6 = r6 ^ ds[r1 & mask]; // 23 load + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r6, 2); // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ hot[__umulhi(r0, HOT_WORDS)]; // 32 hot + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[__umulhi(r1, HOT_WORDS)]; // 34 hot + r0 = __umulhi(r0, r5); // 35 mulhi + r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // 36 shfl + r7 = r7 ^ ds[r0 & mask]; // 37 load + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 4); // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ ds[r7 & mask]; // 44 load + r2 = r2 * r3; // 45 mul + r1 = __umulhi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ hot[__umulhi(r2, HOT_WORDS)]; // 59 hot + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo; +} + +// Host-side launch wrappers. Declared in program.h, called from host.cu. +cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) { + if (nSegments == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nSegments + block - 1u) / block; + igneum_cache_fill<<>>(cache, nSegments); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) { + if (nItems == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nItems + block - 1u) / block; + igneum_build<<>>(ds, cache, nItems); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments) { + if (nSegments == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nSegments + block - 1u) / block; + igneum_hot_fill<<>>(hot, nSegments); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot, + uint32_t nonces, uint32_t blockWarps) { + if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue; + uint32_t block = 32u * blockWarps; + if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue; + igneum_hash<<>>(ds, out, baseNonce, mask, hot); + return cudaGetLastError(); +} + +cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) { + cudaFuncAttributes attr; + cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash); + if (e != cudaSuccess) return e; + *numRegs = attr.numRegs; + return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0); +} diff --git a/proto-cuda/packs-ca2-hot/hot32k4/kernel_bound.cl b/proto-cuda/packs-ca2-hot/hot32k4/kernel_bound.cl new file mode 100644 index 000000000..9bf292b5e --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4/kernel_bound.cl @@ -0,0 +1,399 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal). +// Built from source at runtime by proto-opencl/host.c, which passes these defines: +// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit) +// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default) +// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32 +// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition +// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units; +// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32. +#ifndef IGNEUM_GROUP +#define IGNEUM_GROUP 32 +#endif +#ifndef IGNEUM_EXCHANGE +#define IGNEUM_EXCHANGE 0 +#endif +#ifdef __OPENCL_VERSION__ +#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1))) +#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n] +#if IGNEUM_EXCHANGE == 1 +#ifdef cl_khr_subgroups +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif +#ifdef cl_khr_subgroup_shuffle +#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable +#endif +#elif IGNEUM_EXCHANGE == 2 +#pragma OPENCL EXTENSION cl_intel_subgroups : enable +#endif +#else +// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros. +#include "emu_opencl.h" +#endif + +#if IGNEUM_EXCHANGE == 1 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#elif IGNEUM_EXCHANGE == 2 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#else +// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per +// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane +// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's +// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier. +#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; } +#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; } +#endif + +// Hot table (32 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x00800000u +static inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32. +static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); } +// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x. +static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); } +static inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +static inline void mh_chacha_block(const uint* x, uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +static inline void mh_cache_segment(__global uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +static inline void mh_mixer(uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +static inline void mh_item(__global const uint* cache, uint t, uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + __global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item. +// The same constants as memhard.h in this pack (one emitter, three dialects). +__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) { + uint t = (uint)get_global_id(0); + if (t < nItems) { + uint s[16]; + mh_item(cache, t, s); + __global uint* d = ds + ((ulong)t * 16u); + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (docs/plans/hot-table.md): 32 MiB = 8192 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +static inline void ht_segment(__global uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one work-item per segment. +__kernel void igneum_hot_fill(__global uint* hot, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the +// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and +// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). +IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ ds[r2 & mask]; // 4 load + r5 = r5 ^ ds[r7 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl + r1 = mul_hi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = mul_hi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl + r5 = r5 * r0; // 19 mul + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl + r6 = mul_hi(r6, r2); // 22 mulhi + r6 = r6 ^ ds[r1 & mask]; // 23 load + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ hot[mul_hi(r0, HOT_WORDS)]; // 32 hot + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[mul_hi(r1, HOT_WORDS)]; // 34 hot + r0 = mul_hi(r0, r5); // 35 mulhi + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl + r7 = r7 ^ ds[r0 & mask]; // 37 load + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ ds[r7 & mask]; // 44 load + r2 = r2 * r3; // 45 mul + r1 = mul_hi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ hot[mul_hi(r2, HOT_WORDS)]; // 59 hot + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} + +#if IGNEUM_EXCHANGE != 0 +// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the +// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a +// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md. +IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) { + if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); } +} +#endif + +// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash. +IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7]; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; } + { uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; } + { uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; } + { uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; } + { uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; } + { uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; } + { uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; } + { uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ ds[r2 & mask]; // 4 load + r5 = r5 ^ ds[r7 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl + r1 = mul_hi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = mul_hi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl + r5 = r5 * r0; // 19 mul + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl + r6 = mul_hi(r6, r2); // 22 mulhi + r6 = r6 ^ ds[r1 & mask]; // 23 load + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ hot[mul_hi(r0, HOT_WORDS)]; // 32 hot + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[mul_hi(r1, HOT_WORDS)]; // 34 hot + r0 = mul_hi(r0, r5); // 35 mulhi + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl + r7 = r7 ^ ds[r0 & mask]; // 37 load + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ ds[r7 & mask]; // 44 load + r2 = r2 * r3; // 45 mul + r1 = mul_hi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ hot[mul_hi(r2, HOT_WORDS)]; // 59 hot + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot32k4/kernel_bound.cu b/proto-cuda/packs-ca2-hot/hot32k4/kernel_bound.cu new file mode 100644 index 000000000..a86e26c2c --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4/kernel_bound.cu @@ -0,0 +1,125 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW. +// Host declarations (also in program_bound.h if present): +// struct IgneumInitWords { uint32_t w[8]; }; +// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, +// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps); +// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps); +#include +#include +#include "program.h" + +struct IgneumInitWords { uint32_t w[8]; }; + +// Hot table (32 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x00800000u +__device__ __forceinline__ uint32_t splitmix32(uint32_t x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } +__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } + +__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw, const uint32_t* hot) { + uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x; + uint32_t nonce = baseNonce + gid; + uint32_t r0, r1, r2, r3, r4, r5, r6, r7; + { uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; } + { uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; } + { uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; } + { uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; } + { uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; } + { uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; } + { uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; } + { uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; } + + for (uint32_t it = 0u; it < 8u; ++it) { + uint32_t sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ ds[r2 & mask]; // 4 load + r5 = r5 ^ ds[r7 & mask]; // 5 load + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 6 shfl + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 8); // 7 shfl + r1 = __umulhi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = __umulhi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[__umulhi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 18 shfl + r5 = r5 * r0; // 19 mul + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 20 shfl + r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 16); // 21 shfl + r6 = __umulhi(r6, r2); // 22 mulhi + r6 = r6 ^ ds[r1 & mask]; // 23 load + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r6, 2); // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ hot[__umulhi(r0, HOT_WORDS)]; // 32 hot + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[__umulhi(r1, HOT_WORDS)]; // 34 hot + r0 = __umulhi(r0, r5); // 35 mulhi + r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // 36 shfl + r7 = r7 ^ ds[r0 & mask]; // 37 load + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 4); // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ ds[r7 & mask]; // 44 load + r2 = r2 * r3; // 45 mul + r1 = __umulhi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ hot[__umulhi(r2, HOT_WORDS)]; // 59 hot + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo; +} + +cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, + IgneumInitWords iw, const uint32_t* hot, uint32_t nonces, uint32_t blockWarps) { + if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue; + uint32_t block = 32u * blockWarps; + if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue; + igneum_hash_bound<<>>(ds, out, baseNonce, mask, iw, hot); + return cudaGetLastError(); +} + +cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) { + cudaFuncAttributes attr; + cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound); + if (e != cudaSuccess) return e; + *numRegs = attr.numRegs; + return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0); +} diff --git a/proto-cuda/packs-ca2-hot/hot32k4/memhard.h b/proto-cuda/packs-ca2-hot/hot32k4/memhard.h new file mode 100644 index 000000000..4e0eee822 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4/memhard.h @@ -0,0 +1,129 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against. +// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference). +// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C. +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif +#if defined(__CUDACC__) +#define IGNEUM_HD __host__ __device__ __forceinline__ +#elif defined(_MSC_VER) && !defined(__cplusplus) +#define IGNEUM_HD static __inline +#else +#define IGNEUM_HD static inline +#endif +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) { + for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint32_t r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) { + uint32_t prev[16]; uint32_t x[16]; uint32_t y[16]; + for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +IGNEUM_HD void mh_item(const uint32_t* cache, uint32_t t, uint32_t* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint32_t r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +IGNEUM_HD uint32_t mh_word(const uint32_t* cache, uint32_t w) { uint32_t s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Hot table (docs/plans/hot-table.md): 32 MiB = 8192 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +IGNEUM_HD void ht_segment(uint32_t* hot, uint32_t seg) { + uint32_t prev[16]; uint32_t x[16]; uint32_t y[16]; + for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + uint32_t* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} diff --git a/proto-cuda/packs-ca2-hot/hot32k4/memhard.metal b/proto-cuda/packs-ca2-hot/hot32k4/memhard.metal new file mode 100644 index 000000000..4325ec795 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4/memhard.metal @@ -0,0 +1,131 @@ +#include +using namespace metal; +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +inline void mh_chacha_block(const thread uint* x, thread uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +inline void mh_cache_segment(device uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +inline void mh_mixer(thread uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +inline void mh_item(device const uint* cache, uint t, thread uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +inline uint mh_word(device const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// One thread per segment (2^16 threads). +kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) { + mh_cache_segment(cache, gid); +} +// One thread per 64-byte item (dataset words / 16 threads). +kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]], + uint gid [[thread_position_in_grid]]) { + uint s[16]; + mh_item(cache, gid, s); + device uint* d = dataset + gid * 16u; + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; +} + +// Hot table (docs/plans/hot-table.md): 32 MiB = 8192 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +inline void ht_segment(device uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + device uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one thread per segment (IGNEUM_HOT_SEGMENTS threads). +kernel void igneum_hot_fill(device uint* hot [[buffer(0)]], uint gid [[thread_position_in_grid]]) { + ht_segment(hot, gid); +} diff --git a/proto-cuda/packs-ca2-hot/hot32k4/program.h b/proto-cuda/packs-ca2-hot/hot32k4/program.h new file mode 100644 index 000000000..f28f48724 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4/program.h @@ -0,0 +1,70 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Program metadata for host.cu plus the launch wrappers defined in kernel.cu. +// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros. +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif +#ifndef IGNEUM_NO_CUDA +#include +#endif + +#define IGNEUM_SEED_STRING "igneum-genesis" +#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973" +#define IGNEUM_GENERATOR 2 +#define IGNEUM_PROGRAM_ATTEMPT 0 +#define IGNEUM_PROGRAM_ID 0x3370a4466e5f322cull +#define IGNEUM_DAY_STRING "2026-10-03" +#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033" +#define IGNEUM_DAY0 0x3067619fu +#define IGNEUM_DAY1 0x3c269176u +#define IGNEUM_DATASET_LOG2 28 +#define IGNEUM_MASK 0x0fffffffu +#define IGNEUM_LANES 32 +#define IGNEUM_ITERATIONS 8 +#define IGNEUM_INSTR_COUNT 64 +#define IGNEUM_LOADS_PER_HASH 128 +#define IGNEUM_WIDE_LOADS_PER_HASH 0 +#define IGNEUM_OP_MIX "load=12 add=8 shfl=8 xor=6 mad=5 mul=5 mulhi=5 hot=4 sub=4 rotl=3 rotr=3 or=1" +// Class v3 construction (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): version 2 loads; the dataset item +// derivation applies the mixer IGNEUM_MIXER_MULT times per round (memhard.h), and the cache follows the growth rule. +#define IGNEUM_LOAD_CLASS "hot32k4" +#define IGNEUM_LOAD_SLOTS 16 +#define IGNEUM_LOAD_MIX { 100, 0, 0 } +#define IGNEUM_LOAD_WIDTH_COUNTS { 12, 0, 0 } // loads of 4, 16, 64 bytes per program +#define IGNEUM_BYTES_PER_HASH 384 +#define IGNEUM_FOLD_ROT 11 +#define IGNEUM_FOLD_MUL 0x9e3779b1u +// Hot table (hot-table experiment, 5 October 2026, docs/plans/hot-table.md): NOT the lottery hash. IGNEUM_HOT_SLOTS of the +// 16 load slots read H[mulhi(src, IGNEUM_HOT_WORDS)], H = IGNEUM_HOT_MB MiB of chained ChaCha12 lines (the cache chain) under +// KH = seed_words("igneum-hot/" || epoch seed bytes) and the tag "Igne" "umHT"; filled by igneum_hot_fill once per epoch. +#define IGNEUM_HOT_MB 32 +#define IGNEUM_HOT_WORDS 0x00800000u +#define IGNEUM_HOT_SEGMENTS 8192u +#define IGNEUM_HOT_SLOTS 4 // hot loads per program (32 per hash), replacing the dataset loads (12 of them) +#define IGNEUM_HOT_ADDED 0 +#define IGNEUM_HOT_KEY_INIT { 0x3a48bef5u, 0x6b54b1a1u, 0x9ff897c3u, 0x7d5d85e4u, 0xcd35379fu, 0x9d76bb86u, 0xbe2affb1u, 0x0f8f80b4u } +// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h) +#define IGNEUM_DATASET_MODE 1 + +#define IGNEUM_SEEDW_INIT { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u } +#define IGNEUM_KEY_INIT { 0x3067619fu, 0x3c269176u, 0x84a03b03u, 0xf8c63294u, 0xff977c5bu, 0xe60def3eu, 0x63630141u, 0xb8fbcb58u } +#define IGNEUM_CACHE_LOG2_WORDS 26 +#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6 +#define IGNEUM_CACHE_SEGMENTS 65536u +#define IGNEUM_ITEM_ROUNDS 8 +#define IGNEUM_MIX_ROT_INIT { 20u, 20u, 19u, 4u, 26u, 3u, 3u, 27u } +#define IGNEUM_MIX_MUL_INIT { 0x42146205u, 0x52cbe0fbu, 0x7ecf4a03u, 0x6728907fu, 0xd81d9751u, 0x132952c3u, 0xf60de277u, 0x05358035u, 0xbaf6499du, 0xe4db9667u, 0x3e98f45du, 0xd0004eddu, 0x2691630du, 0x9beb3bcfu, 0xab310379u, 0x99cfb423u } +#define IGNEUM_MIX_RC_INIT { 0xbab68293u, 0xcc162340u, 0x6ce151ccu, 0xe62b8997u, 0xc9c80297u, 0xf74a1654u, 0x3d704af5u, 0x3cf522b7u, 0x2b9cac04u, 0xa880ac10u, 0x13e5dd1du, 0x6fc3e233u, 0x2d83eeacu, 0x9006e8bfu, 0x2c4b5362u, 0x31b49ee2u } + +#ifndef IGNEUM_NO_CUDA +// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError(). +cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments); +cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems); +cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments); +cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot, + uint32_t nonces, uint32_t blockWarps); +cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps); +#endif diff --git a/proto-cuda/packs-ca2-hot/hot32k4/program.json b/proto-cuda/packs-ca2-hot/hot32k4/program.json new file mode 100644 index 000000000..6521857a1 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4/program.json @@ -0,0 +1,129 @@ +{ + "format": "igneum-program-pack-3", + "generator": 2, + "attempt": 0, + "program_id": "0x3370a4466e5f322c", + "program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32", + "dataset_mode": "memory-hard", + "seed": "igneum-genesis", + "seed_bytes": "69676e65756d2d67656e65736973", + "seed_words": ["0x67a9a7be", "0x1a155b25", "0xfddfb732", "0x4b5af2e8", "0xc55caf33", "0xa27c13b7", "0x06628a48", "0x03852469"], + "seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32", + "generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried", + "lanes": 32, + "registers": 8, + "iterations": 8, + "instruction_count": 64, + "loads_per_hash": 128, + "load_class": "hot32k4", + "load_slots": 16, + "load_mix_percent_4_16_64": [100, 0, 0], + "load_width_counts_4_16_64": [12, 0, 0], + "bytes_per_hash": 384, + "wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots", + "hot_table": {"mb": 32, "words": 8388608, "segments": 8192, "slots": 4, "form": "replaced: k of the class's load slots read the table", "dataset_slots": 12, "hot_loads_per_hash": 32, "key": ["0x3a48bef5", "0x6b54b1a1", "0x9ff897c3", "0x7d5d85e4", "0xcd35379f", "0x9d76bb86", "0xbe2affb1", "0x0f8f80b4"], "key_derivation": "seed_words_from_bytes('igneum-hot/' || seed_bytes), the same for every attempt of the epoch", "tag": ["0x49676e65", "0x756d4854"], "chain": "the cache chain of dataset.cache with the hot key and tag: in_j = prev ^ (sigma || key || seg || j || tag); line_j = block(in_j); prev_0 = 0", "load": "dst = dst ^ hot[mulhi(src, words)] (the high 32 bits of the 64-bit product; the index lies in [0, words) for any size)", "slots_rule": "the first k load slots in the Fisher-Yates draw order after the scratch slots (a uniform k-subset); no extra draw, so with version 2 widths and no scratch the program is the version 2 program with k loads redirected", "acceptance_stand_in": "dataset_elem(mulhi(src, words), seed_words[2], seed_words[3])", "program_id": "the read-width id with 'hot/' || mb || k appended", "spec": "docs/plans/hot-table.md"}, + "op_mix": {"load": 12, "add": 8, "shfl": 8, "xor": 6, "mad": 5, "mul": 5, "mulhi": 5, "hot": 4, "sub": 4, "rotl": 3, "rotr": 3, "or": 1}, + "register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]", + "splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16", + "iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order", + "output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo", + "op_semantics": { + "add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)", + "sub": "dst = dst - src", + "mul": "dst = dst * src (low 32)", + "mulhi": "dst = high 32 bits of dst * src", + "xor": "dst = dst ^ src", + "or": "dst = dst | src", + "rotl": "dst = rotl(dst, rot), rot in 1..31", + "rotr": "dst = rotr(dst, src & 31)", + "mad": "dst = src * src2 + dst", + "shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp", + "load": "dst = dst ^ dataset[src & dataset.mask]", + "wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)", + "hot": "dst = dst ^ hot[mulhi(src, hot_table.words)] (hot-table experiment)" + }, + "dataset": { + "log2_words": 28, + "bytes": 1073741824, + "mask": "0x0fffffff", + "day": "2026-10-03", + "day_bytes": "6461792f323032362d31302d3033", + "day_words_from": "seed_words_from_bytes(day_bytes)", + "d0": "0x3067619f", + "d1": "0x3c269176", + "mode": "memory-hard", + "spec": "proto-metal/MEMHARD.md", + "key": ["0x3067619f", "0x3c269176", "0x84a03b03", "0xf8c63294", "0xff977c5b", "0xe60def3e", "0x63630141", "0xb8fbcb58"], + "key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]", + "cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"}, + "mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [20, 20, 19, 4, 26, 3, 3, 27], "mul": ["0x42146205", "0x52cbe0fb", "0x7ecf4a03", "0x6728907f", "0xd81d9751", "0x132952c3", "0xf60de277", "0x05358035", "0xbaf6499d", "0xe4db9667", "0x3e98f45d", "0xd0004edd", "0x2691630d", "0x9beb3bcf", "0xab310379", "0x99cfb423"], "rc": ["0xbab68293", "0xcc162340", "0x6ce151cc", "0xe62b8997", "0xc9c80297", "0xf74a1654", "0x3d704af5", "0x3cf522b7", "0x2b9cac04", "0xa880ac10", "0x13e5dd1d", "0x6fc3e233", "0x2d83eeac", "0x9006e8bf", "0x2c4b5362", "0x31b49ee2"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"}, + "item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: s = M_r(s); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then s = M_8(s); item(t) = s", + "word": "dataset[w] = item(w >> 4)[w & 15]" + }, + "instructions": [ + {"i": 0, "op": "mad", "dst": 2, "src": 3, "src2": 4, "imm": "0xbf7b174d", "imm2": "0x337b762e", "rot": 17, "bit": 2, "mask": 2, "width": 1}, + {"i": 1, "op": "mad", "dst": 2, "src": 1, "src2": 1, "imm": "0xdd04a5da", "imm2": "0x42da7657", "rot": 15, "bit": 30, "mask": 16, "width": 1}, + {"i": 2, "op": "mad", "dst": 2, "src": 3, "src2": 2, "imm": "0x734003fa", "imm2": "0x5bb67700", "rot": 3, "bit": 20, "mask": 1, "width": 1}, + {"i": 3, "op": "xor", "dst": 3, "src": 5, "src2": 5, "imm": "0xc55a1b1c", "imm2": "0xa19720f3", "rot": 7, "bit": 8, "mask": 1, "width": 1}, + {"i": 4, "op": "load", "dst": 7, "src": 2, "src2": 5, "imm": "0xad572dd7", "imm2": "0x9ceb3ea7", "rot": 18, "bit": 30, "mask": 2, "width": 1}, + {"i": 5, "op": "load", "dst": 5, "src": 7, "src2": 3, "imm": "0x769a53be", "imm2": "0x80f9067e", "rot": 12, "bit": 22, "mask": 1, "width": 1}, + {"i": 6, "op": "shfl", "dst": 1, "src": 4, "src2": 0, "imm": "0xd3613d88", "imm2": "0x262fb219", "rot": 10, "bit": 30, "mask": 8, "width": 1}, + {"i": 7, "op": "shfl", "dst": 7, "src": 3, "src2": 4, "imm": "0xce38e42f", "imm2": "0xb868b818", "rot": 11, "bit": 8, "mask": 8, "width": 1}, + {"i": 8, "op": "mulhi", "dst": 1, "src": 5, "src2": 2, "imm": "0xa5eebca5", "imm2": "0x5703a72b", "rot": 13, "bit": 13, "mask": 16, "width": 1}, + {"i": 9, "op": "rotr", "dst": 6, "src": 3, "src2": 4, "imm": "0x17a5a9c7", "imm2": "0xdcfb93a1", "rot": 20, "bit": 27, "mask": 2, "width": 1}, + {"i": 10, "op": "or", "dst": 3, "src": 4, "src2": 1, "imm": "0xccb7d785", "imm2": "0xc335364c", "rot": 14, "bit": 12, "mask": 4, "width": 1}, + {"i": 11, "op": "load", "dst": 4, "src": 3, "src2": 2, "imm": "0x88cb9af3", "imm2": "0x4e7dc10d", "rot": 24, "bit": 17, "mask": 4, "width": 1}, + {"i": 12, "op": "mulhi", "dst": 0, "src": 4, "src2": 4, "imm": "0x45374321", "imm2": "0x3cd91989", "rot": 11, "bit": 4, "mask": 2, "width": 1}, + {"i": 13, "op": "add", "dst": 5, "src": 1, "src2": 2, "imm": "0xc7934706", "imm2": "0xd3177981", "rot": 16, "bit": 30, "mask": 2, "width": 1}, + {"i": 14, "op": "load", "dst": 0, "src": 4, "src2": 3, "imm": "0xfd7f56bb", "imm2": "0x65e14f52", "rot": 13, "bit": 22, "mask": 2, "width": 1}, + {"i": 15, "op": "sub", "dst": 2, "src": 4, "src2": 4, "imm": "0x35a80b49", "imm2": "0x060f2d13", "rot": 16, "bit": 20, "mask": 16, "width": 1}, + {"i": 16, "op": "hot", "dst": 2, "src": 0, "src2": 2, "imm": "0xae0a32c2", "imm2": "0x4c2a4cfe", "rot": 8, "bit": 31, "mask": 16, "width": 1}, + {"i": 17, "op": "load", "dst": 7, "src": 2, "src2": 6, "imm": "0x82a84cc3", "imm2": "0x21a38d68", "rot": 15, "bit": 21, "mask": 2, "width": 1}, + {"i": 18, "op": "shfl", "dst": 7, "src": 3, "src2": 3, "imm": "0xa3818806", "imm2": "0x8f66b5c8", "rot": 14, "bit": 6, "mask": 4, "width": 1}, + {"i": 19, "op": "mul", "dst": 5, "src": 0, "src2": 1, "imm": "0xa00de107", "imm2": "0x77bfcaa5", "rot": 3, "bit": 10, "mask": 2, "width": 1}, + {"i": 20, "op": "shfl", "dst": 3, "src": 4, "src2": 7, "imm": "0x1d2b8cab", "imm2": "0x80b4f9a2", "rot": 14, "bit": 25, "mask": 2, "width": 1}, + {"i": 21, "op": "shfl", "dst": 2, "src": 4, "src2": 5, "imm": "0x3ac915d2", "imm2": "0x5fba7bc2", "rot": 16, "bit": 1, "mask": 16, "width": 1}, + {"i": 22, "op": "mulhi", "dst": 6, "src": 2, "src2": 0, "imm": "0xdc3ec8fd", "imm2": "0x599e2fa3", "rot": 22, "bit": 3, "mask": 2, "width": 1}, + {"i": 23, "op": "load", "dst": 6, "src": 1, "src2": 5, "imm": "0x2a6b16d5", "imm2": "0xd73e396f", "rot": 28, "bit": 29, "mask": 2, "width": 1}, + {"i": 24, "op": "mul", "dst": 5, "src": 0, "src2": 7, "imm": "0x376d0223", "imm2": "0xe1c2169a", "rot": 4, "bit": 16, "mask": 16, "width": 1}, + {"i": 25, "op": "rotl", "dst": 5, "src": 7, "src2": 3, "imm": "0x78ad8c60", "imm2": "0x6f5b77d5", "rot": 19, "bit": 11, "mask": 16, "width": 1}, + {"i": 26, "op": "shfl", "dst": 7, "src": 6, "src2": 5, "imm": "0x93915b9f", "imm2": "0x1e61fb6b", "rot": 28, "bit": 23, "mask": 2, "width": 1}, + {"i": 27, "op": "xor", "dst": 0, "src": 5, "src2": 7, "imm": "0x6378fe15", "imm2": "0x66c78f42", "rot": 12, "bit": 31, "mask": 8, "width": 1}, + {"i": 28, "op": "xor", "dst": 0, "src": 4, "src2": 7, "imm": "0x20a57fda", "imm2": "0x088c848e", "rot": 16, "bit": 13, "mask": 4, "width": 1}, + {"i": 29, "op": "sub", "dst": 3, "src": 0, "src2": 2, "imm": "0x49d95fd5", "imm2": "0x1a5f946a", "rot": 6, "bit": 12, "mask": 1, "width": 1}, + {"i": 30, "op": "mul", "dst": 5, "src": 1, "src2": 7, "imm": "0x0a816217", "imm2": "0x405c4f73", "rot": 13, "bit": 27, "mask": 4, "width": 1}, + {"i": 31, "op": "load", "dst": 7, "src": 2, "src2": 2, "imm": "0x09ed045e", "imm2": "0xd69c4715", "rot": 5, "bit": 9, "mask": 2, "width": 1}, + {"i": 32, "op": "hot", "dst": 1, "src": 0, "src2": 6, "imm": "0xeb79ea49", "imm2": "0xcc587f5a", "rot": 6, "bit": 8, "mask": 16, "width": 1}, + {"i": 33, "op": "xor", "dst": 5, "src": 6, "src2": 1, "imm": "0x3027401e", "imm2": "0x5f20c27e", "rot": 18, "bit": 9, "mask": 2, "width": 1}, + {"i": 34, "op": "hot", "dst": 5, "src": 1, "src2": 3, "imm": "0x0e1cab07", "imm2": "0x09356c5b", "rot": 19, "bit": 31, "mask": 1, "width": 1}, + {"i": 35, "op": "mulhi", "dst": 0, "src": 5, "src2": 2, "imm": "0x90e31357", "imm2": "0xabd32484", "rot": 26, "bit": 5, "mask": 8, "width": 1}, + {"i": 36, "op": "shfl", "dst": 5, "src": 2, "src2": 5, "imm": "0xee9a955f", "imm2": "0x31b3faed", "rot": 8, "bit": 24, "mask": 4, "width": 1}, + {"i": 37, "op": "load", "dst": 7, "src": 0, "src2": 7, "imm": "0x3ba2f832", "imm2": "0x1160dcd3", "rot": 4, "bit": 29, "mask": 1, "width": 1}, + {"i": 38, "op": "add", "dst": 3, "src": 1, "src2": 7, "imm": "0x75ba2fad", "imm2": "0x230c005c", "rot": 4, "bit": 27, "mask": 1, "width": 1}, + {"i": 39, "op": "shfl", "dst": 1, "src": 5, "src2": 3, "imm": "0xdbf37e75", "imm2": "0xb5ac1969", "rot": 30, "bit": 13, "mask": 4, "width": 1}, + {"i": 40, "op": "xor", "dst": 2, "src": 5, "src2": 5, "imm": "0x47f136c5", "imm2": "0x06ce9153", "rot": 19, "bit": 10, "mask": 2, "width": 1}, + {"i": 41, "op": "mad", "dst": 3, "src": 6, "src2": 3, "imm": "0xce13eff8", "imm2": "0x04cc1d55", "rot": 3, "bit": 1, "mask": 4, "width": 1}, + {"i": 42, "op": "sub", "dst": 6, "src": 7, "src2": 1, "imm": "0x6a65ab71", "imm2": "0x8fbc1bcd", "rot": 4, "bit": 1, "mask": 8, "width": 1}, + {"i": 43, "op": "xor", "dst": 7, "src": 0, "src2": 7, "imm": "0xdaeb4928", "imm2": "0xc0423027", "rot": 24, "bit": 11, "mask": 8, "width": 1}, + {"i": 44, "op": "load", "dst": 1, "src": 7, "src2": 7, "imm": "0x778f01c9", "imm2": "0x28cedcea", "rot": 12, "bit": 4, "mask": 16, "width": 1}, + {"i": 45, "op": "mul", "dst": 2, "src": 3, "src2": 2, "imm": "0xf4264f1b", "imm2": "0x0f627d56", "rot": 5, "bit": 28, "mask": 8, "width": 1}, + {"i": 46, "op": "mulhi", "dst": 1, "src": 5, "src2": 1, "imm": "0xffb2147a", "imm2": "0xccde9b05", "rot": 13, "bit": 9, "mask": 2, "width": 1}, + {"i": 47, "op": "sub", "dst": 4, "src": 3, "src2": 5, "imm": "0x73b36234", "imm2": "0x3f5d5997", "rot": 7, "bit": 18, "mask": 2, "width": 1}, + {"i": 48, "op": "rotr", "dst": 2, "src": 6, "src2": 3, "imm": "0x3a4d9aa9", "imm2": "0x212bec7b", "rot": 4, "bit": 29, "mask": 16, "width": 1}, + {"i": 49, "op": "load", "dst": 3, "src": 5, "src2": 2, "imm": "0x626f11df", "imm2": "0x56cd5bfd", "rot": 7, "bit": 1, "mask": 1, "width": 1}, + {"i": 50, "op": "add", "dst": 1, "src": 5, "src2": 6, "imm": "0x81b8bc2c", "imm2": "0x1907970c", "rot": 28, "bit": 7, "mask": 4, "width": 1}, + {"i": 51, "op": "mul", "dst": 0, "src": 2, "src2": 2, "imm": "0xa8848b30", "imm2": "0xef6ac348", "rot": 9, "bit": 15, "mask": 8, "width": 1}, + {"i": 52, "op": "add", "dst": 0, "src": 2, "src2": 0, "imm": "0x4f92b968", "imm2": "0x699fd448", "rot": 22, "bit": 6, "mask": 4, "width": 1}, + {"i": 53, "op": "add", "dst": 1, "src": 0, "src2": 2, "imm": "0x2bb965af", "imm2": "0x77b1520d", "rot": 2, "bit": 12, "mask": 8, "width": 1}, + {"i": 54, "op": "rotl", "dst": 7, "src": 1, "src2": 0, "imm": "0x553e678b", "imm2": "0x3cc8eae0", "rot": 14, "bit": 20, "mask": 2, "width": 1}, + {"i": 55, "op": "add", "dst": 3, "src": 7, "src2": 2, "imm": "0x7b0fe07a", "imm2": "0xa54c55a0", "rot": 10, "bit": 1, "mask": 1, "width": 1}, + {"i": 56, "op": "load", "dst": 6, "src": 7, "src2": 4, "imm": "0x01eba9aa", "imm2": "0x2758c0f7", "rot": 14, "bit": 15, "mask": 4, "width": 1}, + {"i": 57, "op": "rotr", "dst": 1, "src": 5, "src2": 2, "imm": "0x1f5267b3", "imm2": "0x236f5a27", "rot": 2, "bit": 31, "mask": 16, "width": 1}, + {"i": 58, "op": "load", "dst": 5, "src": 4, "src2": 3, "imm": "0xa9954a9b", "imm2": "0x6a54d4e8", "rot": 11, "bit": 10, "mask": 16, "width": 1}, + {"i": 59, "op": "hot", "dst": 6, "src": 2, "src2": 4, "imm": "0x9923ff88", "imm2": "0x9357254e", "rot": 16, "bit": 1, "mask": 16, "width": 1}, + {"i": 60, "op": "mad", "dst": 3, "src": 5, "src2": 0, "imm": "0xc0cc51a6", "imm2": "0x3fd7701b", "rot": 20, "bit": 1, "mask": 4, "width": 1}, + {"i": 61, "op": "add", "dst": 5, "src": 7, "src2": 7, "imm": "0xaf9dd72d", "imm2": "0xad7493e7", "rot": 7, "bit": 31, "mask": 16, "width": 1}, + {"i": 62, "op": "add", "dst": 4, "src": 6, "src2": 2, "imm": "0x89841d87", "imm2": "0x1e07c3d9", "rot": 6, "bit": 27, "mask": 1, "width": 1}, + {"i": 63, "op": "rotl", "dst": 5, "src": 4, "src2": 2, "imm": "0xae210f8d", "imm2": "0x8e499ba4", "rot": 19, "bit": 9, "mask": 1, "width": 1} + ] +} diff --git a/proto-cuda/packs-ca2-hot/hot32k4/program.metal b/proto-cuda/packs-ca2-hot/hot32k4/program.metal new file mode 100644 index 000000000..254382798 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4/program.metal @@ -0,0 +1,112 @@ +#include +using namespace metal; + +#define MASK 0x0fffffffu +// Hot table (32 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x00800000u +constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }; + +inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 +inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +kernel void igneum_hash(device const uint* dataset [[buffer(0)]], + device ulong* out [[buffer(1)]], + constant uint& baseNonce [[buffer(2)]], + device const uint* hot [[buffer(3)]], + uint gid [[thread_position_in_grid]]) { + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + { uint x = nonce ^ SEEDW[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ SEEDW[1]; } + { uint x = nonce ^ SEEDW[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ SEEDW[2]; } + { uint x = nonce ^ SEEDW[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ SEEDW[3]; } + { uint x = nonce ^ SEEDW[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ SEEDW[4]; } + { uint x = nonce ^ SEEDW[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ SEEDW[5]; } + { uint x = nonce ^ SEEDW[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ SEEDW[6]; } + { uint x = nonce ^ SEEDW[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ SEEDW[7]; } + { uint x = nonce ^ SEEDW[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ SEEDW[0]; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 + r2 = r1 * r1 + r2; // 1 + r2 = r3 * r2 + r2; // 2 + r3 = r3 ^ r5; // 3 + r7 = r7 ^ dataset[r2 & MASK]; // 4 + r5 = r5 ^ dataset[r7 & MASK]; // 5 + r1 = r1 ^ simd_shuffle_xor(r4, (ushort)8); // 6 + r7 = r7 ^ simd_shuffle_xor(r3, (ushort)8); // 7 + r1 = mulhi(r1, r5); // 8 + r6 = rotr_var(r6, r3); // 9 + r3 = r3 | r4; // 10 + r4 = r4 ^ dataset[r3 & MASK]; // 11 + r0 = mulhi(r0, r4); // 12 + r5 = r5 + r1 + select(0xc7934706u, 0xd3177981u, ((sel >> 30u) & 1u) != 0u); // 13 + r0 = r0 ^ dataset[r4 & MASK]; // 14 + r2 = r2 - r4; // 15 + r2 = r2 ^ hot[mulhi(r0, HOT_WORDS)]; // 16 + r7 = r7 ^ dataset[r2 & MASK]; // 17 + r7 = r7 ^ simd_shuffle_xor(r3, (ushort)4); // 18 + r5 = r5 * r0; // 19 + r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 20 + r2 = r2 ^ simd_shuffle_xor(r4, (ushort)16); // 21 + r6 = mulhi(r6, r2); // 22 + r6 = r6 ^ dataset[r1 & MASK]; // 23 + r5 = r5 * r0; // 24 + r5 = rotl_imm(r5, 19u); // 25 + r7 = r7 ^ simd_shuffle_xor(r6, (ushort)2); // 26 + r0 = r0 ^ r5; // 27 + r0 = r0 ^ r4; // 28 + r3 = r3 - r0; // 29 + r5 = r5 * r1; // 30 + r7 = r7 ^ dataset[r2 & MASK]; // 31 + r1 = r1 ^ hot[mulhi(r0, HOT_WORDS)]; // 32 + r5 = r5 ^ r6; // 33 + r5 = r5 ^ hot[mulhi(r1, HOT_WORDS)]; // 34 + r0 = mulhi(r0, r5); // 35 + r5 = r5 ^ simd_shuffle_xor(r2, (ushort)4); // 36 + r7 = r7 ^ dataset[r0 & MASK]; // 37 + r3 = r3 + r1 + select(0x75ba2fadu, 0x230c005cu, ((sel >> 27u) & 1u) != 0u); // 38 + r1 = r1 ^ simd_shuffle_xor(r5, (ushort)4); // 39 + r2 = r2 ^ r5; // 40 + r3 = r6 * r3 + r3; // 41 + r6 = r6 - r7; // 42 + r7 = r7 ^ r0; // 43 + r1 = r1 ^ dataset[r7 & MASK]; // 44 + r2 = r2 * r3; // 45 + r1 = mulhi(r1, r5); // 46 + r4 = r4 - r3; // 47 + r2 = rotr_var(r2, r6); // 48 + r3 = r3 ^ dataset[r5 & MASK]; // 49 + r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 50 + r0 = r0 * r2; // 51 + r0 = r0 + r2 + select(0x4f92b968u, 0x699fd448u, ((sel >> 6u) & 1u) != 0u); // 52 + r1 = r1 + r0 + select(0x2bb965afu, 0x77b1520du, ((sel >> 12u) & 1u) != 0u); // 53 + r7 = rotl_imm(r7, 14u); // 54 + r3 = r3 + r7 + select(0x7b0fe07au, 0xa54c55a0u, ((sel >> 1u) & 1u) != 0u); // 55 + r6 = r6 ^ dataset[r7 & MASK]; // 56 + r1 = rotr_var(r1, r5); // 57 + r5 = r5 ^ dataset[r4 & MASK]; // 58 + r6 = r6 ^ hot[mulhi(r2, HOT_WORDS)]; // 59 + r3 = r5 * r0 + r3; // 60 + r5 = r5 + r7 + select(0xaf9dd72du, 0xad7493e7u, ((sel >> 31u) & 1u) != 0u); // 61 + r4 = r4 + r6 + select(0x89841d87u, 0x1e07c3d9u, ((sel >> 27u) & 1u) != 0u); // 62 + r5 = rotl_imm(r5, 19u); // 63 + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot32k4/program_bound.metal b/proto-cuda/packs-ca2-hot/hot32k4/program_bound.metal new file mode 100644 index 000000000..3e5f69990 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4/program_bound.metal @@ -0,0 +1,114 @@ +#include +using namespace metal; + +#define MASK 0x0fffffffu +// Hot table (32 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x00800000u +constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }; + +inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 +inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW. +kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], + device ulong* out [[buffer(1)]], + constant uint& baseNonce [[buffer(2)]], + constant uint* initw [[buffer(3)]], + device const uint* hot [[buffer(4)]], + uint gid [[thread_position_in_grid]]) { + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + { uint x = nonce ^ initw[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ initw[1]; } + { uint x = nonce ^ initw[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ initw[2]; } + { uint x = nonce ^ initw[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ initw[3]; } + { uint x = nonce ^ initw[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ initw[4]; } + { uint x = nonce ^ initw[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ initw[5]; } + { uint x = nonce ^ initw[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ initw[6]; } + { uint x = nonce ^ initw[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ initw[7]; } + { uint x = nonce ^ initw[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ initw[0]; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 + r2 = r1 * r1 + r2; // 1 + r2 = r3 * r2 + r2; // 2 + r3 = r3 ^ r5; // 3 + r7 = r7 ^ dataset[r2 & MASK]; // 4 + r5 = r5 ^ dataset[r7 & MASK]; // 5 + r1 = r1 ^ simd_shuffle_xor(r4, (ushort)8); // 6 + r7 = r7 ^ simd_shuffle_xor(r3, (ushort)8); // 7 + r1 = mulhi(r1, r5); // 8 + r6 = rotr_var(r6, r3); // 9 + r3 = r3 | r4; // 10 + r4 = r4 ^ dataset[r3 & MASK]; // 11 + r0 = mulhi(r0, r4); // 12 + r5 = r5 + r1 + select(0xc7934706u, 0xd3177981u, ((sel >> 30u) & 1u) != 0u); // 13 + r0 = r0 ^ dataset[r4 & MASK]; // 14 + r2 = r2 - r4; // 15 + r2 = r2 ^ hot[mulhi(r0, HOT_WORDS)]; // 16 + r7 = r7 ^ dataset[r2 & MASK]; // 17 + r7 = r7 ^ simd_shuffle_xor(r3, (ushort)4); // 18 + r5 = r5 * r0; // 19 + r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 20 + r2 = r2 ^ simd_shuffle_xor(r4, (ushort)16); // 21 + r6 = mulhi(r6, r2); // 22 + r6 = r6 ^ dataset[r1 & MASK]; // 23 + r5 = r5 * r0; // 24 + r5 = rotl_imm(r5, 19u); // 25 + r7 = r7 ^ simd_shuffle_xor(r6, (ushort)2); // 26 + r0 = r0 ^ r5; // 27 + r0 = r0 ^ r4; // 28 + r3 = r3 - r0; // 29 + r5 = r5 * r1; // 30 + r7 = r7 ^ dataset[r2 & MASK]; // 31 + r1 = r1 ^ hot[mulhi(r0, HOT_WORDS)]; // 32 + r5 = r5 ^ r6; // 33 + r5 = r5 ^ hot[mulhi(r1, HOT_WORDS)]; // 34 + r0 = mulhi(r0, r5); // 35 + r5 = r5 ^ simd_shuffle_xor(r2, (ushort)4); // 36 + r7 = r7 ^ dataset[r0 & MASK]; // 37 + r3 = r3 + r1 + select(0x75ba2fadu, 0x230c005cu, ((sel >> 27u) & 1u) != 0u); // 38 + r1 = r1 ^ simd_shuffle_xor(r5, (ushort)4); // 39 + r2 = r2 ^ r5; // 40 + r3 = r6 * r3 + r3; // 41 + r6 = r6 - r7; // 42 + r7 = r7 ^ r0; // 43 + r1 = r1 ^ dataset[r7 & MASK]; // 44 + r2 = r2 * r3; // 45 + r1 = mulhi(r1, r5); // 46 + r4 = r4 - r3; // 47 + r2 = rotr_var(r2, r6); // 48 + r3 = r3 ^ dataset[r5 & MASK]; // 49 + r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 50 + r0 = r0 * r2; // 51 + r0 = r0 + r2 + select(0x4f92b968u, 0x699fd448u, ((sel >> 6u) & 1u) != 0u); // 52 + r1 = r1 + r0 + select(0x2bb965afu, 0x77b1520du, ((sel >> 12u) & 1u) != 0u); // 53 + r7 = rotl_imm(r7, 14u); // 54 + r3 = r3 + r7 + select(0x7b0fe07au, 0xa54c55a0u, ((sel >> 1u) & 1u) != 0u); // 55 + r6 = r6 ^ dataset[r7 & MASK]; // 56 + r1 = rotr_var(r1, r5); // 57 + r5 = r5 ^ dataset[r4 & MASK]; // 58 + r6 = r6 ^ hot[mulhi(r2, HOT_WORDS)]; // 59 + r3 = r5 * r0 + r3; // 60 + r5 = r5 + r7 + select(0xaf9dd72du, 0xad7493e7u, ((sel >> 31u) & 1u) != 0u); // 61 + r4 = r4 + r6 + select(0x89841d87u, 0x1e07c3d9u, ((sel >> 27u) & 1u) != 0u); // 62 + r5 = rotl_imm(r5, 19u); // 63 + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot32k4/vectors.h b/proto-cuda/packs-ca2-hot/hot32k4/vectors.h new file mode 100644 index 000000000..48402daac --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4/vectors.h @@ -0,0 +1,67 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Expected outputs: igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif + +#define IGNEUM_VEC_WARPS 3 +static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u }; +static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = { + { // base nonce 0 + 0xd6dd7fa4e8ce0412ull, 0x7227f09f836fdfddull, 0xf278ed201cceafbfull, 0xec020afe5a5250acull, 0xf665fb744a23508dull, 0x279db29eb25d9ed7ull, 0x94acf8ad2077b200ull, 0x4570de146f552f46ull, + 0x41af921801576a43ull, 0xbcedf5682351398eull, 0x2f85798eaf3888d2ull, 0xb1973b681b608696ull, 0x43e637d721e2e02cull, 0x348f3e941642b441ull, 0x7a6d0c65ef9b3d99ull, 0xd0b3d21c07182e2aull, + 0x381c06bc6910e404ull, 0x2d6c2d4c8c64a835ull, 0x2a86874d214946a3ull, 0x2b59e5cd9a5b7b34ull, 0x59410384d5464398ull, 0x8a849a3e01c9c3a2ull, 0xa125cf7866802a38ull, 0xe1a8bbe849d936c8ull, + 0x0ca4566095a358ceull, 0xffe671a4d5e8ccc8ull, 0x57d840333cb092a6ull, 0xf97100913713d0deull, 0x442b713409774c64ull, 0xa10cf4e4c635ed10ull, 0xcf114a69a2f009b2ull, 0x36cd1a0b5f16adacull + }, + { // base nonce 4096 + 0x207cc1a5be16289full, 0x9ed2b155c8fc75e3ull, 0x2166ea7f0e13ac0dull, 0x44e3ee81290df4c9ull, 0xc2ecf745975f1401ull, 0x92be86c70096f95dull, 0xfd379260776410f7ull, 0x6cf6e19d5970bbb3ull, + 0x3a8b0506bcbf743cull, 0x10874f7196f213a0ull, 0x694b4277c2aab714ull, 0x343a854f40a54ea9ull, 0xedd91ea3095028cdull, 0x159ff1bc3d1dbf23ull, 0x337fccf9a5f8b745ull, 0x64e49f51dceeb174ull, + 0xd9e29b85844c0b6cull, 0x1223fdc9653f8a81ull, 0xfe4012a89d8cba42ull, 0x366e927907cee509ull, 0xe3b28f6d222d2a24ull, 0x45670ab50f08b3a3ull, 0xb18c0b1390d13480ull, 0xe8e65e2c34e96e4aull, + 0x105bdc8261ccafefull, 0xdf0ab212de234c1cull, 0x13ce611089323eafull, 0x8b964cb8e8ccdba2ull, 0xee088976c82e24e1ull, 0x4a12e75839901f2aull, 0x393d6ceb206e9f2dull, 0xd609e6da965d5304ull + }, + { // base nonce 1000000 + 0xca3bb360bd104e97ull, 0x534f429fa0c22461ull, 0xb7c03deeb1e3f559ull, 0xf1f803cd6f5f31e5ull, 0x4f68db8c38c02ea1ull, 0x3049cbc39de3c35cull, 0x903fdd301faa7a1full, 0xffb430e35c2c4a6full, + 0x0382a5e9c1446e5cull, 0xbec1b6d8a8f169f5ull, 0x447215c419241bc4ull, 0xa70e933c1419bd7eull, 0xac89eacda4f55c87ull, 0x3d0ca0cd39354441ull, 0xf0052aa94cc5abcfull, 0xfd6faddd678d1b18ull, + 0x8c1d8aac97e5ffd4ull, 0x7f831c277f83614full, 0xce48cf6eafce98c6ull, 0xaba6c8080267b4a5ull, 0x1afbd998e0cd01d6ull, 0xc1eac18a208d66cbull, 0x27f5f333f9f753ffull, 0x7e42e3cc90be0cd7ull, + 0x69694bcd49ff23c9ull, 0xc8abe7af7b18359dull, 0x54597507488baadfull, 0x2b2550309446c26bull, 0x63dbc8a53eb904e3ull, 0x5c6193647ccec1a9ull, 0xe7fa7ad65a4ca5bbull, 0x04d2f4278f7272aeull + } +}; + +// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (268435455). +static const uint32_t IGNEUM_DS_HEAD[16] = { + 0xffc3cd94u, 0x5920ccd8u, 0x392f44bbu, 0x5e57f67au, 0x2f2bc2a9u, 0x620b0e36u, 0xbdc09014u, 0x436654bfu, + 0x311e0b48u, 0x1abd93adu, 0x59cc7ce8u, 0xee5247b2u, 0x86171fe8u, 0x6d874751u, 0xc9f7728fu, 0x7c2a435du +}; +static const uint32_t IGNEUM_DS_LAST_INDEX = 268435455u; +static const uint32_t IGNEUM_DS_LAST = 0xa33ada72u; +// 64 sampled dataset words (index, value) computed on the Mac. +#define IGNEUM_DS_SAMPLES 64 +static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = { + 59471966u, 217795994u, 208353206u, 42483309u, 172547758u, 148076330u, 183853158u, 214389424u, 267488061u, 169781097u, 184093494u, 153880993u, 84977930u, 46426879u, 3093825u, 225364072u, 44593546u, 260713159u, 168250303u, 52384140u, 223401610u, 45554030u, 95410555u, 175039924u, 79171087u, 267580473u, 24168642u, 37981670u, 171551130u, 195559979u, 204611762u, 140997658u, 138925853u, 86637313u, 20736778u, 219665210u, 160430336u, 264654675u, 8013395u, 228945585u, 213884386u, 104419827u, 44185464u, 142737231u, 99284897u, 132475900u, 61861762u, 132056166u, 262388043u, 91878046u, 117353561u, 124768597u, 71352993u, 190698941u, 46055428u, 55281366u, 165145231u, 106810753u, 171985651u, 232085256u, 159510492u, 40072060u, 209107596u, 39023794u +}; +static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = { + 0xe8b73d94u, 0x337028b5u, 0xafe148c9u, 0xab99f7aeu, 0x434ea619u, 0xd85cb880u, 0x54764c7fu, 0x82c7e420u, 0xedf4cb9eu, 0x9884c959u, 0x223ee793u, 0x3a9ccf69u, 0x81da4fd2u, 0xd6ce8cb9u, 0xe3922dcau, 0x3e7e6bdeu, 0x382a3acau, 0x567e7f7fu, 0x25a0f084u, 0xbfeef128u, 0xe338abfbu, 0x7c3b5280u, 0x909bc5f1u, 0xd8b74b9cu, 0x8e31a22eu, 0x26b5f1d8u, 0x79122c00u, 0xcafc3340u, 0xd5e02ea3u, 0x1aee1afdu, 0xdb090d9au, 0xb049f435u, 0x4954d8bau, 0x03797ba0u, 0x196eefbdu, 0xd153412au, 0xbe5d2c4bu, 0xdaa14f0eu, 0x8e61ed07u, 0x9e9a64c6u, 0x2e29ff36u, 0x392a8589u, 0xb56a5912u, 0xfa6e8b57u, 0xd1a737cbu, 0xb0fa841au, 0xbe1c341fu, 0xe25be0f1u, 0xe937f543u, 0xebab2248u, 0x8e1b607au, 0x202a2fedu, 0x95e2819cu, 0x9c9652d4u, 0x32fedef0u, 0xdecfff82u, 0xcb5d43e5u, 0xb735806au, 0x8905939cu, 0xfbf8472du, 0xada74e5du, 0x7ebdeeeau, 0x0119f2b3u, 0xa9a376b8u +}; +// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^26 words. +static const uint32_t IGNEUM_CACHE_HEAD[16] = { + 0x355a86d2u, 0x7957db1cu, 0xd21772afu, 0x6fc1e09bu, 0xd55ce61du, 0x6e6a278bu, 0xd3f543ceu, 0x223d8e82u, + 0x143ab337u, 0x2e9f05bdu, 0x2eb389bfu, 0x0c6e449eu, 0x5cfa4222u, 0xba6560feu, 0x8e3e1aa4u, 0xdbcc1d53u +}; +static const uint32_t IGNEUM_CACHE_LAST[16] = { + 0x41190d91u, 0xbd277957u, 0x22ddbb49u, 0x6986f207u, 0xdf69a4d6u, 0x26401a3au, 0x818230fbu, 0xc417122du, + 0x3597b211u, 0xb553ce55u, 0xcf39cc0du, 0x3b7fc43au, 0x3fd43b00u, 0x67e1c80eu, 0xffa7ea7du, 0xca2960abu +}; +static const uint64_t IGNEUM_CACHE_FNV64 = 0x48c4f5bf24166b2eull; +// Hot table self-test (docs/plans/hot-table.md): hot[0..15], the last 16 words, and FNV-1a 64 over all IGNEUM_HOT_WORDS words. +static const uint32_t IGNEUM_HOT_HEAD[16] = { + 0x8068cc73u, 0x6036ebf9u, 0xb604cd25u, 0x8ffb840eu, 0xc54074a2u, 0x285c0695u, 0x77512425u, 0xc26a58a7u, + 0x72c88757u, 0xc10fca78u, 0x513825ddu, 0x30d6ccc8u, 0x9a05e7cfu, 0xb9533f50u, 0x4bac3ba0u, 0xa5c19528u +}; +static const uint32_t IGNEUM_HOT_LAST[16] = { + 0x2e5537fcu, 0x5e1daf5au, 0x4068fac1u, 0xe48688a5u, 0x38ff563au, 0xbd50595eu, 0x4fd1cffdu, 0xcbdad89au, + 0xecd4eaebu, 0xfd452ab4u, 0xcb2ba071u, 0xc147b12cu, 0x2863ad2au, 0x0f58974fu, 0xe0804105u, 0x7a2298aau +}; +static const uint64_t IGNEUM_HOT_FNV64 = 0xc1767ba3ef02719full; diff --git a/proto-cuda/packs-ca2-hot/hot32k4/vectors.json b/proto-cuda/packs-ca2-hot/hot32k4/vectors.json new file mode 100644 index 000000000..97fa3028e --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4/vectors.json @@ -0,0 +1,39 @@ +{ + "seed": "igneum-genesis", + "day": "2026-10-03", + "dataset_mode": "memory-hard", + "dataset_log2_words": 28, + "mask": "0x0fffffff", + "lanes": 32, + "source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset", + "warps": [ + {"base_nonce": 0, "expected": [ + "0xd6dd7fa4e8ce0412", "0x7227f09f836fdfdd", "0xf278ed201cceafbf", "0xec020afe5a5250ac", "0xf665fb744a23508d", "0x279db29eb25d9ed7", "0x94acf8ad2077b200", "0x4570de146f552f46", + "0x41af921801576a43", "0xbcedf5682351398e", "0x2f85798eaf3888d2", "0xb1973b681b608696", "0x43e637d721e2e02c", "0x348f3e941642b441", "0x7a6d0c65ef9b3d99", "0xd0b3d21c07182e2a", + "0x381c06bc6910e404", "0x2d6c2d4c8c64a835", "0x2a86874d214946a3", "0x2b59e5cd9a5b7b34", "0x59410384d5464398", "0x8a849a3e01c9c3a2", "0xa125cf7866802a38", "0xe1a8bbe849d936c8", + "0x0ca4566095a358ce", "0xffe671a4d5e8ccc8", "0x57d840333cb092a6", "0xf97100913713d0de", "0x442b713409774c64", "0xa10cf4e4c635ed10", "0xcf114a69a2f009b2", "0x36cd1a0b5f16adac" + ]}, + {"base_nonce": 4096, "expected": [ + "0x207cc1a5be16289f", "0x9ed2b155c8fc75e3", "0x2166ea7f0e13ac0d", "0x44e3ee81290df4c9", "0xc2ecf745975f1401", "0x92be86c70096f95d", "0xfd379260776410f7", "0x6cf6e19d5970bbb3", + "0x3a8b0506bcbf743c", "0x10874f7196f213a0", "0x694b4277c2aab714", "0x343a854f40a54ea9", "0xedd91ea3095028cd", "0x159ff1bc3d1dbf23", "0x337fccf9a5f8b745", "0x64e49f51dceeb174", + "0xd9e29b85844c0b6c", "0x1223fdc9653f8a81", "0xfe4012a89d8cba42", "0x366e927907cee509", "0xe3b28f6d222d2a24", "0x45670ab50f08b3a3", "0xb18c0b1390d13480", "0xe8e65e2c34e96e4a", + "0x105bdc8261ccafef", "0xdf0ab212de234c1c", "0x13ce611089323eaf", "0x8b964cb8e8ccdba2", "0xee088976c82e24e1", "0x4a12e75839901f2a", "0x393d6ceb206e9f2d", "0xd609e6da965d5304" + ]}, + {"base_nonce": 1000000, "expected": [ + "0xca3bb360bd104e97", "0x534f429fa0c22461", "0xb7c03deeb1e3f559", "0xf1f803cd6f5f31e5", "0x4f68db8c38c02ea1", "0x3049cbc39de3c35c", "0x903fdd301faa7a1f", "0xffb430e35c2c4a6f", + "0x0382a5e9c1446e5c", "0xbec1b6d8a8f169f5", "0x447215c419241bc4", "0xa70e933c1419bd7e", "0xac89eacda4f55c87", "0x3d0ca0cd39354441", "0xf0052aa94cc5abcf", "0xfd6faddd678d1b18", + "0x8c1d8aac97e5ffd4", "0x7f831c277f83614f", "0xce48cf6eafce98c6", "0xaba6c8080267b4a5", "0x1afbd998e0cd01d6", "0xc1eac18a208d66cb", "0x27f5f333f9f753ff", "0x7e42e3cc90be0cd7", + "0x69694bcd49ff23c9", "0xc8abe7af7b18359d", "0x54597507488baadf", "0x2b2550309446c26b", "0x63dbc8a53eb904e3", "0x5c6193647ccec1a9", "0xe7fa7ad65a4ca5bb", "0x04d2f4278f7272ae" + ]} + ], + "dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"], + "dataset_last_index": 268435455, + "dataset_last": "0xa33ada72", + "dataset_samples": [{"index": 59471966, "value": "0xe8b73d94"}, {"index": 217795994, "value": "0x337028b5"}, {"index": 208353206, "value": "0xafe148c9"}, {"index": 42483309, "value": "0xab99f7ae"}, {"index": 172547758, "value": "0x434ea619"}, {"index": 148076330, "value": "0xd85cb880"}, {"index": 183853158, "value": "0x54764c7f"}, {"index": 214389424, "value": "0x82c7e420"}, {"index": 267488061, "value": "0xedf4cb9e"}, {"index": 169781097, "value": "0x9884c959"}, {"index": 184093494, "value": "0x223ee793"}, {"index": 153880993, "value": "0x3a9ccf69"}, {"index": 84977930, "value": "0x81da4fd2"}, {"index": 46426879, "value": "0xd6ce8cb9"}, {"index": 3093825, "value": "0xe3922dca"}, {"index": 225364072, "value": "0x3e7e6bde"}, {"index": 44593546, "value": "0x382a3aca"}, {"index": 260713159, "value": "0x567e7f7f"}, {"index": 168250303, "value": "0x25a0f084"}, {"index": 52384140, "value": "0xbfeef128"}, {"index": 223401610, "value": "0xe338abfb"}, {"index": 45554030, "value": "0x7c3b5280"}, {"index": 95410555, "value": "0x909bc5f1"}, {"index": 175039924, "value": "0xd8b74b9c"}, {"index": 79171087, "value": "0x8e31a22e"}, {"index": 267580473, "value": "0x26b5f1d8"}, {"index": 24168642, "value": "0x79122c00"}, {"index": 37981670, "value": "0xcafc3340"}, {"index": 171551130, "value": "0xd5e02ea3"}, {"index": 195559979, "value": "0x1aee1afd"}, {"index": 204611762, "value": "0xdb090d9a"}, {"index": 140997658, "value": "0xb049f435"}, {"index": 138925853, "value": "0x4954d8ba"}, {"index": 86637313, "value": "0x03797ba0"}, {"index": 20736778, "value": "0x196eefbd"}, {"index": 219665210, "value": "0xd153412a"}, {"index": 160430336, "value": "0xbe5d2c4b"}, {"index": 264654675, "value": "0xdaa14f0e"}, {"index": 8013395, "value": "0x8e61ed07"}, {"index": 228945585, "value": "0x9e9a64c6"}, {"index": 213884386, "value": "0x2e29ff36"}, {"index": 104419827, "value": "0x392a8589"}, {"index": 44185464, "value": "0xb56a5912"}, {"index": 142737231, "value": "0xfa6e8b57"}, {"index": 99284897, "value": "0xd1a737cb"}, {"index": 132475900, "value": "0xb0fa841a"}, {"index": 61861762, "value": "0xbe1c341f"}, {"index": 132056166, "value": "0xe25be0f1"}, {"index": 262388043, "value": "0xe937f543"}, {"index": 91878046, "value": "0xebab2248"}, {"index": 117353561, "value": "0x8e1b607a"}, {"index": 124768597, "value": "0x202a2fed"}, {"index": 71352993, "value": "0x95e2819c"}, {"index": 190698941, "value": "0x9c9652d4"}, {"index": 46055428, "value": "0x32fedef0"}, {"index": 55281366, "value": "0xdecfff82"}, {"index": 165145231, "value": "0xcb5d43e5"}, {"index": 106810753, "value": "0xb735806a"}, {"index": 171985651, "value": "0x8905939c"}, {"index": 232085256, "value": "0xfbf8472d"}, {"index": 159510492, "value": "0xada74e5d"}, {"index": 40072060, "value": "0x7ebdeeea"}, {"index": 209107596, "value": "0x0119f2b3"}, {"index": 39023794, "value": "0xa9a376b8"}], + "cache_head": ["0x355a86d2", "0x7957db1c", "0xd21772af", "0x6fc1e09b", "0xd55ce61d", "0x6e6a278b", "0xd3f543ce", "0x223d8e82", "0x143ab337", "0x2e9f05bd", "0x2eb389bf", "0x0c6e449e", "0x5cfa4222", "0xba6560fe", "0x8e3e1aa4", "0xdbcc1d53"], + "cache_last_line": ["0x41190d91", "0xbd277957", "0x22ddbb49", "0x6986f207", "0xdf69a4d6", "0x26401a3a", "0x818230fb", "0xc417122d", "0x3597b211", "0xb553ce55", "0xcf39cc0d", "0x3b7fc43a", "0x3fd43b00", "0x67e1c80e", "0xffa7ea7d", "0xca2960ab"], + "cache_fnv1a64": "0x48c4f5bf24166b2e", + "hot_head": ["0x8068cc73", "0x6036ebf9", "0xb604cd25", "0x8ffb840e", "0xc54074a2", "0x285c0695", "0x77512425", "0xc26a58a7", "0x72c88757", "0xc10fca78", "0x513825dd", "0x30d6ccc8", "0x9a05e7cf", "0xb9533f50", "0x4bac3ba0", "0xa5c19528"], + "hot_last_line": ["0x2e5537fc", "0x5e1daf5a", "0x4068fac1", "0xe48688a5", "0x38ff563a", "0xbd50595e", "0x4fd1cffd", "0xcbdad89a", "0xecd4eaeb", "0xfd452ab4", "0xcb2ba071", "0xc147b12c", "0x2863ad2a", "0x0f58974f", "0xe0804105", "0x7a2298aa"], + "hot_fnv1a64": "0xc1767ba3ef02719f" +} diff --git a/proto-cuda/packs-ca2-hot/hot32k4a/kernel.cl b/proto-cuda/packs-ca2-hot/hot32k4a/kernel.cl new file mode 100644 index 000000000..b2c90a66a --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4a/kernel.cl @@ -0,0 +1,305 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal). +// Built from source at runtime by proto-opencl/host.c, which passes these defines: +// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit) +// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default) +// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32 +// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition +// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units; +// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32. +#ifndef IGNEUM_GROUP +#define IGNEUM_GROUP 32 +#endif +#ifndef IGNEUM_EXCHANGE +#define IGNEUM_EXCHANGE 0 +#endif +#ifdef __OPENCL_VERSION__ +#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1))) +#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n] +#if IGNEUM_EXCHANGE == 1 +#ifdef cl_khr_subgroups +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif +#ifdef cl_khr_subgroup_shuffle +#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable +#endif +#elif IGNEUM_EXCHANGE == 2 +#pragma OPENCL EXTENSION cl_intel_subgroups : enable +#endif +#else +// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros. +#include "emu_opencl.h" +#endif + +#if IGNEUM_EXCHANGE == 1 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#elif IGNEUM_EXCHANGE == 2 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#else +// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per +// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane +// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's +// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier. +#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; } +#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; } +#endif + +// Hot table (32 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x00800000u +static inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32. +static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); } +// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x. +static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); } +static inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +static inline void mh_chacha_block(const uint* x, uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +static inline void mh_cache_segment(__global uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +static inline void mh_mixer(uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +static inline void mh_item(__global const uint* cache, uint t, uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + __global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item. +// The same constants as memhard.h in this pack (one emitter, three dialects). +__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) { + uint t = (uint)get_global_id(0); + if (t < nItems) { + uint s[16]; + mh_item(cache, t, s); + __global uint* d = ds + ((ulong)t * 16u); + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (docs/plans/hot-table.md): 32 MiB = 8192 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +static inline void ht_segment(__global uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one work-item per segment. +__kernel void igneum_hot_fill(__global uint* hot, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the +// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and +// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). +IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r6 = r6 | r4; // 0 or + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 1u); r7 = r7 ^ t_; } // 1 shfl + r0 = r0 | r2; // 2 or + { uint t_; IGNEUM_SHFL_XOR(t_, r0, 16u); r3 = r3 ^ t_; } // 3 shfl + r7 = r7 ^ ds[r3 & mask]; // 4 load + r6 = r6 ^ ds[r0 & mask]; // 5 load + r1 = r1 * r4; // 6 mul + r0 = r0 - r1; // 7 sub + r3 = r3 ^ ds[r7 & mask]; // 8 load + r1 = r0 * r3 + r1; // 9 mad + r4 = r4 * r0; // 10 mul + r5 = r5 ^ ds[r4 & mask]; // 11 load + r1 = r1 ^ r7; // 12 xor + r1 = r1 ^ ds[r6 & mask]; // 13 load + r2 = r2 ^ ds[r1 & mask]; // 14 load + r3 = r3 * r1; // 15 mul + r6 = r6 ^ hot[mul_hi(r5, HOT_WORDS)]; // 16 hot + r0 = r0 ^ ds[r2 & mask]; // 17 load + r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x535b545fu : 0xc035a4e6u); // 18 add + r5 = r5 - r0; // 19 sub + r2 = r2 * r4; // 20 mul + r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x925d5e63u : 0x1ecdd2cfu); // 21 add + r3 = r3 ^ r7; // 22 xor + r7 = r7 ^ ds[r2 & mask]; // 23 load + r2 = mul_hi(r2, r5); // 24 mulhi + r5 = rotr_var(r5, r2); // 25 rotr + r3 = r2 * r7 + r3; // 26 mad + r2 = r2 - r0; // 27 sub + r6 = r1 * r5 + r6; // 28 mad + r2 = r2 + r3 + ((((sel >> 16u) & 1u) != 0u) ? 0x46b1b505u : 0xbdc6da76u); // 29 add + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r3 = r3 ^ t_; } // 30 shfl + r5 = r5 ^ ds[r7 & mask]; // 31 load + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 32 hot + r6 = r6 | r3; // 33 or + r3 = r3 ^ hot[mul_hi(r6, HOT_WORDS)]; // 34 hot + { uint t_; IGNEUM_SHFL_XOR(t_, r0, 2u); r4 = r4 ^ t_; } // 35 shfl + r5 = r5 ^ r2; // 36 xor + r3 = r3 ^ ds[r4 & mask]; // 37 load + r4 = r4 ^ ds[r3 & mask]; // 38 load + r1 = rotr_var(r1, r4); // 39 rotr + r3 = r3 + r6 + ((((sel >> 28u) & 1u) != 0u) ? 0x6328cb2cu : 0xbc3ff65fu); // 40 add + r5 = r5 ^ r1; // 41 xor + r5 = r5 | r0; // 42 or + r7 = r7 ^ r3; // 43 xor + r2 = r2 ^ ds[r5 & mask]; // 44 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r6 = r6 ^ t_; } // 45 shfl + r5 = r5 ^ r3; // 46 xor + r7 = r7 + r6 + ((((sel >> 22u) & 1u) != 0u) ? 0x9342df0du : 0x5af3bd5bu); // 47 add + r3 = r3 + r4 + ((((sel >> 23u) & 1u) != 0u) ? 0x485614dbu : 0x9ff95776u); // 48 add + r5 = r5 ^ ds[r3 & mask]; // 49 load + r4 = rotr_var(r4, r5); // 50 rotr + r0 = r1 * r7 + r0; // 51 mad + r0 = r0 * r3; // 52 mul + r5 = rotr_var(r5, r4); // 53 rotr + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r0 = r0 ^ t_; } // 54 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r7, 16u); r0 = r0 ^ t_; } // 55 shfl + r7 = r7 ^ ds[r4 & mask]; // 56 load + r7 = rotr_var(r7, r3); // 57 rotr + r0 = r0 ^ ds[r6 & mask]; // 58 load + r6 = r6 ^ hot[mul_hi(r2, HOT_WORDS)]; // 59 hot + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 1u); r3 = r3 ^ t_; } // 60 shfl + r7 = r7 - r5; // 61 sub + r1 = rotl_imm(r1, 4u); // 62 rotl + r4 = r4 ^ ds[r5 & mask]; // 63 load + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} + +#if IGNEUM_EXCHANGE != 0 +// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the +// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a +// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md. +IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) { + if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); } +} +#endif diff --git a/proto-cuda/packs-ca2-hot/hot32k4a/kernel.cu b/proto-cuda/packs-ca2-hot/hot32k4a/kernel.cu new file mode 100644 index 000000000..1ad7393a3 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4a/kernel.cu @@ -0,0 +1,179 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal). +// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC. +#include +#include +#include "program.h" +#include "memhard.h" + +// Hot table (32 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x00800000u +__device__ __forceinline__ uint32_t splitmix32(uint32_t x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31. +__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } +// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x. +__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) { + uint32_t x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item. +// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host. +__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) { + uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x; + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) { + uint32_t t = blockIdx.x * blockDim.x + threadIdx.x; + if (t < nItems) { + uint32_t s[16]; + mh_item(cache, t, s); + uint32_t* d = ds + (size_t)t * 16u; + for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (ht_segment is in memhard.h): one thread per segment. +__global__ void igneum_hot_fill(uint32_t* hot, uint32_t nSegments) { + uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x; + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every +// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a +// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid. +__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot) { + uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x; + uint32_t nonce = baseNonce + gid; + uint32_t r0, r1, r2, r3, r4, r5, r6, r7; + { uint32_t x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint32_t x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint32_t x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint32_t x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint32_t x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint32_t x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint32_t x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint32_t x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint32_t it = 0u; it < 8u; ++it) { + uint32_t sel = r0; + r6 = r6 | r4; // 0 or + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r4, 1); // 1 shfl + r0 = r0 | r2; // 2 or + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r0, 16); // 3 shfl + r7 = r7 ^ ds[r3 & mask]; // 4 load + r6 = r6 ^ ds[r0 & mask]; // 5 load + r1 = r1 * r4; // 6 mul + r0 = r0 - r1; // 7 sub + r3 = r3 ^ ds[r7 & mask]; // 8 load + r1 = r0 * r3 + r1; // 9 mad + r4 = r4 * r0; // 10 mul + r5 = r5 ^ ds[r4 & mask]; // 11 load + r1 = r1 ^ r7; // 12 xor + r1 = r1 ^ ds[r6 & mask]; // 13 load + r2 = r2 ^ ds[r1 & mask]; // 14 load + r3 = r3 * r1; // 15 mul + r6 = r6 ^ hot[__umulhi(r5, HOT_WORDS)]; // 16 hot + r0 = r0 ^ ds[r2 & mask]; // 17 load + r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x535b545fu : 0xc035a4e6u); // 18 add + r5 = r5 - r0; // 19 sub + r2 = r2 * r4; // 20 mul + r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x925d5e63u : 0x1ecdd2cfu); // 21 add + r3 = r3 ^ r7; // 22 xor + r7 = r7 ^ ds[r2 & mask]; // 23 load + r2 = __umulhi(r2, r5); // 24 mulhi + r5 = rotr_var(r5, r2); // 25 rotr + r3 = r2 * r7 + r3; // 26 mad + r2 = r2 - r0; // 27 sub + r6 = r1 * r5 + r6; // 28 mad + r2 = r2 + r3 + ((((sel >> 16u) & 1u) != 0u) ? 0x46b1b505u : 0xbdc6da76u); // 29 add + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r2, 8); // 30 shfl + r5 = r5 ^ ds[r7 & mask]; // 31 load + r2 = r2 ^ hot[__umulhi(r0, HOT_WORDS)]; // 32 hot + r6 = r6 | r3; // 33 or + r3 = r3 ^ hot[__umulhi(r6, HOT_WORDS)]; // 34 hot + r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r0, 2); // 35 shfl + r5 = r5 ^ r2; // 36 xor + r3 = r3 ^ ds[r4 & mask]; // 37 load + r4 = r4 ^ ds[r3 & mask]; // 38 load + r1 = rotr_var(r1, r4); // 39 rotr + r3 = r3 + r6 + ((((sel >> 28u) & 1u) != 0u) ? 0x6328cb2cu : 0xbc3ff65fu); // 40 add + r5 = r5 ^ r1; // 41 xor + r5 = r5 | r0; // 42 or + r7 = r7 ^ r3; // 43 xor + r2 = r2 ^ ds[r5 & mask]; // 44 load + r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 45 shfl + r5 = r5 ^ r3; // 46 xor + r7 = r7 + r6 + ((((sel >> 22u) & 1u) != 0u) ? 0x9342df0du : 0x5af3bd5bu); // 47 add + r3 = r3 + r4 + ((((sel >> 23u) & 1u) != 0u) ? 0x485614dbu : 0x9ff95776u); // 48 add + r5 = r5 ^ ds[r3 & mask]; // 49 load + r4 = rotr_var(r4, r5); // 50 rotr + r0 = r1 * r7 + r0; // 51 mad + r0 = r0 * r3; // 52 mul + r5 = rotr_var(r5, r4); // 53 rotr + r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 1); // 54 shfl + r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r7, 16); // 55 shfl + r7 = r7 ^ ds[r4 & mask]; // 56 load + r7 = rotr_var(r7, r3); // 57 rotr + r0 = r0 ^ ds[r6 & mask]; // 58 load + r6 = r6 ^ hot[__umulhi(r2, HOT_WORDS)]; // 59 hot + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r2, 1); // 60 shfl + r7 = r7 - r5; // 61 sub + r1 = rotl_imm(r1, 4u); // 62 rotl + r4 = r4 ^ ds[r5 & mask]; // 63 load + } + uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo; +} + +// Host-side launch wrappers. Declared in program.h, called from host.cu. +cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) { + if (nSegments == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nSegments + block - 1u) / block; + igneum_cache_fill<<>>(cache, nSegments); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) { + if (nItems == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nItems + block - 1u) / block; + igneum_build<<>>(ds, cache, nItems); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments) { + if (nSegments == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nSegments + block - 1u) / block; + igneum_hot_fill<<>>(hot, nSegments); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot, + uint32_t nonces, uint32_t blockWarps) { + if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue; + uint32_t block = 32u * blockWarps; + if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue; + igneum_hash<<>>(ds, out, baseNonce, mask, hot); + return cudaGetLastError(); +} + +cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) { + cudaFuncAttributes attr; + cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash); + if (e != cudaSuccess) return e; + *numRegs = attr.numRegs; + return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0); +} diff --git a/proto-cuda/packs-ca2-hot/hot32k4a/kernel_bound.cl b/proto-cuda/packs-ca2-hot/hot32k4a/kernel_bound.cl new file mode 100644 index 000000000..c8e463b0d --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4a/kernel_bound.cl @@ -0,0 +1,399 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal). +// Built from source at runtime by proto-opencl/host.c, which passes these defines: +// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit) +// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default) +// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32 +// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition +// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units; +// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32. +#ifndef IGNEUM_GROUP +#define IGNEUM_GROUP 32 +#endif +#ifndef IGNEUM_EXCHANGE +#define IGNEUM_EXCHANGE 0 +#endif +#ifdef __OPENCL_VERSION__ +#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1))) +#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n] +#if IGNEUM_EXCHANGE == 1 +#ifdef cl_khr_subgroups +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif +#ifdef cl_khr_subgroup_shuffle +#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable +#endif +#elif IGNEUM_EXCHANGE == 2 +#pragma OPENCL EXTENSION cl_intel_subgroups : enable +#endif +#else +// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros. +#include "emu_opencl.h" +#endif + +#if IGNEUM_EXCHANGE == 1 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#elif IGNEUM_EXCHANGE == 2 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#else +// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per +// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane +// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's +// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier. +#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; } +#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; } +#endif + +// Hot table (32 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x00800000u +static inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32. +static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); } +// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x. +static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); } +static inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +static inline void mh_chacha_block(const uint* x, uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +static inline void mh_cache_segment(__global uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +static inline void mh_mixer(uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +static inline void mh_item(__global const uint* cache, uint t, uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + __global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item. +// The same constants as memhard.h in this pack (one emitter, three dialects). +__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) { + uint t = (uint)get_global_id(0); + if (t < nItems) { + uint s[16]; + mh_item(cache, t, s); + __global uint* d = ds + ((ulong)t * 16u); + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (docs/plans/hot-table.md): 32 MiB = 8192 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +static inline void ht_segment(__global uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one work-item per segment. +__kernel void igneum_hot_fill(__global uint* hot, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the +// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and +// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). +IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r6 = r6 | r4; // 0 or + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 1u); r7 = r7 ^ t_; } // 1 shfl + r0 = r0 | r2; // 2 or + { uint t_; IGNEUM_SHFL_XOR(t_, r0, 16u); r3 = r3 ^ t_; } // 3 shfl + r7 = r7 ^ ds[r3 & mask]; // 4 load + r6 = r6 ^ ds[r0 & mask]; // 5 load + r1 = r1 * r4; // 6 mul + r0 = r0 - r1; // 7 sub + r3 = r3 ^ ds[r7 & mask]; // 8 load + r1 = r0 * r3 + r1; // 9 mad + r4 = r4 * r0; // 10 mul + r5 = r5 ^ ds[r4 & mask]; // 11 load + r1 = r1 ^ r7; // 12 xor + r1 = r1 ^ ds[r6 & mask]; // 13 load + r2 = r2 ^ ds[r1 & mask]; // 14 load + r3 = r3 * r1; // 15 mul + r6 = r6 ^ hot[mul_hi(r5, HOT_WORDS)]; // 16 hot + r0 = r0 ^ ds[r2 & mask]; // 17 load + r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x535b545fu : 0xc035a4e6u); // 18 add + r5 = r5 - r0; // 19 sub + r2 = r2 * r4; // 20 mul + r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x925d5e63u : 0x1ecdd2cfu); // 21 add + r3 = r3 ^ r7; // 22 xor + r7 = r7 ^ ds[r2 & mask]; // 23 load + r2 = mul_hi(r2, r5); // 24 mulhi + r5 = rotr_var(r5, r2); // 25 rotr + r3 = r2 * r7 + r3; // 26 mad + r2 = r2 - r0; // 27 sub + r6 = r1 * r5 + r6; // 28 mad + r2 = r2 + r3 + ((((sel >> 16u) & 1u) != 0u) ? 0x46b1b505u : 0xbdc6da76u); // 29 add + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r3 = r3 ^ t_; } // 30 shfl + r5 = r5 ^ ds[r7 & mask]; // 31 load + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 32 hot + r6 = r6 | r3; // 33 or + r3 = r3 ^ hot[mul_hi(r6, HOT_WORDS)]; // 34 hot + { uint t_; IGNEUM_SHFL_XOR(t_, r0, 2u); r4 = r4 ^ t_; } // 35 shfl + r5 = r5 ^ r2; // 36 xor + r3 = r3 ^ ds[r4 & mask]; // 37 load + r4 = r4 ^ ds[r3 & mask]; // 38 load + r1 = rotr_var(r1, r4); // 39 rotr + r3 = r3 + r6 + ((((sel >> 28u) & 1u) != 0u) ? 0x6328cb2cu : 0xbc3ff65fu); // 40 add + r5 = r5 ^ r1; // 41 xor + r5 = r5 | r0; // 42 or + r7 = r7 ^ r3; // 43 xor + r2 = r2 ^ ds[r5 & mask]; // 44 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r6 = r6 ^ t_; } // 45 shfl + r5 = r5 ^ r3; // 46 xor + r7 = r7 + r6 + ((((sel >> 22u) & 1u) != 0u) ? 0x9342df0du : 0x5af3bd5bu); // 47 add + r3 = r3 + r4 + ((((sel >> 23u) & 1u) != 0u) ? 0x485614dbu : 0x9ff95776u); // 48 add + r5 = r5 ^ ds[r3 & mask]; // 49 load + r4 = rotr_var(r4, r5); // 50 rotr + r0 = r1 * r7 + r0; // 51 mad + r0 = r0 * r3; // 52 mul + r5 = rotr_var(r5, r4); // 53 rotr + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r0 = r0 ^ t_; } // 54 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r7, 16u); r0 = r0 ^ t_; } // 55 shfl + r7 = r7 ^ ds[r4 & mask]; // 56 load + r7 = rotr_var(r7, r3); // 57 rotr + r0 = r0 ^ ds[r6 & mask]; // 58 load + r6 = r6 ^ hot[mul_hi(r2, HOT_WORDS)]; // 59 hot + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 1u); r3 = r3 ^ t_; } // 60 shfl + r7 = r7 - r5; // 61 sub + r1 = rotl_imm(r1, 4u); // 62 rotl + r4 = r4 ^ ds[r5 & mask]; // 63 load + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} + +#if IGNEUM_EXCHANGE != 0 +// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the +// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a +// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md. +IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) { + if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); } +} +#endif + +// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash. +IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7]; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; } + { uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; } + { uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; } + { uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; } + { uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; } + { uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; } + { uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; } + { uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r6 = r6 | r4; // 0 or + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 1u); r7 = r7 ^ t_; } // 1 shfl + r0 = r0 | r2; // 2 or + { uint t_; IGNEUM_SHFL_XOR(t_, r0, 16u); r3 = r3 ^ t_; } // 3 shfl + r7 = r7 ^ ds[r3 & mask]; // 4 load + r6 = r6 ^ ds[r0 & mask]; // 5 load + r1 = r1 * r4; // 6 mul + r0 = r0 - r1; // 7 sub + r3 = r3 ^ ds[r7 & mask]; // 8 load + r1 = r0 * r3 + r1; // 9 mad + r4 = r4 * r0; // 10 mul + r5 = r5 ^ ds[r4 & mask]; // 11 load + r1 = r1 ^ r7; // 12 xor + r1 = r1 ^ ds[r6 & mask]; // 13 load + r2 = r2 ^ ds[r1 & mask]; // 14 load + r3 = r3 * r1; // 15 mul + r6 = r6 ^ hot[mul_hi(r5, HOT_WORDS)]; // 16 hot + r0 = r0 ^ ds[r2 & mask]; // 17 load + r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x535b545fu : 0xc035a4e6u); // 18 add + r5 = r5 - r0; // 19 sub + r2 = r2 * r4; // 20 mul + r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x925d5e63u : 0x1ecdd2cfu); // 21 add + r3 = r3 ^ r7; // 22 xor + r7 = r7 ^ ds[r2 & mask]; // 23 load + r2 = mul_hi(r2, r5); // 24 mulhi + r5 = rotr_var(r5, r2); // 25 rotr + r3 = r2 * r7 + r3; // 26 mad + r2 = r2 - r0; // 27 sub + r6 = r1 * r5 + r6; // 28 mad + r2 = r2 + r3 + ((((sel >> 16u) & 1u) != 0u) ? 0x46b1b505u : 0xbdc6da76u); // 29 add + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r3 = r3 ^ t_; } // 30 shfl + r5 = r5 ^ ds[r7 & mask]; // 31 load + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 32 hot + r6 = r6 | r3; // 33 or + r3 = r3 ^ hot[mul_hi(r6, HOT_WORDS)]; // 34 hot + { uint t_; IGNEUM_SHFL_XOR(t_, r0, 2u); r4 = r4 ^ t_; } // 35 shfl + r5 = r5 ^ r2; // 36 xor + r3 = r3 ^ ds[r4 & mask]; // 37 load + r4 = r4 ^ ds[r3 & mask]; // 38 load + r1 = rotr_var(r1, r4); // 39 rotr + r3 = r3 + r6 + ((((sel >> 28u) & 1u) != 0u) ? 0x6328cb2cu : 0xbc3ff65fu); // 40 add + r5 = r5 ^ r1; // 41 xor + r5 = r5 | r0; // 42 or + r7 = r7 ^ r3; // 43 xor + r2 = r2 ^ ds[r5 & mask]; // 44 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r6 = r6 ^ t_; } // 45 shfl + r5 = r5 ^ r3; // 46 xor + r7 = r7 + r6 + ((((sel >> 22u) & 1u) != 0u) ? 0x9342df0du : 0x5af3bd5bu); // 47 add + r3 = r3 + r4 + ((((sel >> 23u) & 1u) != 0u) ? 0x485614dbu : 0x9ff95776u); // 48 add + r5 = r5 ^ ds[r3 & mask]; // 49 load + r4 = rotr_var(r4, r5); // 50 rotr + r0 = r1 * r7 + r0; // 51 mad + r0 = r0 * r3; // 52 mul + r5 = rotr_var(r5, r4); // 53 rotr + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r0 = r0 ^ t_; } // 54 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r7, 16u); r0 = r0 ^ t_; } // 55 shfl + r7 = r7 ^ ds[r4 & mask]; // 56 load + r7 = rotr_var(r7, r3); // 57 rotr + r0 = r0 ^ ds[r6 & mask]; // 58 load + r6 = r6 ^ hot[mul_hi(r2, HOT_WORDS)]; // 59 hot + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 1u); r3 = r3 ^ t_; } // 60 shfl + r7 = r7 - r5; // 61 sub + r1 = rotl_imm(r1, 4u); // 62 rotl + r4 = r4 ^ ds[r5 & mask]; // 63 load + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot32k4a/kernel_bound.cu b/proto-cuda/packs-ca2-hot/hot32k4a/kernel_bound.cu new file mode 100644 index 000000000..1e164c426 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4a/kernel_bound.cu @@ -0,0 +1,125 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW. +// Host declarations (also in program_bound.h if present): +// struct IgneumInitWords { uint32_t w[8]; }; +// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, +// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps); +// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps); +#include +#include +#include "program.h" + +struct IgneumInitWords { uint32_t w[8]; }; + +// Hot table (32 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x00800000u +__device__ __forceinline__ uint32_t splitmix32(uint32_t x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } +__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } + +__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw, const uint32_t* hot) { + uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x; + uint32_t nonce = baseNonce + gid; + uint32_t r0, r1, r2, r3, r4, r5, r6, r7; + { uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; } + { uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; } + { uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; } + { uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; } + { uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; } + { uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; } + { uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; } + { uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; } + + for (uint32_t it = 0u; it < 8u; ++it) { + uint32_t sel = r0; + r6 = r6 | r4; // 0 or + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r4, 1); // 1 shfl + r0 = r0 | r2; // 2 or + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r0, 16); // 3 shfl + r7 = r7 ^ ds[r3 & mask]; // 4 load + r6 = r6 ^ ds[r0 & mask]; // 5 load + r1 = r1 * r4; // 6 mul + r0 = r0 - r1; // 7 sub + r3 = r3 ^ ds[r7 & mask]; // 8 load + r1 = r0 * r3 + r1; // 9 mad + r4 = r4 * r0; // 10 mul + r5 = r5 ^ ds[r4 & mask]; // 11 load + r1 = r1 ^ r7; // 12 xor + r1 = r1 ^ ds[r6 & mask]; // 13 load + r2 = r2 ^ ds[r1 & mask]; // 14 load + r3 = r3 * r1; // 15 mul + r6 = r6 ^ hot[__umulhi(r5, HOT_WORDS)]; // 16 hot + r0 = r0 ^ ds[r2 & mask]; // 17 load + r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x535b545fu : 0xc035a4e6u); // 18 add + r5 = r5 - r0; // 19 sub + r2 = r2 * r4; // 20 mul + r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x925d5e63u : 0x1ecdd2cfu); // 21 add + r3 = r3 ^ r7; // 22 xor + r7 = r7 ^ ds[r2 & mask]; // 23 load + r2 = __umulhi(r2, r5); // 24 mulhi + r5 = rotr_var(r5, r2); // 25 rotr + r3 = r2 * r7 + r3; // 26 mad + r2 = r2 - r0; // 27 sub + r6 = r1 * r5 + r6; // 28 mad + r2 = r2 + r3 + ((((sel >> 16u) & 1u) != 0u) ? 0x46b1b505u : 0xbdc6da76u); // 29 add + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r2, 8); // 30 shfl + r5 = r5 ^ ds[r7 & mask]; // 31 load + r2 = r2 ^ hot[__umulhi(r0, HOT_WORDS)]; // 32 hot + r6 = r6 | r3; // 33 or + r3 = r3 ^ hot[__umulhi(r6, HOT_WORDS)]; // 34 hot + r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r0, 2); // 35 shfl + r5 = r5 ^ r2; // 36 xor + r3 = r3 ^ ds[r4 & mask]; // 37 load + r4 = r4 ^ ds[r3 & mask]; // 38 load + r1 = rotr_var(r1, r4); // 39 rotr + r3 = r3 + r6 + ((((sel >> 28u) & 1u) != 0u) ? 0x6328cb2cu : 0xbc3ff65fu); // 40 add + r5 = r5 ^ r1; // 41 xor + r5 = r5 | r0; // 42 or + r7 = r7 ^ r3; // 43 xor + r2 = r2 ^ ds[r5 & mask]; // 44 load + r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 45 shfl + r5 = r5 ^ r3; // 46 xor + r7 = r7 + r6 + ((((sel >> 22u) & 1u) != 0u) ? 0x9342df0du : 0x5af3bd5bu); // 47 add + r3 = r3 + r4 + ((((sel >> 23u) & 1u) != 0u) ? 0x485614dbu : 0x9ff95776u); // 48 add + r5 = r5 ^ ds[r3 & mask]; // 49 load + r4 = rotr_var(r4, r5); // 50 rotr + r0 = r1 * r7 + r0; // 51 mad + r0 = r0 * r3; // 52 mul + r5 = rotr_var(r5, r4); // 53 rotr + r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 1); // 54 shfl + r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r7, 16); // 55 shfl + r7 = r7 ^ ds[r4 & mask]; // 56 load + r7 = rotr_var(r7, r3); // 57 rotr + r0 = r0 ^ ds[r6 & mask]; // 58 load + r6 = r6 ^ hot[__umulhi(r2, HOT_WORDS)]; // 59 hot + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r2, 1); // 60 shfl + r7 = r7 - r5; // 61 sub + r1 = rotl_imm(r1, 4u); // 62 rotl + r4 = r4 ^ ds[r5 & mask]; // 63 load + } + uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo; +} + +cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, + IgneumInitWords iw, const uint32_t* hot, uint32_t nonces, uint32_t blockWarps) { + if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue; + uint32_t block = 32u * blockWarps; + if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue; + igneum_hash_bound<<>>(ds, out, baseNonce, mask, iw, hot); + return cudaGetLastError(); +} + +cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) { + cudaFuncAttributes attr; + cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound); + if (e != cudaSuccess) return e; + *numRegs = attr.numRegs; + return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0); +} diff --git a/proto-cuda/packs-ca2-hot/hot32k4a/memhard.h b/proto-cuda/packs-ca2-hot/hot32k4a/memhard.h new file mode 100644 index 000000000..4e0eee822 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4a/memhard.h @@ -0,0 +1,129 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against. +// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference). +// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C. +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif +#if defined(__CUDACC__) +#define IGNEUM_HD __host__ __device__ __forceinline__ +#elif defined(_MSC_VER) && !defined(__cplusplus) +#define IGNEUM_HD static __inline +#else +#define IGNEUM_HD static inline +#endif +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) { + for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint32_t r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) { + uint32_t prev[16]; uint32_t x[16]; uint32_t y[16]; + for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +IGNEUM_HD void mh_item(const uint32_t* cache, uint32_t t, uint32_t* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint32_t r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +IGNEUM_HD uint32_t mh_word(const uint32_t* cache, uint32_t w) { uint32_t s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Hot table (docs/plans/hot-table.md): 32 MiB = 8192 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +IGNEUM_HD void ht_segment(uint32_t* hot, uint32_t seg) { + uint32_t prev[16]; uint32_t x[16]; uint32_t y[16]; + for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + uint32_t* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} diff --git a/proto-cuda/packs-ca2-hot/hot32k4a/memhard.metal b/proto-cuda/packs-ca2-hot/hot32k4a/memhard.metal new file mode 100644 index 000000000..4325ec795 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4a/memhard.metal @@ -0,0 +1,131 @@ +#include +using namespace metal; +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +inline void mh_chacha_block(const thread uint* x, thread uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +inline void mh_cache_segment(device uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +inline void mh_mixer(thread uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +inline void mh_item(device const uint* cache, uint t, thread uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +inline uint mh_word(device const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// One thread per segment (2^16 threads). +kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) { + mh_cache_segment(cache, gid); +} +// One thread per 64-byte item (dataset words / 16 threads). +kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]], + uint gid [[thread_position_in_grid]]) { + uint s[16]; + mh_item(cache, gid, s); + device uint* d = dataset + gid * 16u; + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; +} + +// Hot table (docs/plans/hot-table.md): 32 MiB = 8192 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +inline void ht_segment(device uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + device uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one thread per segment (IGNEUM_HOT_SEGMENTS threads). +kernel void igneum_hot_fill(device uint* hot [[buffer(0)]], uint gid [[thread_position_in_grid]]) { + ht_segment(hot, gid); +} diff --git a/proto-cuda/packs-ca2-hot/hot32k4a/program.h b/proto-cuda/packs-ca2-hot/hot32k4a/program.h new file mode 100644 index 000000000..d79dcf8bf --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4a/program.h @@ -0,0 +1,70 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Program metadata for host.cu plus the launch wrappers defined in kernel.cu. +// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros. +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif +#ifndef IGNEUM_NO_CUDA +#include +#endif + +#define IGNEUM_SEED_STRING "igneum-genesis" +#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973" +#define IGNEUM_GENERATOR 2 +#define IGNEUM_PROGRAM_ATTEMPT 0 +#define IGNEUM_PROGRAM_ID 0xd40c36ab07d82002ull +#define IGNEUM_DAY_STRING "2026-10-03" +#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033" +#define IGNEUM_DAY0 0x3067619fu +#define IGNEUM_DAY1 0x3c269176u +#define IGNEUM_DATASET_LOG2 28 +#define IGNEUM_MASK 0x0fffffffu +#define IGNEUM_LANES 32 +#define IGNEUM_ITERATIONS 8 +#define IGNEUM_INSTR_COUNT 64 +#define IGNEUM_LOADS_PER_HASH 160 +#define IGNEUM_WIDE_LOADS_PER_HASH 0 +#define IGNEUM_OP_MIX "load=16 shfl=8 add=6 xor=6 mul=5 rotr=5 hot=4 mad=4 or=4 sub=4 mulhi=1 rotl=1" +// Class v3 construction (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): version 2 loads; the dataset item +// derivation applies the mixer IGNEUM_MIXER_MULT times per round (memhard.h), and the cache follows the growth rule. +#define IGNEUM_LOAD_CLASS "hot32k4a" +#define IGNEUM_LOAD_SLOTS 20 +#define IGNEUM_LOAD_MIX { 100, 0, 0 } +#define IGNEUM_LOAD_WIDTH_COUNTS { 16, 0, 0 } // loads of 4, 16, 64 bytes per program +#define IGNEUM_BYTES_PER_HASH 512 +#define IGNEUM_FOLD_ROT 11 +#define IGNEUM_FOLD_MUL 0x9e3779b1u +// Hot table (hot-table experiment, 5 October 2026, docs/plans/hot-table.md): NOT the lottery hash. IGNEUM_HOT_SLOTS of the +// 16 load slots read H[mulhi(src, IGNEUM_HOT_WORDS)], H = IGNEUM_HOT_MB MiB of chained ChaCha12 lines (the cache chain) under +// KH = seed_words("igneum-hot/" || epoch seed bytes) and the tag "Igne" "umHT"; filled by igneum_hot_fill once per epoch. +#define IGNEUM_HOT_MB 32 +#define IGNEUM_HOT_WORDS 0x00800000u +#define IGNEUM_HOT_SEGMENTS 8192u +#define IGNEUM_HOT_SLOTS 4 // hot loads per program (32 per hash), added beside the dataset loads (16 of them) +#define IGNEUM_HOT_ADDED 1 +#define IGNEUM_HOT_KEY_INIT { 0x3a48bef5u, 0x6b54b1a1u, 0x9ff897c3u, 0x7d5d85e4u, 0xcd35379fu, 0x9d76bb86u, 0xbe2affb1u, 0x0f8f80b4u } +// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h) +#define IGNEUM_DATASET_MODE 1 + +#define IGNEUM_SEEDW_INIT { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u } +#define IGNEUM_KEY_INIT { 0x3067619fu, 0x3c269176u, 0x84a03b03u, 0xf8c63294u, 0xff977c5bu, 0xe60def3eu, 0x63630141u, 0xb8fbcb58u } +#define IGNEUM_CACHE_LOG2_WORDS 26 +#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6 +#define IGNEUM_CACHE_SEGMENTS 65536u +#define IGNEUM_ITEM_ROUNDS 8 +#define IGNEUM_MIX_ROT_INIT { 20u, 20u, 19u, 4u, 26u, 3u, 3u, 27u } +#define IGNEUM_MIX_MUL_INIT { 0x42146205u, 0x52cbe0fbu, 0x7ecf4a03u, 0x6728907fu, 0xd81d9751u, 0x132952c3u, 0xf60de277u, 0x05358035u, 0xbaf6499du, 0xe4db9667u, 0x3e98f45du, 0xd0004eddu, 0x2691630du, 0x9beb3bcfu, 0xab310379u, 0x99cfb423u } +#define IGNEUM_MIX_RC_INIT { 0xbab68293u, 0xcc162340u, 0x6ce151ccu, 0xe62b8997u, 0xc9c80297u, 0xf74a1654u, 0x3d704af5u, 0x3cf522b7u, 0x2b9cac04u, 0xa880ac10u, 0x13e5dd1du, 0x6fc3e233u, 0x2d83eeacu, 0x9006e8bfu, 0x2c4b5362u, 0x31b49ee2u } + +#ifndef IGNEUM_NO_CUDA +// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError(). +cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments); +cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems); +cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments); +cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot, + uint32_t nonces, uint32_t blockWarps); +cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps); +#endif diff --git a/proto-cuda/packs-ca2-hot/hot32k4a/program.json b/proto-cuda/packs-ca2-hot/hot32k4a/program.json new file mode 100644 index 000000000..0d46fc054 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4a/program.json @@ -0,0 +1,129 @@ +{ + "format": "igneum-program-pack-3", + "generator": 2, + "attempt": 0, + "program_id": "0xd40c36ab07d82002", + "program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32", + "dataset_mode": "memory-hard", + "seed": "igneum-genesis", + "seed_bytes": "69676e65756d2d67656e65736973", + "seed_words": ["0x67a9a7be", "0x1a155b25", "0xfddfb732", "0x4b5af2e8", "0xc55caf33", "0xa27c13b7", "0x06628a48", "0x03852469"], + "seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32", + "generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried", + "lanes": 32, + "registers": 8, + "iterations": 8, + "instruction_count": 64, + "loads_per_hash": 160, + "load_class": "hot32k4a", + "load_slots": 20, + "load_mix_percent_4_16_64": [100, 0, 0], + "load_width_counts_4_16_64": [16, 0, 0], + "bytes_per_hash": 512, + "wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots", + "hot_table": {"mb": 32, "words": 8388608, "segments": 8192, "slots": 4, "form": "added: k load slots added beside the class's, the dataset loads unchanged", "dataset_slots": 16, "hot_loads_per_hash": 32, "key": ["0x3a48bef5", "0x6b54b1a1", "0x9ff897c3", "0x7d5d85e4", "0xcd35379f", "0x9d76bb86", "0xbe2affb1", "0x0f8f80b4"], "key_derivation": "seed_words_from_bytes('igneum-hot/' || seed_bytes), the same for every attempt of the epoch", "tag": ["0x49676e65", "0x756d4854"], "chain": "the cache chain of dataset.cache with the hot key and tag: in_j = prev ^ (sigma || key || seg || j || tag); line_j = block(in_j); prev_0 = 0", "load": "dst = dst ^ hot[mulhi(src, words)] (the high 32 bits of the 64-bit product; the index lies in [0, words) for any size)", "slots_rule": "the first k load slots in the Fisher-Yates draw order after the scratch slots (a uniform k-subset); no extra draw, so with version 2 widths and no scratch the program is the version 2 program with k loads redirected", "acceptance_stand_in": "dataset_elem(mulhi(src, words), seed_words[2], seed_words[3])", "program_id": "the read-width id with 'hot/' || mb || k appended", "spec": "docs/plans/hot-table.md"}, + "op_mix": {"load": 16, "shfl": 8, "add": 6, "xor": 6, "mul": 5, "rotr": 5, "hot": 4, "mad": 4, "or": 4, "sub": 4, "mulhi": 1, "rotl": 1}, + "register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]", + "splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16", + "iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order", + "output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo", + "op_semantics": { + "add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)", + "sub": "dst = dst - src", + "mul": "dst = dst * src (low 32)", + "mulhi": "dst = high 32 bits of dst * src", + "xor": "dst = dst ^ src", + "or": "dst = dst | src", + "rotl": "dst = rotl(dst, rot), rot in 1..31", + "rotr": "dst = rotr(dst, src & 31)", + "mad": "dst = src * src2 + dst", + "shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp", + "load": "dst = dst ^ dataset[src & dataset.mask]", + "wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)", + "hot": "dst = dst ^ hot[mulhi(src, hot_table.words)] (hot-table experiment)" + }, + "dataset": { + "log2_words": 28, + "bytes": 1073741824, + "mask": "0x0fffffff", + "day": "2026-10-03", + "day_bytes": "6461792f323032362d31302d3033", + "day_words_from": "seed_words_from_bytes(day_bytes)", + "d0": "0x3067619f", + "d1": "0x3c269176", + "mode": "memory-hard", + "spec": "proto-metal/MEMHARD.md", + "key": ["0x3067619f", "0x3c269176", "0x84a03b03", "0xf8c63294", "0xff977c5b", "0xe60def3e", "0x63630141", "0xb8fbcb58"], + "key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]", + "cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"}, + "mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [20, 20, 19, 4, 26, 3, 3, 27], "mul": ["0x42146205", "0x52cbe0fb", "0x7ecf4a03", "0x6728907f", "0xd81d9751", "0x132952c3", "0xf60de277", "0x05358035", "0xbaf6499d", "0xe4db9667", "0x3e98f45d", "0xd0004edd", "0x2691630d", "0x9beb3bcf", "0xab310379", "0x99cfb423"], "rc": ["0xbab68293", "0xcc162340", "0x6ce151cc", "0xe62b8997", "0xc9c80297", "0xf74a1654", "0x3d704af5", "0x3cf522b7", "0x2b9cac04", "0xa880ac10", "0x13e5dd1d", "0x6fc3e233", "0x2d83eeac", "0x9006e8bf", "0x2c4b5362", "0x31b49ee2"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"}, + "item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: s = M_r(s); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then s = M_8(s); item(t) = s", + "word": "dataset[w] = item(w >> 4)[w & 15]" + }, + "instructions": [ + {"i": 0, "op": "or", "dst": 6, "src": 4, "src2": 2, "imm": "0x535c5dd3", "imm2": "0xf8694615", "rot": 3, "bit": 1, "mask": 16, "width": 1}, + {"i": 1, "op": "shfl", "dst": 7, "src": 4, "src2": 6, "imm": "0x87bc9ee4", "imm2": "0xfdb0c856", "rot": 9, "bit": 25, "mask": 1, "width": 1}, + {"i": 2, "op": "or", "dst": 0, "src": 2, "src2": 4, "imm": "0xaf91f0c8", "imm2": "0xe2d5d1fa", "rot": 1, "bit": 24, "mask": 2, "width": 1}, + {"i": 3, "op": "shfl", "dst": 3, "src": 0, "src2": 0, "imm": "0x3044ba32", "imm2": "0x7b1a7ffe", "rot": 14, "bit": 10, "mask": 16, "width": 1}, + {"i": 4, "op": "load", "dst": 7, "src": 3, "src2": 6, "imm": "0x5d1ca2a2", "imm2": "0xe2481807", "rot": 24, "bit": 3, "mask": 1, "width": 1}, + {"i": 5, "op": "load", "dst": 6, "src": 0, "src2": 6, "imm": "0xaba3dbaa", "imm2": "0x987c017a", "rot": 16, "bit": 30, "mask": 1, "width": 1}, + {"i": 6, "op": "mul", "dst": 1, "src": 4, "src2": 6, "imm": "0x12a93d05", "imm2": "0x10761047", "rot": 17, "bit": 17, "mask": 4, "width": 1}, + {"i": 7, "op": "sub", "dst": 0, "src": 1, "src2": 0, "imm": "0x8884e389", "imm2": "0x31009b67", "rot": 5, "bit": 30, "mask": 1, "width": 1}, + {"i": 8, "op": "load", "dst": 3, "src": 7, "src2": 5, "imm": "0xfa6358b3", "imm2": "0xfea9038f", "rot": 17, "bit": 18, "mask": 2, "width": 1}, + {"i": 9, "op": "mad", "dst": 1, "src": 0, "src2": 3, "imm": "0xb154ae4f", "imm2": "0xe85f13f6", "rot": 6, "bit": 18, "mask": 16, "width": 1}, + {"i": 10, "op": "mul", "dst": 4, "src": 0, "src2": 4, "imm": "0x9edffbc3", "imm2": "0x643812d4", "rot": 24, "bit": 25, "mask": 4, "width": 1}, + {"i": 11, "op": "load", "dst": 5, "src": 4, "src2": 1, "imm": "0x163dbb1b", "imm2": "0x20c1f743", "rot": 27, "bit": 25, "mask": 16, "width": 1}, + {"i": 12, "op": "xor", "dst": 1, "src": 7, "src2": 4, "imm": "0x9cadb4f8", "imm2": "0xd9da785b", "rot": 8, "bit": 26, "mask": 4, "width": 1}, + {"i": 13, "op": "load", "dst": 1, "src": 6, "src2": 6, "imm": "0x1bb1b429", "imm2": "0x33d1e391", "rot": 20, "bit": 19, "mask": 16, "width": 1}, + {"i": 14, "op": "load", "dst": 2, "src": 1, "src2": 6, "imm": "0xa672cdd3", "imm2": "0x59a4829c", "rot": 22, "bit": 13, "mask": 16, "width": 1}, + {"i": 15, "op": "mul", "dst": 3, "src": 1, "src2": 4, "imm": "0xe3c4bf4d", "imm2": "0x028b4d37", "rot": 11, "bit": 28, "mask": 8, "width": 1}, + {"i": 16, "op": "hot", "dst": 6, "src": 5, "src2": 7, "imm": "0x71fbe6f2", "imm2": "0x61b6261e", "rot": 15, "bit": 9, "mask": 1, "width": 1}, + {"i": 17, "op": "load", "dst": 0, "src": 2, "src2": 5, "imm": "0x8e516fd3", "imm2": "0x2aa480c2", "rot": 19, "bit": 1, "mask": 8, "width": 1}, + {"i": 18, "op": "add", "dst": 0, "src": 7, "src2": 6, "imm": "0xc035a4e6", "imm2": "0x535b545f", "rot": 18, "bit": 26, "mask": 8, "width": 1}, + {"i": 19, "op": "sub", "dst": 5, "src": 0, "src2": 2, "imm": "0xde38f954", "imm2": "0xde1507bc", "rot": 31, "bit": 21, "mask": 4, "width": 1}, + {"i": 20, "op": "mul", "dst": 2, "src": 4, "src2": 1, "imm": "0x3f64b7c9", "imm2": "0x1f07754a", "rot": 23, "bit": 28, "mask": 4, "width": 1}, + {"i": 21, "op": "add", "dst": 2, "src": 1, "src2": 1, "imm": "0x1ecdd2cf", "imm2": "0x925d5e63", "rot": 6, "bit": 6, "mask": 8, "width": 1}, + {"i": 22, "op": "xor", "dst": 3, "src": 7, "src2": 3, "imm": "0x63c7f533", "imm2": "0xd312164d", "rot": 4, "bit": 14, "mask": 2, "width": 1}, + {"i": 23, "op": "load", "dst": 7, "src": 2, "src2": 5, "imm": "0x6d3ddc5a", "imm2": "0x433d8c2a", "rot": 28, "bit": 7, "mask": 2, "width": 1}, + {"i": 24, "op": "mulhi", "dst": 2, "src": 5, "src2": 0, "imm": "0xc7e9887a", "imm2": "0x19ec898f", "rot": 14, "bit": 9, "mask": 1, "width": 1}, + {"i": 25, "op": "rotr", "dst": 5, "src": 2, "src2": 3, "imm": "0x97acf9f2", "imm2": "0xc7fcfc8f", "rot": 8, "bit": 25, "mask": 1, "width": 1}, + {"i": 26, "op": "mad", "dst": 3, "src": 2, "src2": 7, "imm": "0x400383b6", "imm2": "0xe9bab735", "rot": 16, "bit": 18, "mask": 1, "width": 1}, + {"i": 27, "op": "sub", "dst": 2, "src": 0, "src2": 7, "imm": "0x8ea6cd8d", "imm2": "0x55b67f9f", "rot": 19, "bit": 8, "mask": 8, "width": 1}, + {"i": 28, "op": "mad", "dst": 6, "src": 1, "src2": 5, "imm": "0x17723e5a", "imm2": "0x00779664", "rot": 19, "bit": 2, "mask": 4, "width": 1}, + {"i": 29, "op": "add", "dst": 2, "src": 3, "src2": 4, "imm": "0xbdc6da76", "imm2": "0x46b1b505", "rot": 28, "bit": 16, "mask": 16, "width": 1}, + {"i": 30, "op": "shfl", "dst": 3, "src": 2, "src2": 3, "imm": "0xe586702c", "imm2": "0xd83a1462", "rot": 24, "bit": 6, "mask": 8, "width": 1}, + {"i": 31, "op": "load", "dst": 5, "src": 7, "src2": 1, "imm": "0x1cddad42", "imm2": "0xb0c6b0d0", "rot": 25, "bit": 25, "mask": 8, "width": 1}, + {"i": 32, "op": "hot", "dst": 2, "src": 0, "src2": 0, "imm": "0x2abcfdc6", "imm2": "0xfa4cc809", "rot": 22, "bit": 24, "mask": 2, "width": 1}, + {"i": 33, "op": "or", "dst": 6, "src": 3, "src2": 1, "imm": "0x61c9a38d", "imm2": "0x86597500", "rot": 15, "bit": 8, "mask": 2, "width": 1}, + {"i": 34, "op": "hot", "dst": 3, "src": 6, "src2": 7, "imm": "0xc37723fa", "imm2": "0xf3b024da", "rot": 16, "bit": 27, "mask": 16, "width": 1}, + {"i": 35, "op": "shfl", "dst": 4, "src": 0, "src2": 5, "imm": "0x84cad367", "imm2": "0xcc7972c4", "rot": 21, "bit": 17, "mask": 2, "width": 1}, + {"i": 36, "op": "xor", "dst": 5, "src": 2, "src2": 0, "imm": "0xc24a7d70", "imm2": "0x6591d24c", "rot": 21, "bit": 24, "mask": 1, "width": 1}, + {"i": 37, "op": "load", "dst": 3, "src": 4, "src2": 5, "imm": "0xd404cfe8", "imm2": "0x144ca538", "rot": 10, "bit": 11, "mask": 2, "width": 1}, + {"i": 38, "op": "load", "dst": 4, "src": 3, "src2": 3, "imm": "0x04b0080c", "imm2": "0xb938c290", "rot": 10, "bit": 1, "mask": 8, "width": 1}, + {"i": 39, "op": "rotr", "dst": 1, "src": 4, "src2": 5, "imm": "0x9fe93344", "imm2": "0xff7296e4", "rot": 29, "bit": 14, "mask": 1, "width": 1}, + {"i": 40, "op": "add", "dst": 3, "src": 6, "src2": 2, "imm": "0xbc3ff65f", "imm2": "0x6328cb2c", "rot": 4, "bit": 28, "mask": 16, "width": 1}, + {"i": 41, "op": "xor", "dst": 5, "src": 1, "src2": 1, "imm": "0xdcf4a02e", "imm2": "0x5ee3a976", "rot": 17, "bit": 0, "mask": 8, "width": 1}, + {"i": 42, "op": "or", "dst": 5, "src": 0, "src2": 1, "imm": "0x44846c7a", "imm2": "0x67338877", "rot": 27, "bit": 18, "mask": 1, "width": 1}, + {"i": 43, "op": "xor", "dst": 7, "src": 3, "src2": 3, "imm": "0x1b053acf", "imm2": "0x32e2d23d", "rot": 31, "bit": 23, "mask": 8, "width": 1}, + {"i": 44, "op": "load", "dst": 2, "src": 5, "src2": 4, "imm": "0xf8662282", "imm2": "0x10bb9e30", "rot": 8, "bit": 6, "mask": 2, "width": 1}, + {"i": 45, "op": "shfl", "dst": 6, "src": 4, "src2": 4, "imm": "0x87d9ef84", "imm2": "0x49087d74", "rot": 1, "bit": 2, "mask": 8, "width": 1}, + {"i": 46, "op": "xor", "dst": 5, "src": 3, "src2": 1, "imm": "0x74a59b7d", "imm2": "0xc4766ff1", "rot": 30, "bit": 3, "mask": 16, "width": 1}, + {"i": 47, "op": "add", "dst": 7, "src": 6, "src2": 2, "imm": "0x5af3bd5b", "imm2": "0x9342df0d", "rot": 26, "bit": 22, "mask": 2, "width": 1}, + {"i": 48, "op": "add", "dst": 3, "src": 4, "src2": 5, "imm": "0x9ff95776", "imm2": "0x485614db", "rot": 27, "bit": 23, "mask": 1, "width": 1}, + {"i": 49, "op": "load", "dst": 5, "src": 3, "src2": 1, "imm": "0xd022a812", "imm2": "0xfbe147d6", "rot": 15, "bit": 29, "mask": 4, "width": 1}, + {"i": 50, "op": "rotr", "dst": 4, "src": 5, "src2": 7, "imm": "0x7e099325", "imm2": "0x40d55d10", "rot": 29, "bit": 15, "mask": 16, "width": 1}, + {"i": 51, "op": "mad", "dst": 0, "src": 1, "src2": 7, "imm": "0xd8ab8843", "imm2": "0x7f842b90", "rot": 16, "bit": 9, "mask": 1, "width": 1}, + {"i": 52, "op": "mul", "dst": 0, "src": 3, "src2": 6, "imm": "0x4c30250b", "imm2": "0x9ee1681f", "rot": 20, "bit": 3, "mask": 8, "width": 1}, + {"i": 53, "op": "rotr", "dst": 5, "src": 4, "src2": 4, "imm": "0xc1c15026", "imm2": "0x588915e7", "rot": 1, "bit": 29, "mask": 16, "width": 1}, + {"i": 54, "op": "shfl", "dst": 0, "src": 6, "src2": 4, "imm": "0x636a9dc4", "imm2": "0xac023d9b", "rot": 22, "bit": 29, "mask": 1, "width": 1}, + {"i": 55, "op": "shfl", "dst": 0, "src": 7, "src2": 1, "imm": "0x5c933bd0", "imm2": "0x2baec8c9", "rot": 25, "bit": 27, "mask": 16, "width": 1}, + {"i": 56, "op": "load", "dst": 7, "src": 4, "src2": 7, "imm": "0xf62515d5", "imm2": "0x4a164f9f", "rot": 28, "bit": 30, "mask": 2, "width": 1}, + {"i": 57, "op": "rotr", "dst": 7, "src": 3, "src2": 7, "imm": "0x15ef64da", "imm2": "0x7149e3c9", "rot": 28, "bit": 30, "mask": 2, "width": 1}, + {"i": 58, "op": "load", "dst": 0, "src": 6, "src2": 2, "imm": "0xcde7100c", "imm2": "0x040fc0cf", "rot": 8, "bit": 28, "mask": 2, "width": 1}, + {"i": 59, "op": "hot", "dst": 6, "src": 2, "src2": 1, "imm": "0xcf8b48be", "imm2": "0x14cb1d3f", "rot": 16, "bit": 19, "mask": 1, "width": 1}, + {"i": 60, "op": "shfl", "dst": 3, "src": 2, "src2": 1, "imm": "0xae1529ee", "imm2": "0x732bc114", "rot": 4, "bit": 23, "mask": 1, "width": 1}, + {"i": 61, "op": "sub", "dst": 7, "src": 5, "src2": 7, "imm": "0x0a8f8168", "imm2": "0xf1c9b0ed", "rot": 29, "bit": 2, "mask": 8, "width": 1}, + {"i": 62, "op": "rotl", "dst": 1, "src": 2, "src2": 3, "imm": "0x9b7914dd", "imm2": "0xd14b33a3", "rot": 4, "bit": 16, "mask": 16, "width": 1}, + {"i": 63, "op": "load", "dst": 4, "src": 5, "src2": 1, "imm": "0xbbaa8e24", "imm2": "0xd15ec6a6", "rot": 24, "bit": 2, "mask": 2, "width": 1} + ] +} diff --git a/proto-cuda/packs-ca2-hot/hot32k4a/program.metal b/proto-cuda/packs-ca2-hot/hot32k4a/program.metal new file mode 100644 index 000000000..9e29a8446 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4a/program.metal @@ -0,0 +1,112 @@ +#include +using namespace metal; + +#define MASK 0x0fffffffu +// Hot table (32 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x00800000u +constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }; + +inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 +inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +kernel void igneum_hash(device const uint* dataset [[buffer(0)]], + device ulong* out [[buffer(1)]], + constant uint& baseNonce [[buffer(2)]], + device const uint* hot [[buffer(3)]], + uint gid [[thread_position_in_grid]]) { + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + { uint x = nonce ^ SEEDW[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ SEEDW[1]; } + { uint x = nonce ^ SEEDW[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ SEEDW[2]; } + { uint x = nonce ^ SEEDW[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ SEEDW[3]; } + { uint x = nonce ^ SEEDW[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ SEEDW[4]; } + { uint x = nonce ^ SEEDW[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ SEEDW[5]; } + { uint x = nonce ^ SEEDW[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ SEEDW[6]; } + { uint x = nonce ^ SEEDW[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ SEEDW[7]; } + { uint x = nonce ^ SEEDW[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ SEEDW[0]; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r6 = r6 | r4; // 0 + r7 = r7 ^ simd_shuffle_xor(r4, (ushort)1); // 1 + r0 = r0 | r2; // 2 + r3 = r3 ^ simd_shuffle_xor(r0, (ushort)16); // 3 + r7 = r7 ^ dataset[r3 & MASK]; // 4 + r6 = r6 ^ dataset[r0 & MASK]; // 5 + r1 = r1 * r4; // 6 + r0 = r0 - r1; // 7 + r3 = r3 ^ dataset[r7 & MASK]; // 8 + r1 = r0 * r3 + r1; // 9 + r4 = r4 * r0; // 10 + r5 = r5 ^ dataset[r4 & MASK]; // 11 + r1 = r1 ^ r7; // 12 + r1 = r1 ^ dataset[r6 & MASK]; // 13 + r2 = r2 ^ dataset[r1 & MASK]; // 14 + r3 = r3 * r1; // 15 + r6 = r6 ^ hot[mulhi(r5, HOT_WORDS)]; // 16 + r0 = r0 ^ dataset[r2 & MASK]; // 17 + r0 = r0 + r7 + select(0xc035a4e6u, 0x535b545fu, ((sel >> 26u) & 1u) != 0u); // 18 + r5 = r5 - r0; // 19 + r2 = r2 * r4; // 20 + r2 = r2 + r1 + select(0x1ecdd2cfu, 0x925d5e63u, ((sel >> 6u) & 1u) != 0u); // 21 + r3 = r3 ^ r7; // 22 + r7 = r7 ^ dataset[r2 & MASK]; // 23 + r2 = mulhi(r2, r5); // 24 + r5 = rotr_var(r5, r2); // 25 + r3 = r2 * r7 + r3; // 26 + r2 = r2 - r0; // 27 + r6 = r1 * r5 + r6; // 28 + r2 = r2 + r3 + select(0xbdc6da76u, 0x46b1b505u, ((sel >> 16u) & 1u) != 0u); // 29 + r3 = r3 ^ simd_shuffle_xor(r2, (ushort)8); // 30 + r5 = r5 ^ dataset[r7 & MASK]; // 31 + r2 = r2 ^ hot[mulhi(r0, HOT_WORDS)]; // 32 + r6 = r6 | r3; // 33 + r3 = r3 ^ hot[mulhi(r6, HOT_WORDS)]; // 34 + r4 = r4 ^ simd_shuffle_xor(r0, (ushort)2); // 35 + r5 = r5 ^ r2; // 36 + r3 = r3 ^ dataset[r4 & MASK]; // 37 + r4 = r4 ^ dataset[r3 & MASK]; // 38 + r1 = rotr_var(r1, r4); // 39 + r3 = r3 + r6 + select(0xbc3ff65fu, 0x6328cb2cu, ((sel >> 28u) & 1u) != 0u); // 40 + r5 = r5 ^ r1; // 41 + r5 = r5 | r0; // 42 + r7 = r7 ^ r3; // 43 + r2 = r2 ^ dataset[r5 & MASK]; // 44 + r6 = r6 ^ simd_shuffle_xor(r4, (ushort)8); // 45 + r5 = r5 ^ r3; // 46 + r7 = r7 + r6 + select(0x5af3bd5bu, 0x9342df0du, ((sel >> 22u) & 1u) != 0u); // 47 + r3 = r3 + r4 + select(0x9ff95776u, 0x485614dbu, ((sel >> 23u) & 1u) != 0u); // 48 + r5 = r5 ^ dataset[r3 & MASK]; // 49 + r4 = rotr_var(r4, r5); // 50 + r0 = r1 * r7 + r0; // 51 + r0 = r0 * r3; // 52 + r5 = rotr_var(r5, r4); // 53 + r0 = r0 ^ simd_shuffle_xor(r6, (ushort)1); // 54 + r0 = r0 ^ simd_shuffle_xor(r7, (ushort)16); // 55 + r7 = r7 ^ dataset[r4 & MASK]; // 56 + r7 = rotr_var(r7, r3); // 57 + r0 = r0 ^ dataset[r6 & MASK]; // 58 + r6 = r6 ^ hot[mulhi(r2, HOT_WORDS)]; // 59 + r3 = r3 ^ simd_shuffle_xor(r2, (ushort)1); // 60 + r7 = r7 - r5; // 61 + r1 = rotl_imm(r1, 4u); // 62 + r4 = r4 ^ dataset[r5 & MASK]; // 63 + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot32k4a/program_bound.metal b/proto-cuda/packs-ca2-hot/hot32k4a/program_bound.metal new file mode 100644 index 000000000..79e8d649e --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4a/program_bound.metal @@ -0,0 +1,114 @@ +#include +using namespace metal; + +#define MASK 0x0fffffffu +// Hot table (32 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x00800000u +constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }; + +inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 +inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW. +kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], + device ulong* out [[buffer(1)]], + constant uint& baseNonce [[buffer(2)]], + constant uint* initw [[buffer(3)]], + device const uint* hot [[buffer(4)]], + uint gid [[thread_position_in_grid]]) { + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + { uint x = nonce ^ initw[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ initw[1]; } + { uint x = nonce ^ initw[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ initw[2]; } + { uint x = nonce ^ initw[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ initw[3]; } + { uint x = nonce ^ initw[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ initw[4]; } + { uint x = nonce ^ initw[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ initw[5]; } + { uint x = nonce ^ initw[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ initw[6]; } + { uint x = nonce ^ initw[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ initw[7]; } + { uint x = nonce ^ initw[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ initw[0]; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r6 = r6 | r4; // 0 + r7 = r7 ^ simd_shuffle_xor(r4, (ushort)1); // 1 + r0 = r0 | r2; // 2 + r3 = r3 ^ simd_shuffle_xor(r0, (ushort)16); // 3 + r7 = r7 ^ dataset[r3 & MASK]; // 4 + r6 = r6 ^ dataset[r0 & MASK]; // 5 + r1 = r1 * r4; // 6 + r0 = r0 - r1; // 7 + r3 = r3 ^ dataset[r7 & MASK]; // 8 + r1 = r0 * r3 + r1; // 9 + r4 = r4 * r0; // 10 + r5 = r5 ^ dataset[r4 & MASK]; // 11 + r1 = r1 ^ r7; // 12 + r1 = r1 ^ dataset[r6 & MASK]; // 13 + r2 = r2 ^ dataset[r1 & MASK]; // 14 + r3 = r3 * r1; // 15 + r6 = r6 ^ hot[mulhi(r5, HOT_WORDS)]; // 16 + r0 = r0 ^ dataset[r2 & MASK]; // 17 + r0 = r0 + r7 + select(0xc035a4e6u, 0x535b545fu, ((sel >> 26u) & 1u) != 0u); // 18 + r5 = r5 - r0; // 19 + r2 = r2 * r4; // 20 + r2 = r2 + r1 + select(0x1ecdd2cfu, 0x925d5e63u, ((sel >> 6u) & 1u) != 0u); // 21 + r3 = r3 ^ r7; // 22 + r7 = r7 ^ dataset[r2 & MASK]; // 23 + r2 = mulhi(r2, r5); // 24 + r5 = rotr_var(r5, r2); // 25 + r3 = r2 * r7 + r3; // 26 + r2 = r2 - r0; // 27 + r6 = r1 * r5 + r6; // 28 + r2 = r2 + r3 + select(0xbdc6da76u, 0x46b1b505u, ((sel >> 16u) & 1u) != 0u); // 29 + r3 = r3 ^ simd_shuffle_xor(r2, (ushort)8); // 30 + r5 = r5 ^ dataset[r7 & MASK]; // 31 + r2 = r2 ^ hot[mulhi(r0, HOT_WORDS)]; // 32 + r6 = r6 | r3; // 33 + r3 = r3 ^ hot[mulhi(r6, HOT_WORDS)]; // 34 + r4 = r4 ^ simd_shuffle_xor(r0, (ushort)2); // 35 + r5 = r5 ^ r2; // 36 + r3 = r3 ^ dataset[r4 & MASK]; // 37 + r4 = r4 ^ dataset[r3 & MASK]; // 38 + r1 = rotr_var(r1, r4); // 39 + r3 = r3 + r6 + select(0xbc3ff65fu, 0x6328cb2cu, ((sel >> 28u) & 1u) != 0u); // 40 + r5 = r5 ^ r1; // 41 + r5 = r5 | r0; // 42 + r7 = r7 ^ r3; // 43 + r2 = r2 ^ dataset[r5 & MASK]; // 44 + r6 = r6 ^ simd_shuffle_xor(r4, (ushort)8); // 45 + r5 = r5 ^ r3; // 46 + r7 = r7 + r6 + select(0x5af3bd5bu, 0x9342df0du, ((sel >> 22u) & 1u) != 0u); // 47 + r3 = r3 + r4 + select(0x9ff95776u, 0x485614dbu, ((sel >> 23u) & 1u) != 0u); // 48 + r5 = r5 ^ dataset[r3 & MASK]; // 49 + r4 = rotr_var(r4, r5); // 50 + r0 = r1 * r7 + r0; // 51 + r0 = r0 * r3; // 52 + r5 = rotr_var(r5, r4); // 53 + r0 = r0 ^ simd_shuffle_xor(r6, (ushort)1); // 54 + r0 = r0 ^ simd_shuffle_xor(r7, (ushort)16); // 55 + r7 = r7 ^ dataset[r4 & MASK]; // 56 + r7 = rotr_var(r7, r3); // 57 + r0 = r0 ^ dataset[r6 & MASK]; // 58 + r6 = r6 ^ hot[mulhi(r2, HOT_WORDS)]; // 59 + r3 = r3 ^ simd_shuffle_xor(r2, (ushort)1); // 60 + r7 = r7 - r5; // 61 + r1 = rotl_imm(r1, 4u); // 62 + r4 = r4 ^ dataset[r5 & MASK]; // 63 + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot32k4a/vectors.h b/proto-cuda/packs-ca2-hot/hot32k4a/vectors.h new file mode 100644 index 000000000..5fe569dcb --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4a/vectors.h @@ -0,0 +1,67 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Expected outputs: igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif + +#define IGNEUM_VEC_WARPS 3 +static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u }; +static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = { + { // base nonce 0 + 0x38c6c58d536902f6ull, 0xeca297b8b15ee71dull, 0x5de01e156c268debull, 0x1e5cd7af09839b5eull, 0x426a782b31120694ull, 0xc8ab7a4e0afb32cbull, 0xcb837c845978ac7dull, 0x23b49de45ebc7c11ull, + 0x7c0c5cf7351c2d8dull, 0xf0b045dc0ee668f2ull, 0x2c5fa9b30927a776ull, 0x2aa40e68443d17b4ull, 0xb26d47f8416fbae3ull, 0x9a89690fe2ec670eull, 0x3c0e6e8eeefadc75ull, 0x051c17f50deeac6bull, + 0x6dea6ee96547b120ull, 0xc77505c9a33bfd3eull, 0x01053c0daf25675eull, 0xfc848d11c52e74b0ull, 0x62c401ad3901bb70ull, 0x814874675dd17b2full, 0x4f9f5bd5ad5654a1ull, 0x4eff35d3051104c4ull, + 0xb43c9692e05e06c9ull, 0x83d5eae233ab4ef8ull, 0x4eec3566c6704968ull, 0x6898ee32326466b7ull, 0xbae78f0e96eaee59ull, 0x6fb1ca15f2821951ull, 0xb921ea490246b277ull, 0x65191db7bb92704dull + }, + { // base nonce 4096 + 0x48173935e10a494aull, 0x066dd67c062c745dull, 0xbbd46811f0e3b431ull, 0xc2bea4945c842815ull, 0x4b551ffd6ce1ecfaull, 0x8b66b40f1f943c26ull, 0x963e4d85b6f859f7ull, 0xf9074c2c4c1ea3dfull, + 0x3f636d3d52790daaull, 0x7262e3045c51b1a4ull, 0x553fa2addcdea901ull, 0xa1b3cd0352d644f0ull, 0x9a552108f5e8d60dull, 0xfa36f2ee4133a5abull, 0xdf948681ab41478aull, 0xa751577e6ae0dcadull, + 0x52021cc6ca69492full, 0x2a00c1d68707da0eull, 0xf4bed9e03a59067full, 0xa771bd7f7474b8b1ull, 0x7e1b2cd186e17fa5ull, 0x14da55f44e6d124aull, 0x75764180011d0091ull, 0xf7ab7903fc13e1a2ull, + 0xa8f872e445ffcf33ull, 0x60ff937b67bb6402ull, 0x6b49537c29d3718eull, 0xf69d979ac3a5d090ull, 0x6a55aa7340b6165dull, 0xdbbcb64c2a24c99cull, 0xa268f115f13fe127ull, 0xcd5fdcbfcaec70c7ull + }, + { // base nonce 1000000 + 0x76bf4ad8a26dbba4ull, 0x2e9fbdd49584b54cull, 0x987ed8813b3f0081ull, 0xf731f37a24cb3aecull, 0x423835c437849ecfull, 0x254e0f1383ecd39cull, 0xbaf5f2d123ded780ull, 0xb80051f68ec6d039ull, + 0xcd1b2835e6a6dc7aull, 0xc4fadc46cf78c3f1ull, 0x9ff6e37471b41d78ull, 0xf80e51b0a2f2a594ull, 0xa2a5e9f0e5f5a987ull, 0xb088c5e41a0249d4ull, 0xe3b2437cd2722245ull, 0x5bfdcfea07b9a065ull, + 0x4e07cfad12ab7371ull, 0xca412ad441fb07baull, 0x602d2b491952aab8ull, 0xb4c1862ac545813bull, 0x1cda1df3eb9bfc85ull, 0x0d215ce8fd579321ull, 0x40882ecd21fe8998ull, 0xb6367f56fdf2043dull, + 0x43722735a16493f8ull, 0x319091f0e4aa9ef3ull, 0x2143e27fbfd051efull, 0xbcaf43b30ae26f99ull, 0xdf31272fcb9eb458ull, 0x845ee6177cbc27c2ull, 0xba4d0c371ca002f0ull, 0x13e29438461f5f80ull + } +}; + +// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (268435455). +static const uint32_t IGNEUM_DS_HEAD[16] = { + 0xffc3cd94u, 0x5920ccd8u, 0x392f44bbu, 0x5e57f67au, 0x2f2bc2a9u, 0x620b0e36u, 0xbdc09014u, 0x436654bfu, + 0x311e0b48u, 0x1abd93adu, 0x59cc7ce8u, 0xee5247b2u, 0x86171fe8u, 0x6d874751u, 0xc9f7728fu, 0x7c2a435du +}; +static const uint32_t IGNEUM_DS_LAST_INDEX = 268435455u; +static const uint32_t IGNEUM_DS_LAST = 0xa33ada72u; +// 64 sampled dataset words (index, value) computed on the Mac. +#define IGNEUM_DS_SAMPLES 64 +static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = { + 59471966u, 217795994u, 208353206u, 42483309u, 172547758u, 148076330u, 183853158u, 214389424u, 267488061u, 169781097u, 184093494u, 153880993u, 84977930u, 46426879u, 3093825u, 225364072u, 44593546u, 260713159u, 168250303u, 52384140u, 223401610u, 45554030u, 95410555u, 175039924u, 79171087u, 267580473u, 24168642u, 37981670u, 171551130u, 195559979u, 204611762u, 140997658u, 138925853u, 86637313u, 20736778u, 219665210u, 160430336u, 264654675u, 8013395u, 228945585u, 213884386u, 104419827u, 44185464u, 142737231u, 99284897u, 132475900u, 61861762u, 132056166u, 262388043u, 91878046u, 117353561u, 124768597u, 71352993u, 190698941u, 46055428u, 55281366u, 165145231u, 106810753u, 171985651u, 232085256u, 159510492u, 40072060u, 209107596u, 39023794u +}; +static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = { + 0xe8b73d94u, 0x337028b5u, 0xafe148c9u, 0xab99f7aeu, 0x434ea619u, 0xd85cb880u, 0x54764c7fu, 0x82c7e420u, 0xedf4cb9eu, 0x9884c959u, 0x223ee793u, 0x3a9ccf69u, 0x81da4fd2u, 0xd6ce8cb9u, 0xe3922dcau, 0x3e7e6bdeu, 0x382a3acau, 0x567e7f7fu, 0x25a0f084u, 0xbfeef128u, 0xe338abfbu, 0x7c3b5280u, 0x909bc5f1u, 0xd8b74b9cu, 0x8e31a22eu, 0x26b5f1d8u, 0x79122c00u, 0xcafc3340u, 0xd5e02ea3u, 0x1aee1afdu, 0xdb090d9au, 0xb049f435u, 0x4954d8bau, 0x03797ba0u, 0x196eefbdu, 0xd153412au, 0xbe5d2c4bu, 0xdaa14f0eu, 0x8e61ed07u, 0x9e9a64c6u, 0x2e29ff36u, 0x392a8589u, 0xb56a5912u, 0xfa6e8b57u, 0xd1a737cbu, 0xb0fa841au, 0xbe1c341fu, 0xe25be0f1u, 0xe937f543u, 0xebab2248u, 0x8e1b607au, 0x202a2fedu, 0x95e2819cu, 0x9c9652d4u, 0x32fedef0u, 0xdecfff82u, 0xcb5d43e5u, 0xb735806au, 0x8905939cu, 0xfbf8472du, 0xada74e5du, 0x7ebdeeeau, 0x0119f2b3u, 0xa9a376b8u +}; +// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^26 words. +static const uint32_t IGNEUM_CACHE_HEAD[16] = { + 0x355a86d2u, 0x7957db1cu, 0xd21772afu, 0x6fc1e09bu, 0xd55ce61du, 0x6e6a278bu, 0xd3f543ceu, 0x223d8e82u, + 0x143ab337u, 0x2e9f05bdu, 0x2eb389bfu, 0x0c6e449eu, 0x5cfa4222u, 0xba6560feu, 0x8e3e1aa4u, 0xdbcc1d53u +}; +static const uint32_t IGNEUM_CACHE_LAST[16] = { + 0x41190d91u, 0xbd277957u, 0x22ddbb49u, 0x6986f207u, 0xdf69a4d6u, 0x26401a3au, 0x818230fbu, 0xc417122du, + 0x3597b211u, 0xb553ce55u, 0xcf39cc0du, 0x3b7fc43au, 0x3fd43b00u, 0x67e1c80eu, 0xffa7ea7du, 0xca2960abu +}; +static const uint64_t IGNEUM_CACHE_FNV64 = 0x48c4f5bf24166b2eull; +// Hot table self-test (docs/plans/hot-table.md): hot[0..15], the last 16 words, and FNV-1a 64 over all IGNEUM_HOT_WORDS words. +static const uint32_t IGNEUM_HOT_HEAD[16] = { + 0x8068cc73u, 0x6036ebf9u, 0xb604cd25u, 0x8ffb840eu, 0xc54074a2u, 0x285c0695u, 0x77512425u, 0xc26a58a7u, + 0x72c88757u, 0xc10fca78u, 0x513825ddu, 0x30d6ccc8u, 0x9a05e7cfu, 0xb9533f50u, 0x4bac3ba0u, 0xa5c19528u +}; +static const uint32_t IGNEUM_HOT_LAST[16] = { + 0x2e5537fcu, 0x5e1daf5au, 0x4068fac1u, 0xe48688a5u, 0x38ff563au, 0xbd50595eu, 0x4fd1cffdu, 0xcbdad89au, + 0xecd4eaebu, 0xfd452ab4u, 0xcb2ba071u, 0xc147b12cu, 0x2863ad2au, 0x0f58974fu, 0xe0804105u, 0x7a2298aau +}; +static const uint64_t IGNEUM_HOT_FNV64 = 0xc1767ba3ef02719full; diff --git a/proto-cuda/packs-ca2-hot/hot32k4a/vectors.json b/proto-cuda/packs-ca2-hot/hot32k4a/vectors.json new file mode 100644 index 000000000..d4fbee877 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot32k4a/vectors.json @@ -0,0 +1,39 @@ +{ + "seed": "igneum-genesis", + "day": "2026-10-03", + "dataset_mode": "memory-hard", + "dataset_log2_words": 28, + "mask": "0x0fffffff", + "lanes": 32, + "source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset", + "warps": [ + {"base_nonce": 0, "expected": [ + "0x38c6c58d536902f6", "0xeca297b8b15ee71d", "0x5de01e156c268deb", "0x1e5cd7af09839b5e", "0x426a782b31120694", "0xc8ab7a4e0afb32cb", "0xcb837c845978ac7d", "0x23b49de45ebc7c11", + "0x7c0c5cf7351c2d8d", "0xf0b045dc0ee668f2", "0x2c5fa9b30927a776", "0x2aa40e68443d17b4", "0xb26d47f8416fbae3", "0x9a89690fe2ec670e", "0x3c0e6e8eeefadc75", "0x051c17f50deeac6b", + "0x6dea6ee96547b120", "0xc77505c9a33bfd3e", "0x01053c0daf25675e", "0xfc848d11c52e74b0", "0x62c401ad3901bb70", "0x814874675dd17b2f", "0x4f9f5bd5ad5654a1", "0x4eff35d3051104c4", + "0xb43c9692e05e06c9", "0x83d5eae233ab4ef8", "0x4eec3566c6704968", "0x6898ee32326466b7", "0xbae78f0e96eaee59", "0x6fb1ca15f2821951", "0xb921ea490246b277", "0x65191db7bb92704d" + ]}, + {"base_nonce": 4096, "expected": [ + "0x48173935e10a494a", "0x066dd67c062c745d", "0xbbd46811f0e3b431", "0xc2bea4945c842815", "0x4b551ffd6ce1ecfa", "0x8b66b40f1f943c26", "0x963e4d85b6f859f7", "0xf9074c2c4c1ea3df", + "0x3f636d3d52790daa", "0x7262e3045c51b1a4", "0x553fa2addcdea901", "0xa1b3cd0352d644f0", "0x9a552108f5e8d60d", "0xfa36f2ee4133a5ab", "0xdf948681ab41478a", "0xa751577e6ae0dcad", + "0x52021cc6ca69492f", "0x2a00c1d68707da0e", "0xf4bed9e03a59067f", "0xa771bd7f7474b8b1", "0x7e1b2cd186e17fa5", "0x14da55f44e6d124a", "0x75764180011d0091", "0xf7ab7903fc13e1a2", + "0xa8f872e445ffcf33", "0x60ff937b67bb6402", "0x6b49537c29d3718e", "0xf69d979ac3a5d090", "0x6a55aa7340b6165d", "0xdbbcb64c2a24c99c", "0xa268f115f13fe127", "0xcd5fdcbfcaec70c7" + ]}, + {"base_nonce": 1000000, "expected": [ + "0x76bf4ad8a26dbba4", "0x2e9fbdd49584b54c", "0x987ed8813b3f0081", "0xf731f37a24cb3aec", "0x423835c437849ecf", "0x254e0f1383ecd39c", "0xbaf5f2d123ded780", "0xb80051f68ec6d039", + "0xcd1b2835e6a6dc7a", "0xc4fadc46cf78c3f1", "0x9ff6e37471b41d78", "0xf80e51b0a2f2a594", "0xa2a5e9f0e5f5a987", "0xb088c5e41a0249d4", "0xe3b2437cd2722245", "0x5bfdcfea07b9a065", + "0x4e07cfad12ab7371", "0xca412ad441fb07ba", "0x602d2b491952aab8", "0xb4c1862ac545813b", "0x1cda1df3eb9bfc85", "0x0d215ce8fd579321", "0x40882ecd21fe8998", "0xb6367f56fdf2043d", + "0x43722735a16493f8", "0x319091f0e4aa9ef3", "0x2143e27fbfd051ef", "0xbcaf43b30ae26f99", "0xdf31272fcb9eb458", "0x845ee6177cbc27c2", "0xba4d0c371ca002f0", "0x13e29438461f5f80" + ]} + ], + "dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"], + "dataset_last_index": 268435455, + "dataset_last": "0xa33ada72", + "dataset_samples": [{"index": 59471966, "value": "0xe8b73d94"}, {"index": 217795994, "value": "0x337028b5"}, {"index": 208353206, "value": "0xafe148c9"}, {"index": 42483309, "value": "0xab99f7ae"}, {"index": 172547758, "value": "0x434ea619"}, {"index": 148076330, "value": "0xd85cb880"}, {"index": 183853158, "value": "0x54764c7f"}, {"index": 214389424, "value": "0x82c7e420"}, {"index": 267488061, "value": "0xedf4cb9e"}, {"index": 169781097, "value": "0x9884c959"}, {"index": 184093494, "value": "0x223ee793"}, {"index": 153880993, "value": "0x3a9ccf69"}, {"index": 84977930, "value": "0x81da4fd2"}, {"index": 46426879, "value": "0xd6ce8cb9"}, {"index": 3093825, "value": "0xe3922dca"}, {"index": 225364072, "value": "0x3e7e6bde"}, {"index": 44593546, "value": "0x382a3aca"}, {"index": 260713159, "value": "0x567e7f7f"}, {"index": 168250303, "value": "0x25a0f084"}, {"index": 52384140, "value": "0xbfeef128"}, {"index": 223401610, "value": "0xe338abfb"}, {"index": 45554030, "value": "0x7c3b5280"}, {"index": 95410555, "value": "0x909bc5f1"}, {"index": 175039924, "value": "0xd8b74b9c"}, {"index": 79171087, "value": "0x8e31a22e"}, {"index": 267580473, "value": "0x26b5f1d8"}, {"index": 24168642, "value": "0x79122c00"}, {"index": 37981670, "value": "0xcafc3340"}, {"index": 171551130, "value": "0xd5e02ea3"}, {"index": 195559979, "value": "0x1aee1afd"}, {"index": 204611762, "value": "0xdb090d9a"}, {"index": 140997658, "value": "0xb049f435"}, {"index": 138925853, "value": "0x4954d8ba"}, {"index": 86637313, "value": "0x03797ba0"}, {"index": 20736778, "value": "0x196eefbd"}, {"index": 219665210, "value": "0xd153412a"}, {"index": 160430336, "value": "0xbe5d2c4b"}, {"index": 264654675, "value": "0xdaa14f0e"}, {"index": 8013395, "value": "0x8e61ed07"}, {"index": 228945585, "value": "0x9e9a64c6"}, {"index": 213884386, "value": "0x2e29ff36"}, {"index": 104419827, "value": "0x392a8589"}, {"index": 44185464, "value": "0xb56a5912"}, {"index": 142737231, "value": "0xfa6e8b57"}, {"index": 99284897, "value": "0xd1a737cb"}, {"index": 132475900, "value": "0xb0fa841a"}, {"index": 61861762, "value": "0xbe1c341f"}, {"index": 132056166, "value": "0xe25be0f1"}, {"index": 262388043, "value": "0xe937f543"}, {"index": 91878046, "value": "0xebab2248"}, {"index": 117353561, "value": "0x8e1b607a"}, {"index": 124768597, "value": "0x202a2fed"}, {"index": 71352993, "value": "0x95e2819c"}, {"index": 190698941, "value": "0x9c9652d4"}, {"index": 46055428, "value": "0x32fedef0"}, {"index": 55281366, "value": "0xdecfff82"}, {"index": 165145231, "value": "0xcb5d43e5"}, {"index": 106810753, "value": "0xb735806a"}, {"index": 171985651, "value": "0x8905939c"}, {"index": 232085256, "value": "0xfbf8472d"}, {"index": 159510492, "value": "0xada74e5d"}, {"index": 40072060, "value": "0x7ebdeeea"}, {"index": 209107596, "value": "0x0119f2b3"}, {"index": 39023794, "value": "0xa9a376b8"}], + "cache_head": ["0x355a86d2", "0x7957db1c", "0xd21772af", "0x6fc1e09b", "0xd55ce61d", "0x6e6a278b", "0xd3f543ce", "0x223d8e82", "0x143ab337", "0x2e9f05bd", "0x2eb389bf", "0x0c6e449e", "0x5cfa4222", "0xba6560fe", "0x8e3e1aa4", "0xdbcc1d53"], + "cache_last_line": ["0x41190d91", "0xbd277957", "0x22ddbb49", "0x6986f207", "0xdf69a4d6", "0x26401a3a", "0x818230fb", "0xc417122d", "0x3597b211", "0xb553ce55", "0xcf39cc0d", "0x3b7fc43a", "0x3fd43b00", "0x67e1c80e", "0xffa7ea7d", "0xca2960ab"], + "cache_fnv1a64": "0x48c4f5bf24166b2e", + "hot_head": ["0x8068cc73", "0x6036ebf9", "0xb604cd25", "0x8ffb840e", "0xc54074a2", "0x285c0695", "0x77512425", "0xc26a58a7", "0x72c88757", "0xc10fca78", "0x513825dd", "0x30d6ccc8", "0x9a05e7cf", "0xb9533f50", "0x4bac3ba0", "0xa5c19528"], + "hot_last_line": ["0x2e5537fc", "0x5e1daf5a", "0x4068fac1", "0xe48688a5", "0x38ff563a", "0xbd50595e", "0x4fd1cffd", "0xcbdad89a", "0xecd4eaeb", "0xfd452ab4", "0xcb2ba071", "0xc147b12c", "0x2863ad2a", "0x0f58974f", "0xe0804105", "0x7a2298aa"], + "hot_fnv1a64": "0xc1767ba3ef02719f" +} diff --git a/proto-cuda/packs-ca2-hot/hot64k2/kernel.cl b/proto-cuda/packs-ca2-hot/hot64k2/kernel.cl new file mode 100644 index 000000000..93db63ec3 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k2/kernel.cl @@ -0,0 +1,305 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal). +// Built from source at runtime by proto-opencl/host.c, which passes these defines: +// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit) +// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default) +// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32 +// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition +// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units; +// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32. +#ifndef IGNEUM_GROUP +#define IGNEUM_GROUP 32 +#endif +#ifndef IGNEUM_EXCHANGE +#define IGNEUM_EXCHANGE 0 +#endif +#ifdef __OPENCL_VERSION__ +#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1))) +#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n] +#if IGNEUM_EXCHANGE == 1 +#ifdef cl_khr_subgroups +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif +#ifdef cl_khr_subgroup_shuffle +#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable +#endif +#elif IGNEUM_EXCHANGE == 2 +#pragma OPENCL EXTENSION cl_intel_subgroups : enable +#endif +#else +// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros. +#include "emu_opencl.h" +#endif + +#if IGNEUM_EXCHANGE == 1 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#elif IGNEUM_EXCHANGE == 2 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#else +// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per +// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane +// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's +// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier. +#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; } +#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; } +#endif + +// Hot table (64 MiB, 2 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +static inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32. +static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); } +// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x. +static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); } +static inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +static inline void mh_chacha_block(const uint* x, uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +static inline void mh_cache_segment(__global uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +static inline void mh_mixer(uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +static inline void mh_item(__global const uint* cache, uint t, uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + __global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item. +// The same constants as memhard.h in this pack (one emitter, three dialects). +__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) { + uint t = (uint)get_global_id(0); + if (t < nItems) { + uint s[16]; + mh_item(cache, t, s); + __global uint* d = ds + ((ulong)t * 16u); + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (docs/plans/hot-table.md): 64 MiB = 16384 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +static inline void ht_segment(__global uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one work-item per segment. +__kernel void igneum_hot_fill(__global uint* hot, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the +// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and +// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). +IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ ds[r2 & mask]; // 4 load + r5 = r5 ^ ds[r7 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl + r1 = mul_hi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = mul_hi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl + r5 = r5 * r0; // 19 mul + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl + r6 = mul_hi(r6, r2); // 22 mulhi + r6 = r6 ^ ds[r1 & mask]; // 23 load + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ ds[r0 & mask]; // 32 load + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[mul_hi(r1, HOT_WORDS)]; // 34 hot + r0 = mul_hi(r0, r5); // 35 mulhi + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl + r7 = r7 ^ ds[r0 & mask]; // 37 load + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ ds[r7 & mask]; // 44 load + r2 = r2 * r3; // 45 mul + r1 = mul_hi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ ds[r2 & mask]; // 59 load + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} + +#if IGNEUM_EXCHANGE != 0 +// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the +// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a +// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md. +IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) { + if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); } +} +#endif diff --git a/proto-cuda/packs-ca2-hot/hot64k2/kernel.cu b/proto-cuda/packs-ca2-hot/hot64k2/kernel.cu new file mode 100644 index 000000000..877ecbf74 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k2/kernel.cu @@ -0,0 +1,179 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal). +// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC. +#include +#include +#include "program.h" +#include "memhard.h" + +// Hot table (64 MiB, 2 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +__device__ __forceinline__ uint32_t splitmix32(uint32_t x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31. +__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } +// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x. +__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) { + uint32_t x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item. +// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host. +__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) { + uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x; + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) { + uint32_t t = blockIdx.x * blockDim.x + threadIdx.x; + if (t < nItems) { + uint32_t s[16]; + mh_item(cache, t, s); + uint32_t* d = ds + (size_t)t * 16u; + for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (ht_segment is in memhard.h): one thread per segment. +__global__ void igneum_hot_fill(uint32_t* hot, uint32_t nSegments) { + uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x; + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every +// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a +// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid. +__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot) { + uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x; + uint32_t nonce = baseNonce + gid; + uint32_t r0, r1, r2, r3, r4, r5, r6, r7; + { uint32_t x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint32_t x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint32_t x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint32_t x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint32_t x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint32_t x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint32_t x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint32_t x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint32_t it = 0u; it < 8u; ++it) { + uint32_t sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ ds[r2 & mask]; // 4 load + r5 = r5 ^ ds[r7 & mask]; // 5 load + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 6 shfl + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 8); // 7 shfl + r1 = __umulhi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = __umulhi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[__umulhi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 18 shfl + r5 = r5 * r0; // 19 mul + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 20 shfl + r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 16); // 21 shfl + r6 = __umulhi(r6, r2); // 22 mulhi + r6 = r6 ^ ds[r1 & mask]; // 23 load + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r6, 2); // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ ds[r0 & mask]; // 32 load + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[__umulhi(r1, HOT_WORDS)]; // 34 hot + r0 = __umulhi(r0, r5); // 35 mulhi + r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // 36 shfl + r7 = r7 ^ ds[r0 & mask]; // 37 load + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 4); // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ ds[r7 & mask]; // 44 load + r2 = r2 * r3; // 45 mul + r1 = __umulhi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ ds[r2 & mask]; // 59 load + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo; +} + +// Host-side launch wrappers. Declared in program.h, called from host.cu. +cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) { + if (nSegments == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nSegments + block - 1u) / block; + igneum_cache_fill<<>>(cache, nSegments); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) { + if (nItems == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nItems + block - 1u) / block; + igneum_build<<>>(ds, cache, nItems); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments) { + if (nSegments == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nSegments + block - 1u) / block; + igneum_hot_fill<<>>(hot, nSegments); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot, + uint32_t nonces, uint32_t blockWarps) { + if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue; + uint32_t block = 32u * blockWarps; + if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue; + igneum_hash<<>>(ds, out, baseNonce, mask, hot); + return cudaGetLastError(); +} + +cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) { + cudaFuncAttributes attr; + cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash); + if (e != cudaSuccess) return e; + *numRegs = attr.numRegs; + return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0); +} diff --git a/proto-cuda/packs-ca2-hot/hot64k2/kernel_bound.cl b/proto-cuda/packs-ca2-hot/hot64k2/kernel_bound.cl new file mode 100644 index 000000000..80ad5f3ce --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k2/kernel_bound.cl @@ -0,0 +1,399 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal). +// Built from source at runtime by proto-opencl/host.c, which passes these defines: +// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit) +// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default) +// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32 +// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition +// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units; +// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32. +#ifndef IGNEUM_GROUP +#define IGNEUM_GROUP 32 +#endif +#ifndef IGNEUM_EXCHANGE +#define IGNEUM_EXCHANGE 0 +#endif +#ifdef __OPENCL_VERSION__ +#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1))) +#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n] +#if IGNEUM_EXCHANGE == 1 +#ifdef cl_khr_subgroups +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif +#ifdef cl_khr_subgroup_shuffle +#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable +#endif +#elif IGNEUM_EXCHANGE == 2 +#pragma OPENCL EXTENSION cl_intel_subgroups : enable +#endif +#else +// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros. +#include "emu_opencl.h" +#endif + +#if IGNEUM_EXCHANGE == 1 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#elif IGNEUM_EXCHANGE == 2 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#else +// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per +// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane +// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's +// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier. +#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; } +#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; } +#endif + +// Hot table (64 MiB, 2 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +static inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32. +static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); } +// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x. +static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); } +static inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +static inline void mh_chacha_block(const uint* x, uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +static inline void mh_cache_segment(__global uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +static inline void mh_mixer(uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +static inline void mh_item(__global const uint* cache, uint t, uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + __global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item. +// The same constants as memhard.h in this pack (one emitter, three dialects). +__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) { + uint t = (uint)get_global_id(0); + if (t < nItems) { + uint s[16]; + mh_item(cache, t, s); + __global uint* d = ds + ((ulong)t * 16u); + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (docs/plans/hot-table.md): 64 MiB = 16384 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +static inline void ht_segment(__global uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one work-item per segment. +__kernel void igneum_hot_fill(__global uint* hot, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the +// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and +// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). +IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ ds[r2 & mask]; // 4 load + r5 = r5 ^ ds[r7 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl + r1 = mul_hi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = mul_hi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl + r5 = r5 * r0; // 19 mul + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl + r6 = mul_hi(r6, r2); // 22 mulhi + r6 = r6 ^ ds[r1 & mask]; // 23 load + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ ds[r0 & mask]; // 32 load + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[mul_hi(r1, HOT_WORDS)]; // 34 hot + r0 = mul_hi(r0, r5); // 35 mulhi + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl + r7 = r7 ^ ds[r0 & mask]; // 37 load + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ ds[r7 & mask]; // 44 load + r2 = r2 * r3; // 45 mul + r1 = mul_hi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ ds[r2 & mask]; // 59 load + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} + +#if IGNEUM_EXCHANGE != 0 +// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the +// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a +// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md. +IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) { + if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); } +} +#endif + +// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash. +IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7]; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; } + { uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; } + { uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; } + { uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; } + { uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; } + { uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; } + { uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; } + { uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ ds[r2 & mask]; // 4 load + r5 = r5 ^ ds[r7 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl + r1 = mul_hi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = mul_hi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl + r5 = r5 * r0; // 19 mul + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl + r6 = mul_hi(r6, r2); // 22 mulhi + r6 = r6 ^ ds[r1 & mask]; // 23 load + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ ds[r0 & mask]; // 32 load + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[mul_hi(r1, HOT_WORDS)]; // 34 hot + r0 = mul_hi(r0, r5); // 35 mulhi + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl + r7 = r7 ^ ds[r0 & mask]; // 37 load + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ ds[r7 & mask]; // 44 load + r2 = r2 * r3; // 45 mul + r1 = mul_hi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ ds[r2 & mask]; // 59 load + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot64k2/kernel_bound.cu b/proto-cuda/packs-ca2-hot/hot64k2/kernel_bound.cu new file mode 100644 index 000000000..c2f89653b --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k2/kernel_bound.cu @@ -0,0 +1,125 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW. +// Host declarations (also in program_bound.h if present): +// struct IgneumInitWords { uint32_t w[8]; }; +// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, +// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps); +// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps); +#include +#include +#include "program.h" + +struct IgneumInitWords { uint32_t w[8]; }; + +// Hot table (64 MiB, 2 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +__device__ __forceinline__ uint32_t splitmix32(uint32_t x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } +__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } + +__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw, const uint32_t* hot) { + uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x; + uint32_t nonce = baseNonce + gid; + uint32_t r0, r1, r2, r3, r4, r5, r6, r7; + { uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; } + { uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; } + { uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; } + { uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; } + { uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; } + { uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; } + { uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; } + { uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; } + + for (uint32_t it = 0u; it < 8u; ++it) { + uint32_t sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ ds[r2 & mask]; // 4 load + r5 = r5 ^ ds[r7 & mask]; // 5 load + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 6 shfl + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 8); // 7 shfl + r1 = __umulhi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = __umulhi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[__umulhi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 18 shfl + r5 = r5 * r0; // 19 mul + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 20 shfl + r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 16); // 21 shfl + r6 = __umulhi(r6, r2); // 22 mulhi + r6 = r6 ^ ds[r1 & mask]; // 23 load + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r6, 2); // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ ds[r0 & mask]; // 32 load + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[__umulhi(r1, HOT_WORDS)]; // 34 hot + r0 = __umulhi(r0, r5); // 35 mulhi + r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // 36 shfl + r7 = r7 ^ ds[r0 & mask]; // 37 load + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 4); // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ ds[r7 & mask]; // 44 load + r2 = r2 * r3; // 45 mul + r1 = __umulhi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ ds[r2 & mask]; // 59 load + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo; +} + +cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, + IgneumInitWords iw, const uint32_t* hot, uint32_t nonces, uint32_t blockWarps) { + if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue; + uint32_t block = 32u * blockWarps; + if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue; + igneum_hash_bound<<>>(ds, out, baseNonce, mask, iw, hot); + return cudaGetLastError(); +} + +cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) { + cudaFuncAttributes attr; + cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound); + if (e != cudaSuccess) return e; + *numRegs = attr.numRegs; + return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0); +} diff --git a/proto-cuda/packs-ca2-hot/hot64k2/memhard.h b/proto-cuda/packs-ca2-hot/hot64k2/memhard.h new file mode 100644 index 000000000..180f678b7 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k2/memhard.h @@ -0,0 +1,129 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against. +// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference). +// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C. +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif +#if defined(__CUDACC__) +#define IGNEUM_HD __host__ __device__ __forceinline__ +#elif defined(_MSC_VER) && !defined(__cplusplus) +#define IGNEUM_HD static __inline +#else +#define IGNEUM_HD static inline +#endif +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) { + for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint32_t r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) { + uint32_t prev[16]; uint32_t x[16]; uint32_t y[16]; + for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +IGNEUM_HD void mh_item(const uint32_t* cache, uint32_t t, uint32_t* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint32_t r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +IGNEUM_HD uint32_t mh_word(const uint32_t* cache, uint32_t w) { uint32_t s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Hot table (docs/plans/hot-table.md): 64 MiB = 16384 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +IGNEUM_HD void ht_segment(uint32_t* hot, uint32_t seg) { + uint32_t prev[16]; uint32_t x[16]; uint32_t y[16]; + for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + uint32_t* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} diff --git a/proto-cuda/packs-ca2-hot/hot64k2/memhard.metal b/proto-cuda/packs-ca2-hot/hot64k2/memhard.metal new file mode 100644 index 000000000..c2ec5702e --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k2/memhard.metal @@ -0,0 +1,131 @@ +#include +using namespace metal; +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +inline void mh_chacha_block(const thread uint* x, thread uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +inline void mh_cache_segment(device uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +inline void mh_mixer(thread uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +inline void mh_item(device const uint* cache, uint t, thread uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +inline uint mh_word(device const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// One thread per segment (2^16 threads). +kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) { + mh_cache_segment(cache, gid); +} +// One thread per 64-byte item (dataset words / 16 threads). +kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]], + uint gid [[thread_position_in_grid]]) { + uint s[16]; + mh_item(cache, gid, s); + device uint* d = dataset + gid * 16u; + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; +} + +// Hot table (docs/plans/hot-table.md): 64 MiB = 16384 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +inline void ht_segment(device uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + device uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one thread per segment (IGNEUM_HOT_SEGMENTS threads). +kernel void igneum_hot_fill(device uint* hot [[buffer(0)]], uint gid [[thread_position_in_grid]]) { + ht_segment(hot, gid); +} diff --git a/proto-cuda/packs-ca2-hot/hot64k2/program.h b/proto-cuda/packs-ca2-hot/hot64k2/program.h new file mode 100644 index 000000000..81ba8d056 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k2/program.h @@ -0,0 +1,70 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Program metadata for host.cu plus the launch wrappers defined in kernel.cu. +// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros. +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif +#ifndef IGNEUM_NO_CUDA +#include +#endif + +#define IGNEUM_SEED_STRING "igneum-genesis" +#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973" +#define IGNEUM_GENERATOR 2 +#define IGNEUM_PROGRAM_ATTEMPT 0 +#define IGNEUM_PROGRAM_ID 0x322a62466d49ff66ull +#define IGNEUM_DAY_STRING "2026-10-03" +#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033" +#define IGNEUM_DAY0 0x3067619fu +#define IGNEUM_DAY1 0x3c269176u +#define IGNEUM_DATASET_LOG2 28 +#define IGNEUM_MASK 0x0fffffffu +#define IGNEUM_LANES 32 +#define IGNEUM_ITERATIONS 8 +#define IGNEUM_INSTR_COUNT 64 +#define IGNEUM_LOADS_PER_HASH 128 +#define IGNEUM_WIDE_LOADS_PER_HASH 0 +#define IGNEUM_OP_MIX "load=14 add=8 shfl=8 xor=6 mad=5 mul=5 mulhi=5 sub=4 rotl=3 rotr=3 hot=2 or=1" +// Class v3 construction (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): version 2 loads; the dataset item +// derivation applies the mixer IGNEUM_MIXER_MULT times per round (memhard.h), and the cache follows the growth rule. +#define IGNEUM_LOAD_CLASS "hot64k2" +#define IGNEUM_LOAD_SLOTS 16 +#define IGNEUM_LOAD_MIX { 100, 0, 0 } +#define IGNEUM_LOAD_WIDTH_COUNTS { 14, 0, 0 } // loads of 4, 16, 64 bytes per program +#define IGNEUM_BYTES_PER_HASH 448 +#define IGNEUM_FOLD_ROT 11 +#define IGNEUM_FOLD_MUL 0x9e3779b1u +// Hot table (hot-table experiment, 5 October 2026, docs/plans/hot-table.md): NOT the lottery hash. IGNEUM_HOT_SLOTS of the +// 16 load slots read H[mulhi(src, IGNEUM_HOT_WORDS)], H = IGNEUM_HOT_MB MiB of chained ChaCha12 lines (the cache chain) under +// KH = seed_words("igneum-hot/" || epoch seed bytes) and the tag "Igne" "umHT"; filled by igneum_hot_fill once per epoch. +#define IGNEUM_HOT_MB 64 +#define IGNEUM_HOT_WORDS 0x01000000u +#define IGNEUM_HOT_SEGMENTS 16384u +#define IGNEUM_HOT_SLOTS 2 // hot loads per program (16 per hash), replacing the dataset loads (14 of them) +#define IGNEUM_HOT_ADDED 0 +#define IGNEUM_HOT_KEY_INIT { 0x3a48bef5u, 0x6b54b1a1u, 0x9ff897c3u, 0x7d5d85e4u, 0xcd35379fu, 0x9d76bb86u, 0xbe2affb1u, 0x0f8f80b4u } +// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h) +#define IGNEUM_DATASET_MODE 1 + +#define IGNEUM_SEEDW_INIT { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u } +#define IGNEUM_KEY_INIT { 0x3067619fu, 0x3c269176u, 0x84a03b03u, 0xf8c63294u, 0xff977c5bu, 0xe60def3eu, 0x63630141u, 0xb8fbcb58u } +#define IGNEUM_CACHE_LOG2_WORDS 26 +#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6 +#define IGNEUM_CACHE_SEGMENTS 65536u +#define IGNEUM_ITEM_ROUNDS 8 +#define IGNEUM_MIX_ROT_INIT { 20u, 20u, 19u, 4u, 26u, 3u, 3u, 27u } +#define IGNEUM_MIX_MUL_INIT { 0x42146205u, 0x52cbe0fbu, 0x7ecf4a03u, 0x6728907fu, 0xd81d9751u, 0x132952c3u, 0xf60de277u, 0x05358035u, 0xbaf6499du, 0xe4db9667u, 0x3e98f45du, 0xd0004eddu, 0x2691630du, 0x9beb3bcfu, 0xab310379u, 0x99cfb423u } +#define IGNEUM_MIX_RC_INIT { 0xbab68293u, 0xcc162340u, 0x6ce151ccu, 0xe62b8997u, 0xc9c80297u, 0xf74a1654u, 0x3d704af5u, 0x3cf522b7u, 0x2b9cac04u, 0xa880ac10u, 0x13e5dd1du, 0x6fc3e233u, 0x2d83eeacu, 0x9006e8bfu, 0x2c4b5362u, 0x31b49ee2u } + +#ifndef IGNEUM_NO_CUDA +// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError(). +cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments); +cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems); +cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments); +cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot, + uint32_t nonces, uint32_t blockWarps); +cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps); +#endif diff --git a/proto-cuda/packs-ca2-hot/hot64k2/program.json b/proto-cuda/packs-ca2-hot/hot64k2/program.json new file mode 100644 index 000000000..5d402be4d --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k2/program.json @@ -0,0 +1,129 @@ +{ + "format": "igneum-program-pack-3", + "generator": 2, + "attempt": 0, + "program_id": "0x322a62466d49ff66", + "program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32", + "dataset_mode": "memory-hard", + "seed": "igneum-genesis", + "seed_bytes": "69676e65756d2d67656e65736973", + "seed_words": ["0x67a9a7be", "0x1a155b25", "0xfddfb732", "0x4b5af2e8", "0xc55caf33", "0xa27c13b7", "0x06628a48", "0x03852469"], + "seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32", + "generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried", + "lanes": 32, + "registers": 8, + "iterations": 8, + "instruction_count": 64, + "loads_per_hash": 128, + "load_class": "hot64k2", + "load_slots": 16, + "load_mix_percent_4_16_64": [100, 0, 0], + "load_width_counts_4_16_64": [14, 0, 0], + "bytes_per_hash": 448, + "wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots", + "hot_table": {"mb": 64, "words": 16777216, "segments": 16384, "slots": 2, "form": "replaced: k of the class's load slots read the table", "dataset_slots": 14, "hot_loads_per_hash": 16, "key": ["0x3a48bef5", "0x6b54b1a1", "0x9ff897c3", "0x7d5d85e4", "0xcd35379f", "0x9d76bb86", "0xbe2affb1", "0x0f8f80b4"], "key_derivation": "seed_words_from_bytes('igneum-hot/' || seed_bytes), the same for every attempt of the epoch", "tag": ["0x49676e65", "0x756d4854"], "chain": "the cache chain of dataset.cache with the hot key and tag: in_j = prev ^ (sigma || key || seg || j || tag); line_j = block(in_j); prev_0 = 0", "load": "dst = dst ^ hot[mulhi(src, words)] (the high 32 bits of the 64-bit product; the index lies in [0, words) for any size)", "slots_rule": "the first k load slots in the Fisher-Yates draw order after the scratch slots (a uniform k-subset); no extra draw, so with version 2 widths and no scratch the program is the version 2 program with k loads redirected", "acceptance_stand_in": "dataset_elem(mulhi(src, words), seed_words[2], seed_words[3])", "program_id": "the read-width id with 'hot/' || mb || k appended", "spec": "docs/plans/hot-table.md"}, + "op_mix": {"load": 14, "add": 8, "shfl": 8, "xor": 6, "mad": 5, "mul": 5, "mulhi": 5, "sub": 4, "rotl": 3, "rotr": 3, "hot": 2, "or": 1}, + "register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]", + "splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16", + "iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order", + "output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo", + "op_semantics": { + "add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)", + "sub": "dst = dst - src", + "mul": "dst = dst * src (low 32)", + "mulhi": "dst = high 32 bits of dst * src", + "xor": "dst = dst ^ src", + "or": "dst = dst | src", + "rotl": "dst = rotl(dst, rot), rot in 1..31", + "rotr": "dst = rotr(dst, src & 31)", + "mad": "dst = src * src2 + dst", + "shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp", + "load": "dst = dst ^ dataset[src & dataset.mask]", + "wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)", + "hot": "dst = dst ^ hot[mulhi(src, hot_table.words)] (hot-table experiment)" + }, + "dataset": { + "log2_words": 28, + "bytes": 1073741824, + "mask": "0x0fffffff", + "day": "2026-10-03", + "day_bytes": "6461792f323032362d31302d3033", + "day_words_from": "seed_words_from_bytes(day_bytes)", + "d0": "0x3067619f", + "d1": "0x3c269176", + "mode": "memory-hard", + "spec": "proto-metal/MEMHARD.md", + "key": ["0x3067619f", "0x3c269176", "0x84a03b03", "0xf8c63294", "0xff977c5b", "0xe60def3e", "0x63630141", "0xb8fbcb58"], + "key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]", + "cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"}, + "mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [20, 20, 19, 4, 26, 3, 3, 27], "mul": ["0x42146205", "0x52cbe0fb", "0x7ecf4a03", "0x6728907f", "0xd81d9751", "0x132952c3", "0xf60de277", "0x05358035", "0xbaf6499d", "0xe4db9667", "0x3e98f45d", "0xd0004edd", "0x2691630d", "0x9beb3bcf", "0xab310379", "0x99cfb423"], "rc": ["0xbab68293", "0xcc162340", "0x6ce151cc", "0xe62b8997", "0xc9c80297", "0xf74a1654", "0x3d704af5", "0x3cf522b7", "0x2b9cac04", "0xa880ac10", "0x13e5dd1d", "0x6fc3e233", "0x2d83eeac", "0x9006e8bf", "0x2c4b5362", "0x31b49ee2"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"}, + "item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: s = M_r(s); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then s = M_8(s); item(t) = s", + "word": "dataset[w] = item(w >> 4)[w & 15]" + }, + "instructions": [ + {"i": 0, "op": "mad", "dst": 2, "src": 3, "src2": 4, "imm": "0xbf7b174d", "imm2": "0x337b762e", "rot": 17, "bit": 2, "mask": 2, "width": 1}, + {"i": 1, "op": "mad", "dst": 2, "src": 1, "src2": 1, "imm": "0xdd04a5da", "imm2": "0x42da7657", "rot": 15, "bit": 30, "mask": 16, "width": 1}, + {"i": 2, "op": "mad", "dst": 2, "src": 3, "src2": 2, "imm": "0x734003fa", "imm2": "0x5bb67700", "rot": 3, "bit": 20, "mask": 1, "width": 1}, + {"i": 3, "op": "xor", "dst": 3, "src": 5, "src2": 5, "imm": "0xc55a1b1c", "imm2": "0xa19720f3", "rot": 7, "bit": 8, "mask": 1, "width": 1}, + {"i": 4, "op": "load", "dst": 7, "src": 2, "src2": 5, "imm": "0xad572dd7", "imm2": "0x9ceb3ea7", "rot": 18, "bit": 30, "mask": 2, "width": 1}, + {"i": 5, "op": "load", "dst": 5, "src": 7, "src2": 3, "imm": "0x769a53be", "imm2": "0x80f9067e", "rot": 12, "bit": 22, "mask": 1, "width": 1}, + {"i": 6, "op": "shfl", "dst": 1, "src": 4, "src2": 0, "imm": "0xd3613d88", "imm2": "0x262fb219", "rot": 10, "bit": 30, "mask": 8, "width": 1}, + {"i": 7, "op": "shfl", "dst": 7, "src": 3, "src2": 4, "imm": "0xce38e42f", "imm2": "0xb868b818", "rot": 11, "bit": 8, "mask": 8, "width": 1}, + {"i": 8, "op": "mulhi", "dst": 1, "src": 5, "src2": 2, "imm": "0xa5eebca5", "imm2": "0x5703a72b", "rot": 13, "bit": 13, "mask": 16, "width": 1}, + {"i": 9, "op": "rotr", "dst": 6, "src": 3, "src2": 4, "imm": "0x17a5a9c7", "imm2": "0xdcfb93a1", "rot": 20, "bit": 27, "mask": 2, "width": 1}, + {"i": 10, "op": "or", "dst": 3, "src": 4, "src2": 1, "imm": "0xccb7d785", "imm2": "0xc335364c", "rot": 14, "bit": 12, "mask": 4, "width": 1}, + {"i": 11, "op": "load", "dst": 4, "src": 3, "src2": 2, "imm": "0x88cb9af3", "imm2": "0x4e7dc10d", "rot": 24, "bit": 17, "mask": 4, "width": 1}, + {"i": 12, "op": "mulhi", "dst": 0, "src": 4, "src2": 4, "imm": "0x45374321", "imm2": "0x3cd91989", "rot": 11, "bit": 4, "mask": 2, "width": 1}, + {"i": 13, "op": "add", "dst": 5, "src": 1, "src2": 2, "imm": "0xc7934706", "imm2": "0xd3177981", "rot": 16, "bit": 30, "mask": 2, "width": 1}, + {"i": 14, "op": "load", "dst": 0, "src": 4, "src2": 3, "imm": "0xfd7f56bb", "imm2": "0x65e14f52", "rot": 13, "bit": 22, "mask": 2, "width": 1}, + {"i": 15, "op": "sub", "dst": 2, "src": 4, "src2": 4, "imm": "0x35a80b49", "imm2": "0x060f2d13", "rot": 16, "bit": 20, "mask": 16, "width": 1}, + {"i": 16, "op": "hot", "dst": 2, "src": 0, "src2": 2, "imm": "0xae0a32c2", "imm2": "0x4c2a4cfe", "rot": 8, "bit": 31, "mask": 16, "width": 1}, + {"i": 17, "op": "load", "dst": 7, "src": 2, "src2": 6, "imm": "0x82a84cc3", "imm2": "0x21a38d68", "rot": 15, "bit": 21, "mask": 2, "width": 1}, + {"i": 18, "op": "shfl", "dst": 7, "src": 3, "src2": 3, "imm": "0xa3818806", "imm2": "0x8f66b5c8", "rot": 14, "bit": 6, "mask": 4, "width": 1}, + {"i": 19, "op": "mul", "dst": 5, "src": 0, "src2": 1, "imm": "0xa00de107", "imm2": "0x77bfcaa5", "rot": 3, "bit": 10, "mask": 2, "width": 1}, + {"i": 20, "op": "shfl", "dst": 3, "src": 4, "src2": 7, "imm": "0x1d2b8cab", "imm2": "0x80b4f9a2", "rot": 14, "bit": 25, "mask": 2, "width": 1}, + {"i": 21, "op": "shfl", "dst": 2, "src": 4, "src2": 5, "imm": "0x3ac915d2", "imm2": "0x5fba7bc2", "rot": 16, "bit": 1, "mask": 16, "width": 1}, + {"i": 22, "op": "mulhi", "dst": 6, "src": 2, "src2": 0, "imm": "0xdc3ec8fd", "imm2": "0x599e2fa3", "rot": 22, "bit": 3, "mask": 2, "width": 1}, + {"i": 23, "op": "load", "dst": 6, "src": 1, "src2": 5, "imm": "0x2a6b16d5", "imm2": "0xd73e396f", "rot": 28, "bit": 29, "mask": 2, "width": 1}, + {"i": 24, "op": "mul", "dst": 5, "src": 0, "src2": 7, "imm": "0x376d0223", "imm2": "0xe1c2169a", "rot": 4, "bit": 16, "mask": 16, "width": 1}, + {"i": 25, "op": "rotl", "dst": 5, "src": 7, "src2": 3, "imm": "0x78ad8c60", "imm2": "0x6f5b77d5", "rot": 19, "bit": 11, "mask": 16, "width": 1}, + {"i": 26, "op": "shfl", "dst": 7, "src": 6, "src2": 5, "imm": "0x93915b9f", "imm2": "0x1e61fb6b", "rot": 28, "bit": 23, "mask": 2, "width": 1}, + {"i": 27, "op": "xor", "dst": 0, "src": 5, "src2": 7, "imm": "0x6378fe15", "imm2": "0x66c78f42", "rot": 12, "bit": 31, "mask": 8, "width": 1}, + {"i": 28, "op": "xor", "dst": 0, "src": 4, "src2": 7, "imm": "0x20a57fda", "imm2": "0x088c848e", "rot": 16, "bit": 13, "mask": 4, "width": 1}, + {"i": 29, "op": "sub", "dst": 3, "src": 0, "src2": 2, "imm": "0x49d95fd5", "imm2": "0x1a5f946a", "rot": 6, "bit": 12, "mask": 1, "width": 1}, + {"i": 30, "op": "mul", "dst": 5, "src": 1, "src2": 7, "imm": "0x0a816217", "imm2": "0x405c4f73", "rot": 13, "bit": 27, "mask": 4, "width": 1}, + {"i": 31, "op": "load", "dst": 7, "src": 2, "src2": 2, "imm": "0x09ed045e", "imm2": "0xd69c4715", "rot": 5, "bit": 9, "mask": 2, "width": 1}, + {"i": 32, "op": "load", "dst": 1, "src": 0, "src2": 6, "imm": "0xeb79ea49", "imm2": "0xcc587f5a", "rot": 6, "bit": 8, "mask": 16, "width": 1}, + {"i": 33, "op": "xor", "dst": 5, "src": 6, "src2": 1, "imm": "0x3027401e", "imm2": "0x5f20c27e", "rot": 18, "bit": 9, "mask": 2, "width": 1}, + {"i": 34, "op": "hot", "dst": 5, "src": 1, "src2": 3, "imm": "0x0e1cab07", "imm2": "0x09356c5b", "rot": 19, "bit": 31, "mask": 1, "width": 1}, + {"i": 35, "op": "mulhi", "dst": 0, "src": 5, "src2": 2, "imm": "0x90e31357", "imm2": "0xabd32484", "rot": 26, "bit": 5, "mask": 8, "width": 1}, + {"i": 36, "op": "shfl", "dst": 5, "src": 2, "src2": 5, "imm": "0xee9a955f", "imm2": "0x31b3faed", "rot": 8, "bit": 24, "mask": 4, "width": 1}, + {"i": 37, "op": "load", "dst": 7, "src": 0, "src2": 7, "imm": "0x3ba2f832", "imm2": "0x1160dcd3", "rot": 4, "bit": 29, "mask": 1, "width": 1}, + {"i": 38, "op": "add", "dst": 3, "src": 1, "src2": 7, "imm": "0x75ba2fad", "imm2": "0x230c005c", "rot": 4, "bit": 27, "mask": 1, "width": 1}, + {"i": 39, "op": "shfl", "dst": 1, "src": 5, "src2": 3, "imm": "0xdbf37e75", "imm2": "0xb5ac1969", "rot": 30, "bit": 13, "mask": 4, "width": 1}, + {"i": 40, "op": "xor", "dst": 2, "src": 5, "src2": 5, "imm": "0x47f136c5", "imm2": "0x06ce9153", "rot": 19, "bit": 10, "mask": 2, "width": 1}, + {"i": 41, "op": "mad", "dst": 3, "src": 6, "src2": 3, "imm": "0xce13eff8", "imm2": "0x04cc1d55", "rot": 3, "bit": 1, "mask": 4, "width": 1}, + {"i": 42, "op": "sub", "dst": 6, "src": 7, "src2": 1, "imm": "0x6a65ab71", "imm2": "0x8fbc1bcd", "rot": 4, "bit": 1, "mask": 8, "width": 1}, + {"i": 43, "op": "xor", "dst": 7, "src": 0, "src2": 7, "imm": "0xdaeb4928", "imm2": "0xc0423027", "rot": 24, "bit": 11, "mask": 8, "width": 1}, + {"i": 44, "op": "load", "dst": 1, "src": 7, "src2": 7, "imm": "0x778f01c9", "imm2": "0x28cedcea", "rot": 12, "bit": 4, "mask": 16, "width": 1}, + {"i": 45, "op": "mul", "dst": 2, "src": 3, "src2": 2, "imm": "0xf4264f1b", "imm2": "0x0f627d56", "rot": 5, "bit": 28, "mask": 8, "width": 1}, + {"i": 46, "op": "mulhi", "dst": 1, "src": 5, "src2": 1, "imm": "0xffb2147a", "imm2": "0xccde9b05", "rot": 13, "bit": 9, "mask": 2, "width": 1}, + {"i": 47, "op": "sub", "dst": 4, "src": 3, "src2": 5, "imm": "0x73b36234", "imm2": "0x3f5d5997", "rot": 7, "bit": 18, "mask": 2, "width": 1}, + {"i": 48, "op": "rotr", "dst": 2, "src": 6, "src2": 3, "imm": "0x3a4d9aa9", "imm2": "0x212bec7b", "rot": 4, "bit": 29, "mask": 16, "width": 1}, + {"i": 49, "op": "load", "dst": 3, "src": 5, "src2": 2, "imm": "0x626f11df", "imm2": "0x56cd5bfd", "rot": 7, "bit": 1, "mask": 1, "width": 1}, + {"i": 50, "op": "add", "dst": 1, "src": 5, "src2": 6, "imm": "0x81b8bc2c", "imm2": "0x1907970c", "rot": 28, "bit": 7, "mask": 4, "width": 1}, + {"i": 51, "op": "mul", "dst": 0, "src": 2, "src2": 2, "imm": "0xa8848b30", "imm2": "0xef6ac348", "rot": 9, "bit": 15, "mask": 8, "width": 1}, + {"i": 52, "op": "add", "dst": 0, "src": 2, "src2": 0, "imm": "0x4f92b968", "imm2": "0x699fd448", "rot": 22, "bit": 6, "mask": 4, "width": 1}, + {"i": 53, "op": "add", "dst": 1, "src": 0, "src2": 2, "imm": "0x2bb965af", "imm2": "0x77b1520d", "rot": 2, "bit": 12, "mask": 8, "width": 1}, + {"i": 54, "op": "rotl", "dst": 7, "src": 1, "src2": 0, "imm": "0x553e678b", "imm2": "0x3cc8eae0", "rot": 14, "bit": 20, "mask": 2, "width": 1}, + {"i": 55, "op": "add", "dst": 3, "src": 7, "src2": 2, "imm": "0x7b0fe07a", "imm2": "0xa54c55a0", "rot": 10, "bit": 1, "mask": 1, "width": 1}, + {"i": 56, "op": "load", "dst": 6, "src": 7, "src2": 4, "imm": "0x01eba9aa", "imm2": "0x2758c0f7", "rot": 14, "bit": 15, "mask": 4, "width": 1}, + {"i": 57, "op": "rotr", "dst": 1, "src": 5, "src2": 2, "imm": "0x1f5267b3", "imm2": "0x236f5a27", "rot": 2, "bit": 31, "mask": 16, "width": 1}, + {"i": 58, "op": "load", "dst": 5, "src": 4, "src2": 3, "imm": "0xa9954a9b", "imm2": "0x6a54d4e8", "rot": 11, "bit": 10, "mask": 16, "width": 1}, + {"i": 59, "op": "load", "dst": 6, "src": 2, "src2": 4, "imm": "0x9923ff88", "imm2": "0x9357254e", "rot": 16, "bit": 1, "mask": 16, "width": 1}, + {"i": 60, "op": "mad", "dst": 3, "src": 5, "src2": 0, "imm": "0xc0cc51a6", "imm2": "0x3fd7701b", "rot": 20, "bit": 1, "mask": 4, "width": 1}, + {"i": 61, "op": "add", "dst": 5, "src": 7, "src2": 7, "imm": "0xaf9dd72d", "imm2": "0xad7493e7", "rot": 7, "bit": 31, "mask": 16, "width": 1}, + {"i": 62, "op": "add", "dst": 4, "src": 6, "src2": 2, "imm": "0x89841d87", "imm2": "0x1e07c3d9", "rot": 6, "bit": 27, "mask": 1, "width": 1}, + {"i": 63, "op": "rotl", "dst": 5, "src": 4, "src2": 2, "imm": "0xae210f8d", "imm2": "0x8e499ba4", "rot": 19, "bit": 9, "mask": 1, "width": 1} + ] +} diff --git a/proto-cuda/packs-ca2-hot/hot64k2/program.metal b/proto-cuda/packs-ca2-hot/hot64k2/program.metal new file mode 100644 index 000000000..ae2552c01 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k2/program.metal @@ -0,0 +1,112 @@ +#include +using namespace metal; + +#define MASK 0x0fffffffu +// Hot table (64 MiB, 2 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }; + +inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 +inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +kernel void igneum_hash(device const uint* dataset [[buffer(0)]], + device ulong* out [[buffer(1)]], + constant uint& baseNonce [[buffer(2)]], + device const uint* hot [[buffer(3)]], + uint gid [[thread_position_in_grid]]) { + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + { uint x = nonce ^ SEEDW[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ SEEDW[1]; } + { uint x = nonce ^ SEEDW[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ SEEDW[2]; } + { uint x = nonce ^ SEEDW[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ SEEDW[3]; } + { uint x = nonce ^ SEEDW[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ SEEDW[4]; } + { uint x = nonce ^ SEEDW[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ SEEDW[5]; } + { uint x = nonce ^ SEEDW[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ SEEDW[6]; } + { uint x = nonce ^ SEEDW[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ SEEDW[7]; } + { uint x = nonce ^ SEEDW[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ SEEDW[0]; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 + r2 = r1 * r1 + r2; // 1 + r2 = r3 * r2 + r2; // 2 + r3 = r3 ^ r5; // 3 + r7 = r7 ^ dataset[r2 & MASK]; // 4 + r5 = r5 ^ dataset[r7 & MASK]; // 5 + r1 = r1 ^ simd_shuffle_xor(r4, (ushort)8); // 6 + r7 = r7 ^ simd_shuffle_xor(r3, (ushort)8); // 7 + r1 = mulhi(r1, r5); // 8 + r6 = rotr_var(r6, r3); // 9 + r3 = r3 | r4; // 10 + r4 = r4 ^ dataset[r3 & MASK]; // 11 + r0 = mulhi(r0, r4); // 12 + r5 = r5 + r1 + select(0xc7934706u, 0xd3177981u, ((sel >> 30u) & 1u) != 0u); // 13 + r0 = r0 ^ dataset[r4 & MASK]; // 14 + r2 = r2 - r4; // 15 + r2 = r2 ^ hot[mulhi(r0, HOT_WORDS)]; // 16 + r7 = r7 ^ dataset[r2 & MASK]; // 17 + r7 = r7 ^ simd_shuffle_xor(r3, (ushort)4); // 18 + r5 = r5 * r0; // 19 + r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 20 + r2 = r2 ^ simd_shuffle_xor(r4, (ushort)16); // 21 + r6 = mulhi(r6, r2); // 22 + r6 = r6 ^ dataset[r1 & MASK]; // 23 + r5 = r5 * r0; // 24 + r5 = rotl_imm(r5, 19u); // 25 + r7 = r7 ^ simd_shuffle_xor(r6, (ushort)2); // 26 + r0 = r0 ^ r5; // 27 + r0 = r0 ^ r4; // 28 + r3 = r3 - r0; // 29 + r5 = r5 * r1; // 30 + r7 = r7 ^ dataset[r2 & MASK]; // 31 + r1 = r1 ^ dataset[r0 & MASK]; // 32 + r5 = r5 ^ r6; // 33 + r5 = r5 ^ hot[mulhi(r1, HOT_WORDS)]; // 34 + r0 = mulhi(r0, r5); // 35 + r5 = r5 ^ simd_shuffle_xor(r2, (ushort)4); // 36 + r7 = r7 ^ dataset[r0 & MASK]; // 37 + r3 = r3 + r1 + select(0x75ba2fadu, 0x230c005cu, ((sel >> 27u) & 1u) != 0u); // 38 + r1 = r1 ^ simd_shuffle_xor(r5, (ushort)4); // 39 + r2 = r2 ^ r5; // 40 + r3 = r6 * r3 + r3; // 41 + r6 = r6 - r7; // 42 + r7 = r7 ^ r0; // 43 + r1 = r1 ^ dataset[r7 & MASK]; // 44 + r2 = r2 * r3; // 45 + r1 = mulhi(r1, r5); // 46 + r4 = r4 - r3; // 47 + r2 = rotr_var(r2, r6); // 48 + r3 = r3 ^ dataset[r5 & MASK]; // 49 + r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 50 + r0 = r0 * r2; // 51 + r0 = r0 + r2 + select(0x4f92b968u, 0x699fd448u, ((sel >> 6u) & 1u) != 0u); // 52 + r1 = r1 + r0 + select(0x2bb965afu, 0x77b1520du, ((sel >> 12u) & 1u) != 0u); // 53 + r7 = rotl_imm(r7, 14u); // 54 + r3 = r3 + r7 + select(0x7b0fe07au, 0xa54c55a0u, ((sel >> 1u) & 1u) != 0u); // 55 + r6 = r6 ^ dataset[r7 & MASK]; // 56 + r1 = rotr_var(r1, r5); // 57 + r5 = r5 ^ dataset[r4 & MASK]; // 58 + r6 = r6 ^ dataset[r2 & MASK]; // 59 + r3 = r5 * r0 + r3; // 60 + r5 = r5 + r7 + select(0xaf9dd72du, 0xad7493e7u, ((sel >> 31u) & 1u) != 0u); // 61 + r4 = r4 + r6 + select(0x89841d87u, 0x1e07c3d9u, ((sel >> 27u) & 1u) != 0u); // 62 + r5 = rotl_imm(r5, 19u); // 63 + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot64k2/program_bound.metal b/proto-cuda/packs-ca2-hot/hot64k2/program_bound.metal new file mode 100644 index 000000000..7fbf2e65f --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k2/program_bound.metal @@ -0,0 +1,114 @@ +#include +using namespace metal; + +#define MASK 0x0fffffffu +// Hot table (64 MiB, 2 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }; + +inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 +inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW. +kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], + device ulong* out [[buffer(1)]], + constant uint& baseNonce [[buffer(2)]], + constant uint* initw [[buffer(3)]], + device const uint* hot [[buffer(4)]], + uint gid [[thread_position_in_grid]]) { + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + { uint x = nonce ^ initw[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ initw[1]; } + { uint x = nonce ^ initw[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ initw[2]; } + { uint x = nonce ^ initw[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ initw[3]; } + { uint x = nonce ^ initw[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ initw[4]; } + { uint x = nonce ^ initw[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ initw[5]; } + { uint x = nonce ^ initw[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ initw[6]; } + { uint x = nonce ^ initw[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ initw[7]; } + { uint x = nonce ^ initw[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ initw[0]; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 + r2 = r1 * r1 + r2; // 1 + r2 = r3 * r2 + r2; // 2 + r3 = r3 ^ r5; // 3 + r7 = r7 ^ dataset[r2 & MASK]; // 4 + r5 = r5 ^ dataset[r7 & MASK]; // 5 + r1 = r1 ^ simd_shuffle_xor(r4, (ushort)8); // 6 + r7 = r7 ^ simd_shuffle_xor(r3, (ushort)8); // 7 + r1 = mulhi(r1, r5); // 8 + r6 = rotr_var(r6, r3); // 9 + r3 = r3 | r4; // 10 + r4 = r4 ^ dataset[r3 & MASK]; // 11 + r0 = mulhi(r0, r4); // 12 + r5 = r5 + r1 + select(0xc7934706u, 0xd3177981u, ((sel >> 30u) & 1u) != 0u); // 13 + r0 = r0 ^ dataset[r4 & MASK]; // 14 + r2 = r2 - r4; // 15 + r2 = r2 ^ hot[mulhi(r0, HOT_WORDS)]; // 16 + r7 = r7 ^ dataset[r2 & MASK]; // 17 + r7 = r7 ^ simd_shuffle_xor(r3, (ushort)4); // 18 + r5 = r5 * r0; // 19 + r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 20 + r2 = r2 ^ simd_shuffle_xor(r4, (ushort)16); // 21 + r6 = mulhi(r6, r2); // 22 + r6 = r6 ^ dataset[r1 & MASK]; // 23 + r5 = r5 * r0; // 24 + r5 = rotl_imm(r5, 19u); // 25 + r7 = r7 ^ simd_shuffle_xor(r6, (ushort)2); // 26 + r0 = r0 ^ r5; // 27 + r0 = r0 ^ r4; // 28 + r3 = r3 - r0; // 29 + r5 = r5 * r1; // 30 + r7 = r7 ^ dataset[r2 & MASK]; // 31 + r1 = r1 ^ dataset[r0 & MASK]; // 32 + r5 = r5 ^ r6; // 33 + r5 = r5 ^ hot[mulhi(r1, HOT_WORDS)]; // 34 + r0 = mulhi(r0, r5); // 35 + r5 = r5 ^ simd_shuffle_xor(r2, (ushort)4); // 36 + r7 = r7 ^ dataset[r0 & MASK]; // 37 + r3 = r3 + r1 + select(0x75ba2fadu, 0x230c005cu, ((sel >> 27u) & 1u) != 0u); // 38 + r1 = r1 ^ simd_shuffle_xor(r5, (ushort)4); // 39 + r2 = r2 ^ r5; // 40 + r3 = r6 * r3 + r3; // 41 + r6 = r6 - r7; // 42 + r7 = r7 ^ r0; // 43 + r1 = r1 ^ dataset[r7 & MASK]; // 44 + r2 = r2 * r3; // 45 + r1 = mulhi(r1, r5); // 46 + r4 = r4 - r3; // 47 + r2 = rotr_var(r2, r6); // 48 + r3 = r3 ^ dataset[r5 & MASK]; // 49 + r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 50 + r0 = r0 * r2; // 51 + r0 = r0 + r2 + select(0x4f92b968u, 0x699fd448u, ((sel >> 6u) & 1u) != 0u); // 52 + r1 = r1 + r0 + select(0x2bb965afu, 0x77b1520du, ((sel >> 12u) & 1u) != 0u); // 53 + r7 = rotl_imm(r7, 14u); // 54 + r3 = r3 + r7 + select(0x7b0fe07au, 0xa54c55a0u, ((sel >> 1u) & 1u) != 0u); // 55 + r6 = r6 ^ dataset[r7 & MASK]; // 56 + r1 = rotr_var(r1, r5); // 57 + r5 = r5 ^ dataset[r4 & MASK]; // 58 + r6 = r6 ^ dataset[r2 & MASK]; // 59 + r3 = r5 * r0 + r3; // 60 + r5 = r5 + r7 + select(0xaf9dd72du, 0xad7493e7u, ((sel >> 31u) & 1u) != 0u); // 61 + r4 = r4 + r6 + select(0x89841d87u, 0x1e07c3d9u, ((sel >> 27u) & 1u) != 0u); // 62 + r5 = rotl_imm(r5, 19u); // 63 + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot64k2/vectors.h b/proto-cuda/packs-ca2-hot/hot64k2/vectors.h new file mode 100644 index 000000000..fd8e01caa --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k2/vectors.h @@ -0,0 +1,67 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Expected outputs: igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif + +#define IGNEUM_VEC_WARPS 3 +static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u }; +static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = { + { // base nonce 0 + 0x65b0158a345b67e7ull, 0x8c0df6208996119dull, 0xa82920dfd2564384ull, 0x3f98635d88e71980ull, 0x9330bc0a2c56e1a8ull, 0x2f909a91ce4c3f70ull, 0xaefbb1536b8d51bcull, 0x339807367ad7a031ull, + 0xfbfffb50795b4050ull, 0x36038d518a9f7b87ull, 0x56c10e7bbcc1b46cull, 0x7ee8f70d9c40a0aeull, 0x430641c5c64e7f33ull, 0x1cf9748074f98ec1ull, 0x013ecaf70fe24220ull, 0x1e1d2b948428b87aull, + 0x4d681ef40907816cull, 0xf594910d3fd5fc06ull, 0x29b8b25ba39f2ccfull, 0x5bc340343406553full, 0x05f87eb5c67aa7d6ull, 0x8d3b61236cb38cb5ull, 0xc056056ed6154abaull, 0xdbaf8e4ace0da7aaull, + 0x551d2ef6ff72a674ull, 0x99613e85f5c9187eull, 0x9597e6e00fac1a7dull, 0x4806742b7b0869a9ull, 0x685f008c389db453ull, 0x4ec754ef00d2026dull, 0xbf0c4e9528978691ull, 0x3d2cd0c742789be9ull + }, + { // base nonce 4096 + 0x18e0a4279770b00bull, 0xb5b83f4d0edb6b8eull, 0x024399529627448aull, 0xc83b95ce143a618full, 0x4dbabfa14be76be6ull, 0x207079f745606905ull, 0x8e920df676583078ull, 0xd8499ebc77afe275ull, + 0x9d8c674573683174ull, 0x346de7a99fdb40b6ull, 0x924a7287c62be081ull, 0x6baa55408c5b2901ull, 0x73003fd8f521c962ull, 0x8ed90cefc544e916ull, 0x4056df74a34656eaull, 0x6dfe2ec0874d6872ull, + 0x38d8f3ab3b5a5414ull, 0x8f082c3e5af03d0cull, 0x659be876ab2fb5e1ull, 0x6796490dba2bde16ull, 0xcd389d4acf88a6b3ull, 0xa72e8293a53777a3ull, 0x27688d5f693b358bull, 0x7925eab9cdc62dc1ull, + 0x7e18db5099e624c5ull, 0xd18bba6cd4629ab5ull, 0xecde3dc17c4796feull, 0xe5386f3afa28f523ull, 0x16b4ca1e85042299ull, 0xb4411635998d0bdfull, 0xbc4793e8fd01d963ull, 0xab67165dcfe87a9dull + }, + { // base nonce 1000000 + 0xd2eb4ca6afb04f3aull, 0xef1f5c1416c10822ull, 0x0ac3034339afe313ull, 0x3e29eb72d7492296ull, 0x00e70e214591abdcull, 0x71ddb7760dd8e86aull, 0xb9f6dc2158b0c389ull, 0xc4d5a7905b475bd9ull, + 0x6510ed7b7976c952ull, 0x18f41cf0fad53752ull, 0x3b01c44ed18ee920ull, 0xea78aa293930586full, 0xc7eb0149acd0991bull, 0x640d3e9176c9dd09ull, 0x3f1cc7bb223f56e0ull, 0xdcc6b668b2872187ull, + 0xa4a124ab824d5ea9ull, 0xbb95eb77c20d50e8ull, 0xcdae9c9edd32978bull, 0x5675a40696b5666bull, 0xdc619039a79b0d7dull, 0x52b23d4117deb142ull, 0xc1dddbbd50a34c20ull, 0xd0f54b71e0e4ce96ull, + 0x4334aed9d2294383ull, 0x519ac5278300d625ull, 0xadc42115aa1f645bull, 0xb67280570485b157ull, 0x2d72f2b3a89fdaffull, 0x39456ffbfcb86733ull, 0xaaebb5e9afba3337ull, 0xf092fdfb9df66980ull + } +}; + +// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (268435455). +static const uint32_t IGNEUM_DS_HEAD[16] = { + 0xffc3cd94u, 0x5920ccd8u, 0x392f44bbu, 0x5e57f67au, 0x2f2bc2a9u, 0x620b0e36u, 0xbdc09014u, 0x436654bfu, + 0x311e0b48u, 0x1abd93adu, 0x59cc7ce8u, 0xee5247b2u, 0x86171fe8u, 0x6d874751u, 0xc9f7728fu, 0x7c2a435du +}; +static const uint32_t IGNEUM_DS_LAST_INDEX = 268435455u; +static const uint32_t IGNEUM_DS_LAST = 0xa33ada72u; +// 64 sampled dataset words (index, value) computed on the Mac. +#define IGNEUM_DS_SAMPLES 64 +static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = { + 59471966u, 217795994u, 208353206u, 42483309u, 172547758u, 148076330u, 183853158u, 214389424u, 267488061u, 169781097u, 184093494u, 153880993u, 84977930u, 46426879u, 3093825u, 225364072u, 44593546u, 260713159u, 168250303u, 52384140u, 223401610u, 45554030u, 95410555u, 175039924u, 79171087u, 267580473u, 24168642u, 37981670u, 171551130u, 195559979u, 204611762u, 140997658u, 138925853u, 86637313u, 20736778u, 219665210u, 160430336u, 264654675u, 8013395u, 228945585u, 213884386u, 104419827u, 44185464u, 142737231u, 99284897u, 132475900u, 61861762u, 132056166u, 262388043u, 91878046u, 117353561u, 124768597u, 71352993u, 190698941u, 46055428u, 55281366u, 165145231u, 106810753u, 171985651u, 232085256u, 159510492u, 40072060u, 209107596u, 39023794u +}; +static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = { + 0xe8b73d94u, 0x337028b5u, 0xafe148c9u, 0xab99f7aeu, 0x434ea619u, 0xd85cb880u, 0x54764c7fu, 0x82c7e420u, 0xedf4cb9eu, 0x9884c959u, 0x223ee793u, 0x3a9ccf69u, 0x81da4fd2u, 0xd6ce8cb9u, 0xe3922dcau, 0x3e7e6bdeu, 0x382a3acau, 0x567e7f7fu, 0x25a0f084u, 0xbfeef128u, 0xe338abfbu, 0x7c3b5280u, 0x909bc5f1u, 0xd8b74b9cu, 0x8e31a22eu, 0x26b5f1d8u, 0x79122c00u, 0xcafc3340u, 0xd5e02ea3u, 0x1aee1afdu, 0xdb090d9au, 0xb049f435u, 0x4954d8bau, 0x03797ba0u, 0x196eefbdu, 0xd153412au, 0xbe5d2c4bu, 0xdaa14f0eu, 0x8e61ed07u, 0x9e9a64c6u, 0x2e29ff36u, 0x392a8589u, 0xb56a5912u, 0xfa6e8b57u, 0xd1a737cbu, 0xb0fa841au, 0xbe1c341fu, 0xe25be0f1u, 0xe937f543u, 0xebab2248u, 0x8e1b607au, 0x202a2fedu, 0x95e2819cu, 0x9c9652d4u, 0x32fedef0u, 0xdecfff82u, 0xcb5d43e5u, 0xb735806au, 0x8905939cu, 0xfbf8472du, 0xada74e5du, 0x7ebdeeeau, 0x0119f2b3u, 0xa9a376b8u +}; +// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^26 words. +static const uint32_t IGNEUM_CACHE_HEAD[16] = { + 0x355a86d2u, 0x7957db1cu, 0xd21772afu, 0x6fc1e09bu, 0xd55ce61du, 0x6e6a278bu, 0xd3f543ceu, 0x223d8e82u, + 0x143ab337u, 0x2e9f05bdu, 0x2eb389bfu, 0x0c6e449eu, 0x5cfa4222u, 0xba6560feu, 0x8e3e1aa4u, 0xdbcc1d53u +}; +static const uint32_t IGNEUM_CACHE_LAST[16] = { + 0x41190d91u, 0xbd277957u, 0x22ddbb49u, 0x6986f207u, 0xdf69a4d6u, 0x26401a3au, 0x818230fbu, 0xc417122du, + 0x3597b211u, 0xb553ce55u, 0xcf39cc0du, 0x3b7fc43au, 0x3fd43b00u, 0x67e1c80eu, 0xffa7ea7du, 0xca2960abu +}; +static const uint64_t IGNEUM_CACHE_FNV64 = 0x48c4f5bf24166b2eull; +// Hot table self-test (docs/plans/hot-table.md): hot[0..15], the last 16 words, and FNV-1a 64 over all IGNEUM_HOT_WORDS words. +static const uint32_t IGNEUM_HOT_HEAD[16] = { + 0x8068cc73u, 0x6036ebf9u, 0xb604cd25u, 0x8ffb840eu, 0xc54074a2u, 0x285c0695u, 0x77512425u, 0xc26a58a7u, + 0x72c88757u, 0xc10fca78u, 0x513825ddu, 0x30d6ccc8u, 0x9a05e7cfu, 0xb9533f50u, 0x4bac3ba0u, 0xa5c19528u +}; +static const uint32_t IGNEUM_HOT_LAST[16] = { + 0x7c6d7cebu, 0xe24fcd46u, 0xe5cce976u, 0x3ecffe59u, 0x94e98b66u, 0x3f6ef45cu, 0xa3715aa1u, 0xdbe35281u, + 0xba05d27bu, 0x00541963u, 0xe636f453u, 0xd3972366u, 0x529599b8u, 0x79ae3c8cu, 0x48436863u, 0x897a8bf3u +}; +static const uint64_t IGNEUM_HOT_FNV64 = 0x77ca4b9527104530ull; diff --git a/proto-cuda/packs-ca2-hot/hot64k2/vectors.json b/proto-cuda/packs-ca2-hot/hot64k2/vectors.json new file mode 100644 index 000000000..d886fb83f --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k2/vectors.json @@ -0,0 +1,39 @@ +{ + "seed": "igneum-genesis", + "day": "2026-10-03", + "dataset_mode": "memory-hard", + "dataset_log2_words": 28, + "mask": "0x0fffffff", + "lanes": 32, + "source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset", + "warps": [ + {"base_nonce": 0, "expected": [ + "0x65b0158a345b67e7", "0x8c0df6208996119d", "0xa82920dfd2564384", "0x3f98635d88e71980", "0x9330bc0a2c56e1a8", "0x2f909a91ce4c3f70", "0xaefbb1536b8d51bc", "0x339807367ad7a031", + "0xfbfffb50795b4050", "0x36038d518a9f7b87", "0x56c10e7bbcc1b46c", "0x7ee8f70d9c40a0ae", "0x430641c5c64e7f33", "0x1cf9748074f98ec1", "0x013ecaf70fe24220", "0x1e1d2b948428b87a", + "0x4d681ef40907816c", "0xf594910d3fd5fc06", "0x29b8b25ba39f2ccf", "0x5bc340343406553f", "0x05f87eb5c67aa7d6", "0x8d3b61236cb38cb5", "0xc056056ed6154aba", "0xdbaf8e4ace0da7aa", + "0x551d2ef6ff72a674", "0x99613e85f5c9187e", "0x9597e6e00fac1a7d", "0x4806742b7b0869a9", "0x685f008c389db453", "0x4ec754ef00d2026d", "0xbf0c4e9528978691", "0x3d2cd0c742789be9" + ]}, + {"base_nonce": 4096, "expected": [ + "0x18e0a4279770b00b", "0xb5b83f4d0edb6b8e", "0x024399529627448a", "0xc83b95ce143a618f", "0x4dbabfa14be76be6", "0x207079f745606905", "0x8e920df676583078", "0xd8499ebc77afe275", + "0x9d8c674573683174", "0x346de7a99fdb40b6", "0x924a7287c62be081", "0x6baa55408c5b2901", "0x73003fd8f521c962", "0x8ed90cefc544e916", "0x4056df74a34656ea", "0x6dfe2ec0874d6872", + "0x38d8f3ab3b5a5414", "0x8f082c3e5af03d0c", "0x659be876ab2fb5e1", "0x6796490dba2bde16", "0xcd389d4acf88a6b3", "0xa72e8293a53777a3", "0x27688d5f693b358b", "0x7925eab9cdc62dc1", + "0x7e18db5099e624c5", "0xd18bba6cd4629ab5", "0xecde3dc17c4796fe", "0xe5386f3afa28f523", "0x16b4ca1e85042299", "0xb4411635998d0bdf", "0xbc4793e8fd01d963", "0xab67165dcfe87a9d" + ]}, + {"base_nonce": 1000000, "expected": [ + "0xd2eb4ca6afb04f3a", "0xef1f5c1416c10822", "0x0ac3034339afe313", "0x3e29eb72d7492296", "0x00e70e214591abdc", "0x71ddb7760dd8e86a", "0xb9f6dc2158b0c389", "0xc4d5a7905b475bd9", + "0x6510ed7b7976c952", "0x18f41cf0fad53752", "0x3b01c44ed18ee920", "0xea78aa293930586f", "0xc7eb0149acd0991b", "0x640d3e9176c9dd09", "0x3f1cc7bb223f56e0", "0xdcc6b668b2872187", + "0xa4a124ab824d5ea9", "0xbb95eb77c20d50e8", "0xcdae9c9edd32978b", "0x5675a40696b5666b", "0xdc619039a79b0d7d", "0x52b23d4117deb142", "0xc1dddbbd50a34c20", "0xd0f54b71e0e4ce96", + "0x4334aed9d2294383", "0x519ac5278300d625", "0xadc42115aa1f645b", "0xb67280570485b157", "0x2d72f2b3a89fdaff", "0x39456ffbfcb86733", "0xaaebb5e9afba3337", "0xf092fdfb9df66980" + ]} + ], + "dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"], + "dataset_last_index": 268435455, + "dataset_last": "0xa33ada72", + "dataset_samples": [{"index": 59471966, "value": "0xe8b73d94"}, {"index": 217795994, "value": "0x337028b5"}, {"index": 208353206, "value": "0xafe148c9"}, {"index": 42483309, "value": "0xab99f7ae"}, {"index": 172547758, "value": "0x434ea619"}, {"index": 148076330, "value": "0xd85cb880"}, {"index": 183853158, "value": "0x54764c7f"}, {"index": 214389424, "value": "0x82c7e420"}, {"index": 267488061, "value": "0xedf4cb9e"}, {"index": 169781097, "value": "0x9884c959"}, {"index": 184093494, "value": "0x223ee793"}, {"index": 153880993, "value": "0x3a9ccf69"}, {"index": 84977930, "value": "0x81da4fd2"}, {"index": 46426879, "value": "0xd6ce8cb9"}, {"index": 3093825, "value": "0xe3922dca"}, {"index": 225364072, "value": "0x3e7e6bde"}, {"index": 44593546, "value": "0x382a3aca"}, {"index": 260713159, "value": "0x567e7f7f"}, {"index": 168250303, "value": "0x25a0f084"}, {"index": 52384140, "value": "0xbfeef128"}, {"index": 223401610, "value": "0xe338abfb"}, {"index": 45554030, "value": "0x7c3b5280"}, {"index": 95410555, "value": "0x909bc5f1"}, {"index": 175039924, "value": "0xd8b74b9c"}, {"index": 79171087, "value": "0x8e31a22e"}, {"index": 267580473, "value": "0x26b5f1d8"}, {"index": 24168642, "value": "0x79122c00"}, {"index": 37981670, "value": "0xcafc3340"}, {"index": 171551130, "value": "0xd5e02ea3"}, {"index": 195559979, "value": "0x1aee1afd"}, {"index": 204611762, "value": "0xdb090d9a"}, {"index": 140997658, "value": "0xb049f435"}, {"index": 138925853, "value": "0x4954d8ba"}, {"index": 86637313, "value": "0x03797ba0"}, {"index": 20736778, "value": "0x196eefbd"}, {"index": 219665210, "value": "0xd153412a"}, {"index": 160430336, "value": "0xbe5d2c4b"}, {"index": 264654675, "value": "0xdaa14f0e"}, {"index": 8013395, "value": "0x8e61ed07"}, {"index": 228945585, "value": "0x9e9a64c6"}, {"index": 213884386, "value": "0x2e29ff36"}, {"index": 104419827, "value": "0x392a8589"}, {"index": 44185464, "value": "0xb56a5912"}, {"index": 142737231, "value": "0xfa6e8b57"}, {"index": 99284897, "value": "0xd1a737cb"}, {"index": 132475900, "value": "0xb0fa841a"}, {"index": 61861762, "value": "0xbe1c341f"}, {"index": 132056166, "value": "0xe25be0f1"}, {"index": 262388043, "value": "0xe937f543"}, {"index": 91878046, "value": "0xebab2248"}, {"index": 117353561, "value": "0x8e1b607a"}, {"index": 124768597, "value": "0x202a2fed"}, {"index": 71352993, "value": "0x95e2819c"}, {"index": 190698941, "value": "0x9c9652d4"}, {"index": 46055428, "value": "0x32fedef0"}, {"index": 55281366, "value": "0xdecfff82"}, {"index": 165145231, "value": "0xcb5d43e5"}, {"index": 106810753, "value": "0xb735806a"}, {"index": 171985651, "value": "0x8905939c"}, {"index": 232085256, "value": "0xfbf8472d"}, {"index": 159510492, "value": "0xada74e5d"}, {"index": 40072060, "value": "0x7ebdeeea"}, {"index": 209107596, "value": "0x0119f2b3"}, {"index": 39023794, "value": "0xa9a376b8"}], + "cache_head": ["0x355a86d2", "0x7957db1c", "0xd21772af", "0x6fc1e09b", "0xd55ce61d", "0x6e6a278b", "0xd3f543ce", "0x223d8e82", "0x143ab337", "0x2e9f05bd", "0x2eb389bf", "0x0c6e449e", "0x5cfa4222", "0xba6560fe", "0x8e3e1aa4", "0xdbcc1d53"], + "cache_last_line": ["0x41190d91", "0xbd277957", "0x22ddbb49", "0x6986f207", "0xdf69a4d6", "0x26401a3a", "0x818230fb", "0xc417122d", "0x3597b211", "0xb553ce55", "0xcf39cc0d", "0x3b7fc43a", "0x3fd43b00", "0x67e1c80e", "0xffa7ea7d", "0xca2960ab"], + "cache_fnv1a64": "0x48c4f5bf24166b2e", + "hot_head": ["0x8068cc73", "0x6036ebf9", "0xb604cd25", "0x8ffb840e", "0xc54074a2", "0x285c0695", "0x77512425", "0xc26a58a7", "0x72c88757", "0xc10fca78", "0x513825dd", "0x30d6ccc8", "0x9a05e7cf", "0xb9533f50", "0x4bac3ba0", "0xa5c19528"], + "hot_last_line": ["0x7c6d7ceb", "0xe24fcd46", "0xe5cce976", "0x3ecffe59", "0x94e98b66", "0x3f6ef45c", "0xa3715aa1", "0xdbe35281", "0xba05d27b", "0x00541963", "0xe636f453", "0xd3972366", "0x529599b8", "0x79ae3c8c", "0x48436863", "0x897a8bf3"], + "hot_fnv1a64": "0x77ca4b9527104530" +} diff --git a/proto-cuda/packs-ca2-hot/hot64k4/kernel.cl b/proto-cuda/packs-ca2-hot/hot64k4/kernel.cl new file mode 100644 index 000000000..5efed6def --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4/kernel.cl @@ -0,0 +1,305 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal). +// Built from source at runtime by proto-opencl/host.c, which passes these defines: +// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit) +// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default) +// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32 +// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition +// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units; +// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32. +#ifndef IGNEUM_GROUP +#define IGNEUM_GROUP 32 +#endif +#ifndef IGNEUM_EXCHANGE +#define IGNEUM_EXCHANGE 0 +#endif +#ifdef __OPENCL_VERSION__ +#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1))) +#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n] +#if IGNEUM_EXCHANGE == 1 +#ifdef cl_khr_subgroups +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif +#ifdef cl_khr_subgroup_shuffle +#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable +#endif +#elif IGNEUM_EXCHANGE == 2 +#pragma OPENCL EXTENSION cl_intel_subgroups : enable +#endif +#else +// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros. +#include "emu_opencl.h" +#endif + +#if IGNEUM_EXCHANGE == 1 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#elif IGNEUM_EXCHANGE == 2 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#else +// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per +// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane +// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's +// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier. +#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; } +#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; } +#endif + +// Hot table (64 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +static inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32. +static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); } +// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x. +static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); } +static inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +static inline void mh_chacha_block(const uint* x, uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +static inline void mh_cache_segment(__global uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +static inline void mh_mixer(uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +static inline void mh_item(__global const uint* cache, uint t, uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + __global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item. +// The same constants as memhard.h in this pack (one emitter, three dialects). +__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) { + uint t = (uint)get_global_id(0); + if (t < nItems) { + uint s[16]; + mh_item(cache, t, s); + __global uint* d = ds + ((ulong)t * 16u); + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (docs/plans/hot-table.md): 64 MiB = 16384 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +static inline void ht_segment(__global uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one work-item per segment. +__kernel void igneum_hot_fill(__global uint* hot, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the +// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and +// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). +IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ ds[r2 & mask]; // 4 load + r5 = r5 ^ ds[r7 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl + r1 = mul_hi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = mul_hi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl + r5 = r5 * r0; // 19 mul + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl + r6 = mul_hi(r6, r2); // 22 mulhi + r6 = r6 ^ ds[r1 & mask]; // 23 load + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ hot[mul_hi(r0, HOT_WORDS)]; // 32 hot + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[mul_hi(r1, HOT_WORDS)]; // 34 hot + r0 = mul_hi(r0, r5); // 35 mulhi + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl + r7 = r7 ^ ds[r0 & mask]; // 37 load + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ ds[r7 & mask]; // 44 load + r2 = r2 * r3; // 45 mul + r1 = mul_hi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ hot[mul_hi(r2, HOT_WORDS)]; // 59 hot + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} + +#if IGNEUM_EXCHANGE != 0 +// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the +// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a +// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md. +IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) { + if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); } +} +#endif diff --git a/proto-cuda/packs-ca2-hot/hot64k4/kernel.cu b/proto-cuda/packs-ca2-hot/hot64k4/kernel.cu new file mode 100644 index 000000000..092127f11 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4/kernel.cu @@ -0,0 +1,179 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal). +// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC. +#include +#include +#include "program.h" +#include "memhard.h" + +// Hot table (64 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +__device__ __forceinline__ uint32_t splitmix32(uint32_t x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31. +__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } +// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x. +__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) { + uint32_t x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item. +// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host. +__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) { + uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x; + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) { + uint32_t t = blockIdx.x * blockDim.x + threadIdx.x; + if (t < nItems) { + uint32_t s[16]; + mh_item(cache, t, s); + uint32_t* d = ds + (size_t)t * 16u; + for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (ht_segment is in memhard.h): one thread per segment. +__global__ void igneum_hot_fill(uint32_t* hot, uint32_t nSegments) { + uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x; + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every +// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a +// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid. +__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot) { + uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x; + uint32_t nonce = baseNonce + gid; + uint32_t r0, r1, r2, r3, r4, r5, r6, r7; + { uint32_t x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint32_t x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint32_t x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint32_t x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint32_t x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint32_t x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint32_t x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint32_t x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint32_t it = 0u; it < 8u; ++it) { + uint32_t sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ ds[r2 & mask]; // 4 load + r5 = r5 ^ ds[r7 & mask]; // 5 load + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 6 shfl + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 8); // 7 shfl + r1 = __umulhi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = __umulhi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[__umulhi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 18 shfl + r5 = r5 * r0; // 19 mul + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 20 shfl + r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 16); // 21 shfl + r6 = __umulhi(r6, r2); // 22 mulhi + r6 = r6 ^ ds[r1 & mask]; // 23 load + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r6, 2); // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ hot[__umulhi(r0, HOT_WORDS)]; // 32 hot + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[__umulhi(r1, HOT_WORDS)]; // 34 hot + r0 = __umulhi(r0, r5); // 35 mulhi + r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // 36 shfl + r7 = r7 ^ ds[r0 & mask]; // 37 load + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 4); // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ ds[r7 & mask]; // 44 load + r2 = r2 * r3; // 45 mul + r1 = __umulhi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ hot[__umulhi(r2, HOT_WORDS)]; // 59 hot + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo; +} + +// Host-side launch wrappers. Declared in program.h, called from host.cu. +cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) { + if (nSegments == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nSegments + block - 1u) / block; + igneum_cache_fill<<>>(cache, nSegments); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) { + if (nItems == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nItems + block - 1u) / block; + igneum_build<<>>(ds, cache, nItems); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments) { + if (nSegments == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nSegments + block - 1u) / block; + igneum_hot_fill<<>>(hot, nSegments); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot, + uint32_t nonces, uint32_t blockWarps) { + if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue; + uint32_t block = 32u * blockWarps; + if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue; + igneum_hash<<>>(ds, out, baseNonce, mask, hot); + return cudaGetLastError(); +} + +cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) { + cudaFuncAttributes attr; + cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash); + if (e != cudaSuccess) return e; + *numRegs = attr.numRegs; + return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0); +} diff --git a/proto-cuda/packs-ca2-hot/hot64k4/kernel_bound.cl b/proto-cuda/packs-ca2-hot/hot64k4/kernel_bound.cl new file mode 100644 index 000000000..b22617173 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4/kernel_bound.cl @@ -0,0 +1,399 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal). +// Built from source at runtime by proto-opencl/host.c, which passes these defines: +// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit) +// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default) +// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32 +// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition +// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units; +// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32. +#ifndef IGNEUM_GROUP +#define IGNEUM_GROUP 32 +#endif +#ifndef IGNEUM_EXCHANGE +#define IGNEUM_EXCHANGE 0 +#endif +#ifdef __OPENCL_VERSION__ +#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1))) +#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n] +#if IGNEUM_EXCHANGE == 1 +#ifdef cl_khr_subgroups +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif +#ifdef cl_khr_subgroup_shuffle +#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable +#endif +#elif IGNEUM_EXCHANGE == 2 +#pragma OPENCL EXTENSION cl_intel_subgroups : enable +#endif +#else +// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros. +#include "emu_opencl.h" +#endif + +#if IGNEUM_EXCHANGE == 1 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#elif IGNEUM_EXCHANGE == 2 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#else +// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per +// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane +// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's +// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier. +#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; } +#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; } +#endif + +// Hot table (64 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +static inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32. +static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); } +// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x. +static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); } +static inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +static inline void mh_chacha_block(const uint* x, uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +static inline void mh_cache_segment(__global uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +static inline void mh_mixer(uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +static inline void mh_item(__global const uint* cache, uint t, uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + __global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item. +// The same constants as memhard.h in this pack (one emitter, three dialects). +__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) { + uint t = (uint)get_global_id(0); + if (t < nItems) { + uint s[16]; + mh_item(cache, t, s); + __global uint* d = ds + ((ulong)t * 16u); + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (docs/plans/hot-table.md): 64 MiB = 16384 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +static inline void ht_segment(__global uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one work-item per segment. +__kernel void igneum_hot_fill(__global uint* hot, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the +// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and +// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). +IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ ds[r2 & mask]; // 4 load + r5 = r5 ^ ds[r7 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl + r1 = mul_hi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = mul_hi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl + r5 = r5 * r0; // 19 mul + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl + r6 = mul_hi(r6, r2); // 22 mulhi + r6 = r6 ^ ds[r1 & mask]; // 23 load + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ hot[mul_hi(r0, HOT_WORDS)]; // 32 hot + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[mul_hi(r1, HOT_WORDS)]; // 34 hot + r0 = mul_hi(r0, r5); // 35 mulhi + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl + r7 = r7 ^ ds[r0 & mask]; // 37 load + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ ds[r7 & mask]; // 44 load + r2 = r2 * r3; // 45 mul + r1 = mul_hi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ hot[mul_hi(r2, HOT_WORDS)]; // 59 hot + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} + +#if IGNEUM_EXCHANGE != 0 +// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the +// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a +// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md. +IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) { + if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); } +} +#endif + +// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash. +IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7]; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; } + { uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; } + { uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; } + { uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; } + { uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; } + { uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; } + { uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; } + { uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ ds[r2 & mask]; // 4 load + r5 = r5 ^ ds[r7 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl + r1 = mul_hi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = mul_hi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl + r5 = r5 * r0; // 19 mul + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl + r6 = mul_hi(r6, r2); // 22 mulhi + r6 = r6 ^ ds[r1 & mask]; // 23 load + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ hot[mul_hi(r0, HOT_WORDS)]; // 32 hot + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[mul_hi(r1, HOT_WORDS)]; // 34 hot + r0 = mul_hi(r0, r5); // 35 mulhi + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl + r7 = r7 ^ ds[r0 & mask]; // 37 load + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ ds[r7 & mask]; // 44 load + r2 = r2 * r3; // 45 mul + r1 = mul_hi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ hot[mul_hi(r2, HOT_WORDS)]; // 59 hot + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot64k4/kernel_bound.cu b/proto-cuda/packs-ca2-hot/hot64k4/kernel_bound.cu new file mode 100644 index 000000000..733413064 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4/kernel_bound.cu @@ -0,0 +1,125 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW. +// Host declarations (also in program_bound.h if present): +// struct IgneumInitWords { uint32_t w[8]; }; +// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, +// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps); +// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps); +#include +#include +#include "program.h" + +struct IgneumInitWords { uint32_t w[8]; }; + +// Hot table (64 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +__device__ __forceinline__ uint32_t splitmix32(uint32_t x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } +__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } + +__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw, const uint32_t* hot) { + uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x; + uint32_t nonce = baseNonce + gid; + uint32_t r0, r1, r2, r3, r4, r5, r6, r7; + { uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; } + { uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; } + { uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; } + { uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; } + { uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; } + { uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; } + { uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; } + { uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; } + + for (uint32_t it = 0u; it < 8u; ++it) { + uint32_t sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ ds[r2 & mask]; // 4 load + r5 = r5 ^ ds[r7 & mask]; // 5 load + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 6 shfl + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 8); // 7 shfl + r1 = __umulhi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = __umulhi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[__umulhi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 18 shfl + r5 = r5 * r0; // 19 mul + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 20 shfl + r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 16); // 21 shfl + r6 = __umulhi(r6, r2); // 22 mulhi + r6 = r6 ^ ds[r1 & mask]; // 23 load + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r6, 2); // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ hot[__umulhi(r0, HOT_WORDS)]; // 32 hot + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[__umulhi(r1, HOT_WORDS)]; // 34 hot + r0 = __umulhi(r0, r5); // 35 mulhi + r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // 36 shfl + r7 = r7 ^ ds[r0 & mask]; // 37 load + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 4); // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ ds[r7 & mask]; // 44 load + r2 = r2 * r3; // 45 mul + r1 = __umulhi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ hot[__umulhi(r2, HOT_WORDS)]; // 59 hot + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo; +} + +cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, + IgneumInitWords iw, const uint32_t* hot, uint32_t nonces, uint32_t blockWarps) { + if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue; + uint32_t block = 32u * blockWarps; + if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue; + igneum_hash_bound<<>>(ds, out, baseNonce, mask, iw, hot); + return cudaGetLastError(); +} + +cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) { + cudaFuncAttributes attr; + cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound); + if (e != cudaSuccess) return e; + *numRegs = attr.numRegs; + return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0); +} diff --git a/proto-cuda/packs-ca2-hot/hot64k4/memhard.h b/proto-cuda/packs-ca2-hot/hot64k4/memhard.h new file mode 100644 index 000000000..180f678b7 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4/memhard.h @@ -0,0 +1,129 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against. +// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference). +// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C. +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif +#if defined(__CUDACC__) +#define IGNEUM_HD __host__ __device__ __forceinline__ +#elif defined(_MSC_VER) && !defined(__cplusplus) +#define IGNEUM_HD static __inline +#else +#define IGNEUM_HD static inline +#endif +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) { + for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint32_t r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) { + uint32_t prev[16]; uint32_t x[16]; uint32_t y[16]; + for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +IGNEUM_HD void mh_item(const uint32_t* cache, uint32_t t, uint32_t* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint32_t r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +IGNEUM_HD uint32_t mh_word(const uint32_t* cache, uint32_t w) { uint32_t s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Hot table (docs/plans/hot-table.md): 64 MiB = 16384 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +IGNEUM_HD void ht_segment(uint32_t* hot, uint32_t seg) { + uint32_t prev[16]; uint32_t x[16]; uint32_t y[16]; + for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + uint32_t* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} diff --git a/proto-cuda/packs-ca2-hot/hot64k4/memhard.metal b/proto-cuda/packs-ca2-hot/hot64k4/memhard.metal new file mode 100644 index 000000000..c2ec5702e --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4/memhard.metal @@ -0,0 +1,131 @@ +#include +using namespace metal; +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +inline void mh_chacha_block(const thread uint* x, thread uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +inline void mh_cache_segment(device uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +inline void mh_mixer(thread uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +inline void mh_item(device const uint* cache, uint t, thread uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +inline uint mh_word(device const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// One thread per segment (2^16 threads). +kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) { + mh_cache_segment(cache, gid); +} +// One thread per 64-byte item (dataset words / 16 threads). +kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]], + uint gid [[thread_position_in_grid]]) { + uint s[16]; + mh_item(cache, gid, s); + device uint* d = dataset + gid * 16u; + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; +} + +// Hot table (docs/plans/hot-table.md): 64 MiB = 16384 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +inline void ht_segment(device uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + device uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one thread per segment (IGNEUM_HOT_SEGMENTS threads). +kernel void igneum_hot_fill(device uint* hot [[buffer(0)]], uint gid [[thread_position_in_grid]]) { + ht_segment(hot, gid); +} diff --git a/proto-cuda/packs-ca2-hot/hot64k4/program.h b/proto-cuda/packs-ca2-hot/hot64k4/program.h new file mode 100644 index 000000000..a52279c4f --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4/program.h @@ -0,0 +1,70 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Program metadata for host.cu plus the launch wrappers defined in kernel.cu. +// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros. +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif +#ifndef IGNEUM_NO_CUDA +#include +#endif + +#define IGNEUM_SEED_STRING "igneum-genesis" +#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973" +#define IGNEUM_GENERATOR 2 +#define IGNEUM_PROGRAM_ATTEMPT 0 +#define IGNEUM_PROGRAM_ID 0x322a64466d4a02ccull +#define IGNEUM_DAY_STRING "2026-10-03" +#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033" +#define IGNEUM_DAY0 0x3067619fu +#define IGNEUM_DAY1 0x3c269176u +#define IGNEUM_DATASET_LOG2 28 +#define IGNEUM_MASK 0x0fffffffu +#define IGNEUM_LANES 32 +#define IGNEUM_ITERATIONS 8 +#define IGNEUM_INSTR_COUNT 64 +#define IGNEUM_LOADS_PER_HASH 128 +#define IGNEUM_WIDE_LOADS_PER_HASH 0 +#define IGNEUM_OP_MIX "load=12 add=8 shfl=8 xor=6 mad=5 mul=5 mulhi=5 hot=4 sub=4 rotl=3 rotr=3 or=1" +// Class v3 construction (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): version 2 loads; the dataset item +// derivation applies the mixer IGNEUM_MIXER_MULT times per round (memhard.h), and the cache follows the growth rule. +#define IGNEUM_LOAD_CLASS "hot64k4" +#define IGNEUM_LOAD_SLOTS 16 +#define IGNEUM_LOAD_MIX { 100, 0, 0 } +#define IGNEUM_LOAD_WIDTH_COUNTS { 12, 0, 0 } // loads of 4, 16, 64 bytes per program +#define IGNEUM_BYTES_PER_HASH 384 +#define IGNEUM_FOLD_ROT 11 +#define IGNEUM_FOLD_MUL 0x9e3779b1u +// Hot table (hot-table experiment, 5 October 2026, docs/plans/hot-table.md): NOT the lottery hash. IGNEUM_HOT_SLOTS of the +// 16 load slots read H[mulhi(src, IGNEUM_HOT_WORDS)], H = IGNEUM_HOT_MB MiB of chained ChaCha12 lines (the cache chain) under +// KH = seed_words("igneum-hot/" || epoch seed bytes) and the tag "Igne" "umHT"; filled by igneum_hot_fill once per epoch. +#define IGNEUM_HOT_MB 64 +#define IGNEUM_HOT_WORDS 0x01000000u +#define IGNEUM_HOT_SEGMENTS 16384u +#define IGNEUM_HOT_SLOTS 4 // hot loads per program (32 per hash), replacing the dataset loads (12 of them) +#define IGNEUM_HOT_ADDED 0 +#define IGNEUM_HOT_KEY_INIT { 0x3a48bef5u, 0x6b54b1a1u, 0x9ff897c3u, 0x7d5d85e4u, 0xcd35379fu, 0x9d76bb86u, 0xbe2affb1u, 0x0f8f80b4u } +// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h) +#define IGNEUM_DATASET_MODE 1 + +#define IGNEUM_SEEDW_INIT { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u } +#define IGNEUM_KEY_INIT { 0x3067619fu, 0x3c269176u, 0x84a03b03u, 0xf8c63294u, 0xff977c5bu, 0xe60def3eu, 0x63630141u, 0xb8fbcb58u } +#define IGNEUM_CACHE_LOG2_WORDS 26 +#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6 +#define IGNEUM_CACHE_SEGMENTS 65536u +#define IGNEUM_ITEM_ROUNDS 8 +#define IGNEUM_MIX_ROT_INIT { 20u, 20u, 19u, 4u, 26u, 3u, 3u, 27u } +#define IGNEUM_MIX_MUL_INIT { 0x42146205u, 0x52cbe0fbu, 0x7ecf4a03u, 0x6728907fu, 0xd81d9751u, 0x132952c3u, 0xf60de277u, 0x05358035u, 0xbaf6499du, 0xe4db9667u, 0x3e98f45du, 0xd0004eddu, 0x2691630du, 0x9beb3bcfu, 0xab310379u, 0x99cfb423u } +#define IGNEUM_MIX_RC_INIT { 0xbab68293u, 0xcc162340u, 0x6ce151ccu, 0xe62b8997u, 0xc9c80297u, 0xf74a1654u, 0x3d704af5u, 0x3cf522b7u, 0x2b9cac04u, 0xa880ac10u, 0x13e5dd1du, 0x6fc3e233u, 0x2d83eeacu, 0x9006e8bfu, 0x2c4b5362u, 0x31b49ee2u } + +#ifndef IGNEUM_NO_CUDA +// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError(). +cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments); +cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems); +cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments); +cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot, + uint32_t nonces, uint32_t blockWarps); +cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps); +#endif diff --git a/proto-cuda/packs-ca2-hot/hot64k4/program.json b/proto-cuda/packs-ca2-hot/hot64k4/program.json new file mode 100644 index 000000000..436d58d65 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4/program.json @@ -0,0 +1,129 @@ +{ + "format": "igneum-program-pack-3", + "generator": 2, + "attempt": 0, + "program_id": "0x322a64466d4a02cc", + "program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32", + "dataset_mode": "memory-hard", + "seed": "igneum-genesis", + "seed_bytes": "69676e65756d2d67656e65736973", + "seed_words": ["0x67a9a7be", "0x1a155b25", "0xfddfb732", "0x4b5af2e8", "0xc55caf33", "0xa27c13b7", "0x06628a48", "0x03852469"], + "seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32", + "generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried", + "lanes": 32, + "registers": 8, + "iterations": 8, + "instruction_count": 64, + "loads_per_hash": 128, + "load_class": "hot64k4", + "load_slots": 16, + "load_mix_percent_4_16_64": [100, 0, 0], + "load_width_counts_4_16_64": [12, 0, 0], + "bytes_per_hash": 384, + "wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots", + "hot_table": {"mb": 64, "words": 16777216, "segments": 16384, "slots": 4, "form": "replaced: k of the class's load slots read the table", "dataset_slots": 12, "hot_loads_per_hash": 32, "key": ["0x3a48bef5", "0x6b54b1a1", "0x9ff897c3", "0x7d5d85e4", "0xcd35379f", "0x9d76bb86", "0xbe2affb1", "0x0f8f80b4"], "key_derivation": "seed_words_from_bytes('igneum-hot/' || seed_bytes), the same for every attempt of the epoch", "tag": ["0x49676e65", "0x756d4854"], "chain": "the cache chain of dataset.cache with the hot key and tag: in_j = prev ^ (sigma || key || seg || j || tag); line_j = block(in_j); prev_0 = 0", "load": "dst = dst ^ hot[mulhi(src, words)] (the high 32 bits of the 64-bit product; the index lies in [0, words) for any size)", "slots_rule": "the first k load slots in the Fisher-Yates draw order after the scratch slots (a uniform k-subset); no extra draw, so with version 2 widths and no scratch the program is the version 2 program with k loads redirected", "acceptance_stand_in": "dataset_elem(mulhi(src, words), seed_words[2], seed_words[3])", "program_id": "the read-width id with 'hot/' || mb || k appended", "spec": "docs/plans/hot-table.md"}, + "op_mix": {"load": 12, "add": 8, "shfl": 8, "xor": 6, "mad": 5, "mul": 5, "mulhi": 5, "hot": 4, "sub": 4, "rotl": 3, "rotr": 3, "or": 1}, + "register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]", + "splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16", + "iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order", + "output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo", + "op_semantics": { + "add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)", + "sub": "dst = dst - src", + "mul": "dst = dst * src (low 32)", + "mulhi": "dst = high 32 bits of dst * src", + "xor": "dst = dst ^ src", + "or": "dst = dst | src", + "rotl": "dst = rotl(dst, rot), rot in 1..31", + "rotr": "dst = rotr(dst, src & 31)", + "mad": "dst = src * src2 + dst", + "shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp", + "load": "dst = dst ^ dataset[src & dataset.mask]", + "wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)", + "hot": "dst = dst ^ hot[mulhi(src, hot_table.words)] (hot-table experiment)" + }, + "dataset": { + "log2_words": 28, + "bytes": 1073741824, + "mask": "0x0fffffff", + "day": "2026-10-03", + "day_bytes": "6461792f323032362d31302d3033", + "day_words_from": "seed_words_from_bytes(day_bytes)", + "d0": "0x3067619f", + "d1": "0x3c269176", + "mode": "memory-hard", + "spec": "proto-metal/MEMHARD.md", + "key": ["0x3067619f", "0x3c269176", "0x84a03b03", "0xf8c63294", "0xff977c5b", "0xe60def3e", "0x63630141", "0xb8fbcb58"], + "key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]", + "cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"}, + "mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [20, 20, 19, 4, 26, 3, 3, 27], "mul": ["0x42146205", "0x52cbe0fb", "0x7ecf4a03", "0x6728907f", "0xd81d9751", "0x132952c3", "0xf60de277", "0x05358035", "0xbaf6499d", "0xe4db9667", "0x3e98f45d", "0xd0004edd", "0x2691630d", "0x9beb3bcf", "0xab310379", "0x99cfb423"], "rc": ["0xbab68293", "0xcc162340", "0x6ce151cc", "0xe62b8997", "0xc9c80297", "0xf74a1654", "0x3d704af5", "0x3cf522b7", "0x2b9cac04", "0xa880ac10", "0x13e5dd1d", "0x6fc3e233", "0x2d83eeac", "0x9006e8bf", "0x2c4b5362", "0x31b49ee2"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"}, + "item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: s = M_r(s); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then s = M_8(s); item(t) = s", + "word": "dataset[w] = item(w >> 4)[w & 15]" + }, + "instructions": [ + {"i": 0, "op": "mad", "dst": 2, "src": 3, "src2": 4, "imm": "0xbf7b174d", "imm2": "0x337b762e", "rot": 17, "bit": 2, "mask": 2, "width": 1}, + {"i": 1, "op": "mad", "dst": 2, "src": 1, "src2": 1, "imm": "0xdd04a5da", "imm2": "0x42da7657", "rot": 15, "bit": 30, "mask": 16, "width": 1}, + {"i": 2, "op": "mad", "dst": 2, "src": 3, "src2": 2, "imm": "0x734003fa", "imm2": "0x5bb67700", "rot": 3, "bit": 20, "mask": 1, "width": 1}, + {"i": 3, "op": "xor", "dst": 3, "src": 5, "src2": 5, "imm": "0xc55a1b1c", "imm2": "0xa19720f3", "rot": 7, "bit": 8, "mask": 1, "width": 1}, + {"i": 4, "op": "load", "dst": 7, "src": 2, "src2": 5, "imm": "0xad572dd7", "imm2": "0x9ceb3ea7", "rot": 18, "bit": 30, "mask": 2, "width": 1}, + {"i": 5, "op": "load", "dst": 5, "src": 7, "src2": 3, "imm": "0x769a53be", "imm2": "0x80f9067e", "rot": 12, "bit": 22, "mask": 1, "width": 1}, + {"i": 6, "op": "shfl", "dst": 1, "src": 4, "src2": 0, "imm": "0xd3613d88", "imm2": "0x262fb219", "rot": 10, "bit": 30, "mask": 8, "width": 1}, + {"i": 7, "op": "shfl", "dst": 7, "src": 3, "src2": 4, "imm": "0xce38e42f", "imm2": "0xb868b818", "rot": 11, "bit": 8, "mask": 8, "width": 1}, + {"i": 8, "op": "mulhi", "dst": 1, "src": 5, "src2": 2, "imm": "0xa5eebca5", "imm2": "0x5703a72b", "rot": 13, "bit": 13, "mask": 16, "width": 1}, + {"i": 9, "op": "rotr", "dst": 6, "src": 3, "src2": 4, "imm": "0x17a5a9c7", "imm2": "0xdcfb93a1", "rot": 20, "bit": 27, "mask": 2, "width": 1}, + {"i": 10, "op": "or", "dst": 3, "src": 4, "src2": 1, "imm": "0xccb7d785", "imm2": "0xc335364c", "rot": 14, "bit": 12, "mask": 4, "width": 1}, + {"i": 11, "op": "load", "dst": 4, "src": 3, "src2": 2, "imm": "0x88cb9af3", "imm2": "0x4e7dc10d", "rot": 24, "bit": 17, "mask": 4, "width": 1}, + {"i": 12, "op": "mulhi", "dst": 0, "src": 4, "src2": 4, "imm": "0x45374321", "imm2": "0x3cd91989", "rot": 11, "bit": 4, "mask": 2, "width": 1}, + {"i": 13, "op": "add", "dst": 5, "src": 1, "src2": 2, "imm": "0xc7934706", "imm2": "0xd3177981", "rot": 16, "bit": 30, "mask": 2, "width": 1}, + {"i": 14, "op": "load", "dst": 0, "src": 4, "src2": 3, "imm": "0xfd7f56bb", "imm2": "0x65e14f52", "rot": 13, "bit": 22, "mask": 2, "width": 1}, + {"i": 15, "op": "sub", "dst": 2, "src": 4, "src2": 4, "imm": "0x35a80b49", "imm2": "0x060f2d13", "rot": 16, "bit": 20, "mask": 16, "width": 1}, + {"i": 16, "op": "hot", "dst": 2, "src": 0, "src2": 2, "imm": "0xae0a32c2", "imm2": "0x4c2a4cfe", "rot": 8, "bit": 31, "mask": 16, "width": 1}, + {"i": 17, "op": "load", "dst": 7, "src": 2, "src2": 6, "imm": "0x82a84cc3", "imm2": "0x21a38d68", "rot": 15, "bit": 21, "mask": 2, "width": 1}, + {"i": 18, "op": "shfl", "dst": 7, "src": 3, "src2": 3, "imm": "0xa3818806", "imm2": "0x8f66b5c8", "rot": 14, "bit": 6, "mask": 4, "width": 1}, + {"i": 19, "op": "mul", "dst": 5, "src": 0, "src2": 1, "imm": "0xa00de107", "imm2": "0x77bfcaa5", "rot": 3, "bit": 10, "mask": 2, "width": 1}, + {"i": 20, "op": "shfl", "dst": 3, "src": 4, "src2": 7, "imm": "0x1d2b8cab", "imm2": "0x80b4f9a2", "rot": 14, "bit": 25, "mask": 2, "width": 1}, + {"i": 21, "op": "shfl", "dst": 2, "src": 4, "src2": 5, "imm": "0x3ac915d2", "imm2": "0x5fba7bc2", "rot": 16, "bit": 1, "mask": 16, "width": 1}, + {"i": 22, "op": "mulhi", "dst": 6, "src": 2, "src2": 0, "imm": "0xdc3ec8fd", "imm2": "0x599e2fa3", "rot": 22, "bit": 3, "mask": 2, "width": 1}, + {"i": 23, "op": "load", "dst": 6, "src": 1, "src2": 5, "imm": "0x2a6b16d5", "imm2": "0xd73e396f", "rot": 28, "bit": 29, "mask": 2, "width": 1}, + {"i": 24, "op": "mul", "dst": 5, "src": 0, "src2": 7, "imm": "0x376d0223", "imm2": "0xe1c2169a", "rot": 4, "bit": 16, "mask": 16, "width": 1}, + {"i": 25, "op": "rotl", "dst": 5, "src": 7, "src2": 3, "imm": "0x78ad8c60", "imm2": "0x6f5b77d5", "rot": 19, "bit": 11, "mask": 16, "width": 1}, + {"i": 26, "op": "shfl", "dst": 7, "src": 6, "src2": 5, "imm": "0x93915b9f", "imm2": "0x1e61fb6b", "rot": 28, "bit": 23, "mask": 2, "width": 1}, + {"i": 27, "op": "xor", "dst": 0, "src": 5, "src2": 7, "imm": "0x6378fe15", "imm2": "0x66c78f42", "rot": 12, "bit": 31, "mask": 8, "width": 1}, + {"i": 28, "op": "xor", "dst": 0, "src": 4, "src2": 7, "imm": "0x20a57fda", "imm2": "0x088c848e", "rot": 16, "bit": 13, "mask": 4, "width": 1}, + {"i": 29, "op": "sub", "dst": 3, "src": 0, "src2": 2, "imm": "0x49d95fd5", "imm2": "0x1a5f946a", "rot": 6, "bit": 12, "mask": 1, "width": 1}, + {"i": 30, "op": "mul", "dst": 5, "src": 1, "src2": 7, "imm": "0x0a816217", "imm2": "0x405c4f73", "rot": 13, "bit": 27, "mask": 4, "width": 1}, + {"i": 31, "op": "load", "dst": 7, "src": 2, "src2": 2, "imm": "0x09ed045e", "imm2": "0xd69c4715", "rot": 5, "bit": 9, "mask": 2, "width": 1}, + {"i": 32, "op": "hot", "dst": 1, "src": 0, "src2": 6, "imm": "0xeb79ea49", "imm2": "0xcc587f5a", "rot": 6, "bit": 8, "mask": 16, "width": 1}, + {"i": 33, "op": "xor", "dst": 5, "src": 6, "src2": 1, "imm": "0x3027401e", "imm2": "0x5f20c27e", "rot": 18, "bit": 9, "mask": 2, "width": 1}, + {"i": 34, "op": "hot", "dst": 5, "src": 1, "src2": 3, "imm": "0x0e1cab07", "imm2": "0x09356c5b", "rot": 19, "bit": 31, "mask": 1, "width": 1}, + {"i": 35, "op": "mulhi", "dst": 0, "src": 5, "src2": 2, "imm": "0x90e31357", "imm2": "0xabd32484", "rot": 26, "bit": 5, "mask": 8, "width": 1}, + {"i": 36, "op": "shfl", "dst": 5, "src": 2, "src2": 5, "imm": "0xee9a955f", "imm2": "0x31b3faed", "rot": 8, "bit": 24, "mask": 4, "width": 1}, + {"i": 37, "op": "load", "dst": 7, "src": 0, "src2": 7, "imm": "0x3ba2f832", "imm2": "0x1160dcd3", "rot": 4, "bit": 29, "mask": 1, "width": 1}, + {"i": 38, "op": "add", "dst": 3, "src": 1, "src2": 7, "imm": "0x75ba2fad", "imm2": "0x230c005c", "rot": 4, "bit": 27, "mask": 1, "width": 1}, + {"i": 39, "op": "shfl", "dst": 1, "src": 5, "src2": 3, "imm": "0xdbf37e75", "imm2": "0xb5ac1969", "rot": 30, "bit": 13, "mask": 4, "width": 1}, + {"i": 40, "op": "xor", "dst": 2, "src": 5, "src2": 5, "imm": "0x47f136c5", "imm2": "0x06ce9153", "rot": 19, "bit": 10, "mask": 2, "width": 1}, + {"i": 41, "op": "mad", "dst": 3, "src": 6, "src2": 3, "imm": "0xce13eff8", "imm2": "0x04cc1d55", "rot": 3, "bit": 1, "mask": 4, "width": 1}, + {"i": 42, "op": "sub", "dst": 6, "src": 7, "src2": 1, "imm": "0x6a65ab71", "imm2": "0x8fbc1bcd", "rot": 4, "bit": 1, "mask": 8, "width": 1}, + {"i": 43, "op": "xor", "dst": 7, "src": 0, "src2": 7, "imm": "0xdaeb4928", "imm2": "0xc0423027", "rot": 24, "bit": 11, "mask": 8, "width": 1}, + {"i": 44, "op": "load", "dst": 1, "src": 7, "src2": 7, "imm": "0x778f01c9", "imm2": "0x28cedcea", "rot": 12, "bit": 4, "mask": 16, "width": 1}, + {"i": 45, "op": "mul", "dst": 2, "src": 3, "src2": 2, "imm": "0xf4264f1b", "imm2": "0x0f627d56", "rot": 5, "bit": 28, "mask": 8, "width": 1}, + {"i": 46, "op": "mulhi", "dst": 1, "src": 5, "src2": 1, "imm": "0xffb2147a", "imm2": "0xccde9b05", "rot": 13, "bit": 9, "mask": 2, "width": 1}, + {"i": 47, "op": "sub", "dst": 4, "src": 3, "src2": 5, "imm": "0x73b36234", "imm2": "0x3f5d5997", "rot": 7, "bit": 18, "mask": 2, "width": 1}, + {"i": 48, "op": "rotr", "dst": 2, "src": 6, "src2": 3, "imm": "0x3a4d9aa9", "imm2": "0x212bec7b", "rot": 4, "bit": 29, "mask": 16, "width": 1}, + {"i": 49, "op": "load", "dst": 3, "src": 5, "src2": 2, "imm": "0x626f11df", "imm2": "0x56cd5bfd", "rot": 7, "bit": 1, "mask": 1, "width": 1}, + {"i": 50, "op": "add", "dst": 1, "src": 5, "src2": 6, "imm": "0x81b8bc2c", "imm2": "0x1907970c", "rot": 28, "bit": 7, "mask": 4, "width": 1}, + {"i": 51, "op": "mul", "dst": 0, "src": 2, "src2": 2, "imm": "0xa8848b30", "imm2": "0xef6ac348", "rot": 9, "bit": 15, "mask": 8, "width": 1}, + {"i": 52, "op": "add", "dst": 0, "src": 2, "src2": 0, "imm": "0x4f92b968", "imm2": "0x699fd448", "rot": 22, "bit": 6, "mask": 4, "width": 1}, + {"i": 53, "op": "add", "dst": 1, "src": 0, "src2": 2, "imm": "0x2bb965af", "imm2": "0x77b1520d", "rot": 2, "bit": 12, "mask": 8, "width": 1}, + {"i": 54, "op": "rotl", "dst": 7, "src": 1, "src2": 0, "imm": "0x553e678b", "imm2": "0x3cc8eae0", "rot": 14, "bit": 20, "mask": 2, "width": 1}, + {"i": 55, "op": "add", "dst": 3, "src": 7, "src2": 2, "imm": "0x7b0fe07a", "imm2": "0xa54c55a0", "rot": 10, "bit": 1, "mask": 1, "width": 1}, + {"i": 56, "op": "load", "dst": 6, "src": 7, "src2": 4, "imm": "0x01eba9aa", "imm2": "0x2758c0f7", "rot": 14, "bit": 15, "mask": 4, "width": 1}, + {"i": 57, "op": "rotr", "dst": 1, "src": 5, "src2": 2, "imm": "0x1f5267b3", "imm2": "0x236f5a27", "rot": 2, "bit": 31, "mask": 16, "width": 1}, + {"i": 58, "op": "load", "dst": 5, "src": 4, "src2": 3, "imm": "0xa9954a9b", "imm2": "0x6a54d4e8", "rot": 11, "bit": 10, "mask": 16, "width": 1}, + {"i": 59, "op": "hot", "dst": 6, "src": 2, "src2": 4, "imm": "0x9923ff88", "imm2": "0x9357254e", "rot": 16, "bit": 1, "mask": 16, "width": 1}, + {"i": 60, "op": "mad", "dst": 3, "src": 5, "src2": 0, "imm": "0xc0cc51a6", "imm2": "0x3fd7701b", "rot": 20, "bit": 1, "mask": 4, "width": 1}, + {"i": 61, "op": "add", "dst": 5, "src": 7, "src2": 7, "imm": "0xaf9dd72d", "imm2": "0xad7493e7", "rot": 7, "bit": 31, "mask": 16, "width": 1}, + {"i": 62, "op": "add", "dst": 4, "src": 6, "src2": 2, "imm": "0x89841d87", "imm2": "0x1e07c3d9", "rot": 6, "bit": 27, "mask": 1, "width": 1}, + {"i": 63, "op": "rotl", "dst": 5, "src": 4, "src2": 2, "imm": "0xae210f8d", "imm2": "0x8e499ba4", "rot": 19, "bit": 9, "mask": 1, "width": 1} + ] +} diff --git a/proto-cuda/packs-ca2-hot/hot64k4/program.metal b/proto-cuda/packs-ca2-hot/hot64k4/program.metal new file mode 100644 index 000000000..0f37469b1 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4/program.metal @@ -0,0 +1,112 @@ +#include +using namespace metal; + +#define MASK 0x0fffffffu +// Hot table (64 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }; + +inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 +inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +kernel void igneum_hash(device const uint* dataset [[buffer(0)]], + device ulong* out [[buffer(1)]], + constant uint& baseNonce [[buffer(2)]], + device const uint* hot [[buffer(3)]], + uint gid [[thread_position_in_grid]]) { + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + { uint x = nonce ^ SEEDW[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ SEEDW[1]; } + { uint x = nonce ^ SEEDW[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ SEEDW[2]; } + { uint x = nonce ^ SEEDW[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ SEEDW[3]; } + { uint x = nonce ^ SEEDW[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ SEEDW[4]; } + { uint x = nonce ^ SEEDW[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ SEEDW[5]; } + { uint x = nonce ^ SEEDW[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ SEEDW[6]; } + { uint x = nonce ^ SEEDW[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ SEEDW[7]; } + { uint x = nonce ^ SEEDW[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ SEEDW[0]; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 + r2 = r1 * r1 + r2; // 1 + r2 = r3 * r2 + r2; // 2 + r3 = r3 ^ r5; // 3 + r7 = r7 ^ dataset[r2 & MASK]; // 4 + r5 = r5 ^ dataset[r7 & MASK]; // 5 + r1 = r1 ^ simd_shuffle_xor(r4, (ushort)8); // 6 + r7 = r7 ^ simd_shuffle_xor(r3, (ushort)8); // 7 + r1 = mulhi(r1, r5); // 8 + r6 = rotr_var(r6, r3); // 9 + r3 = r3 | r4; // 10 + r4 = r4 ^ dataset[r3 & MASK]; // 11 + r0 = mulhi(r0, r4); // 12 + r5 = r5 + r1 + select(0xc7934706u, 0xd3177981u, ((sel >> 30u) & 1u) != 0u); // 13 + r0 = r0 ^ dataset[r4 & MASK]; // 14 + r2 = r2 - r4; // 15 + r2 = r2 ^ hot[mulhi(r0, HOT_WORDS)]; // 16 + r7 = r7 ^ dataset[r2 & MASK]; // 17 + r7 = r7 ^ simd_shuffle_xor(r3, (ushort)4); // 18 + r5 = r5 * r0; // 19 + r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 20 + r2 = r2 ^ simd_shuffle_xor(r4, (ushort)16); // 21 + r6 = mulhi(r6, r2); // 22 + r6 = r6 ^ dataset[r1 & MASK]; // 23 + r5 = r5 * r0; // 24 + r5 = rotl_imm(r5, 19u); // 25 + r7 = r7 ^ simd_shuffle_xor(r6, (ushort)2); // 26 + r0 = r0 ^ r5; // 27 + r0 = r0 ^ r4; // 28 + r3 = r3 - r0; // 29 + r5 = r5 * r1; // 30 + r7 = r7 ^ dataset[r2 & MASK]; // 31 + r1 = r1 ^ hot[mulhi(r0, HOT_WORDS)]; // 32 + r5 = r5 ^ r6; // 33 + r5 = r5 ^ hot[mulhi(r1, HOT_WORDS)]; // 34 + r0 = mulhi(r0, r5); // 35 + r5 = r5 ^ simd_shuffle_xor(r2, (ushort)4); // 36 + r7 = r7 ^ dataset[r0 & MASK]; // 37 + r3 = r3 + r1 + select(0x75ba2fadu, 0x230c005cu, ((sel >> 27u) & 1u) != 0u); // 38 + r1 = r1 ^ simd_shuffle_xor(r5, (ushort)4); // 39 + r2 = r2 ^ r5; // 40 + r3 = r6 * r3 + r3; // 41 + r6 = r6 - r7; // 42 + r7 = r7 ^ r0; // 43 + r1 = r1 ^ dataset[r7 & MASK]; // 44 + r2 = r2 * r3; // 45 + r1 = mulhi(r1, r5); // 46 + r4 = r4 - r3; // 47 + r2 = rotr_var(r2, r6); // 48 + r3 = r3 ^ dataset[r5 & MASK]; // 49 + r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 50 + r0 = r0 * r2; // 51 + r0 = r0 + r2 + select(0x4f92b968u, 0x699fd448u, ((sel >> 6u) & 1u) != 0u); // 52 + r1 = r1 + r0 + select(0x2bb965afu, 0x77b1520du, ((sel >> 12u) & 1u) != 0u); // 53 + r7 = rotl_imm(r7, 14u); // 54 + r3 = r3 + r7 + select(0x7b0fe07au, 0xa54c55a0u, ((sel >> 1u) & 1u) != 0u); // 55 + r6 = r6 ^ dataset[r7 & MASK]; // 56 + r1 = rotr_var(r1, r5); // 57 + r5 = r5 ^ dataset[r4 & MASK]; // 58 + r6 = r6 ^ hot[mulhi(r2, HOT_WORDS)]; // 59 + r3 = r5 * r0 + r3; // 60 + r5 = r5 + r7 + select(0xaf9dd72du, 0xad7493e7u, ((sel >> 31u) & 1u) != 0u); // 61 + r4 = r4 + r6 + select(0x89841d87u, 0x1e07c3d9u, ((sel >> 27u) & 1u) != 0u); // 62 + r5 = rotl_imm(r5, 19u); // 63 + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot64k4/program_bound.metal b/proto-cuda/packs-ca2-hot/hot64k4/program_bound.metal new file mode 100644 index 000000000..48eec87a7 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4/program_bound.metal @@ -0,0 +1,114 @@ +#include +using namespace metal; + +#define MASK 0x0fffffffu +// Hot table (64 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }; + +inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 +inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW. +kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], + device ulong* out [[buffer(1)]], + constant uint& baseNonce [[buffer(2)]], + constant uint* initw [[buffer(3)]], + device const uint* hot [[buffer(4)]], + uint gid [[thread_position_in_grid]]) { + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + { uint x = nonce ^ initw[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ initw[1]; } + { uint x = nonce ^ initw[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ initw[2]; } + { uint x = nonce ^ initw[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ initw[3]; } + { uint x = nonce ^ initw[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ initw[4]; } + { uint x = nonce ^ initw[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ initw[5]; } + { uint x = nonce ^ initw[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ initw[6]; } + { uint x = nonce ^ initw[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ initw[7]; } + { uint x = nonce ^ initw[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ initw[0]; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 + r2 = r1 * r1 + r2; // 1 + r2 = r3 * r2 + r2; // 2 + r3 = r3 ^ r5; // 3 + r7 = r7 ^ dataset[r2 & MASK]; // 4 + r5 = r5 ^ dataset[r7 & MASK]; // 5 + r1 = r1 ^ simd_shuffle_xor(r4, (ushort)8); // 6 + r7 = r7 ^ simd_shuffle_xor(r3, (ushort)8); // 7 + r1 = mulhi(r1, r5); // 8 + r6 = rotr_var(r6, r3); // 9 + r3 = r3 | r4; // 10 + r4 = r4 ^ dataset[r3 & MASK]; // 11 + r0 = mulhi(r0, r4); // 12 + r5 = r5 + r1 + select(0xc7934706u, 0xd3177981u, ((sel >> 30u) & 1u) != 0u); // 13 + r0 = r0 ^ dataset[r4 & MASK]; // 14 + r2 = r2 - r4; // 15 + r2 = r2 ^ hot[mulhi(r0, HOT_WORDS)]; // 16 + r7 = r7 ^ dataset[r2 & MASK]; // 17 + r7 = r7 ^ simd_shuffle_xor(r3, (ushort)4); // 18 + r5 = r5 * r0; // 19 + r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 20 + r2 = r2 ^ simd_shuffle_xor(r4, (ushort)16); // 21 + r6 = mulhi(r6, r2); // 22 + r6 = r6 ^ dataset[r1 & MASK]; // 23 + r5 = r5 * r0; // 24 + r5 = rotl_imm(r5, 19u); // 25 + r7 = r7 ^ simd_shuffle_xor(r6, (ushort)2); // 26 + r0 = r0 ^ r5; // 27 + r0 = r0 ^ r4; // 28 + r3 = r3 - r0; // 29 + r5 = r5 * r1; // 30 + r7 = r7 ^ dataset[r2 & MASK]; // 31 + r1 = r1 ^ hot[mulhi(r0, HOT_WORDS)]; // 32 + r5 = r5 ^ r6; // 33 + r5 = r5 ^ hot[mulhi(r1, HOT_WORDS)]; // 34 + r0 = mulhi(r0, r5); // 35 + r5 = r5 ^ simd_shuffle_xor(r2, (ushort)4); // 36 + r7 = r7 ^ dataset[r0 & MASK]; // 37 + r3 = r3 + r1 + select(0x75ba2fadu, 0x230c005cu, ((sel >> 27u) & 1u) != 0u); // 38 + r1 = r1 ^ simd_shuffle_xor(r5, (ushort)4); // 39 + r2 = r2 ^ r5; // 40 + r3 = r6 * r3 + r3; // 41 + r6 = r6 - r7; // 42 + r7 = r7 ^ r0; // 43 + r1 = r1 ^ dataset[r7 & MASK]; // 44 + r2 = r2 * r3; // 45 + r1 = mulhi(r1, r5); // 46 + r4 = r4 - r3; // 47 + r2 = rotr_var(r2, r6); // 48 + r3 = r3 ^ dataset[r5 & MASK]; // 49 + r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 50 + r0 = r0 * r2; // 51 + r0 = r0 + r2 + select(0x4f92b968u, 0x699fd448u, ((sel >> 6u) & 1u) != 0u); // 52 + r1 = r1 + r0 + select(0x2bb965afu, 0x77b1520du, ((sel >> 12u) & 1u) != 0u); // 53 + r7 = rotl_imm(r7, 14u); // 54 + r3 = r3 + r7 + select(0x7b0fe07au, 0xa54c55a0u, ((sel >> 1u) & 1u) != 0u); // 55 + r6 = r6 ^ dataset[r7 & MASK]; // 56 + r1 = rotr_var(r1, r5); // 57 + r5 = r5 ^ dataset[r4 & MASK]; // 58 + r6 = r6 ^ hot[mulhi(r2, HOT_WORDS)]; // 59 + r3 = r5 * r0 + r3; // 60 + r5 = r5 + r7 + select(0xaf9dd72du, 0xad7493e7u, ((sel >> 31u) & 1u) != 0u); // 61 + r4 = r4 + r6 + select(0x89841d87u, 0x1e07c3d9u, ((sel >> 27u) & 1u) != 0u); // 62 + r5 = rotl_imm(r5, 19u); // 63 + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot64k4/vectors.h b/proto-cuda/packs-ca2-hot/hot64k4/vectors.h new file mode 100644 index 000000000..e8e3ddb46 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4/vectors.h @@ -0,0 +1,67 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Expected outputs: igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif + +#define IGNEUM_VEC_WARPS 3 +static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u }; +static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = { + { // base nonce 0 + 0xaa6fadb5bf6a8657ull, 0x72a8df1dfcdc031dull, 0xdfc9bfb4dfc1fe4cull, 0x4f1d147208fe55a8ull, 0x857268e9be0d985aull, 0x2cd63cf21c16a144ull, 0xf0ce56007bef6616ull, 0x390d7530f3749f9dull, + 0xa625999bb3fc4033ull, 0x75277fef995afda6ull, 0xc51f53a3f0fafa5dull, 0x9225edc2f6f599d0ull, 0x8ea8d001890faaebull, 0x5814328cac0dbabfull, 0x2792499724c133b8ull, 0x93676b263a0a932dull, + 0xd04556c931e6cc5cull, 0xac770e5eacfccbfdull, 0x8f639b44b57ffcddull, 0x9faaa47b6a8f7906ull, 0xcf6945dec1692c76ull, 0xbae76864ab2853efull, 0xfdf1a7b9f292e721ull, 0xd1be28c4e06896caull, + 0x524ca1a30966dfc3ull, 0x6d9f82b4a0c49d55ull, 0xe9ccc952024c37a3ull, 0xcb1e2c142fe640fbull, 0xeb6a3ddceff93637ull, 0x9db685563837b969ull, 0x735d0a97397e9a73ull, 0x769e87d4851edc62ull + }, + { // base nonce 4096 + 0x86e437a3824dd195ull, 0xc4c6be5e71c3ce95ull, 0xd5bfb6d3c5607e8full, 0x2d66fc5035f07923ull, 0x2208b36836ddad05ull, 0x475ee8b8358f5d9bull, 0xc7073fa62542094eull, 0x8772a170da76fe86ull, + 0xde59b7b944b9cca2ull, 0xaca1169164b071f2ull, 0xc6b0c77aaf6156d7ull, 0xb269aa74d6ac8bbfull, 0x7813c5f52b4be07bull, 0x47347f1e833a9f8dull, 0xc2279941ae011758ull, 0x2f7e5a51aa488898ull, + 0x31d5b0d6d59bbc55ull, 0xc8428b0ffb32249aull, 0x81ccf818cb08e92aull, 0x3c36ddb655e2ae21ull, 0xaa420ec37a8513c8ull, 0x9d0916264bbc2838ull, 0x524c8e78f960f455ull, 0x2925791f277fbbcfull, + 0x1954e7e4d0621360ull, 0xf9d956934cbe24ecull, 0x59a0d78af50ec50eull, 0xd948051009055877ull, 0x8e250d4154d7799bull, 0x6c239c628f0aae74ull, 0x33b91411356782bcull, 0xfc984387a607428dull + }, + { // base nonce 1000000 + 0xd03a4006cc65d85dull, 0x7543d04a066ffa80ull, 0xbafb3d9d03e61806ull, 0x0511df048363823bull, 0x29e8e974f2559768ull, 0x2c62638edf0e0604ull, 0xb13fb1cc73af933dull, 0x17c0d802411cddddull, + 0x0783fd5e1ede02e0ull, 0x051f25268a6c036bull, 0x4616e4f6ba9f8365ull, 0x250e2bf1b1f533ecull, 0x1d88f9330f831014ull, 0x96ebf15542bf1cd8ull, 0x9c8c7424c761fb7aull, 0xa1cafc96cb1c25dbull, + 0x3fced2ed32d9d18aull, 0xb36a4d069a211707ull, 0x0af3b656c4af55dfull, 0x855756ea7662bf41ull, 0xd1a5ecf9b38f188dull, 0x4829cb01e2dc9865ull, 0x2f19493a1a2598acull, 0x03143778f0d53d3cull, + 0x981e77585c6a5655ull, 0x3bef255f9ed8d0acull, 0x0180828c60148870ull, 0xbf3c07d9686c4c29ull, 0x21475e8dff5f01d9ull, 0xac6a1d4618256c7cull, 0xa1c7ecbbac17bae7ull, 0x879d3d5a5b0407d4ull + } +}; + +// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (268435455). +static const uint32_t IGNEUM_DS_HEAD[16] = { + 0xffc3cd94u, 0x5920ccd8u, 0x392f44bbu, 0x5e57f67au, 0x2f2bc2a9u, 0x620b0e36u, 0xbdc09014u, 0x436654bfu, + 0x311e0b48u, 0x1abd93adu, 0x59cc7ce8u, 0xee5247b2u, 0x86171fe8u, 0x6d874751u, 0xc9f7728fu, 0x7c2a435du +}; +static const uint32_t IGNEUM_DS_LAST_INDEX = 268435455u; +static const uint32_t IGNEUM_DS_LAST = 0xa33ada72u; +// 64 sampled dataset words (index, value) computed on the Mac. +#define IGNEUM_DS_SAMPLES 64 +static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = { + 59471966u, 217795994u, 208353206u, 42483309u, 172547758u, 148076330u, 183853158u, 214389424u, 267488061u, 169781097u, 184093494u, 153880993u, 84977930u, 46426879u, 3093825u, 225364072u, 44593546u, 260713159u, 168250303u, 52384140u, 223401610u, 45554030u, 95410555u, 175039924u, 79171087u, 267580473u, 24168642u, 37981670u, 171551130u, 195559979u, 204611762u, 140997658u, 138925853u, 86637313u, 20736778u, 219665210u, 160430336u, 264654675u, 8013395u, 228945585u, 213884386u, 104419827u, 44185464u, 142737231u, 99284897u, 132475900u, 61861762u, 132056166u, 262388043u, 91878046u, 117353561u, 124768597u, 71352993u, 190698941u, 46055428u, 55281366u, 165145231u, 106810753u, 171985651u, 232085256u, 159510492u, 40072060u, 209107596u, 39023794u +}; +static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = { + 0xe8b73d94u, 0x337028b5u, 0xafe148c9u, 0xab99f7aeu, 0x434ea619u, 0xd85cb880u, 0x54764c7fu, 0x82c7e420u, 0xedf4cb9eu, 0x9884c959u, 0x223ee793u, 0x3a9ccf69u, 0x81da4fd2u, 0xd6ce8cb9u, 0xe3922dcau, 0x3e7e6bdeu, 0x382a3acau, 0x567e7f7fu, 0x25a0f084u, 0xbfeef128u, 0xe338abfbu, 0x7c3b5280u, 0x909bc5f1u, 0xd8b74b9cu, 0x8e31a22eu, 0x26b5f1d8u, 0x79122c00u, 0xcafc3340u, 0xd5e02ea3u, 0x1aee1afdu, 0xdb090d9au, 0xb049f435u, 0x4954d8bau, 0x03797ba0u, 0x196eefbdu, 0xd153412au, 0xbe5d2c4bu, 0xdaa14f0eu, 0x8e61ed07u, 0x9e9a64c6u, 0x2e29ff36u, 0x392a8589u, 0xb56a5912u, 0xfa6e8b57u, 0xd1a737cbu, 0xb0fa841au, 0xbe1c341fu, 0xe25be0f1u, 0xe937f543u, 0xebab2248u, 0x8e1b607au, 0x202a2fedu, 0x95e2819cu, 0x9c9652d4u, 0x32fedef0u, 0xdecfff82u, 0xcb5d43e5u, 0xb735806au, 0x8905939cu, 0xfbf8472du, 0xada74e5du, 0x7ebdeeeau, 0x0119f2b3u, 0xa9a376b8u +}; +// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^26 words. +static const uint32_t IGNEUM_CACHE_HEAD[16] = { + 0x355a86d2u, 0x7957db1cu, 0xd21772afu, 0x6fc1e09bu, 0xd55ce61du, 0x6e6a278bu, 0xd3f543ceu, 0x223d8e82u, + 0x143ab337u, 0x2e9f05bdu, 0x2eb389bfu, 0x0c6e449eu, 0x5cfa4222u, 0xba6560feu, 0x8e3e1aa4u, 0xdbcc1d53u +}; +static const uint32_t IGNEUM_CACHE_LAST[16] = { + 0x41190d91u, 0xbd277957u, 0x22ddbb49u, 0x6986f207u, 0xdf69a4d6u, 0x26401a3au, 0x818230fbu, 0xc417122du, + 0x3597b211u, 0xb553ce55u, 0xcf39cc0du, 0x3b7fc43au, 0x3fd43b00u, 0x67e1c80eu, 0xffa7ea7du, 0xca2960abu +}; +static const uint64_t IGNEUM_CACHE_FNV64 = 0x48c4f5bf24166b2eull; +// Hot table self-test (docs/plans/hot-table.md): hot[0..15], the last 16 words, and FNV-1a 64 over all IGNEUM_HOT_WORDS words. +static const uint32_t IGNEUM_HOT_HEAD[16] = { + 0x8068cc73u, 0x6036ebf9u, 0xb604cd25u, 0x8ffb840eu, 0xc54074a2u, 0x285c0695u, 0x77512425u, 0xc26a58a7u, + 0x72c88757u, 0xc10fca78u, 0x513825ddu, 0x30d6ccc8u, 0x9a05e7cfu, 0xb9533f50u, 0x4bac3ba0u, 0xa5c19528u +}; +static const uint32_t IGNEUM_HOT_LAST[16] = { + 0x7c6d7cebu, 0xe24fcd46u, 0xe5cce976u, 0x3ecffe59u, 0x94e98b66u, 0x3f6ef45cu, 0xa3715aa1u, 0xdbe35281u, + 0xba05d27bu, 0x00541963u, 0xe636f453u, 0xd3972366u, 0x529599b8u, 0x79ae3c8cu, 0x48436863u, 0x897a8bf3u +}; +static const uint64_t IGNEUM_HOT_FNV64 = 0x77ca4b9527104530ull; diff --git a/proto-cuda/packs-ca2-hot/hot64k4/vectors.json b/proto-cuda/packs-ca2-hot/hot64k4/vectors.json new file mode 100644 index 000000000..969e686b2 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4/vectors.json @@ -0,0 +1,39 @@ +{ + "seed": "igneum-genesis", + "day": "2026-10-03", + "dataset_mode": "memory-hard", + "dataset_log2_words": 28, + "mask": "0x0fffffff", + "lanes": 32, + "source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset", + "warps": [ + {"base_nonce": 0, "expected": [ + "0xaa6fadb5bf6a8657", "0x72a8df1dfcdc031d", "0xdfc9bfb4dfc1fe4c", "0x4f1d147208fe55a8", "0x857268e9be0d985a", "0x2cd63cf21c16a144", "0xf0ce56007bef6616", "0x390d7530f3749f9d", + "0xa625999bb3fc4033", "0x75277fef995afda6", "0xc51f53a3f0fafa5d", "0x9225edc2f6f599d0", "0x8ea8d001890faaeb", "0x5814328cac0dbabf", "0x2792499724c133b8", "0x93676b263a0a932d", + "0xd04556c931e6cc5c", "0xac770e5eacfccbfd", "0x8f639b44b57ffcdd", "0x9faaa47b6a8f7906", "0xcf6945dec1692c76", "0xbae76864ab2853ef", "0xfdf1a7b9f292e721", "0xd1be28c4e06896ca", + "0x524ca1a30966dfc3", "0x6d9f82b4a0c49d55", "0xe9ccc952024c37a3", "0xcb1e2c142fe640fb", "0xeb6a3ddceff93637", "0x9db685563837b969", "0x735d0a97397e9a73", "0x769e87d4851edc62" + ]}, + {"base_nonce": 4096, "expected": [ + "0x86e437a3824dd195", "0xc4c6be5e71c3ce95", "0xd5bfb6d3c5607e8f", "0x2d66fc5035f07923", "0x2208b36836ddad05", "0x475ee8b8358f5d9b", "0xc7073fa62542094e", "0x8772a170da76fe86", + "0xde59b7b944b9cca2", "0xaca1169164b071f2", "0xc6b0c77aaf6156d7", "0xb269aa74d6ac8bbf", "0x7813c5f52b4be07b", "0x47347f1e833a9f8d", "0xc2279941ae011758", "0x2f7e5a51aa488898", + "0x31d5b0d6d59bbc55", "0xc8428b0ffb32249a", "0x81ccf818cb08e92a", "0x3c36ddb655e2ae21", "0xaa420ec37a8513c8", "0x9d0916264bbc2838", "0x524c8e78f960f455", "0x2925791f277fbbcf", + "0x1954e7e4d0621360", "0xf9d956934cbe24ec", "0x59a0d78af50ec50e", "0xd948051009055877", "0x8e250d4154d7799b", "0x6c239c628f0aae74", "0x33b91411356782bc", "0xfc984387a607428d" + ]}, + {"base_nonce": 1000000, "expected": [ + "0xd03a4006cc65d85d", "0x7543d04a066ffa80", "0xbafb3d9d03e61806", "0x0511df048363823b", "0x29e8e974f2559768", "0x2c62638edf0e0604", "0xb13fb1cc73af933d", "0x17c0d802411cdddd", + "0x0783fd5e1ede02e0", "0x051f25268a6c036b", "0x4616e4f6ba9f8365", "0x250e2bf1b1f533ec", "0x1d88f9330f831014", "0x96ebf15542bf1cd8", "0x9c8c7424c761fb7a", "0xa1cafc96cb1c25db", + "0x3fced2ed32d9d18a", "0xb36a4d069a211707", "0x0af3b656c4af55df", "0x855756ea7662bf41", "0xd1a5ecf9b38f188d", "0x4829cb01e2dc9865", "0x2f19493a1a2598ac", "0x03143778f0d53d3c", + "0x981e77585c6a5655", "0x3bef255f9ed8d0ac", "0x0180828c60148870", "0xbf3c07d9686c4c29", "0x21475e8dff5f01d9", "0xac6a1d4618256c7c", "0xa1c7ecbbac17bae7", "0x879d3d5a5b0407d4" + ]} + ], + "dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"], + "dataset_last_index": 268435455, + "dataset_last": "0xa33ada72", + "dataset_samples": [{"index": 59471966, "value": "0xe8b73d94"}, {"index": 217795994, "value": "0x337028b5"}, {"index": 208353206, "value": "0xafe148c9"}, {"index": 42483309, "value": "0xab99f7ae"}, {"index": 172547758, "value": "0x434ea619"}, {"index": 148076330, "value": "0xd85cb880"}, {"index": 183853158, "value": "0x54764c7f"}, {"index": 214389424, "value": "0x82c7e420"}, {"index": 267488061, "value": "0xedf4cb9e"}, {"index": 169781097, "value": "0x9884c959"}, {"index": 184093494, "value": "0x223ee793"}, {"index": 153880993, "value": "0x3a9ccf69"}, {"index": 84977930, "value": "0x81da4fd2"}, {"index": 46426879, "value": "0xd6ce8cb9"}, {"index": 3093825, "value": "0xe3922dca"}, {"index": 225364072, "value": "0x3e7e6bde"}, {"index": 44593546, "value": "0x382a3aca"}, {"index": 260713159, "value": "0x567e7f7f"}, {"index": 168250303, "value": "0x25a0f084"}, {"index": 52384140, "value": "0xbfeef128"}, {"index": 223401610, "value": "0xe338abfb"}, {"index": 45554030, "value": "0x7c3b5280"}, {"index": 95410555, "value": "0x909bc5f1"}, {"index": 175039924, "value": "0xd8b74b9c"}, {"index": 79171087, "value": "0x8e31a22e"}, {"index": 267580473, "value": "0x26b5f1d8"}, {"index": 24168642, "value": "0x79122c00"}, {"index": 37981670, "value": "0xcafc3340"}, {"index": 171551130, "value": "0xd5e02ea3"}, {"index": 195559979, "value": "0x1aee1afd"}, {"index": 204611762, "value": "0xdb090d9a"}, {"index": 140997658, "value": "0xb049f435"}, {"index": 138925853, "value": "0x4954d8ba"}, {"index": 86637313, "value": "0x03797ba0"}, {"index": 20736778, "value": "0x196eefbd"}, {"index": 219665210, "value": "0xd153412a"}, {"index": 160430336, "value": "0xbe5d2c4b"}, {"index": 264654675, "value": "0xdaa14f0e"}, {"index": 8013395, "value": "0x8e61ed07"}, {"index": 228945585, "value": "0x9e9a64c6"}, {"index": 213884386, "value": "0x2e29ff36"}, {"index": 104419827, "value": "0x392a8589"}, {"index": 44185464, "value": "0xb56a5912"}, {"index": 142737231, "value": "0xfa6e8b57"}, {"index": 99284897, "value": "0xd1a737cb"}, {"index": 132475900, "value": "0xb0fa841a"}, {"index": 61861762, "value": "0xbe1c341f"}, {"index": 132056166, "value": "0xe25be0f1"}, {"index": 262388043, "value": "0xe937f543"}, {"index": 91878046, "value": "0xebab2248"}, {"index": 117353561, "value": "0x8e1b607a"}, {"index": 124768597, "value": "0x202a2fed"}, {"index": 71352993, "value": "0x95e2819c"}, {"index": 190698941, "value": "0x9c9652d4"}, {"index": 46055428, "value": "0x32fedef0"}, {"index": 55281366, "value": "0xdecfff82"}, {"index": 165145231, "value": "0xcb5d43e5"}, {"index": 106810753, "value": "0xb735806a"}, {"index": 171985651, "value": "0x8905939c"}, {"index": 232085256, "value": "0xfbf8472d"}, {"index": 159510492, "value": "0xada74e5d"}, {"index": 40072060, "value": "0x7ebdeeea"}, {"index": 209107596, "value": "0x0119f2b3"}, {"index": 39023794, "value": "0xa9a376b8"}], + "cache_head": ["0x355a86d2", "0x7957db1c", "0xd21772af", "0x6fc1e09b", "0xd55ce61d", "0x6e6a278b", "0xd3f543ce", "0x223d8e82", "0x143ab337", "0x2e9f05bd", "0x2eb389bf", "0x0c6e449e", "0x5cfa4222", "0xba6560fe", "0x8e3e1aa4", "0xdbcc1d53"], + "cache_last_line": ["0x41190d91", "0xbd277957", "0x22ddbb49", "0x6986f207", "0xdf69a4d6", "0x26401a3a", "0x818230fb", "0xc417122d", "0x3597b211", "0xb553ce55", "0xcf39cc0d", "0x3b7fc43a", "0x3fd43b00", "0x67e1c80e", "0xffa7ea7d", "0xca2960ab"], + "cache_fnv1a64": "0x48c4f5bf24166b2e", + "hot_head": ["0x8068cc73", "0x6036ebf9", "0xb604cd25", "0x8ffb840e", "0xc54074a2", "0x285c0695", "0x77512425", "0xc26a58a7", "0x72c88757", "0xc10fca78", "0x513825dd", "0x30d6ccc8", "0x9a05e7cf", "0xb9533f50", "0x4bac3ba0", "0xa5c19528"], + "hot_last_line": ["0x7c6d7ceb", "0xe24fcd46", "0xe5cce976", "0x3ecffe59", "0x94e98b66", "0x3f6ef45c", "0xa3715aa1", "0xdbe35281", "0xba05d27b", "0x00541963", "0xe636f453", "0xd3972366", "0x529599b8", "0x79ae3c8c", "0x48436863", "0x897a8bf3"], + "hot_fnv1a64": "0x77ca4b9527104530" +} diff --git a/proto-cuda/packs-ca2-hot/hot64k4a/kernel.cl b/proto-cuda/packs-ca2-hot/hot64k4a/kernel.cl new file mode 100644 index 000000000..40ff760ec --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4a/kernel.cl @@ -0,0 +1,305 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal). +// Built from source at runtime by proto-opencl/host.c, which passes these defines: +// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit) +// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default) +// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32 +// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition +// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units; +// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32. +#ifndef IGNEUM_GROUP +#define IGNEUM_GROUP 32 +#endif +#ifndef IGNEUM_EXCHANGE +#define IGNEUM_EXCHANGE 0 +#endif +#ifdef __OPENCL_VERSION__ +#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1))) +#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n] +#if IGNEUM_EXCHANGE == 1 +#ifdef cl_khr_subgroups +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif +#ifdef cl_khr_subgroup_shuffle +#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable +#endif +#elif IGNEUM_EXCHANGE == 2 +#pragma OPENCL EXTENSION cl_intel_subgroups : enable +#endif +#else +// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros. +#include "emu_opencl.h" +#endif + +#if IGNEUM_EXCHANGE == 1 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#elif IGNEUM_EXCHANGE == 2 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#else +// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per +// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane +// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's +// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier. +#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; } +#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; } +#endif + +// Hot table (64 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +static inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32. +static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); } +// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x. +static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); } +static inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +static inline void mh_chacha_block(const uint* x, uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +static inline void mh_cache_segment(__global uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +static inline void mh_mixer(uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +static inline void mh_item(__global const uint* cache, uint t, uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + __global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item. +// The same constants as memhard.h in this pack (one emitter, three dialects). +__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) { + uint t = (uint)get_global_id(0); + if (t < nItems) { + uint s[16]; + mh_item(cache, t, s); + __global uint* d = ds + ((ulong)t * 16u); + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (docs/plans/hot-table.md): 64 MiB = 16384 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +static inline void ht_segment(__global uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one work-item per segment. +__kernel void igneum_hot_fill(__global uint* hot, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the +// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and +// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). +IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r6 = r6 | r4; // 0 or + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 1u); r7 = r7 ^ t_; } // 1 shfl + r0 = r0 | r2; // 2 or + { uint t_; IGNEUM_SHFL_XOR(t_, r0, 16u); r3 = r3 ^ t_; } // 3 shfl + r7 = r7 ^ ds[r3 & mask]; // 4 load + r6 = r6 ^ ds[r0 & mask]; // 5 load + r1 = r1 * r4; // 6 mul + r0 = r0 - r1; // 7 sub + r3 = r3 ^ ds[r7 & mask]; // 8 load + r1 = r0 * r3 + r1; // 9 mad + r4 = r4 * r0; // 10 mul + r5 = r5 ^ ds[r4 & mask]; // 11 load + r1 = r1 ^ r7; // 12 xor + r1 = r1 ^ ds[r6 & mask]; // 13 load + r2 = r2 ^ ds[r1 & mask]; // 14 load + r3 = r3 * r1; // 15 mul + r6 = r6 ^ hot[mul_hi(r5, HOT_WORDS)]; // 16 hot + r0 = r0 ^ ds[r2 & mask]; // 17 load + r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x535b545fu : 0xc035a4e6u); // 18 add + r5 = r5 - r0; // 19 sub + r2 = r2 * r4; // 20 mul + r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x925d5e63u : 0x1ecdd2cfu); // 21 add + r3 = r3 ^ r7; // 22 xor + r7 = r7 ^ ds[r2 & mask]; // 23 load + r2 = mul_hi(r2, r5); // 24 mulhi + r5 = rotr_var(r5, r2); // 25 rotr + r3 = r2 * r7 + r3; // 26 mad + r2 = r2 - r0; // 27 sub + r6 = r1 * r5 + r6; // 28 mad + r2 = r2 + r3 + ((((sel >> 16u) & 1u) != 0u) ? 0x46b1b505u : 0xbdc6da76u); // 29 add + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r3 = r3 ^ t_; } // 30 shfl + r5 = r5 ^ ds[r7 & mask]; // 31 load + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 32 hot + r6 = r6 | r3; // 33 or + r3 = r3 ^ hot[mul_hi(r6, HOT_WORDS)]; // 34 hot + { uint t_; IGNEUM_SHFL_XOR(t_, r0, 2u); r4 = r4 ^ t_; } // 35 shfl + r5 = r5 ^ r2; // 36 xor + r3 = r3 ^ ds[r4 & mask]; // 37 load + r4 = r4 ^ ds[r3 & mask]; // 38 load + r1 = rotr_var(r1, r4); // 39 rotr + r3 = r3 + r6 + ((((sel >> 28u) & 1u) != 0u) ? 0x6328cb2cu : 0xbc3ff65fu); // 40 add + r5 = r5 ^ r1; // 41 xor + r5 = r5 | r0; // 42 or + r7 = r7 ^ r3; // 43 xor + r2 = r2 ^ ds[r5 & mask]; // 44 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r6 = r6 ^ t_; } // 45 shfl + r5 = r5 ^ r3; // 46 xor + r7 = r7 + r6 + ((((sel >> 22u) & 1u) != 0u) ? 0x9342df0du : 0x5af3bd5bu); // 47 add + r3 = r3 + r4 + ((((sel >> 23u) & 1u) != 0u) ? 0x485614dbu : 0x9ff95776u); // 48 add + r5 = r5 ^ ds[r3 & mask]; // 49 load + r4 = rotr_var(r4, r5); // 50 rotr + r0 = r1 * r7 + r0; // 51 mad + r0 = r0 * r3; // 52 mul + r5 = rotr_var(r5, r4); // 53 rotr + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r0 = r0 ^ t_; } // 54 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r7, 16u); r0 = r0 ^ t_; } // 55 shfl + r7 = r7 ^ ds[r4 & mask]; // 56 load + r7 = rotr_var(r7, r3); // 57 rotr + r0 = r0 ^ ds[r6 & mask]; // 58 load + r6 = r6 ^ hot[mul_hi(r2, HOT_WORDS)]; // 59 hot + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 1u); r3 = r3 ^ t_; } // 60 shfl + r7 = r7 - r5; // 61 sub + r1 = rotl_imm(r1, 4u); // 62 rotl + r4 = r4 ^ ds[r5 & mask]; // 63 load + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} + +#if IGNEUM_EXCHANGE != 0 +// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the +// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a +// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md. +IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) { + if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); } +} +#endif diff --git a/proto-cuda/packs-ca2-hot/hot64k4a/kernel.cu b/proto-cuda/packs-ca2-hot/hot64k4a/kernel.cu new file mode 100644 index 000000000..84bdb1a53 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4a/kernel.cu @@ -0,0 +1,179 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal). +// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC. +#include +#include +#include "program.h" +#include "memhard.h" + +// Hot table (64 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +__device__ __forceinline__ uint32_t splitmix32(uint32_t x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31. +__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } +// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x. +__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) { + uint32_t x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item. +// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host. +__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) { + uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x; + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) { + uint32_t t = blockIdx.x * blockDim.x + threadIdx.x; + if (t < nItems) { + uint32_t s[16]; + mh_item(cache, t, s); + uint32_t* d = ds + (size_t)t * 16u; + for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (ht_segment is in memhard.h): one thread per segment. +__global__ void igneum_hot_fill(uint32_t* hot, uint32_t nSegments) { + uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x; + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every +// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a +// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid. +__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot) { + uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x; + uint32_t nonce = baseNonce + gid; + uint32_t r0, r1, r2, r3, r4, r5, r6, r7; + { uint32_t x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint32_t x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint32_t x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint32_t x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint32_t x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint32_t x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint32_t x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint32_t x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint32_t it = 0u; it < 8u; ++it) { + uint32_t sel = r0; + r6 = r6 | r4; // 0 or + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r4, 1); // 1 shfl + r0 = r0 | r2; // 2 or + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r0, 16); // 3 shfl + r7 = r7 ^ ds[r3 & mask]; // 4 load + r6 = r6 ^ ds[r0 & mask]; // 5 load + r1 = r1 * r4; // 6 mul + r0 = r0 - r1; // 7 sub + r3 = r3 ^ ds[r7 & mask]; // 8 load + r1 = r0 * r3 + r1; // 9 mad + r4 = r4 * r0; // 10 mul + r5 = r5 ^ ds[r4 & mask]; // 11 load + r1 = r1 ^ r7; // 12 xor + r1 = r1 ^ ds[r6 & mask]; // 13 load + r2 = r2 ^ ds[r1 & mask]; // 14 load + r3 = r3 * r1; // 15 mul + r6 = r6 ^ hot[__umulhi(r5, HOT_WORDS)]; // 16 hot + r0 = r0 ^ ds[r2 & mask]; // 17 load + r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x535b545fu : 0xc035a4e6u); // 18 add + r5 = r5 - r0; // 19 sub + r2 = r2 * r4; // 20 mul + r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x925d5e63u : 0x1ecdd2cfu); // 21 add + r3 = r3 ^ r7; // 22 xor + r7 = r7 ^ ds[r2 & mask]; // 23 load + r2 = __umulhi(r2, r5); // 24 mulhi + r5 = rotr_var(r5, r2); // 25 rotr + r3 = r2 * r7 + r3; // 26 mad + r2 = r2 - r0; // 27 sub + r6 = r1 * r5 + r6; // 28 mad + r2 = r2 + r3 + ((((sel >> 16u) & 1u) != 0u) ? 0x46b1b505u : 0xbdc6da76u); // 29 add + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r2, 8); // 30 shfl + r5 = r5 ^ ds[r7 & mask]; // 31 load + r2 = r2 ^ hot[__umulhi(r0, HOT_WORDS)]; // 32 hot + r6 = r6 | r3; // 33 or + r3 = r3 ^ hot[__umulhi(r6, HOT_WORDS)]; // 34 hot + r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r0, 2); // 35 shfl + r5 = r5 ^ r2; // 36 xor + r3 = r3 ^ ds[r4 & mask]; // 37 load + r4 = r4 ^ ds[r3 & mask]; // 38 load + r1 = rotr_var(r1, r4); // 39 rotr + r3 = r3 + r6 + ((((sel >> 28u) & 1u) != 0u) ? 0x6328cb2cu : 0xbc3ff65fu); // 40 add + r5 = r5 ^ r1; // 41 xor + r5 = r5 | r0; // 42 or + r7 = r7 ^ r3; // 43 xor + r2 = r2 ^ ds[r5 & mask]; // 44 load + r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 45 shfl + r5 = r5 ^ r3; // 46 xor + r7 = r7 + r6 + ((((sel >> 22u) & 1u) != 0u) ? 0x9342df0du : 0x5af3bd5bu); // 47 add + r3 = r3 + r4 + ((((sel >> 23u) & 1u) != 0u) ? 0x485614dbu : 0x9ff95776u); // 48 add + r5 = r5 ^ ds[r3 & mask]; // 49 load + r4 = rotr_var(r4, r5); // 50 rotr + r0 = r1 * r7 + r0; // 51 mad + r0 = r0 * r3; // 52 mul + r5 = rotr_var(r5, r4); // 53 rotr + r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 1); // 54 shfl + r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r7, 16); // 55 shfl + r7 = r7 ^ ds[r4 & mask]; // 56 load + r7 = rotr_var(r7, r3); // 57 rotr + r0 = r0 ^ ds[r6 & mask]; // 58 load + r6 = r6 ^ hot[__umulhi(r2, HOT_WORDS)]; // 59 hot + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r2, 1); // 60 shfl + r7 = r7 - r5; // 61 sub + r1 = rotl_imm(r1, 4u); // 62 rotl + r4 = r4 ^ ds[r5 & mask]; // 63 load + } + uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo; +} + +// Host-side launch wrappers. Declared in program.h, called from host.cu. +cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) { + if (nSegments == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nSegments + block - 1u) / block; + igneum_cache_fill<<>>(cache, nSegments); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) { + if (nItems == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nItems + block - 1u) / block; + igneum_build<<>>(ds, cache, nItems); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments) { + if (nSegments == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nSegments + block - 1u) / block; + igneum_hot_fill<<>>(hot, nSegments); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot, + uint32_t nonces, uint32_t blockWarps) { + if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue; + uint32_t block = 32u * blockWarps; + if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue; + igneum_hash<<>>(ds, out, baseNonce, mask, hot); + return cudaGetLastError(); +} + +cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) { + cudaFuncAttributes attr; + cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash); + if (e != cudaSuccess) return e; + *numRegs = attr.numRegs; + return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0); +} diff --git a/proto-cuda/packs-ca2-hot/hot64k4a/kernel_bound.cl b/proto-cuda/packs-ca2-hot/hot64k4a/kernel_bound.cl new file mode 100644 index 000000000..e76829789 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4a/kernel_bound.cl @@ -0,0 +1,399 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal). +// Built from source at runtime by proto-opencl/host.c, which passes these defines: +// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit) +// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default) +// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32 +// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition +// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units; +// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32. +#ifndef IGNEUM_GROUP +#define IGNEUM_GROUP 32 +#endif +#ifndef IGNEUM_EXCHANGE +#define IGNEUM_EXCHANGE 0 +#endif +#ifdef __OPENCL_VERSION__ +#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1))) +#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n] +#if IGNEUM_EXCHANGE == 1 +#ifdef cl_khr_subgroups +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif +#ifdef cl_khr_subgroup_shuffle +#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable +#endif +#elif IGNEUM_EXCHANGE == 2 +#pragma OPENCL EXTENSION cl_intel_subgroups : enable +#endif +#else +// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros. +#include "emu_opencl.h" +#endif + +#if IGNEUM_EXCHANGE == 1 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#elif IGNEUM_EXCHANGE == 2 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#else +// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per +// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane +// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's +// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier. +#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; } +#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; } +#endif + +// Hot table (64 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +static inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32. +static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); } +// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x. +static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); } +static inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +static inline void mh_chacha_block(const uint* x, uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +static inline void mh_cache_segment(__global uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +static inline void mh_mixer(uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +static inline void mh_item(__global const uint* cache, uint t, uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + __global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item. +// The same constants as memhard.h in this pack (one emitter, three dialects). +__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) { + uint t = (uint)get_global_id(0); + if (t < nItems) { + uint s[16]; + mh_item(cache, t, s); + __global uint* d = ds + ((ulong)t * 16u); + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (docs/plans/hot-table.md): 64 MiB = 16384 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +static inline void ht_segment(__global uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one work-item per segment. +__kernel void igneum_hot_fill(__global uint* hot, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the +// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and +// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). +IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r6 = r6 | r4; // 0 or + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 1u); r7 = r7 ^ t_; } // 1 shfl + r0 = r0 | r2; // 2 or + { uint t_; IGNEUM_SHFL_XOR(t_, r0, 16u); r3 = r3 ^ t_; } // 3 shfl + r7 = r7 ^ ds[r3 & mask]; // 4 load + r6 = r6 ^ ds[r0 & mask]; // 5 load + r1 = r1 * r4; // 6 mul + r0 = r0 - r1; // 7 sub + r3 = r3 ^ ds[r7 & mask]; // 8 load + r1 = r0 * r3 + r1; // 9 mad + r4 = r4 * r0; // 10 mul + r5 = r5 ^ ds[r4 & mask]; // 11 load + r1 = r1 ^ r7; // 12 xor + r1 = r1 ^ ds[r6 & mask]; // 13 load + r2 = r2 ^ ds[r1 & mask]; // 14 load + r3 = r3 * r1; // 15 mul + r6 = r6 ^ hot[mul_hi(r5, HOT_WORDS)]; // 16 hot + r0 = r0 ^ ds[r2 & mask]; // 17 load + r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x535b545fu : 0xc035a4e6u); // 18 add + r5 = r5 - r0; // 19 sub + r2 = r2 * r4; // 20 mul + r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x925d5e63u : 0x1ecdd2cfu); // 21 add + r3 = r3 ^ r7; // 22 xor + r7 = r7 ^ ds[r2 & mask]; // 23 load + r2 = mul_hi(r2, r5); // 24 mulhi + r5 = rotr_var(r5, r2); // 25 rotr + r3 = r2 * r7 + r3; // 26 mad + r2 = r2 - r0; // 27 sub + r6 = r1 * r5 + r6; // 28 mad + r2 = r2 + r3 + ((((sel >> 16u) & 1u) != 0u) ? 0x46b1b505u : 0xbdc6da76u); // 29 add + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r3 = r3 ^ t_; } // 30 shfl + r5 = r5 ^ ds[r7 & mask]; // 31 load + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 32 hot + r6 = r6 | r3; // 33 or + r3 = r3 ^ hot[mul_hi(r6, HOT_WORDS)]; // 34 hot + { uint t_; IGNEUM_SHFL_XOR(t_, r0, 2u); r4 = r4 ^ t_; } // 35 shfl + r5 = r5 ^ r2; // 36 xor + r3 = r3 ^ ds[r4 & mask]; // 37 load + r4 = r4 ^ ds[r3 & mask]; // 38 load + r1 = rotr_var(r1, r4); // 39 rotr + r3 = r3 + r6 + ((((sel >> 28u) & 1u) != 0u) ? 0x6328cb2cu : 0xbc3ff65fu); // 40 add + r5 = r5 ^ r1; // 41 xor + r5 = r5 | r0; // 42 or + r7 = r7 ^ r3; // 43 xor + r2 = r2 ^ ds[r5 & mask]; // 44 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r6 = r6 ^ t_; } // 45 shfl + r5 = r5 ^ r3; // 46 xor + r7 = r7 + r6 + ((((sel >> 22u) & 1u) != 0u) ? 0x9342df0du : 0x5af3bd5bu); // 47 add + r3 = r3 + r4 + ((((sel >> 23u) & 1u) != 0u) ? 0x485614dbu : 0x9ff95776u); // 48 add + r5 = r5 ^ ds[r3 & mask]; // 49 load + r4 = rotr_var(r4, r5); // 50 rotr + r0 = r1 * r7 + r0; // 51 mad + r0 = r0 * r3; // 52 mul + r5 = rotr_var(r5, r4); // 53 rotr + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r0 = r0 ^ t_; } // 54 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r7, 16u); r0 = r0 ^ t_; } // 55 shfl + r7 = r7 ^ ds[r4 & mask]; // 56 load + r7 = rotr_var(r7, r3); // 57 rotr + r0 = r0 ^ ds[r6 & mask]; // 58 load + r6 = r6 ^ hot[mul_hi(r2, HOT_WORDS)]; // 59 hot + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 1u); r3 = r3 ^ t_; } // 60 shfl + r7 = r7 - r5; // 61 sub + r1 = rotl_imm(r1, 4u); // 62 rotl + r4 = r4 ^ ds[r5 & mask]; // 63 load + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} + +#if IGNEUM_EXCHANGE != 0 +// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the +// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a +// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md. +IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) { + if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); } +} +#endif + +// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash. +IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7]; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; } + { uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; } + { uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; } + { uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; } + { uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; } + { uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; } + { uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; } + { uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r6 = r6 | r4; // 0 or + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 1u); r7 = r7 ^ t_; } // 1 shfl + r0 = r0 | r2; // 2 or + { uint t_; IGNEUM_SHFL_XOR(t_, r0, 16u); r3 = r3 ^ t_; } // 3 shfl + r7 = r7 ^ ds[r3 & mask]; // 4 load + r6 = r6 ^ ds[r0 & mask]; // 5 load + r1 = r1 * r4; // 6 mul + r0 = r0 - r1; // 7 sub + r3 = r3 ^ ds[r7 & mask]; // 8 load + r1 = r0 * r3 + r1; // 9 mad + r4 = r4 * r0; // 10 mul + r5 = r5 ^ ds[r4 & mask]; // 11 load + r1 = r1 ^ r7; // 12 xor + r1 = r1 ^ ds[r6 & mask]; // 13 load + r2 = r2 ^ ds[r1 & mask]; // 14 load + r3 = r3 * r1; // 15 mul + r6 = r6 ^ hot[mul_hi(r5, HOT_WORDS)]; // 16 hot + r0 = r0 ^ ds[r2 & mask]; // 17 load + r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x535b545fu : 0xc035a4e6u); // 18 add + r5 = r5 - r0; // 19 sub + r2 = r2 * r4; // 20 mul + r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x925d5e63u : 0x1ecdd2cfu); // 21 add + r3 = r3 ^ r7; // 22 xor + r7 = r7 ^ ds[r2 & mask]; // 23 load + r2 = mul_hi(r2, r5); // 24 mulhi + r5 = rotr_var(r5, r2); // 25 rotr + r3 = r2 * r7 + r3; // 26 mad + r2 = r2 - r0; // 27 sub + r6 = r1 * r5 + r6; // 28 mad + r2 = r2 + r3 + ((((sel >> 16u) & 1u) != 0u) ? 0x46b1b505u : 0xbdc6da76u); // 29 add + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r3 = r3 ^ t_; } // 30 shfl + r5 = r5 ^ ds[r7 & mask]; // 31 load + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 32 hot + r6 = r6 | r3; // 33 or + r3 = r3 ^ hot[mul_hi(r6, HOT_WORDS)]; // 34 hot + { uint t_; IGNEUM_SHFL_XOR(t_, r0, 2u); r4 = r4 ^ t_; } // 35 shfl + r5 = r5 ^ r2; // 36 xor + r3 = r3 ^ ds[r4 & mask]; // 37 load + r4 = r4 ^ ds[r3 & mask]; // 38 load + r1 = rotr_var(r1, r4); // 39 rotr + r3 = r3 + r6 + ((((sel >> 28u) & 1u) != 0u) ? 0x6328cb2cu : 0xbc3ff65fu); // 40 add + r5 = r5 ^ r1; // 41 xor + r5 = r5 | r0; // 42 or + r7 = r7 ^ r3; // 43 xor + r2 = r2 ^ ds[r5 & mask]; // 44 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r6 = r6 ^ t_; } // 45 shfl + r5 = r5 ^ r3; // 46 xor + r7 = r7 + r6 + ((((sel >> 22u) & 1u) != 0u) ? 0x9342df0du : 0x5af3bd5bu); // 47 add + r3 = r3 + r4 + ((((sel >> 23u) & 1u) != 0u) ? 0x485614dbu : 0x9ff95776u); // 48 add + r5 = r5 ^ ds[r3 & mask]; // 49 load + r4 = rotr_var(r4, r5); // 50 rotr + r0 = r1 * r7 + r0; // 51 mad + r0 = r0 * r3; // 52 mul + r5 = rotr_var(r5, r4); // 53 rotr + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r0 = r0 ^ t_; } // 54 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r7, 16u); r0 = r0 ^ t_; } // 55 shfl + r7 = r7 ^ ds[r4 & mask]; // 56 load + r7 = rotr_var(r7, r3); // 57 rotr + r0 = r0 ^ ds[r6 & mask]; // 58 load + r6 = r6 ^ hot[mul_hi(r2, HOT_WORDS)]; // 59 hot + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 1u); r3 = r3 ^ t_; } // 60 shfl + r7 = r7 - r5; // 61 sub + r1 = rotl_imm(r1, 4u); // 62 rotl + r4 = r4 ^ ds[r5 & mask]; // 63 load + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot64k4a/kernel_bound.cu b/proto-cuda/packs-ca2-hot/hot64k4a/kernel_bound.cu new file mode 100644 index 000000000..57081032b --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4a/kernel_bound.cu @@ -0,0 +1,125 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW. +// Host declarations (also in program_bound.h if present): +// struct IgneumInitWords { uint32_t w[8]; }; +// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, +// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps); +// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps); +#include +#include +#include "program.h" + +struct IgneumInitWords { uint32_t w[8]; }; + +// Hot table (64 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +__device__ __forceinline__ uint32_t splitmix32(uint32_t x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } +__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } + +__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw, const uint32_t* hot) { + uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x; + uint32_t nonce = baseNonce + gid; + uint32_t r0, r1, r2, r3, r4, r5, r6, r7; + { uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; } + { uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; } + { uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; } + { uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; } + { uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; } + { uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; } + { uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; } + { uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; } + + for (uint32_t it = 0u; it < 8u; ++it) { + uint32_t sel = r0; + r6 = r6 | r4; // 0 or + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r4, 1); // 1 shfl + r0 = r0 | r2; // 2 or + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r0, 16); // 3 shfl + r7 = r7 ^ ds[r3 & mask]; // 4 load + r6 = r6 ^ ds[r0 & mask]; // 5 load + r1 = r1 * r4; // 6 mul + r0 = r0 - r1; // 7 sub + r3 = r3 ^ ds[r7 & mask]; // 8 load + r1 = r0 * r3 + r1; // 9 mad + r4 = r4 * r0; // 10 mul + r5 = r5 ^ ds[r4 & mask]; // 11 load + r1 = r1 ^ r7; // 12 xor + r1 = r1 ^ ds[r6 & mask]; // 13 load + r2 = r2 ^ ds[r1 & mask]; // 14 load + r3 = r3 * r1; // 15 mul + r6 = r6 ^ hot[__umulhi(r5, HOT_WORDS)]; // 16 hot + r0 = r0 ^ ds[r2 & mask]; // 17 load + r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x535b545fu : 0xc035a4e6u); // 18 add + r5 = r5 - r0; // 19 sub + r2 = r2 * r4; // 20 mul + r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x925d5e63u : 0x1ecdd2cfu); // 21 add + r3 = r3 ^ r7; // 22 xor + r7 = r7 ^ ds[r2 & mask]; // 23 load + r2 = __umulhi(r2, r5); // 24 mulhi + r5 = rotr_var(r5, r2); // 25 rotr + r3 = r2 * r7 + r3; // 26 mad + r2 = r2 - r0; // 27 sub + r6 = r1 * r5 + r6; // 28 mad + r2 = r2 + r3 + ((((sel >> 16u) & 1u) != 0u) ? 0x46b1b505u : 0xbdc6da76u); // 29 add + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r2, 8); // 30 shfl + r5 = r5 ^ ds[r7 & mask]; // 31 load + r2 = r2 ^ hot[__umulhi(r0, HOT_WORDS)]; // 32 hot + r6 = r6 | r3; // 33 or + r3 = r3 ^ hot[__umulhi(r6, HOT_WORDS)]; // 34 hot + r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r0, 2); // 35 shfl + r5 = r5 ^ r2; // 36 xor + r3 = r3 ^ ds[r4 & mask]; // 37 load + r4 = r4 ^ ds[r3 & mask]; // 38 load + r1 = rotr_var(r1, r4); // 39 rotr + r3 = r3 + r6 + ((((sel >> 28u) & 1u) != 0u) ? 0x6328cb2cu : 0xbc3ff65fu); // 40 add + r5 = r5 ^ r1; // 41 xor + r5 = r5 | r0; // 42 or + r7 = r7 ^ r3; // 43 xor + r2 = r2 ^ ds[r5 & mask]; // 44 load + r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 45 shfl + r5 = r5 ^ r3; // 46 xor + r7 = r7 + r6 + ((((sel >> 22u) & 1u) != 0u) ? 0x9342df0du : 0x5af3bd5bu); // 47 add + r3 = r3 + r4 + ((((sel >> 23u) & 1u) != 0u) ? 0x485614dbu : 0x9ff95776u); // 48 add + r5 = r5 ^ ds[r3 & mask]; // 49 load + r4 = rotr_var(r4, r5); // 50 rotr + r0 = r1 * r7 + r0; // 51 mad + r0 = r0 * r3; // 52 mul + r5 = rotr_var(r5, r4); // 53 rotr + r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 1); // 54 shfl + r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r7, 16); // 55 shfl + r7 = r7 ^ ds[r4 & mask]; // 56 load + r7 = rotr_var(r7, r3); // 57 rotr + r0 = r0 ^ ds[r6 & mask]; // 58 load + r6 = r6 ^ hot[__umulhi(r2, HOT_WORDS)]; // 59 hot + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r2, 1); // 60 shfl + r7 = r7 - r5; // 61 sub + r1 = rotl_imm(r1, 4u); // 62 rotl + r4 = r4 ^ ds[r5 & mask]; // 63 load + } + uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo; +} + +cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, + IgneumInitWords iw, const uint32_t* hot, uint32_t nonces, uint32_t blockWarps) { + if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue; + uint32_t block = 32u * blockWarps; + if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue; + igneum_hash_bound<<>>(ds, out, baseNonce, mask, iw, hot); + return cudaGetLastError(); +} + +cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) { + cudaFuncAttributes attr; + cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound); + if (e != cudaSuccess) return e; + *numRegs = attr.numRegs; + return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0); +} diff --git a/proto-cuda/packs-ca2-hot/hot64k4a/memhard.h b/proto-cuda/packs-ca2-hot/hot64k4a/memhard.h new file mode 100644 index 000000000..180f678b7 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4a/memhard.h @@ -0,0 +1,129 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against. +// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference). +// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C. +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif +#if defined(__CUDACC__) +#define IGNEUM_HD __host__ __device__ __forceinline__ +#elif defined(_MSC_VER) && !defined(__cplusplus) +#define IGNEUM_HD static __inline +#else +#define IGNEUM_HD static inline +#endif +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) { + for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint32_t r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) { + uint32_t prev[16]; uint32_t x[16]; uint32_t y[16]; + for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +IGNEUM_HD void mh_item(const uint32_t* cache, uint32_t t, uint32_t* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint32_t r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +IGNEUM_HD uint32_t mh_word(const uint32_t* cache, uint32_t w) { uint32_t s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Hot table (docs/plans/hot-table.md): 64 MiB = 16384 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +IGNEUM_HD void ht_segment(uint32_t* hot, uint32_t seg) { + uint32_t prev[16]; uint32_t x[16]; uint32_t y[16]; + for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + uint32_t* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} diff --git a/proto-cuda/packs-ca2-hot/hot64k4a/memhard.metal b/proto-cuda/packs-ca2-hot/hot64k4a/memhard.metal new file mode 100644 index 000000000..c2ec5702e --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4a/memhard.metal @@ -0,0 +1,131 @@ +#include +using namespace metal; +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +inline void mh_chacha_block(const thread uint* x, thread uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +inline void mh_cache_segment(device uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +inline void mh_mixer(thread uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +inline void mh_item(device const uint* cache, uint t, thread uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +inline uint mh_word(device const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// One thread per segment (2^16 threads). +kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) { + mh_cache_segment(cache, gid); +} +// One thread per 64-byte item (dataset words / 16 threads). +kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]], + uint gid [[thread_position_in_grid]]) { + uint s[16]; + mh_item(cache, gid, s); + device uint* d = dataset + gid * 16u; + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; +} + +// Hot table (docs/plans/hot-table.md): 64 MiB = 16384 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +inline void ht_segment(device uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + device uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one thread per segment (IGNEUM_HOT_SEGMENTS threads). +kernel void igneum_hot_fill(device uint* hot [[buffer(0)]], uint gid [[thread_position_in_grid]]) { + ht_segment(hot, gid); +} diff --git a/proto-cuda/packs-ca2-hot/hot64k4a/program.h b/proto-cuda/packs-ca2-hot/hot64k4a/program.h new file mode 100644 index 000000000..771be93fb --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4a/program.h @@ -0,0 +1,70 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Program metadata for host.cu plus the launch wrappers defined in kernel.cu. +// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros. +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif +#ifndef IGNEUM_NO_CUDA +#include +#endif + +#define IGNEUM_SEED_STRING "igneum-genesis" +#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973" +#define IGNEUM_GENERATOR 2 +#define IGNEUM_PROGRAM_ATTEMPT 0 +#define IGNEUM_PROGRAM_ID 0x7f581216e95eb062ull +#define IGNEUM_DAY_STRING "2026-10-03" +#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033" +#define IGNEUM_DAY0 0x3067619fu +#define IGNEUM_DAY1 0x3c269176u +#define IGNEUM_DATASET_LOG2 28 +#define IGNEUM_MASK 0x0fffffffu +#define IGNEUM_LANES 32 +#define IGNEUM_ITERATIONS 8 +#define IGNEUM_INSTR_COUNT 64 +#define IGNEUM_LOADS_PER_HASH 160 +#define IGNEUM_WIDE_LOADS_PER_HASH 0 +#define IGNEUM_OP_MIX "load=16 shfl=8 add=6 xor=6 mul=5 rotr=5 hot=4 mad=4 or=4 sub=4 mulhi=1 rotl=1" +// Class v3 construction (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): version 2 loads; the dataset item +// derivation applies the mixer IGNEUM_MIXER_MULT times per round (memhard.h), and the cache follows the growth rule. +#define IGNEUM_LOAD_CLASS "hot64k4a" +#define IGNEUM_LOAD_SLOTS 20 +#define IGNEUM_LOAD_MIX { 100, 0, 0 } +#define IGNEUM_LOAD_WIDTH_COUNTS { 16, 0, 0 } // loads of 4, 16, 64 bytes per program +#define IGNEUM_BYTES_PER_HASH 512 +#define IGNEUM_FOLD_ROT 11 +#define IGNEUM_FOLD_MUL 0x9e3779b1u +// Hot table (hot-table experiment, 5 October 2026, docs/plans/hot-table.md): NOT the lottery hash. IGNEUM_HOT_SLOTS of the +// 16 load slots read H[mulhi(src, IGNEUM_HOT_WORDS)], H = IGNEUM_HOT_MB MiB of chained ChaCha12 lines (the cache chain) under +// KH = seed_words("igneum-hot/" || epoch seed bytes) and the tag "Igne" "umHT"; filled by igneum_hot_fill once per epoch. +#define IGNEUM_HOT_MB 64 +#define IGNEUM_HOT_WORDS 0x01000000u +#define IGNEUM_HOT_SEGMENTS 16384u +#define IGNEUM_HOT_SLOTS 4 // hot loads per program (32 per hash), added beside the dataset loads (16 of them) +#define IGNEUM_HOT_ADDED 1 +#define IGNEUM_HOT_KEY_INIT { 0x3a48bef5u, 0x6b54b1a1u, 0x9ff897c3u, 0x7d5d85e4u, 0xcd35379fu, 0x9d76bb86u, 0xbe2affb1u, 0x0f8f80b4u } +// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h) +#define IGNEUM_DATASET_MODE 1 + +#define IGNEUM_SEEDW_INIT { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u } +#define IGNEUM_KEY_INIT { 0x3067619fu, 0x3c269176u, 0x84a03b03u, 0xf8c63294u, 0xff977c5bu, 0xe60def3eu, 0x63630141u, 0xb8fbcb58u } +#define IGNEUM_CACHE_LOG2_WORDS 26 +#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6 +#define IGNEUM_CACHE_SEGMENTS 65536u +#define IGNEUM_ITEM_ROUNDS 8 +#define IGNEUM_MIX_ROT_INIT { 20u, 20u, 19u, 4u, 26u, 3u, 3u, 27u } +#define IGNEUM_MIX_MUL_INIT { 0x42146205u, 0x52cbe0fbu, 0x7ecf4a03u, 0x6728907fu, 0xd81d9751u, 0x132952c3u, 0xf60de277u, 0x05358035u, 0xbaf6499du, 0xe4db9667u, 0x3e98f45du, 0xd0004eddu, 0x2691630du, 0x9beb3bcfu, 0xab310379u, 0x99cfb423u } +#define IGNEUM_MIX_RC_INIT { 0xbab68293u, 0xcc162340u, 0x6ce151ccu, 0xe62b8997u, 0xc9c80297u, 0xf74a1654u, 0x3d704af5u, 0x3cf522b7u, 0x2b9cac04u, 0xa880ac10u, 0x13e5dd1du, 0x6fc3e233u, 0x2d83eeacu, 0x9006e8bfu, 0x2c4b5362u, 0x31b49ee2u } + +#ifndef IGNEUM_NO_CUDA +// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError(). +cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments); +cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems); +cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments); +cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot, + uint32_t nonces, uint32_t blockWarps); +cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps); +#endif diff --git a/proto-cuda/packs-ca2-hot/hot64k4a/program.json b/proto-cuda/packs-ca2-hot/hot64k4a/program.json new file mode 100644 index 000000000..1fc72c2d6 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4a/program.json @@ -0,0 +1,129 @@ +{ + "format": "igneum-program-pack-3", + "generator": 2, + "attempt": 0, + "program_id": "0x7f581216e95eb062", + "program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32", + "dataset_mode": "memory-hard", + "seed": "igneum-genesis", + "seed_bytes": "69676e65756d2d67656e65736973", + "seed_words": ["0x67a9a7be", "0x1a155b25", "0xfddfb732", "0x4b5af2e8", "0xc55caf33", "0xa27c13b7", "0x06628a48", "0x03852469"], + "seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32", + "generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried", + "lanes": 32, + "registers": 8, + "iterations": 8, + "instruction_count": 64, + "loads_per_hash": 160, + "load_class": "hot64k4a", + "load_slots": 20, + "load_mix_percent_4_16_64": [100, 0, 0], + "load_width_counts_4_16_64": [16, 0, 0], + "bytes_per_hash": 512, + "wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots", + "hot_table": {"mb": 64, "words": 16777216, "segments": 16384, "slots": 4, "form": "added: k load slots added beside the class's, the dataset loads unchanged", "dataset_slots": 16, "hot_loads_per_hash": 32, "key": ["0x3a48bef5", "0x6b54b1a1", "0x9ff897c3", "0x7d5d85e4", "0xcd35379f", "0x9d76bb86", "0xbe2affb1", "0x0f8f80b4"], "key_derivation": "seed_words_from_bytes('igneum-hot/' || seed_bytes), the same for every attempt of the epoch", "tag": ["0x49676e65", "0x756d4854"], "chain": "the cache chain of dataset.cache with the hot key and tag: in_j = prev ^ (sigma || key || seg || j || tag); line_j = block(in_j); prev_0 = 0", "load": "dst = dst ^ hot[mulhi(src, words)] (the high 32 bits of the 64-bit product; the index lies in [0, words) for any size)", "slots_rule": "the first k load slots in the Fisher-Yates draw order after the scratch slots (a uniform k-subset); no extra draw, so with version 2 widths and no scratch the program is the version 2 program with k loads redirected", "acceptance_stand_in": "dataset_elem(mulhi(src, words), seed_words[2], seed_words[3])", "program_id": "the read-width id with 'hot/' || mb || k appended", "spec": "docs/plans/hot-table.md"}, + "op_mix": {"load": 16, "shfl": 8, "add": 6, "xor": 6, "mul": 5, "rotr": 5, "hot": 4, "mad": 4, "or": 4, "sub": 4, "mulhi": 1, "rotl": 1}, + "register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]", + "splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16", + "iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order", + "output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo", + "op_semantics": { + "add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)", + "sub": "dst = dst - src", + "mul": "dst = dst * src (low 32)", + "mulhi": "dst = high 32 bits of dst * src", + "xor": "dst = dst ^ src", + "or": "dst = dst | src", + "rotl": "dst = rotl(dst, rot), rot in 1..31", + "rotr": "dst = rotr(dst, src & 31)", + "mad": "dst = src * src2 + dst", + "shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp", + "load": "dst = dst ^ dataset[src & dataset.mask]", + "wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)", + "hot": "dst = dst ^ hot[mulhi(src, hot_table.words)] (hot-table experiment)" + }, + "dataset": { + "log2_words": 28, + "bytes": 1073741824, + "mask": "0x0fffffff", + "day": "2026-10-03", + "day_bytes": "6461792f323032362d31302d3033", + "day_words_from": "seed_words_from_bytes(day_bytes)", + "d0": "0x3067619f", + "d1": "0x3c269176", + "mode": "memory-hard", + "spec": "proto-metal/MEMHARD.md", + "key": ["0x3067619f", "0x3c269176", "0x84a03b03", "0xf8c63294", "0xff977c5b", "0xe60def3e", "0x63630141", "0xb8fbcb58"], + "key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]", + "cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"}, + "mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [20, 20, 19, 4, 26, 3, 3, 27], "mul": ["0x42146205", "0x52cbe0fb", "0x7ecf4a03", "0x6728907f", "0xd81d9751", "0x132952c3", "0xf60de277", "0x05358035", "0xbaf6499d", "0xe4db9667", "0x3e98f45d", "0xd0004edd", "0x2691630d", "0x9beb3bcf", "0xab310379", "0x99cfb423"], "rc": ["0xbab68293", "0xcc162340", "0x6ce151cc", "0xe62b8997", "0xc9c80297", "0xf74a1654", "0x3d704af5", "0x3cf522b7", "0x2b9cac04", "0xa880ac10", "0x13e5dd1d", "0x6fc3e233", "0x2d83eeac", "0x9006e8bf", "0x2c4b5362", "0x31b49ee2"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"}, + "item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: s = M_r(s); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then s = M_8(s); item(t) = s", + "word": "dataset[w] = item(w >> 4)[w & 15]" + }, + "instructions": [ + {"i": 0, "op": "or", "dst": 6, "src": 4, "src2": 2, "imm": "0x535c5dd3", "imm2": "0xf8694615", "rot": 3, "bit": 1, "mask": 16, "width": 1}, + {"i": 1, "op": "shfl", "dst": 7, "src": 4, "src2": 6, "imm": "0x87bc9ee4", "imm2": "0xfdb0c856", "rot": 9, "bit": 25, "mask": 1, "width": 1}, + {"i": 2, "op": "or", "dst": 0, "src": 2, "src2": 4, "imm": "0xaf91f0c8", "imm2": "0xe2d5d1fa", "rot": 1, "bit": 24, "mask": 2, "width": 1}, + {"i": 3, "op": "shfl", "dst": 3, "src": 0, "src2": 0, "imm": "0x3044ba32", "imm2": "0x7b1a7ffe", "rot": 14, "bit": 10, "mask": 16, "width": 1}, + {"i": 4, "op": "load", "dst": 7, "src": 3, "src2": 6, "imm": "0x5d1ca2a2", "imm2": "0xe2481807", "rot": 24, "bit": 3, "mask": 1, "width": 1}, + {"i": 5, "op": "load", "dst": 6, "src": 0, "src2": 6, "imm": "0xaba3dbaa", "imm2": "0x987c017a", "rot": 16, "bit": 30, "mask": 1, "width": 1}, + {"i": 6, "op": "mul", "dst": 1, "src": 4, "src2": 6, "imm": "0x12a93d05", "imm2": "0x10761047", "rot": 17, "bit": 17, "mask": 4, "width": 1}, + {"i": 7, "op": "sub", "dst": 0, "src": 1, "src2": 0, "imm": "0x8884e389", "imm2": "0x31009b67", "rot": 5, "bit": 30, "mask": 1, "width": 1}, + {"i": 8, "op": "load", "dst": 3, "src": 7, "src2": 5, "imm": "0xfa6358b3", "imm2": "0xfea9038f", "rot": 17, "bit": 18, "mask": 2, "width": 1}, + {"i": 9, "op": "mad", "dst": 1, "src": 0, "src2": 3, "imm": "0xb154ae4f", "imm2": "0xe85f13f6", "rot": 6, "bit": 18, "mask": 16, "width": 1}, + {"i": 10, "op": "mul", "dst": 4, "src": 0, "src2": 4, "imm": "0x9edffbc3", "imm2": "0x643812d4", "rot": 24, "bit": 25, "mask": 4, "width": 1}, + {"i": 11, "op": "load", "dst": 5, "src": 4, "src2": 1, "imm": "0x163dbb1b", "imm2": "0x20c1f743", "rot": 27, "bit": 25, "mask": 16, "width": 1}, + {"i": 12, "op": "xor", "dst": 1, "src": 7, "src2": 4, "imm": "0x9cadb4f8", "imm2": "0xd9da785b", "rot": 8, "bit": 26, "mask": 4, "width": 1}, + {"i": 13, "op": "load", "dst": 1, "src": 6, "src2": 6, "imm": "0x1bb1b429", "imm2": "0x33d1e391", "rot": 20, "bit": 19, "mask": 16, "width": 1}, + {"i": 14, "op": "load", "dst": 2, "src": 1, "src2": 6, "imm": "0xa672cdd3", "imm2": "0x59a4829c", "rot": 22, "bit": 13, "mask": 16, "width": 1}, + {"i": 15, "op": "mul", "dst": 3, "src": 1, "src2": 4, "imm": "0xe3c4bf4d", "imm2": "0x028b4d37", "rot": 11, "bit": 28, "mask": 8, "width": 1}, + {"i": 16, "op": "hot", "dst": 6, "src": 5, "src2": 7, "imm": "0x71fbe6f2", "imm2": "0x61b6261e", "rot": 15, "bit": 9, "mask": 1, "width": 1}, + {"i": 17, "op": "load", "dst": 0, "src": 2, "src2": 5, "imm": "0x8e516fd3", "imm2": "0x2aa480c2", "rot": 19, "bit": 1, "mask": 8, "width": 1}, + {"i": 18, "op": "add", "dst": 0, "src": 7, "src2": 6, "imm": "0xc035a4e6", "imm2": "0x535b545f", "rot": 18, "bit": 26, "mask": 8, "width": 1}, + {"i": 19, "op": "sub", "dst": 5, "src": 0, "src2": 2, "imm": "0xde38f954", "imm2": "0xde1507bc", "rot": 31, "bit": 21, "mask": 4, "width": 1}, + {"i": 20, "op": "mul", "dst": 2, "src": 4, "src2": 1, "imm": "0x3f64b7c9", "imm2": "0x1f07754a", "rot": 23, "bit": 28, "mask": 4, "width": 1}, + {"i": 21, "op": "add", "dst": 2, "src": 1, "src2": 1, "imm": "0x1ecdd2cf", "imm2": "0x925d5e63", "rot": 6, "bit": 6, "mask": 8, "width": 1}, + {"i": 22, "op": "xor", "dst": 3, "src": 7, "src2": 3, "imm": "0x63c7f533", "imm2": "0xd312164d", "rot": 4, "bit": 14, "mask": 2, "width": 1}, + {"i": 23, "op": "load", "dst": 7, "src": 2, "src2": 5, "imm": "0x6d3ddc5a", "imm2": "0x433d8c2a", "rot": 28, "bit": 7, "mask": 2, "width": 1}, + {"i": 24, "op": "mulhi", "dst": 2, "src": 5, "src2": 0, "imm": "0xc7e9887a", "imm2": "0x19ec898f", "rot": 14, "bit": 9, "mask": 1, "width": 1}, + {"i": 25, "op": "rotr", "dst": 5, "src": 2, "src2": 3, "imm": "0x97acf9f2", "imm2": "0xc7fcfc8f", "rot": 8, "bit": 25, "mask": 1, "width": 1}, + {"i": 26, "op": "mad", "dst": 3, "src": 2, "src2": 7, "imm": "0x400383b6", "imm2": "0xe9bab735", "rot": 16, "bit": 18, "mask": 1, "width": 1}, + {"i": 27, "op": "sub", "dst": 2, "src": 0, "src2": 7, "imm": "0x8ea6cd8d", "imm2": "0x55b67f9f", "rot": 19, "bit": 8, "mask": 8, "width": 1}, + {"i": 28, "op": "mad", "dst": 6, "src": 1, "src2": 5, "imm": "0x17723e5a", "imm2": "0x00779664", "rot": 19, "bit": 2, "mask": 4, "width": 1}, + {"i": 29, "op": "add", "dst": 2, "src": 3, "src2": 4, "imm": "0xbdc6da76", "imm2": "0x46b1b505", "rot": 28, "bit": 16, "mask": 16, "width": 1}, + {"i": 30, "op": "shfl", "dst": 3, "src": 2, "src2": 3, "imm": "0xe586702c", "imm2": "0xd83a1462", "rot": 24, "bit": 6, "mask": 8, "width": 1}, + {"i": 31, "op": "load", "dst": 5, "src": 7, "src2": 1, "imm": "0x1cddad42", "imm2": "0xb0c6b0d0", "rot": 25, "bit": 25, "mask": 8, "width": 1}, + {"i": 32, "op": "hot", "dst": 2, "src": 0, "src2": 0, "imm": "0x2abcfdc6", "imm2": "0xfa4cc809", "rot": 22, "bit": 24, "mask": 2, "width": 1}, + {"i": 33, "op": "or", "dst": 6, "src": 3, "src2": 1, "imm": "0x61c9a38d", "imm2": "0x86597500", "rot": 15, "bit": 8, "mask": 2, "width": 1}, + {"i": 34, "op": "hot", "dst": 3, "src": 6, "src2": 7, "imm": "0xc37723fa", "imm2": "0xf3b024da", "rot": 16, "bit": 27, "mask": 16, "width": 1}, + {"i": 35, "op": "shfl", "dst": 4, "src": 0, "src2": 5, "imm": "0x84cad367", "imm2": "0xcc7972c4", "rot": 21, "bit": 17, "mask": 2, "width": 1}, + {"i": 36, "op": "xor", "dst": 5, "src": 2, "src2": 0, "imm": "0xc24a7d70", "imm2": "0x6591d24c", "rot": 21, "bit": 24, "mask": 1, "width": 1}, + {"i": 37, "op": "load", "dst": 3, "src": 4, "src2": 5, "imm": "0xd404cfe8", "imm2": "0x144ca538", "rot": 10, "bit": 11, "mask": 2, "width": 1}, + {"i": 38, "op": "load", "dst": 4, "src": 3, "src2": 3, "imm": "0x04b0080c", "imm2": "0xb938c290", "rot": 10, "bit": 1, "mask": 8, "width": 1}, + {"i": 39, "op": "rotr", "dst": 1, "src": 4, "src2": 5, "imm": "0x9fe93344", "imm2": "0xff7296e4", "rot": 29, "bit": 14, "mask": 1, "width": 1}, + {"i": 40, "op": "add", "dst": 3, "src": 6, "src2": 2, "imm": "0xbc3ff65f", "imm2": "0x6328cb2c", "rot": 4, "bit": 28, "mask": 16, "width": 1}, + {"i": 41, "op": "xor", "dst": 5, "src": 1, "src2": 1, "imm": "0xdcf4a02e", "imm2": "0x5ee3a976", "rot": 17, "bit": 0, "mask": 8, "width": 1}, + {"i": 42, "op": "or", "dst": 5, "src": 0, "src2": 1, "imm": "0x44846c7a", "imm2": "0x67338877", "rot": 27, "bit": 18, "mask": 1, "width": 1}, + {"i": 43, "op": "xor", "dst": 7, "src": 3, "src2": 3, "imm": "0x1b053acf", "imm2": "0x32e2d23d", "rot": 31, "bit": 23, "mask": 8, "width": 1}, + {"i": 44, "op": "load", "dst": 2, "src": 5, "src2": 4, "imm": "0xf8662282", "imm2": "0x10bb9e30", "rot": 8, "bit": 6, "mask": 2, "width": 1}, + {"i": 45, "op": "shfl", "dst": 6, "src": 4, "src2": 4, "imm": "0x87d9ef84", "imm2": "0x49087d74", "rot": 1, "bit": 2, "mask": 8, "width": 1}, + {"i": 46, "op": "xor", "dst": 5, "src": 3, "src2": 1, "imm": "0x74a59b7d", "imm2": "0xc4766ff1", "rot": 30, "bit": 3, "mask": 16, "width": 1}, + {"i": 47, "op": "add", "dst": 7, "src": 6, "src2": 2, "imm": "0x5af3bd5b", "imm2": "0x9342df0d", "rot": 26, "bit": 22, "mask": 2, "width": 1}, + {"i": 48, "op": "add", "dst": 3, "src": 4, "src2": 5, "imm": "0x9ff95776", "imm2": "0x485614db", "rot": 27, "bit": 23, "mask": 1, "width": 1}, + {"i": 49, "op": "load", "dst": 5, "src": 3, "src2": 1, "imm": "0xd022a812", "imm2": "0xfbe147d6", "rot": 15, "bit": 29, "mask": 4, "width": 1}, + {"i": 50, "op": "rotr", "dst": 4, "src": 5, "src2": 7, "imm": "0x7e099325", "imm2": "0x40d55d10", "rot": 29, "bit": 15, "mask": 16, "width": 1}, + {"i": 51, "op": "mad", "dst": 0, "src": 1, "src2": 7, "imm": "0xd8ab8843", "imm2": "0x7f842b90", "rot": 16, "bit": 9, "mask": 1, "width": 1}, + {"i": 52, "op": "mul", "dst": 0, "src": 3, "src2": 6, "imm": "0x4c30250b", "imm2": "0x9ee1681f", "rot": 20, "bit": 3, "mask": 8, "width": 1}, + {"i": 53, "op": "rotr", "dst": 5, "src": 4, "src2": 4, "imm": "0xc1c15026", "imm2": "0x588915e7", "rot": 1, "bit": 29, "mask": 16, "width": 1}, + {"i": 54, "op": "shfl", "dst": 0, "src": 6, "src2": 4, "imm": "0x636a9dc4", "imm2": "0xac023d9b", "rot": 22, "bit": 29, "mask": 1, "width": 1}, + {"i": 55, "op": "shfl", "dst": 0, "src": 7, "src2": 1, "imm": "0x5c933bd0", "imm2": "0x2baec8c9", "rot": 25, "bit": 27, "mask": 16, "width": 1}, + {"i": 56, "op": "load", "dst": 7, "src": 4, "src2": 7, "imm": "0xf62515d5", "imm2": "0x4a164f9f", "rot": 28, "bit": 30, "mask": 2, "width": 1}, + {"i": 57, "op": "rotr", "dst": 7, "src": 3, "src2": 7, "imm": "0x15ef64da", "imm2": "0x7149e3c9", "rot": 28, "bit": 30, "mask": 2, "width": 1}, + {"i": 58, "op": "load", "dst": 0, "src": 6, "src2": 2, "imm": "0xcde7100c", "imm2": "0x040fc0cf", "rot": 8, "bit": 28, "mask": 2, "width": 1}, + {"i": 59, "op": "hot", "dst": 6, "src": 2, "src2": 1, "imm": "0xcf8b48be", "imm2": "0x14cb1d3f", "rot": 16, "bit": 19, "mask": 1, "width": 1}, + {"i": 60, "op": "shfl", "dst": 3, "src": 2, "src2": 1, "imm": "0xae1529ee", "imm2": "0x732bc114", "rot": 4, "bit": 23, "mask": 1, "width": 1}, + {"i": 61, "op": "sub", "dst": 7, "src": 5, "src2": 7, "imm": "0x0a8f8168", "imm2": "0xf1c9b0ed", "rot": 29, "bit": 2, "mask": 8, "width": 1}, + {"i": 62, "op": "rotl", "dst": 1, "src": 2, "src2": 3, "imm": "0x9b7914dd", "imm2": "0xd14b33a3", "rot": 4, "bit": 16, "mask": 16, "width": 1}, + {"i": 63, "op": "load", "dst": 4, "src": 5, "src2": 1, "imm": "0xbbaa8e24", "imm2": "0xd15ec6a6", "rot": 24, "bit": 2, "mask": 2, "width": 1} + ] +} diff --git a/proto-cuda/packs-ca2-hot/hot64k4a/program.metal b/proto-cuda/packs-ca2-hot/hot64k4a/program.metal new file mode 100644 index 000000000..5b1be82fb --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4a/program.metal @@ -0,0 +1,112 @@ +#include +using namespace metal; + +#define MASK 0x0fffffffu +// Hot table (64 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }; + +inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 +inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +kernel void igneum_hash(device const uint* dataset [[buffer(0)]], + device ulong* out [[buffer(1)]], + constant uint& baseNonce [[buffer(2)]], + device const uint* hot [[buffer(3)]], + uint gid [[thread_position_in_grid]]) { + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + { uint x = nonce ^ SEEDW[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ SEEDW[1]; } + { uint x = nonce ^ SEEDW[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ SEEDW[2]; } + { uint x = nonce ^ SEEDW[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ SEEDW[3]; } + { uint x = nonce ^ SEEDW[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ SEEDW[4]; } + { uint x = nonce ^ SEEDW[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ SEEDW[5]; } + { uint x = nonce ^ SEEDW[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ SEEDW[6]; } + { uint x = nonce ^ SEEDW[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ SEEDW[7]; } + { uint x = nonce ^ SEEDW[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ SEEDW[0]; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r6 = r6 | r4; // 0 + r7 = r7 ^ simd_shuffle_xor(r4, (ushort)1); // 1 + r0 = r0 | r2; // 2 + r3 = r3 ^ simd_shuffle_xor(r0, (ushort)16); // 3 + r7 = r7 ^ dataset[r3 & MASK]; // 4 + r6 = r6 ^ dataset[r0 & MASK]; // 5 + r1 = r1 * r4; // 6 + r0 = r0 - r1; // 7 + r3 = r3 ^ dataset[r7 & MASK]; // 8 + r1 = r0 * r3 + r1; // 9 + r4 = r4 * r0; // 10 + r5 = r5 ^ dataset[r4 & MASK]; // 11 + r1 = r1 ^ r7; // 12 + r1 = r1 ^ dataset[r6 & MASK]; // 13 + r2 = r2 ^ dataset[r1 & MASK]; // 14 + r3 = r3 * r1; // 15 + r6 = r6 ^ hot[mulhi(r5, HOT_WORDS)]; // 16 + r0 = r0 ^ dataset[r2 & MASK]; // 17 + r0 = r0 + r7 + select(0xc035a4e6u, 0x535b545fu, ((sel >> 26u) & 1u) != 0u); // 18 + r5 = r5 - r0; // 19 + r2 = r2 * r4; // 20 + r2 = r2 + r1 + select(0x1ecdd2cfu, 0x925d5e63u, ((sel >> 6u) & 1u) != 0u); // 21 + r3 = r3 ^ r7; // 22 + r7 = r7 ^ dataset[r2 & MASK]; // 23 + r2 = mulhi(r2, r5); // 24 + r5 = rotr_var(r5, r2); // 25 + r3 = r2 * r7 + r3; // 26 + r2 = r2 - r0; // 27 + r6 = r1 * r5 + r6; // 28 + r2 = r2 + r3 + select(0xbdc6da76u, 0x46b1b505u, ((sel >> 16u) & 1u) != 0u); // 29 + r3 = r3 ^ simd_shuffle_xor(r2, (ushort)8); // 30 + r5 = r5 ^ dataset[r7 & MASK]; // 31 + r2 = r2 ^ hot[mulhi(r0, HOT_WORDS)]; // 32 + r6 = r6 | r3; // 33 + r3 = r3 ^ hot[mulhi(r6, HOT_WORDS)]; // 34 + r4 = r4 ^ simd_shuffle_xor(r0, (ushort)2); // 35 + r5 = r5 ^ r2; // 36 + r3 = r3 ^ dataset[r4 & MASK]; // 37 + r4 = r4 ^ dataset[r3 & MASK]; // 38 + r1 = rotr_var(r1, r4); // 39 + r3 = r3 + r6 + select(0xbc3ff65fu, 0x6328cb2cu, ((sel >> 28u) & 1u) != 0u); // 40 + r5 = r5 ^ r1; // 41 + r5 = r5 | r0; // 42 + r7 = r7 ^ r3; // 43 + r2 = r2 ^ dataset[r5 & MASK]; // 44 + r6 = r6 ^ simd_shuffle_xor(r4, (ushort)8); // 45 + r5 = r5 ^ r3; // 46 + r7 = r7 + r6 + select(0x5af3bd5bu, 0x9342df0du, ((sel >> 22u) & 1u) != 0u); // 47 + r3 = r3 + r4 + select(0x9ff95776u, 0x485614dbu, ((sel >> 23u) & 1u) != 0u); // 48 + r5 = r5 ^ dataset[r3 & MASK]; // 49 + r4 = rotr_var(r4, r5); // 50 + r0 = r1 * r7 + r0; // 51 + r0 = r0 * r3; // 52 + r5 = rotr_var(r5, r4); // 53 + r0 = r0 ^ simd_shuffle_xor(r6, (ushort)1); // 54 + r0 = r0 ^ simd_shuffle_xor(r7, (ushort)16); // 55 + r7 = r7 ^ dataset[r4 & MASK]; // 56 + r7 = rotr_var(r7, r3); // 57 + r0 = r0 ^ dataset[r6 & MASK]; // 58 + r6 = r6 ^ hot[mulhi(r2, HOT_WORDS)]; // 59 + r3 = r3 ^ simd_shuffle_xor(r2, (ushort)1); // 60 + r7 = r7 - r5; // 61 + r1 = rotl_imm(r1, 4u); // 62 + r4 = r4 ^ dataset[r5 & MASK]; // 63 + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot64k4a/program_bound.metal b/proto-cuda/packs-ca2-hot/hot64k4a/program_bound.metal new file mode 100644 index 000000000..df4eb94ff --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4a/program_bound.metal @@ -0,0 +1,114 @@ +#include +using namespace metal; + +#define MASK 0x0fffffffu +// Hot table (64 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }; + +inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 +inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW. +kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], + device ulong* out [[buffer(1)]], + constant uint& baseNonce [[buffer(2)]], + constant uint* initw [[buffer(3)]], + device const uint* hot [[buffer(4)]], + uint gid [[thread_position_in_grid]]) { + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + { uint x = nonce ^ initw[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ initw[1]; } + { uint x = nonce ^ initw[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ initw[2]; } + { uint x = nonce ^ initw[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ initw[3]; } + { uint x = nonce ^ initw[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ initw[4]; } + { uint x = nonce ^ initw[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ initw[5]; } + { uint x = nonce ^ initw[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ initw[6]; } + { uint x = nonce ^ initw[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ initw[7]; } + { uint x = nonce ^ initw[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ initw[0]; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r6 = r6 | r4; // 0 + r7 = r7 ^ simd_shuffle_xor(r4, (ushort)1); // 1 + r0 = r0 | r2; // 2 + r3 = r3 ^ simd_shuffle_xor(r0, (ushort)16); // 3 + r7 = r7 ^ dataset[r3 & MASK]; // 4 + r6 = r6 ^ dataset[r0 & MASK]; // 5 + r1 = r1 * r4; // 6 + r0 = r0 - r1; // 7 + r3 = r3 ^ dataset[r7 & MASK]; // 8 + r1 = r0 * r3 + r1; // 9 + r4 = r4 * r0; // 10 + r5 = r5 ^ dataset[r4 & MASK]; // 11 + r1 = r1 ^ r7; // 12 + r1 = r1 ^ dataset[r6 & MASK]; // 13 + r2 = r2 ^ dataset[r1 & MASK]; // 14 + r3 = r3 * r1; // 15 + r6 = r6 ^ hot[mulhi(r5, HOT_WORDS)]; // 16 + r0 = r0 ^ dataset[r2 & MASK]; // 17 + r0 = r0 + r7 + select(0xc035a4e6u, 0x535b545fu, ((sel >> 26u) & 1u) != 0u); // 18 + r5 = r5 - r0; // 19 + r2 = r2 * r4; // 20 + r2 = r2 + r1 + select(0x1ecdd2cfu, 0x925d5e63u, ((sel >> 6u) & 1u) != 0u); // 21 + r3 = r3 ^ r7; // 22 + r7 = r7 ^ dataset[r2 & MASK]; // 23 + r2 = mulhi(r2, r5); // 24 + r5 = rotr_var(r5, r2); // 25 + r3 = r2 * r7 + r3; // 26 + r2 = r2 - r0; // 27 + r6 = r1 * r5 + r6; // 28 + r2 = r2 + r3 + select(0xbdc6da76u, 0x46b1b505u, ((sel >> 16u) & 1u) != 0u); // 29 + r3 = r3 ^ simd_shuffle_xor(r2, (ushort)8); // 30 + r5 = r5 ^ dataset[r7 & MASK]; // 31 + r2 = r2 ^ hot[mulhi(r0, HOT_WORDS)]; // 32 + r6 = r6 | r3; // 33 + r3 = r3 ^ hot[mulhi(r6, HOT_WORDS)]; // 34 + r4 = r4 ^ simd_shuffle_xor(r0, (ushort)2); // 35 + r5 = r5 ^ r2; // 36 + r3 = r3 ^ dataset[r4 & MASK]; // 37 + r4 = r4 ^ dataset[r3 & MASK]; // 38 + r1 = rotr_var(r1, r4); // 39 + r3 = r3 + r6 + select(0xbc3ff65fu, 0x6328cb2cu, ((sel >> 28u) & 1u) != 0u); // 40 + r5 = r5 ^ r1; // 41 + r5 = r5 | r0; // 42 + r7 = r7 ^ r3; // 43 + r2 = r2 ^ dataset[r5 & MASK]; // 44 + r6 = r6 ^ simd_shuffle_xor(r4, (ushort)8); // 45 + r5 = r5 ^ r3; // 46 + r7 = r7 + r6 + select(0x5af3bd5bu, 0x9342df0du, ((sel >> 22u) & 1u) != 0u); // 47 + r3 = r3 + r4 + select(0x9ff95776u, 0x485614dbu, ((sel >> 23u) & 1u) != 0u); // 48 + r5 = r5 ^ dataset[r3 & MASK]; // 49 + r4 = rotr_var(r4, r5); // 50 + r0 = r1 * r7 + r0; // 51 + r0 = r0 * r3; // 52 + r5 = rotr_var(r5, r4); // 53 + r0 = r0 ^ simd_shuffle_xor(r6, (ushort)1); // 54 + r0 = r0 ^ simd_shuffle_xor(r7, (ushort)16); // 55 + r7 = r7 ^ dataset[r4 & MASK]; // 56 + r7 = rotr_var(r7, r3); // 57 + r0 = r0 ^ dataset[r6 & MASK]; // 58 + r6 = r6 ^ hot[mulhi(r2, HOT_WORDS)]; // 59 + r3 = r3 ^ simd_shuffle_xor(r2, (ushort)1); // 60 + r7 = r7 - r5; // 61 + r1 = rotl_imm(r1, 4u); // 62 + r4 = r4 ^ dataset[r5 & MASK]; // 63 + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot64k4a/vectors.h b/proto-cuda/packs-ca2-hot/hot64k4a/vectors.h new file mode 100644 index 000000000..b5754ac5f --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4a/vectors.h @@ -0,0 +1,67 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Expected outputs: igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif + +#define IGNEUM_VEC_WARPS 3 +static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u }; +static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = { + { // base nonce 0 + 0x4a6c42fc5764d8daull, 0x37c212a56e0569d3ull, 0x7b7a757a397db3f9ull, 0x5f9239251eba91acull, 0x135c81a6021e6ebaull, 0xdfa3a5e06a090a2dull, 0xa9fb970071c3f44eull, 0x52ea9c6007fab551ull, + 0xd8cf556ecaecf076ull, 0xa62cb0a038cd1b6bull, 0xb9f2deaafe56f529ull, 0x12d5820babcd7ed7ull, 0x2fd74c1f4c263db1ull, 0x0a6d69f426775ccfull, 0x54770096d3830839ull, 0x716fb1b1386bb7c9ull, + 0xabee533d572e49d0ull, 0x1016d59d921ac63aull, 0xcba4535a1aa1cbecull, 0x8bb81e8fc4155979ull, 0x49b38d7f02559f63ull, 0x5d3305c9df5caec5ull, 0x7d27218c53fceaffull, 0xd5227fd7a84cd56bull, + 0x9a8bdfa2072dabfaull, 0x04d0c1de6291afcbull, 0x081e212f940bdcb0ull, 0xd89a68f84756809aull, 0x01c8d632262d9d4bull, 0x8aacf4738a85e560ull, 0x0cd6e71517b63f2eull, 0x16228149075efbbdull + }, + { // base nonce 4096 + 0xacc15aebf682c54eull, 0xb0161dc67381582bull, 0xb6e4fe3e5c7ef39bull, 0x84e8d563e4177b06ull, 0x7a82671908cf3339ull, 0x615d82bfc8764e2bull, 0x5e10090034c4f702ull, 0x6ac4930e87ec07d4ull, + 0xb84b0e2f24efc7cfull, 0xc32dbd5e9ba5d8ceull, 0xc08a54d4cdf36eb1ull, 0xa97d9ed652f10bb6ull, 0x5245ef8bd1f8c7ecull, 0x255f3c03e0aab8d2ull, 0xe13ac7aa4fea498bull, 0x359427f974cefc36ull, + 0x7e1edeceb422b084ull, 0xd52ab3281a61b0bfull, 0xcadfb6c064863f68ull, 0x151dea2e8ba325feull, 0x72765f61e58f7e14ull, 0xc1408a61c1a7a8e9ull, 0xe19b0547c381fcefull, 0x1ffe3c35eaf66da6ull, + 0xa65d3957dded456eull, 0x9d8be52052fdf575ull, 0xf9215ef5fe2cff7dull, 0xdb6a6d7479219649ull, 0x6d7c535bf72c9244ull, 0x3cb662e110255e7bull, 0x10ea30028c05989full, 0x4cd430d0231a71d4ull + }, + { // base nonce 1000000 + 0x9ac3bb238bb22131ull, 0x0bda004ad3f1ff29ull, 0x899ca1164378afb2ull, 0x46b6a4102c6d7a6cull, 0x2af9223204de4911ull, 0x5970c0c0a985a8e7ull, 0x204dd16a10561ecdull, 0x18fb72c36de899eeull, + 0xf1cc27b160f797feull, 0x080b9a4ae81f63a1ull, 0x7ba0935f19aa11cfull, 0xa4e3728850084707ull, 0xcd7be68dcef247a7ull, 0x16ac696dc46ed44eull, 0x05f2629a93b1d468ull, 0x52e6f0ec20da269full, + 0xb2b6041980aea11cull, 0x257d149fbd830dccull, 0x8b176d000043b42eull, 0xb318688b422040d5ull, 0xb1795b8c9a7d7ba9ull, 0x8639442ed4c424caull, 0x4ccf966dd01cf886ull, 0xbc7e653603bf904bull, + 0x25593174d1637b15ull, 0xac52c8645ce16ee0ull, 0x1ab7366733ec0850ull, 0x76d375c623fa7438ull, 0xb8af625ded198b28ull, 0xd34b4bd1e70ea01full, 0xd50747c69aac5f02ull, 0xc3a4c63e9b334ed1ull + } +}; + +// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (268435455). +static const uint32_t IGNEUM_DS_HEAD[16] = { + 0xffc3cd94u, 0x5920ccd8u, 0x392f44bbu, 0x5e57f67au, 0x2f2bc2a9u, 0x620b0e36u, 0xbdc09014u, 0x436654bfu, + 0x311e0b48u, 0x1abd93adu, 0x59cc7ce8u, 0xee5247b2u, 0x86171fe8u, 0x6d874751u, 0xc9f7728fu, 0x7c2a435du +}; +static const uint32_t IGNEUM_DS_LAST_INDEX = 268435455u; +static const uint32_t IGNEUM_DS_LAST = 0xa33ada72u; +// 64 sampled dataset words (index, value) computed on the Mac. +#define IGNEUM_DS_SAMPLES 64 +static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = { + 59471966u, 217795994u, 208353206u, 42483309u, 172547758u, 148076330u, 183853158u, 214389424u, 267488061u, 169781097u, 184093494u, 153880993u, 84977930u, 46426879u, 3093825u, 225364072u, 44593546u, 260713159u, 168250303u, 52384140u, 223401610u, 45554030u, 95410555u, 175039924u, 79171087u, 267580473u, 24168642u, 37981670u, 171551130u, 195559979u, 204611762u, 140997658u, 138925853u, 86637313u, 20736778u, 219665210u, 160430336u, 264654675u, 8013395u, 228945585u, 213884386u, 104419827u, 44185464u, 142737231u, 99284897u, 132475900u, 61861762u, 132056166u, 262388043u, 91878046u, 117353561u, 124768597u, 71352993u, 190698941u, 46055428u, 55281366u, 165145231u, 106810753u, 171985651u, 232085256u, 159510492u, 40072060u, 209107596u, 39023794u +}; +static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = { + 0xe8b73d94u, 0x337028b5u, 0xafe148c9u, 0xab99f7aeu, 0x434ea619u, 0xd85cb880u, 0x54764c7fu, 0x82c7e420u, 0xedf4cb9eu, 0x9884c959u, 0x223ee793u, 0x3a9ccf69u, 0x81da4fd2u, 0xd6ce8cb9u, 0xe3922dcau, 0x3e7e6bdeu, 0x382a3acau, 0x567e7f7fu, 0x25a0f084u, 0xbfeef128u, 0xe338abfbu, 0x7c3b5280u, 0x909bc5f1u, 0xd8b74b9cu, 0x8e31a22eu, 0x26b5f1d8u, 0x79122c00u, 0xcafc3340u, 0xd5e02ea3u, 0x1aee1afdu, 0xdb090d9au, 0xb049f435u, 0x4954d8bau, 0x03797ba0u, 0x196eefbdu, 0xd153412au, 0xbe5d2c4bu, 0xdaa14f0eu, 0x8e61ed07u, 0x9e9a64c6u, 0x2e29ff36u, 0x392a8589u, 0xb56a5912u, 0xfa6e8b57u, 0xd1a737cbu, 0xb0fa841au, 0xbe1c341fu, 0xe25be0f1u, 0xe937f543u, 0xebab2248u, 0x8e1b607au, 0x202a2fedu, 0x95e2819cu, 0x9c9652d4u, 0x32fedef0u, 0xdecfff82u, 0xcb5d43e5u, 0xb735806au, 0x8905939cu, 0xfbf8472du, 0xada74e5du, 0x7ebdeeeau, 0x0119f2b3u, 0xa9a376b8u +}; +// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^26 words. +static const uint32_t IGNEUM_CACHE_HEAD[16] = { + 0x355a86d2u, 0x7957db1cu, 0xd21772afu, 0x6fc1e09bu, 0xd55ce61du, 0x6e6a278bu, 0xd3f543ceu, 0x223d8e82u, + 0x143ab337u, 0x2e9f05bdu, 0x2eb389bfu, 0x0c6e449eu, 0x5cfa4222u, 0xba6560feu, 0x8e3e1aa4u, 0xdbcc1d53u +}; +static const uint32_t IGNEUM_CACHE_LAST[16] = { + 0x41190d91u, 0xbd277957u, 0x22ddbb49u, 0x6986f207u, 0xdf69a4d6u, 0x26401a3au, 0x818230fbu, 0xc417122du, + 0x3597b211u, 0xb553ce55u, 0xcf39cc0du, 0x3b7fc43au, 0x3fd43b00u, 0x67e1c80eu, 0xffa7ea7du, 0xca2960abu +}; +static const uint64_t IGNEUM_CACHE_FNV64 = 0x48c4f5bf24166b2eull; +// Hot table self-test (docs/plans/hot-table.md): hot[0..15], the last 16 words, and FNV-1a 64 over all IGNEUM_HOT_WORDS words. +static const uint32_t IGNEUM_HOT_HEAD[16] = { + 0x8068cc73u, 0x6036ebf9u, 0xb604cd25u, 0x8ffb840eu, 0xc54074a2u, 0x285c0695u, 0x77512425u, 0xc26a58a7u, + 0x72c88757u, 0xc10fca78u, 0x513825ddu, 0x30d6ccc8u, 0x9a05e7cfu, 0xb9533f50u, 0x4bac3ba0u, 0xa5c19528u +}; +static const uint32_t IGNEUM_HOT_LAST[16] = { + 0x7c6d7cebu, 0xe24fcd46u, 0xe5cce976u, 0x3ecffe59u, 0x94e98b66u, 0x3f6ef45cu, 0xa3715aa1u, 0xdbe35281u, + 0xba05d27bu, 0x00541963u, 0xe636f453u, 0xd3972366u, 0x529599b8u, 0x79ae3c8cu, 0x48436863u, 0x897a8bf3u +}; +static const uint64_t IGNEUM_HOT_FNV64 = 0x77ca4b9527104530ull; diff --git a/proto-cuda/packs-ca2-hot/hot64k4a/vectors.json b/proto-cuda/packs-ca2-hot/hot64k4a/vectors.json new file mode 100644 index 000000000..1233f634d --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k4a/vectors.json @@ -0,0 +1,39 @@ +{ + "seed": "igneum-genesis", + "day": "2026-10-03", + "dataset_mode": "memory-hard", + "dataset_log2_words": 28, + "mask": "0x0fffffff", + "lanes": 32, + "source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset", + "warps": [ + {"base_nonce": 0, "expected": [ + "0x4a6c42fc5764d8da", "0x37c212a56e0569d3", "0x7b7a757a397db3f9", "0x5f9239251eba91ac", "0x135c81a6021e6eba", "0xdfa3a5e06a090a2d", "0xa9fb970071c3f44e", "0x52ea9c6007fab551", + "0xd8cf556ecaecf076", "0xa62cb0a038cd1b6b", "0xb9f2deaafe56f529", "0x12d5820babcd7ed7", "0x2fd74c1f4c263db1", "0x0a6d69f426775ccf", "0x54770096d3830839", "0x716fb1b1386bb7c9", + "0xabee533d572e49d0", "0x1016d59d921ac63a", "0xcba4535a1aa1cbec", "0x8bb81e8fc4155979", "0x49b38d7f02559f63", "0x5d3305c9df5caec5", "0x7d27218c53fceaff", "0xd5227fd7a84cd56b", + "0x9a8bdfa2072dabfa", "0x04d0c1de6291afcb", "0x081e212f940bdcb0", "0xd89a68f84756809a", "0x01c8d632262d9d4b", "0x8aacf4738a85e560", "0x0cd6e71517b63f2e", "0x16228149075efbbd" + ]}, + {"base_nonce": 4096, "expected": [ + "0xacc15aebf682c54e", "0xb0161dc67381582b", "0xb6e4fe3e5c7ef39b", "0x84e8d563e4177b06", "0x7a82671908cf3339", "0x615d82bfc8764e2b", "0x5e10090034c4f702", "0x6ac4930e87ec07d4", + "0xb84b0e2f24efc7cf", "0xc32dbd5e9ba5d8ce", "0xc08a54d4cdf36eb1", "0xa97d9ed652f10bb6", "0x5245ef8bd1f8c7ec", "0x255f3c03e0aab8d2", "0xe13ac7aa4fea498b", "0x359427f974cefc36", + "0x7e1edeceb422b084", "0xd52ab3281a61b0bf", "0xcadfb6c064863f68", "0x151dea2e8ba325fe", "0x72765f61e58f7e14", "0xc1408a61c1a7a8e9", "0xe19b0547c381fcef", "0x1ffe3c35eaf66da6", + "0xa65d3957dded456e", "0x9d8be52052fdf575", "0xf9215ef5fe2cff7d", "0xdb6a6d7479219649", "0x6d7c535bf72c9244", "0x3cb662e110255e7b", "0x10ea30028c05989f", "0x4cd430d0231a71d4" + ]}, + {"base_nonce": 1000000, "expected": [ + "0x9ac3bb238bb22131", "0x0bda004ad3f1ff29", "0x899ca1164378afb2", "0x46b6a4102c6d7a6c", "0x2af9223204de4911", "0x5970c0c0a985a8e7", "0x204dd16a10561ecd", "0x18fb72c36de899ee", + "0xf1cc27b160f797fe", "0x080b9a4ae81f63a1", "0x7ba0935f19aa11cf", "0xa4e3728850084707", "0xcd7be68dcef247a7", "0x16ac696dc46ed44e", "0x05f2629a93b1d468", "0x52e6f0ec20da269f", + "0xb2b6041980aea11c", "0x257d149fbd830dcc", "0x8b176d000043b42e", "0xb318688b422040d5", "0xb1795b8c9a7d7ba9", "0x8639442ed4c424ca", "0x4ccf966dd01cf886", "0xbc7e653603bf904b", + "0x25593174d1637b15", "0xac52c8645ce16ee0", "0x1ab7366733ec0850", "0x76d375c623fa7438", "0xb8af625ded198b28", "0xd34b4bd1e70ea01f", "0xd50747c69aac5f02", "0xc3a4c63e9b334ed1" + ]} + ], + "dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"], + "dataset_last_index": 268435455, + "dataset_last": "0xa33ada72", + "dataset_samples": [{"index": 59471966, "value": "0xe8b73d94"}, {"index": 217795994, "value": "0x337028b5"}, {"index": 208353206, "value": "0xafe148c9"}, {"index": 42483309, "value": "0xab99f7ae"}, {"index": 172547758, "value": "0x434ea619"}, {"index": 148076330, "value": "0xd85cb880"}, {"index": 183853158, "value": "0x54764c7f"}, {"index": 214389424, "value": "0x82c7e420"}, {"index": 267488061, "value": "0xedf4cb9e"}, {"index": 169781097, "value": "0x9884c959"}, {"index": 184093494, "value": "0x223ee793"}, {"index": 153880993, "value": "0x3a9ccf69"}, {"index": 84977930, "value": "0x81da4fd2"}, {"index": 46426879, "value": "0xd6ce8cb9"}, {"index": 3093825, "value": "0xe3922dca"}, {"index": 225364072, "value": "0x3e7e6bde"}, {"index": 44593546, "value": "0x382a3aca"}, {"index": 260713159, "value": "0x567e7f7f"}, {"index": 168250303, "value": "0x25a0f084"}, {"index": 52384140, "value": "0xbfeef128"}, {"index": 223401610, "value": "0xe338abfb"}, {"index": 45554030, "value": "0x7c3b5280"}, {"index": 95410555, "value": "0x909bc5f1"}, {"index": 175039924, "value": "0xd8b74b9c"}, {"index": 79171087, "value": "0x8e31a22e"}, {"index": 267580473, "value": "0x26b5f1d8"}, {"index": 24168642, "value": "0x79122c00"}, {"index": 37981670, "value": "0xcafc3340"}, {"index": 171551130, "value": "0xd5e02ea3"}, {"index": 195559979, "value": "0x1aee1afd"}, {"index": 204611762, "value": "0xdb090d9a"}, {"index": 140997658, "value": "0xb049f435"}, {"index": 138925853, "value": "0x4954d8ba"}, {"index": 86637313, "value": "0x03797ba0"}, {"index": 20736778, "value": "0x196eefbd"}, {"index": 219665210, "value": "0xd153412a"}, {"index": 160430336, "value": "0xbe5d2c4b"}, {"index": 264654675, "value": "0xdaa14f0e"}, {"index": 8013395, "value": "0x8e61ed07"}, {"index": 228945585, "value": "0x9e9a64c6"}, {"index": 213884386, "value": "0x2e29ff36"}, {"index": 104419827, "value": "0x392a8589"}, {"index": 44185464, "value": "0xb56a5912"}, {"index": 142737231, "value": "0xfa6e8b57"}, {"index": 99284897, "value": "0xd1a737cb"}, {"index": 132475900, "value": "0xb0fa841a"}, {"index": 61861762, "value": "0xbe1c341f"}, {"index": 132056166, "value": "0xe25be0f1"}, {"index": 262388043, "value": "0xe937f543"}, {"index": 91878046, "value": "0xebab2248"}, {"index": 117353561, "value": "0x8e1b607a"}, {"index": 124768597, "value": "0x202a2fed"}, {"index": 71352993, "value": "0x95e2819c"}, {"index": 190698941, "value": "0x9c9652d4"}, {"index": 46055428, "value": "0x32fedef0"}, {"index": 55281366, "value": "0xdecfff82"}, {"index": 165145231, "value": "0xcb5d43e5"}, {"index": 106810753, "value": "0xb735806a"}, {"index": 171985651, "value": "0x8905939c"}, {"index": 232085256, "value": "0xfbf8472d"}, {"index": 159510492, "value": "0xada74e5d"}, {"index": 40072060, "value": "0x7ebdeeea"}, {"index": 209107596, "value": "0x0119f2b3"}, {"index": 39023794, "value": "0xa9a376b8"}], + "cache_head": ["0x355a86d2", "0x7957db1c", "0xd21772af", "0x6fc1e09b", "0xd55ce61d", "0x6e6a278b", "0xd3f543ce", "0x223d8e82", "0x143ab337", "0x2e9f05bd", "0x2eb389bf", "0x0c6e449e", "0x5cfa4222", "0xba6560fe", "0x8e3e1aa4", "0xdbcc1d53"], + "cache_last_line": ["0x41190d91", "0xbd277957", "0x22ddbb49", "0x6986f207", "0xdf69a4d6", "0x26401a3a", "0x818230fb", "0xc417122d", "0x3597b211", "0xb553ce55", "0xcf39cc0d", "0x3b7fc43a", "0x3fd43b00", "0x67e1c80e", "0xffa7ea7d", "0xca2960ab"], + "cache_fnv1a64": "0x48c4f5bf24166b2e", + "hot_head": ["0x8068cc73", "0x6036ebf9", "0xb604cd25", "0x8ffb840e", "0xc54074a2", "0x285c0695", "0x77512425", "0xc26a58a7", "0x72c88757", "0xc10fca78", "0x513825dd", "0x30d6ccc8", "0x9a05e7cf", "0xb9533f50", "0x4bac3ba0", "0xa5c19528"], + "hot_last_line": ["0x7c6d7ceb", "0xe24fcd46", "0xe5cce976", "0x3ecffe59", "0x94e98b66", "0x3f6ef45c", "0xa3715aa1", "0xdbe35281", "0xba05d27b", "0x00541963", "0xe636f453", "0xd3972366", "0x529599b8", "0x79ae3c8c", "0x48436863", "0x897a8bf3"], + "hot_fnv1a64": "0x77ca4b9527104530" +} diff --git a/proto-cuda/packs-ca2-hot/hot64k8/kernel.cl b/proto-cuda/packs-ca2-hot/hot64k8/kernel.cl new file mode 100644 index 000000000..d02e82dd0 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k8/kernel.cl @@ -0,0 +1,305 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal). +// Built from source at runtime by proto-opencl/host.c, which passes these defines: +// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit) +// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default) +// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32 +// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition +// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units; +// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32. +#ifndef IGNEUM_GROUP +#define IGNEUM_GROUP 32 +#endif +#ifndef IGNEUM_EXCHANGE +#define IGNEUM_EXCHANGE 0 +#endif +#ifdef __OPENCL_VERSION__ +#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1))) +#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n] +#if IGNEUM_EXCHANGE == 1 +#ifdef cl_khr_subgroups +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif +#ifdef cl_khr_subgroup_shuffle +#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable +#endif +#elif IGNEUM_EXCHANGE == 2 +#pragma OPENCL EXTENSION cl_intel_subgroups : enable +#endif +#else +// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros. +#include "emu_opencl.h" +#endif + +#if IGNEUM_EXCHANGE == 1 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#elif IGNEUM_EXCHANGE == 2 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#else +// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per +// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane +// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's +// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier. +#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; } +#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; } +#endif + +// Hot table (64 MiB, 8 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +static inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32. +static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); } +// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x. +static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); } +static inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +static inline void mh_chacha_block(const uint* x, uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +static inline void mh_cache_segment(__global uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +static inline void mh_mixer(uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +static inline void mh_item(__global const uint* cache, uint t, uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + __global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item. +// The same constants as memhard.h in this pack (one emitter, three dialects). +__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) { + uint t = (uint)get_global_id(0); + if (t < nItems) { + uint s[16]; + mh_item(cache, t, s); + __global uint* d = ds + ((ulong)t * 16u); + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (docs/plans/hot-table.md): 64 MiB = 16384 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +static inline void ht_segment(__global uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one work-item per segment. +__kernel void igneum_hot_fill(__global uint* hot, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the +// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and +// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). +IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ hot[mul_hi(r2, HOT_WORDS)]; // 4 hot + r5 = r5 ^ ds[r7 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl + r1 = mul_hi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = mul_hi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl + r5 = r5 * r0; // 19 mul + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl + r6 = mul_hi(r6, r2); // 22 mulhi + r6 = r6 ^ hot[mul_hi(r1, HOT_WORDS)]; // 23 hot + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ hot[mul_hi(r0, HOT_WORDS)]; // 32 hot + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[mul_hi(r1, HOT_WORDS)]; // 34 hot + r0 = mul_hi(r0, r5); // 35 mulhi + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl + r7 = r7 ^ hot[mul_hi(r0, HOT_WORDS)]; // 37 hot + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ hot[mul_hi(r7, HOT_WORDS)]; // 44 hot + r2 = r2 * r3; // 45 mul + r1 = mul_hi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ hot[mul_hi(r2, HOT_WORDS)]; // 59 hot + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} + +#if IGNEUM_EXCHANGE != 0 +// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the +// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a +// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md. +IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) { + if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); } +} +#endif diff --git a/proto-cuda/packs-ca2-hot/hot64k8/kernel.cu b/proto-cuda/packs-ca2-hot/hot64k8/kernel.cu new file mode 100644 index 000000000..b51ebb341 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k8/kernel.cu @@ -0,0 +1,179 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal). +// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC. +#include +#include +#include "program.h" +#include "memhard.h" + +// Hot table (64 MiB, 8 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +__device__ __forceinline__ uint32_t splitmix32(uint32_t x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31. +__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } +// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x. +__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) { + uint32_t x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item. +// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host. +__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) { + uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x; + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) { + uint32_t t = blockIdx.x * blockDim.x + threadIdx.x; + if (t < nItems) { + uint32_t s[16]; + mh_item(cache, t, s); + uint32_t* d = ds + (size_t)t * 16u; + for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (ht_segment is in memhard.h): one thread per segment. +__global__ void igneum_hot_fill(uint32_t* hot, uint32_t nSegments) { + uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x; + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every +// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a +// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid. +__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot) { + uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x; + uint32_t nonce = baseNonce + gid; + uint32_t r0, r1, r2, r3, r4, r5, r6, r7; + { uint32_t x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint32_t x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint32_t x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint32_t x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint32_t x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint32_t x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint32_t x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint32_t x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint32_t it = 0u; it < 8u; ++it) { + uint32_t sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ hot[__umulhi(r2, HOT_WORDS)]; // 4 hot + r5 = r5 ^ ds[r7 & mask]; // 5 load + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 6 shfl + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 8); // 7 shfl + r1 = __umulhi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = __umulhi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[__umulhi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 18 shfl + r5 = r5 * r0; // 19 mul + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 20 shfl + r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 16); // 21 shfl + r6 = __umulhi(r6, r2); // 22 mulhi + r6 = r6 ^ hot[__umulhi(r1, HOT_WORDS)]; // 23 hot + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r6, 2); // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ hot[__umulhi(r0, HOT_WORDS)]; // 32 hot + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[__umulhi(r1, HOT_WORDS)]; // 34 hot + r0 = __umulhi(r0, r5); // 35 mulhi + r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // 36 shfl + r7 = r7 ^ hot[__umulhi(r0, HOT_WORDS)]; // 37 hot + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 4); // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ hot[__umulhi(r7, HOT_WORDS)]; // 44 hot + r2 = r2 * r3; // 45 mul + r1 = __umulhi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ hot[__umulhi(r2, HOT_WORDS)]; // 59 hot + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo; +} + +// Host-side launch wrappers. Declared in program.h, called from host.cu. +cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) { + if (nSegments == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nSegments + block - 1u) / block; + igneum_cache_fill<<>>(cache, nSegments); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) { + if (nItems == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nItems + block - 1u) / block; + igneum_build<<>>(ds, cache, nItems); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments) { + if (nSegments == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nSegments + block - 1u) / block; + igneum_hot_fill<<>>(hot, nSegments); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot, + uint32_t nonces, uint32_t blockWarps) { + if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue; + uint32_t block = 32u * blockWarps; + if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue; + igneum_hash<<>>(ds, out, baseNonce, mask, hot); + return cudaGetLastError(); +} + +cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) { + cudaFuncAttributes attr; + cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash); + if (e != cudaSuccess) return e; + *numRegs = attr.numRegs; + return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0); +} diff --git a/proto-cuda/packs-ca2-hot/hot64k8/kernel_bound.cl b/proto-cuda/packs-ca2-hot/hot64k8/kernel_bound.cl new file mode 100644 index 000000000..5c40a8ea0 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k8/kernel_bound.cl @@ -0,0 +1,399 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal). +// Built from source at runtime by proto-opencl/host.c, which passes these defines: +// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit) +// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default) +// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32 +// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition +// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units; +// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32. +#ifndef IGNEUM_GROUP +#define IGNEUM_GROUP 32 +#endif +#ifndef IGNEUM_EXCHANGE +#define IGNEUM_EXCHANGE 0 +#endif +#ifdef __OPENCL_VERSION__ +#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1))) +#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n] +#if IGNEUM_EXCHANGE == 1 +#ifdef cl_khr_subgroups +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif +#ifdef cl_khr_subgroup_shuffle +#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable +#endif +#elif IGNEUM_EXCHANGE == 2 +#pragma OPENCL EXTENSION cl_intel_subgroups : enable +#endif +#else +// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros. +#include "emu_opencl.h" +#endif + +#if IGNEUM_EXCHANGE == 1 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#elif IGNEUM_EXCHANGE == 2 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#else +// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per +// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane +// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's +// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier. +#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; } +#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; } +#endif + +// Hot table (64 MiB, 8 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +static inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32. +static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); } +// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x. +static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); } +static inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +static inline void mh_chacha_block(const uint* x, uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +static inline void mh_cache_segment(__global uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +static inline void mh_mixer(uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +static inline void mh_item(__global const uint* cache, uint t, uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + __global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item. +// The same constants as memhard.h in this pack (one emitter, three dialects). +__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) { + uint t = (uint)get_global_id(0); + if (t < nItems) { + uint s[16]; + mh_item(cache, t, s); + __global uint* d = ds + ((ulong)t * 16u); + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (docs/plans/hot-table.md): 64 MiB = 16384 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +static inline void ht_segment(__global uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one work-item per segment. +__kernel void igneum_hot_fill(__global uint* hot, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the +// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and +// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). +IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ hot[mul_hi(r2, HOT_WORDS)]; // 4 hot + r5 = r5 ^ ds[r7 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl + r1 = mul_hi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = mul_hi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl + r5 = r5 * r0; // 19 mul + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl + r6 = mul_hi(r6, r2); // 22 mulhi + r6 = r6 ^ hot[mul_hi(r1, HOT_WORDS)]; // 23 hot + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ hot[mul_hi(r0, HOT_WORDS)]; // 32 hot + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[mul_hi(r1, HOT_WORDS)]; // 34 hot + r0 = mul_hi(r0, r5); // 35 mulhi + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl + r7 = r7 ^ hot[mul_hi(r0, HOT_WORDS)]; // 37 hot + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ hot[mul_hi(r7, HOT_WORDS)]; // 44 hot + r2 = r2 * r3; // 45 mul + r1 = mul_hi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ hot[mul_hi(r2, HOT_WORDS)]; // 59 hot + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} + +#if IGNEUM_EXCHANGE != 0 +// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the +// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a +// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md. +IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) { + if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); } +} +#endif + +// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash. +IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7]; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; } + { uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; } + { uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; } + { uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; } + { uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; } + { uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; } + { uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; } + { uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ hot[mul_hi(r2, HOT_WORDS)]; // 4 hot + r5 = r5 ^ ds[r7 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl + r1 = mul_hi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = mul_hi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl + r5 = r5 * r0; // 19 mul + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl + r6 = mul_hi(r6, r2); // 22 mulhi + r6 = r6 ^ hot[mul_hi(r1, HOT_WORDS)]; // 23 hot + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ hot[mul_hi(r0, HOT_WORDS)]; // 32 hot + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[mul_hi(r1, HOT_WORDS)]; // 34 hot + r0 = mul_hi(r0, r5); // 35 mulhi + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl + r7 = r7 ^ hot[mul_hi(r0, HOT_WORDS)]; // 37 hot + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ hot[mul_hi(r7, HOT_WORDS)]; // 44 hot + r2 = r2 * r3; // 45 mul + r1 = mul_hi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ hot[mul_hi(r2, HOT_WORDS)]; // 59 hot + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot64k8/kernel_bound.cu b/proto-cuda/packs-ca2-hot/hot64k8/kernel_bound.cu new file mode 100644 index 000000000..c28cece2e --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k8/kernel_bound.cu @@ -0,0 +1,125 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW. +// Host declarations (also in program_bound.h if present): +// struct IgneumInitWords { uint32_t w[8]; }; +// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, +// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps); +// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps); +#include +#include +#include "program.h" + +struct IgneumInitWords { uint32_t w[8]; }; + +// Hot table (64 MiB, 8 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +__device__ __forceinline__ uint32_t splitmix32(uint32_t x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } +__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } + +__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw, const uint32_t* hot) { + uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x; + uint32_t nonce = baseNonce + gid; + uint32_t r0, r1, r2, r3, r4, r5, r6, r7; + { uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; } + { uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; } + { uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; } + { uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; } + { uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; } + { uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; } + { uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; } + { uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; } + + for (uint32_t it = 0u; it < 8u; ++it) { + uint32_t sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ hot[__umulhi(r2, HOT_WORDS)]; // 4 hot + r5 = r5 ^ ds[r7 & mask]; // 5 load + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 6 shfl + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 8); // 7 shfl + r1 = __umulhi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = __umulhi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[__umulhi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 18 shfl + r5 = r5 * r0; // 19 mul + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 20 shfl + r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 16); // 21 shfl + r6 = __umulhi(r6, r2); // 22 mulhi + r6 = r6 ^ hot[__umulhi(r1, HOT_WORDS)]; // 23 hot + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r6, 2); // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ hot[__umulhi(r0, HOT_WORDS)]; // 32 hot + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[__umulhi(r1, HOT_WORDS)]; // 34 hot + r0 = __umulhi(r0, r5); // 35 mulhi + r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // 36 shfl + r7 = r7 ^ hot[__umulhi(r0, HOT_WORDS)]; // 37 hot + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 4); // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ hot[__umulhi(r7, HOT_WORDS)]; // 44 hot + r2 = r2 * r3; // 45 mul + r1 = __umulhi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ hot[__umulhi(r2, HOT_WORDS)]; // 59 hot + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo; +} + +cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, + IgneumInitWords iw, const uint32_t* hot, uint32_t nonces, uint32_t blockWarps) { + if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue; + uint32_t block = 32u * blockWarps; + if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue; + igneum_hash_bound<<>>(ds, out, baseNonce, mask, iw, hot); + return cudaGetLastError(); +} + +cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) { + cudaFuncAttributes attr; + cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound); + if (e != cudaSuccess) return e; + *numRegs = attr.numRegs; + return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0); +} diff --git a/proto-cuda/packs-ca2-hot/hot64k8/memhard.h b/proto-cuda/packs-ca2-hot/hot64k8/memhard.h new file mode 100644 index 000000000..180f678b7 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k8/memhard.h @@ -0,0 +1,129 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against. +// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference). +// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C. +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif +#if defined(__CUDACC__) +#define IGNEUM_HD __host__ __device__ __forceinline__ +#elif defined(_MSC_VER) && !defined(__cplusplus) +#define IGNEUM_HD static __inline +#else +#define IGNEUM_HD static inline +#endif +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) { + for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint32_t r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) { + uint32_t prev[16]; uint32_t x[16]; uint32_t y[16]; + for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +IGNEUM_HD void mh_item(const uint32_t* cache, uint32_t t, uint32_t* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint32_t r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +IGNEUM_HD uint32_t mh_word(const uint32_t* cache, uint32_t w) { uint32_t s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Hot table (docs/plans/hot-table.md): 64 MiB = 16384 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +IGNEUM_HD void ht_segment(uint32_t* hot, uint32_t seg) { + uint32_t prev[16]; uint32_t x[16]; uint32_t y[16]; + for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + uint32_t* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} diff --git a/proto-cuda/packs-ca2-hot/hot64k8/memhard.metal b/proto-cuda/packs-ca2-hot/hot64k8/memhard.metal new file mode 100644 index 000000000..c2ec5702e --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k8/memhard.metal @@ -0,0 +1,131 @@ +#include +using namespace metal; +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +inline void mh_chacha_block(const thread uint* x, thread uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +inline void mh_cache_segment(device uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +inline void mh_mixer(thread uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +inline void mh_item(device const uint* cache, uint t, thread uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +inline uint mh_word(device const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// One thread per segment (2^16 threads). +kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) { + mh_cache_segment(cache, gid); +} +// One thread per 64-byte item (dataset words / 16 threads). +kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]], + uint gid [[thread_position_in_grid]]) { + uint s[16]; + mh_item(cache, gid, s); + device uint* d = dataset + gid * 16u; + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; +} + +// Hot table (docs/plans/hot-table.md): 64 MiB = 16384 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +inline void ht_segment(device uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + device uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one thread per segment (IGNEUM_HOT_SEGMENTS threads). +kernel void igneum_hot_fill(device uint* hot [[buffer(0)]], uint gid [[thread_position_in_grid]]) { + ht_segment(hot, gid); +} diff --git a/proto-cuda/packs-ca2-hot/hot64k8/program.h b/proto-cuda/packs-ca2-hot/hot64k8/program.h new file mode 100644 index 000000000..88d5df38e --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k8/program.h @@ -0,0 +1,70 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Program metadata for host.cu plus the launch wrappers defined in kernel.cu. +// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros. +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif +#ifndef IGNEUM_NO_CUDA +#include +#endif + +#define IGNEUM_SEED_STRING "igneum-genesis" +#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973" +#define IGNEUM_GENERATOR 2 +#define IGNEUM_PROGRAM_ATTEMPT 0 +#define IGNEUM_PROGRAM_ID 0x322a68466d4a0998ull +#define IGNEUM_DAY_STRING "2026-10-03" +#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033" +#define IGNEUM_DAY0 0x3067619fu +#define IGNEUM_DAY1 0x3c269176u +#define IGNEUM_DATASET_LOG2 28 +#define IGNEUM_MASK 0x0fffffffu +#define IGNEUM_LANES 32 +#define IGNEUM_ITERATIONS 8 +#define IGNEUM_INSTR_COUNT 64 +#define IGNEUM_LOADS_PER_HASH 128 +#define IGNEUM_WIDE_LOADS_PER_HASH 0 +#define IGNEUM_OP_MIX "add=8 hot=8 load=8 shfl=8 xor=6 mad=5 mul=5 mulhi=5 sub=4 rotl=3 rotr=3 or=1" +// Class v3 construction (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): version 2 loads; the dataset item +// derivation applies the mixer IGNEUM_MIXER_MULT times per round (memhard.h), and the cache follows the growth rule. +#define IGNEUM_LOAD_CLASS "hot64k8" +#define IGNEUM_LOAD_SLOTS 16 +#define IGNEUM_LOAD_MIX { 100, 0, 0 } +#define IGNEUM_LOAD_WIDTH_COUNTS { 8, 0, 0 } // loads of 4, 16, 64 bytes per program +#define IGNEUM_BYTES_PER_HASH 256 +#define IGNEUM_FOLD_ROT 11 +#define IGNEUM_FOLD_MUL 0x9e3779b1u +// Hot table (hot-table experiment, 5 October 2026, docs/plans/hot-table.md): NOT the lottery hash. IGNEUM_HOT_SLOTS of the +// 16 load slots read H[mulhi(src, IGNEUM_HOT_WORDS)], H = IGNEUM_HOT_MB MiB of chained ChaCha12 lines (the cache chain) under +// KH = seed_words("igneum-hot/" || epoch seed bytes) and the tag "Igne" "umHT"; filled by igneum_hot_fill once per epoch. +#define IGNEUM_HOT_MB 64 +#define IGNEUM_HOT_WORDS 0x01000000u +#define IGNEUM_HOT_SEGMENTS 16384u +#define IGNEUM_HOT_SLOTS 8 // hot loads per program (64 per hash), replacing the dataset loads (8 of them) +#define IGNEUM_HOT_ADDED 0 +#define IGNEUM_HOT_KEY_INIT { 0x3a48bef5u, 0x6b54b1a1u, 0x9ff897c3u, 0x7d5d85e4u, 0xcd35379fu, 0x9d76bb86u, 0xbe2affb1u, 0x0f8f80b4u } +// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h) +#define IGNEUM_DATASET_MODE 1 + +#define IGNEUM_SEEDW_INIT { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u } +#define IGNEUM_KEY_INIT { 0x3067619fu, 0x3c269176u, 0x84a03b03u, 0xf8c63294u, 0xff977c5bu, 0xe60def3eu, 0x63630141u, 0xb8fbcb58u } +#define IGNEUM_CACHE_LOG2_WORDS 26 +#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6 +#define IGNEUM_CACHE_SEGMENTS 65536u +#define IGNEUM_ITEM_ROUNDS 8 +#define IGNEUM_MIX_ROT_INIT { 20u, 20u, 19u, 4u, 26u, 3u, 3u, 27u } +#define IGNEUM_MIX_MUL_INIT { 0x42146205u, 0x52cbe0fbu, 0x7ecf4a03u, 0x6728907fu, 0xd81d9751u, 0x132952c3u, 0xf60de277u, 0x05358035u, 0xbaf6499du, 0xe4db9667u, 0x3e98f45du, 0xd0004eddu, 0x2691630du, 0x9beb3bcfu, 0xab310379u, 0x99cfb423u } +#define IGNEUM_MIX_RC_INIT { 0xbab68293u, 0xcc162340u, 0x6ce151ccu, 0xe62b8997u, 0xc9c80297u, 0xf74a1654u, 0x3d704af5u, 0x3cf522b7u, 0x2b9cac04u, 0xa880ac10u, 0x13e5dd1du, 0x6fc3e233u, 0x2d83eeacu, 0x9006e8bfu, 0x2c4b5362u, 0x31b49ee2u } + +#ifndef IGNEUM_NO_CUDA +// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError(). +cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments); +cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems); +cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments); +cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot, + uint32_t nonces, uint32_t blockWarps); +cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps); +#endif diff --git a/proto-cuda/packs-ca2-hot/hot64k8/program.json b/proto-cuda/packs-ca2-hot/hot64k8/program.json new file mode 100644 index 000000000..63e52f1c4 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k8/program.json @@ -0,0 +1,129 @@ +{ + "format": "igneum-program-pack-3", + "generator": 2, + "attempt": 0, + "program_id": "0x322a68466d4a0998", + "program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32", + "dataset_mode": "memory-hard", + "seed": "igneum-genesis", + "seed_bytes": "69676e65756d2d67656e65736973", + "seed_words": ["0x67a9a7be", "0x1a155b25", "0xfddfb732", "0x4b5af2e8", "0xc55caf33", "0xa27c13b7", "0x06628a48", "0x03852469"], + "seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32", + "generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried", + "lanes": 32, + "registers": 8, + "iterations": 8, + "instruction_count": 64, + "loads_per_hash": 128, + "load_class": "hot64k8", + "load_slots": 16, + "load_mix_percent_4_16_64": [100, 0, 0], + "load_width_counts_4_16_64": [8, 0, 0], + "bytes_per_hash": 256, + "wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots", + "hot_table": {"mb": 64, "words": 16777216, "segments": 16384, "slots": 8, "form": "replaced: k of the class's load slots read the table", "dataset_slots": 8, "hot_loads_per_hash": 64, "key": ["0x3a48bef5", "0x6b54b1a1", "0x9ff897c3", "0x7d5d85e4", "0xcd35379f", "0x9d76bb86", "0xbe2affb1", "0x0f8f80b4"], "key_derivation": "seed_words_from_bytes('igneum-hot/' || seed_bytes), the same for every attempt of the epoch", "tag": ["0x49676e65", "0x756d4854"], "chain": "the cache chain of dataset.cache with the hot key and tag: in_j = prev ^ (sigma || key || seg || j || tag); line_j = block(in_j); prev_0 = 0", "load": "dst = dst ^ hot[mulhi(src, words)] (the high 32 bits of the 64-bit product; the index lies in [0, words) for any size)", "slots_rule": "the first k load slots in the Fisher-Yates draw order after the scratch slots (a uniform k-subset); no extra draw, so with version 2 widths and no scratch the program is the version 2 program with k loads redirected", "acceptance_stand_in": "dataset_elem(mulhi(src, words), seed_words[2], seed_words[3])", "program_id": "the read-width id with 'hot/' || mb || k appended", "spec": "docs/plans/hot-table.md"}, + "op_mix": {"add": 8, "hot": 8, "load": 8, "shfl": 8, "xor": 6, "mad": 5, "mul": 5, "mulhi": 5, "sub": 4, "rotl": 3, "rotr": 3, "or": 1}, + "register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]", + "splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16", + "iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order", + "output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo", + "op_semantics": { + "add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)", + "sub": "dst = dst - src", + "mul": "dst = dst * src (low 32)", + "mulhi": "dst = high 32 bits of dst * src", + "xor": "dst = dst ^ src", + "or": "dst = dst | src", + "rotl": "dst = rotl(dst, rot), rot in 1..31", + "rotr": "dst = rotr(dst, src & 31)", + "mad": "dst = src * src2 + dst", + "shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp", + "load": "dst = dst ^ dataset[src & dataset.mask]", + "wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)", + "hot": "dst = dst ^ hot[mulhi(src, hot_table.words)] (hot-table experiment)" + }, + "dataset": { + "log2_words": 28, + "bytes": 1073741824, + "mask": "0x0fffffff", + "day": "2026-10-03", + "day_bytes": "6461792f323032362d31302d3033", + "day_words_from": "seed_words_from_bytes(day_bytes)", + "d0": "0x3067619f", + "d1": "0x3c269176", + "mode": "memory-hard", + "spec": "proto-metal/MEMHARD.md", + "key": ["0x3067619f", "0x3c269176", "0x84a03b03", "0xf8c63294", "0xff977c5b", "0xe60def3e", "0x63630141", "0xb8fbcb58"], + "key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]", + "cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"}, + "mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [20, 20, 19, 4, 26, 3, 3, 27], "mul": ["0x42146205", "0x52cbe0fb", "0x7ecf4a03", "0x6728907f", "0xd81d9751", "0x132952c3", "0xf60de277", "0x05358035", "0xbaf6499d", "0xe4db9667", "0x3e98f45d", "0xd0004edd", "0x2691630d", "0x9beb3bcf", "0xab310379", "0x99cfb423"], "rc": ["0xbab68293", "0xcc162340", "0x6ce151cc", "0xe62b8997", "0xc9c80297", "0xf74a1654", "0x3d704af5", "0x3cf522b7", "0x2b9cac04", "0xa880ac10", "0x13e5dd1d", "0x6fc3e233", "0x2d83eeac", "0x9006e8bf", "0x2c4b5362", "0x31b49ee2"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"}, + "item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: s = M_r(s); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then s = M_8(s); item(t) = s", + "word": "dataset[w] = item(w >> 4)[w & 15]" + }, + "instructions": [ + {"i": 0, "op": "mad", "dst": 2, "src": 3, "src2": 4, "imm": "0xbf7b174d", "imm2": "0x337b762e", "rot": 17, "bit": 2, "mask": 2, "width": 1}, + {"i": 1, "op": "mad", "dst": 2, "src": 1, "src2": 1, "imm": "0xdd04a5da", "imm2": "0x42da7657", "rot": 15, "bit": 30, "mask": 16, "width": 1}, + {"i": 2, "op": "mad", "dst": 2, "src": 3, "src2": 2, "imm": "0x734003fa", "imm2": "0x5bb67700", "rot": 3, "bit": 20, "mask": 1, "width": 1}, + {"i": 3, "op": "xor", "dst": 3, "src": 5, "src2": 5, "imm": "0xc55a1b1c", "imm2": "0xa19720f3", "rot": 7, "bit": 8, "mask": 1, "width": 1}, + {"i": 4, "op": "hot", "dst": 7, "src": 2, "src2": 5, "imm": "0xad572dd7", "imm2": "0x9ceb3ea7", "rot": 18, "bit": 30, "mask": 2, "width": 1}, + {"i": 5, "op": "load", "dst": 5, "src": 7, "src2": 3, "imm": "0x769a53be", "imm2": "0x80f9067e", "rot": 12, "bit": 22, "mask": 1, "width": 1}, + {"i": 6, "op": "shfl", "dst": 1, "src": 4, "src2": 0, "imm": "0xd3613d88", "imm2": "0x262fb219", "rot": 10, "bit": 30, "mask": 8, "width": 1}, + {"i": 7, "op": "shfl", "dst": 7, "src": 3, "src2": 4, "imm": "0xce38e42f", "imm2": "0xb868b818", "rot": 11, "bit": 8, "mask": 8, "width": 1}, + {"i": 8, "op": "mulhi", "dst": 1, "src": 5, "src2": 2, "imm": "0xa5eebca5", "imm2": "0x5703a72b", "rot": 13, "bit": 13, "mask": 16, "width": 1}, + {"i": 9, "op": "rotr", "dst": 6, "src": 3, "src2": 4, "imm": "0x17a5a9c7", "imm2": "0xdcfb93a1", "rot": 20, "bit": 27, "mask": 2, "width": 1}, + {"i": 10, "op": "or", "dst": 3, "src": 4, "src2": 1, "imm": "0xccb7d785", "imm2": "0xc335364c", "rot": 14, "bit": 12, "mask": 4, "width": 1}, + {"i": 11, "op": "load", "dst": 4, "src": 3, "src2": 2, "imm": "0x88cb9af3", "imm2": "0x4e7dc10d", "rot": 24, "bit": 17, "mask": 4, "width": 1}, + {"i": 12, "op": "mulhi", "dst": 0, "src": 4, "src2": 4, "imm": "0x45374321", "imm2": "0x3cd91989", "rot": 11, "bit": 4, "mask": 2, "width": 1}, + {"i": 13, "op": "add", "dst": 5, "src": 1, "src2": 2, "imm": "0xc7934706", "imm2": "0xd3177981", "rot": 16, "bit": 30, "mask": 2, "width": 1}, + {"i": 14, "op": "load", "dst": 0, "src": 4, "src2": 3, "imm": "0xfd7f56bb", "imm2": "0x65e14f52", "rot": 13, "bit": 22, "mask": 2, "width": 1}, + {"i": 15, "op": "sub", "dst": 2, "src": 4, "src2": 4, "imm": "0x35a80b49", "imm2": "0x060f2d13", "rot": 16, "bit": 20, "mask": 16, "width": 1}, + {"i": 16, "op": "hot", "dst": 2, "src": 0, "src2": 2, "imm": "0xae0a32c2", "imm2": "0x4c2a4cfe", "rot": 8, "bit": 31, "mask": 16, "width": 1}, + {"i": 17, "op": "load", "dst": 7, "src": 2, "src2": 6, "imm": "0x82a84cc3", "imm2": "0x21a38d68", "rot": 15, "bit": 21, "mask": 2, "width": 1}, + {"i": 18, "op": "shfl", "dst": 7, "src": 3, "src2": 3, "imm": "0xa3818806", "imm2": "0x8f66b5c8", "rot": 14, "bit": 6, "mask": 4, "width": 1}, + {"i": 19, "op": "mul", "dst": 5, "src": 0, "src2": 1, "imm": "0xa00de107", "imm2": "0x77bfcaa5", "rot": 3, "bit": 10, "mask": 2, "width": 1}, + {"i": 20, "op": "shfl", "dst": 3, "src": 4, "src2": 7, "imm": "0x1d2b8cab", "imm2": "0x80b4f9a2", "rot": 14, "bit": 25, "mask": 2, "width": 1}, + {"i": 21, "op": "shfl", "dst": 2, "src": 4, "src2": 5, "imm": "0x3ac915d2", "imm2": "0x5fba7bc2", "rot": 16, "bit": 1, "mask": 16, "width": 1}, + {"i": 22, "op": "mulhi", "dst": 6, "src": 2, "src2": 0, "imm": "0xdc3ec8fd", "imm2": "0x599e2fa3", "rot": 22, "bit": 3, "mask": 2, "width": 1}, + {"i": 23, "op": "hot", "dst": 6, "src": 1, "src2": 5, "imm": "0x2a6b16d5", "imm2": "0xd73e396f", "rot": 28, "bit": 29, "mask": 2, "width": 1}, + {"i": 24, "op": "mul", "dst": 5, "src": 0, "src2": 7, "imm": "0x376d0223", "imm2": "0xe1c2169a", "rot": 4, "bit": 16, "mask": 16, "width": 1}, + {"i": 25, "op": "rotl", "dst": 5, "src": 7, "src2": 3, "imm": "0x78ad8c60", "imm2": "0x6f5b77d5", "rot": 19, "bit": 11, "mask": 16, "width": 1}, + {"i": 26, "op": "shfl", "dst": 7, "src": 6, "src2": 5, "imm": "0x93915b9f", "imm2": "0x1e61fb6b", "rot": 28, "bit": 23, "mask": 2, "width": 1}, + {"i": 27, "op": "xor", "dst": 0, "src": 5, "src2": 7, "imm": "0x6378fe15", "imm2": "0x66c78f42", "rot": 12, "bit": 31, "mask": 8, "width": 1}, + {"i": 28, "op": "xor", "dst": 0, "src": 4, "src2": 7, "imm": "0x20a57fda", "imm2": "0x088c848e", "rot": 16, "bit": 13, "mask": 4, "width": 1}, + {"i": 29, "op": "sub", "dst": 3, "src": 0, "src2": 2, "imm": "0x49d95fd5", "imm2": "0x1a5f946a", "rot": 6, "bit": 12, "mask": 1, "width": 1}, + {"i": 30, "op": "mul", "dst": 5, "src": 1, "src2": 7, "imm": "0x0a816217", "imm2": "0x405c4f73", "rot": 13, "bit": 27, "mask": 4, "width": 1}, + {"i": 31, "op": "load", "dst": 7, "src": 2, "src2": 2, "imm": "0x09ed045e", "imm2": "0xd69c4715", "rot": 5, "bit": 9, "mask": 2, "width": 1}, + {"i": 32, "op": "hot", "dst": 1, "src": 0, "src2": 6, "imm": "0xeb79ea49", "imm2": "0xcc587f5a", "rot": 6, "bit": 8, "mask": 16, "width": 1}, + {"i": 33, "op": "xor", "dst": 5, "src": 6, "src2": 1, "imm": "0x3027401e", "imm2": "0x5f20c27e", "rot": 18, "bit": 9, "mask": 2, "width": 1}, + {"i": 34, "op": "hot", "dst": 5, "src": 1, "src2": 3, "imm": "0x0e1cab07", "imm2": "0x09356c5b", "rot": 19, "bit": 31, "mask": 1, "width": 1}, + {"i": 35, "op": "mulhi", "dst": 0, "src": 5, "src2": 2, "imm": "0x90e31357", "imm2": "0xabd32484", "rot": 26, "bit": 5, "mask": 8, "width": 1}, + {"i": 36, "op": "shfl", "dst": 5, "src": 2, "src2": 5, "imm": "0xee9a955f", "imm2": "0x31b3faed", "rot": 8, "bit": 24, "mask": 4, "width": 1}, + {"i": 37, "op": "hot", "dst": 7, "src": 0, "src2": 7, "imm": "0x3ba2f832", "imm2": "0x1160dcd3", "rot": 4, "bit": 29, "mask": 1, "width": 1}, + {"i": 38, "op": "add", "dst": 3, "src": 1, "src2": 7, "imm": "0x75ba2fad", "imm2": "0x230c005c", "rot": 4, "bit": 27, "mask": 1, "width": 1}, + {"i": 39, "op": "shfl", "dst": 1, "src": 5, "src2": 3, "imm": "0xdbf37e75", "imm2": "0xb5ac1969", "rot": 30, "bit": 13, "mask": 4, "width": 1}, + {"i": 40, "op": "xor", "dst": 2, "src": 5, "src2": 5, "imm": "0x47f136c5", "imm2": "0x06ce9153", "rot": 19, "bit": 10, "mask": 2, "width": 1}, + {"i": 41, "op": "mad", "dst": 3, "src": 6, "src2": 3, "imm": "0xce13eff8", "imm2": "0x04cc1d55", "rot": 3, "bit": 1, "mask": 4, "width": 1}, + {"i": 42, "op": "sub", "dst": 6, "src": 7, "src2": 1, "imm": "0x6a65ab71", "imm2": "0x8fbc1bcd", "rot": 4, "bit": 1, "mask": 8, "width": 1}, + {"i": 43, "op": "xor", "dst": 7, "src": 0, "src2": 7, "imm": "0xdaeb4928", "imm2": "0xc0423027", "rot": 24, "bit": 11, "mask": 8, "width": 1}, + {"i": 44, "op": "hot", "dst": 1, "src": 7, "src2": 7, "imm": "0x778f01c9", "imm2": "0x28cedcea", "rot": 12, "bit": 4, "mask": 16, "width": 1}, + {"i": 45, "op": "mul", "dst": 2, "src": 3, "src2": 2, "imm": "0xf4264f1b", "imm2": "0x0f627d56", "rot": 5, "bit": 28, "mask": 8, "width": 1}, + {"i": 46, "op": "mulhi", "dst": 1, "src": 5, "src2": 1, "imm": "0xffb2147a", "imm2": "0xccde9b05", "rot": 13, "bit": 9, "mask": 2, "width": 1}, + {"i": 47, "op": "sub", "dst": 4, "src": 3, "src2": 5, "imm": "0x73b36234", "imm2": "0x3f5d5997", "rot": 7, "bit": 18, "mask": 2, "width": 1}, + {"i": 48, "op": "rotr", "dst": 2, "src": 6, "src2": 3, "imm": "0x3a4d9aa9", "imm2": "0x212bec7b", "rot": 4, "bit": 29, "mask": 16, "width": 1}, + {"i": 49, "op": "load", "dst": 3, "src": 5, "src2": 2, "imm": "0x626f11df", "imm2": "0x56cd5bfd", "rot": 7, "bit": 1, "mask": 1, "width": 1}, + {"i": 50, "op": "add", "dst": 1, "src": 5, "src2": 6, "imm": "0x81b8bc2c", "imm2": "0x1907970c", "rot": 28, "bit": 7, "mask": 4, "width": 1}, + {"i": 51, "op": "mul", "dst": 0, "src": 2, "src2": 2, "imm": "0xa8848b30", "imm2": "0xef6ac348", "rot": 9, "bit": 15, "mask": 8, "width": 1}, + {"i": 52, "op": "add", "dst": 0, "src": 2, "src2": 0, "imm": "0x4f92b968", "imm2": "0x699fd448", "rot": 22, "bit": 6, "mask": 4, "width": 1}, + {"i": 53, "op": "add", "dst": 1, "src": 0, "src2": 2, "imm": "0x2bb965af", "imm2": "0x77b1520d", "rot": 2, "bit": 12, "mask": 8, "width": 1}, + {"i": 54, "op": "rotl", "dst": 7, "src": 1, "src2": 0, "imm": "0x553e678b", "imm2": "0x3cc8eae0", "rot": 14, "bit": 20, "mask": 2, "width": 1}, + {"i": 55, "op": "add", "dst": 3, "src": 7, "src2": 2, "imm": "0x7b0fe07a", "imm2": "0xa54c55a0", "rot": 10, "bit": 1, "mask": 1, "width": 1}, + {"i": 56, "op": "load", "dst": 6, "src": 7, "src2": 4, "imm": "0x01eba9aa", "imm2": "0x2758c0f7", "rot": 14, "bit": 15, "mask": 4, "width": 1}, + {"i": 57, "op": "rotr", "dst": 1, "src": 5, "src2": 2, "imm": "0x1f5267b3", "imm2": "0x236f5a27", "rot": 2, "bit": 31, "mask": 16, "width": 1}, + {"i": 58, "op": "load", "dst": 5, "src": 4, "src2": 3, "imm": "0xa9954a9b", "imm2": "0x6a54d4e8", "rot": 11, "bit": 10, "mask": 16, "width": 1}, + {"i": 59, "op": "hot", "dst": 6, "src": 2, "src2": 4, "imm": "0x9923ff88", "imm2": "0x9357254e", "rot": 16, "bit": 1, "mask": 16, "width": 1}, + {"i": 60, "op": "mad", "dst": 3, "src": 5, "src2": 0, "imm": "0xc0cc51a6", "imm2": "0x3fd7701b", "rot": 20, "bit": 1, "mask": 4, "width": 1}, + {"i": 61, "op": "add", "dst": 5, "src": 7, "src2": 7, "imm": "0xaf9dd72d", "imm2": "0xad7493e7", "rot": 7, "bit": 31, "mask": 16, "width": 1}, + {"i": 62, "op": "add", "dst": 4, "src": 6, "src2": 2, "imm": "0x89841d87", "imm2": "0x1e07c3d9", "rot": 6, "bit": 27, "mask": 1, "width": 1}, + {"i": 63, "op": "rotl", "dst": 5, "src": 4, "src2": 2, "imm": "0xae210f8d", "imm2": "0x8e499ba4", "rot": 19, "bit": 9, "mask": 1, "width": 1} + ] +} diff --git a/proto-cuda/packs-ca2-hot/hot64k8/program.metal b/proto-cuda/packs-ca2-hot/hot64k8/program.metal new file mode 100644 index 000000000..38e0298f8 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k8/program.metal @@ -0,0 +1,112 @@ +#include +using namespace metal; + +#define MASK 0x0fffffffu +// Hot table (64 MiB, 8 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }; + +inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 +inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +kernel void igneum_hash(device const uint* dataset [[buffer(0)]], + device ulong* out [[buffer(1)]], + constant uint& baseNonce [[buffer(2)]], + device const uint* hot [[buffer(3)]], + uint gid [[thread_position_in_grid]]) { + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + { uint x = nonce ^ SEEDW[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ SEEDW[1]; } + { uint x = nonce ^ SEEDW[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ SEEDW[2]; } + { uint x = nonce ^ SEEDW[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ SEEDW[3]; } + { uint x = nonce ^ SEEDW[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ SEEDW[4]; } + { uint x = nonce ^ SEEDW[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ SEEDW[5]; } + { uint x = nonce ^ SEEDW[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ SEEDW[6]; } + { uint x = nonce ^ SEEDW[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ SEEDW[7]; } + { uint x = nonce ^ SEEDW[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ SEEDW[0]; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 + r2 = r1 * r1 + r2; // 1 + r2 = r3 * r2 + r2; // 2 + r3 = r3 ^ r5; // 3 + r7 = r7 ^ hot[mulhi(r2, HOT_WORDS)]; // 4 + r5 = r5 ^ dataset[r7 & MASK]; // 5 + r1 = r1 ^ simd_shuffle_xor(r4, (ushort)8); // 6 + r7 = r7 ^ simd_shuffle_xor(r3, (ushort)8); // 7 + r1 = mulhi(r1, r5); // 8 + r6 = rotr_var(r6, r3); // 9 + r3 = r3 | r4; // 10 + r4 = r4 ^ dataset[r3 & MASK]; // 11 + r0 = mulhi(r0, r4); // 12 + r5 = r5 + r1 + select(0xc7934706u, 0xd3177981u, ((sel >> 30u) & 1u) != 0u); // 13 + r0 = r0 ^ dataset[r4 & MASK]; // 14 + r2 = r2 - r4; // 15 + r2 = r2 ^ hot[mulhi(r0, HOT_WORDS)]; // 16 + r7 = r7 ^ dataset[r2 & MASK]; // 17 + r7 = r7 ^ simd_shuffle_xor(r3, (ushort)4); // 18 + r5 = r5 * r0; // 19 + r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 20 + r2 = r2 ^ simd_shuffle_xor(r4, (ushort)16); // 21 + r6 = mulhi(r6, r2); // 22 + r6 = r6 ^ hot[mulhi(r1, HOT_WORDS)]; // 23 + r5 = r5 * r0; // 24 + r5 = rotl_imm(r5, 19u); // 25 + r7 = r7 ^ simd_shuffle_xor(r6, (ushort)2); // 26 + r0 = r0 ^ r5; // 27 + r0 = r0 ^ r4; // 28 + r3 = r3 - r0; // 29 + r5 = r5 * r1; // 30 + r7 = r7 ^ dataset[r2 & MASK]; // 31 + r1 = r1 ^ hot[mulhi(r0, HOT_WORDS)]; // 32 + r5 = r5 ^ r6; // 33 + r5 = r5 ^ hot[mulhi(r1, HOT_WORDS)]; // 34 + r0 = mulhi(r0, r5); // 35 + r5 = r5 ^ simd_shuffle_xor(r2, (ushort)4); // 36 + r7 = r7 ^ hot[mulhi(r0, HOT_WORDS)]; // 37 + r3 = r3 + r1 + select(0x75ba2fadu, 0x230c005cu, ((sel >> 27u) & 1u) != 0u); // 38 + r1 = r1 ^ simd_shuffle_xor(r5, (ushort)4); // 39 + r2 = r2 ^ r5; // 40 + r3 = r6 * r3 + r3; // 41 + r6 = r6 - r7; // 42 + r7 = r7 ^ r0; // 43 + r1 = r1 ^ hot[mulhi(r7, HOT_WORDS)]; // 44 + r2 = r2 * r3; // 45 + r1 = mulhi(r1, r5); // 46 + r4 = r4 - r3; // 47 + r2 = rotr_var(r2, r6); // 48 + r3 = r3 ^ dataset[r5 & MASK]; // 49 + r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 50 + r0 = r0 * r2; // 51 + r0 = r0 + r2 + select(0x4f92b968u, 0x699fd448u, ((sel >> 6u) & 1u) != 0u); // 52 + r1 = r1 + r0 + select(0x2bb965afu, 0x77b1520du, ((sel >> 12u) & 1u) != 0u); // 53 + r7 = rotl_imm(r7, 14u); // 54 + r3 = r3 + r7 + select(0x7b0fe07au, 0xa54c55a0u, ((sel >> 1u) & 1u) != 0u); // 55 + r6 = r6 ^ dataset[r7 & MASK]; // 56 + r1 = rotr_var(r1, r5); // 57 + r5 = r5 ^ dataset[r4 & MASK]; // 58 + r6 = r6 ^ hot[mulhi(r2, HOT_WORDS)]; // 59 + r3 = r5 * r0 + r3; // 60 + r5 = r5 + r7 + select(0xaf9dd72du, 0xad7493e7u, ((sel >> 31u) & 1u) != 0u); // 61 + r4 = r4 + r6 + select(0x89841d87u, 0x1e07c3d9u, ((sel >> 27u) & 1u) != 0u); // 62 + r5 = rotl_imm(r5, 19u); // 63 + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot64k8/program_bound.metal b/proto-cuda/packs-ca2-hot/hot64k8/program_bound.metal new file mode 100644 index 000000000..cfb7e40fa --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k8/program_bound.metal @@ -0,0 +1,114 @@ +#include +using namespace metal; + +#define MASK 0x0fffffffu +// Hot table (64 MiB, 8 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01000000u +constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }; + +inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 +inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW. +kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], + device ulong* out [[buffer(1)]], + constant uint& baseNonce [[buffer(2)]], + constant uint* initw [[buffer(3)]], + device const uint* hot [[buffer(4)]], + uint gid [[thread_position_in_grid]]) { + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + { uint x = nonce ^ initw[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ initw[1]; } + { uint x = nonce ^ initw[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ initw[2]; } + { uint x = nonce ^ initw[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ initw[3]; } + { uint x = nonce ^ initw[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ initw[4]; } + { uint x = nonce ^ initw[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ initw[5]; } + { uint x = nonce ^ initw[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ initw[6]; } + { uint x = nonce ^ initw[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ initw[7]; } + { uint x = nonce ^ initw[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ initw[0]; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 + r2 = r1 * r1 + r2; // 1 + r2 = r3 * r2 + r2; // 2 + r3 = r3 ^ r5; // 3 + r7 = r7 ^ hot[mulhi(r2, HOT_WORDS)]; // 4 + r5 = r5 ^ dataset[r7 & MASK]; // 5 + r1 = r1 ^ simd_shuffle_xor(r4, (ushort)8); // 6 + r7 = r7 ^ simd_shuffle_xor(r3, (ushort)8); // 7 + r1 = mulhi(r1, r5); // 8 + r6 = rotr_var(r6, r3); // 9 + r3 = r3 | r4; // 10 + r4 = r4 ^ dataset[r3 & MASK]; // 11 + r0 = mulhi(r0, r4); // 12 + r5 = r5 + r1 + select(0xc7934706u, 0xd3177981u, ((sel >> 30u) & 1u) != 0u); // 13 + r0 = r0 ^ dataset[r4 & MASK]; // 14 + r2 = r2 - r4; // 15 + r2 = r2 ^ hot[mulhi(r0, HOT_WORDS)]; // 16 + r7 = r7 ^ dataset[r2 & MASK]; // 17 + r7 = r7 ^ simd_shuffle_xor(r3, (ushort)4); // 18 + r5 = r5 * r0; // 19 + r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 20 + r2 = r2 ^ simd_shuffle_xor(r4, (ushort)16); // 21 + r6 = mulhi(r6, r2); // 22 + r6 = r6 ^ hot[mulhi(r1, HOT_WORDS)]; // 23 + r5 = r5 * r0; // 24 + r5 = rotl_imm(r5, 19u); // 25 + r7 = r7 ^ simd_shuffle_xor(r6, (ushort)2); // 26 + r0 = r0 ^ r5; // 27 + r0 = r0 ^ r4; // 28 + r3 = r3 - r0; // 29 + r5 = r5 * r1; // 30 + r7 = r7 ^ dataset[r2 & MASK]; // 31 + r1 = r1 ^ hot[mulhi(r0, HOT_WORDS)]; // 32 + r5 = r5 ^ r6; // 33 + r5 = r5 ^ hot[mulhi(r1, HOT_WORDS)]; // 34 + r0 = mulhi(r0, r5); // 35 + r5 = r5 ^ simd_shuffle_xor(r2, (ushort)4); // 36 + r7 = r7 ^ hot[mulhi(r0, HOT_WORDS)]; // 37 + r3 = r3 + r1 + select(0x75ba2fadu, 0x230c005cu, ((sel >> 27u) & 1u) != 0u); // 38 + r1 = r1 ^ simd_shuffle_xor(r5, (ushort)4); // 39 + r2 = r2 ^ r5; // 40 + r3 = r6 * r3 + r3; // 41 + r6 = r6 - r7; // 42 + r7 = r7 ^ r0; // 43 + r1 = r1 ^ hot[mulhi(r7, HOT_WORDS)]; // 44 + r2 = r2 * r3; // 45 + r1 = mulhi(r1, r5); // 46 + r4 = r4 - r3; // 47 + r2 = rotr_var(r2, r6); // 48 + r3 = r3 ^ dataset[r5 & MASK]; // 49 + r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 50 + r0 = r0 * r2; // 51 + r0 = r0 + r2 + select(0x4f92b968u, 0x699fd448u, ((sel >> 6u) & 1u) != 0u); // 52 + r1 = r1 + r0 + select(0x2bb965afu, 0x77b1520du, ((sel >> 12u) & 1u) != 0u); // 53 + r7 = rotl_imm(r7, 14u); // 54 + r3 = r3 + r7 + select(0x7b0fe07au, 0xa54c55a0u, ((sel >> 1u) & 1u) != 0u); // 55 + r6 = r6 ^ dataset[r7 & MASK]; // 56 + r1 = rotr_var(r1, r5); // 57 + r5 = r5 ^ dataset[r4 & MASK]; // 58 + r6 = r6 ^ hot[mulhi(r2, HOT_WORDS)]; // 59 + r3 = r5 * r0 + r3; // 60 + r5 = r5 + r7 + select(0xaf9dd72du, 0xad7493e7u, ((sel >> 31u) & 1u) != 0u); // 61 + r4 = r4 + r6 + select(0x89841d87u, 0x1e07c3d9u, ((sel >> 27u) & 1u) != 0u); // 62 + r5 = rotl_imm(r5, 19u); // 63 + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot64k8/vectors.h b/proto-cuda/packs-ca2-hot/hot64k8/vectors.h new file mode 100644 index 000000000..83b42f862 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k8/vectors.h @@ -0,0 +1,67 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Expected outputs: igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif + +#define IGNEUM_VEC_WARPS 3 +static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u }; +static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = { + { // base nonce 0 + 0xe2e3466ed893b4f8ull, 0x8a8a8c07512484a5ull, 0x2eab76ad896647ebull, 0x308c5fb089674ebdull, 0x6851d37a31b0da79ull, 0xdf80d0cd9d2bbd54ull, 0x0bb440d7e0a0c83bull, 0xde6fb92cd5441665ull, + 0xd2a4449d68eb862cull, 0xf863dc5645b0081bull, 0xb08cdd14d2362b87ull, 0x27be058ae53b67c9ull, 0x898f910ebf79bf14ull, 0x25de3eb6b1834550ull, 0xd5e9f52d4a43dc4bull, 0x17a4229a36fbfe81ull, + 0xb16266ab869066e8ull, 0x036640126ec2de22ull, 0xc32cba038e36202full, 0x0b35fe0444f234d7ull, 0x907e8a28e7216ec1ull, 0x013b9d3e595a08d8ull, 0xf113d862e5d87a49ull, 0xee1f6d8873e2fd6cull, + 0xbc8eeb89f8c2468cull, 0xc893fa696ddb1348ull, 0xf775fe419ea65ea0ull, 0x59daf16ad92be258ull, 0xafc9d2775158e383ull, 0x7bca50ec84d8f4e3ull, 0xb60e23cc4c57ec8bull, 0x14ea22b4a5bf58cfull + }, + { // base nonce 4096 + 0x7493ca83442f0349ull, 0x14d80b5c689132dfull, 0xb3aae72dd710e516ull, 0xf0206e95d8082ec9ull, 0xf7edf2b8c33ed76eull, 0xaa99a579cd72c064ull, 0x13982b590565347full, 0x91ca2fab3e30f7d9ull, + 0x1a23a5a146439384ull, 0x6d7e9a1095396238ull, 0x35f22e5244fd6eeaull, 0x35b7e646dc856457ull, 0x12cb9a082f761df4ull, 0x2dab46f7cee1d833ull, 0xb6b3a052f5f46a2full, 0x1bb92a53c6282e16ull, + 0xc464cdf1a18e398cull, 0xe79e62c2cd81d3ebull, 0x7d269c15d28e6e34ull, 0x635d32a0c93e5e73ull, 0x043280c159109d85ull, 0x34e079fb8358db7eull, 0xa8eaa16f25e34d81ull, 0x862a845e3d6eaea4ull, + 0x173c275d8a9b36eeull, 0x3d4b4470f1a1e240ull, 0xabd1ee2cdc472377ull, 0x1a6bed28465dcc1eull, 0xa30a7ce538dc5919ull, 0x359785e23a8baef5ull, 0x639a2cab894c1953ull, 0xdce43d7b8b0ddfaeull + }, + { // base nonce 1000000 + 0x893e52b4424a3fc3ull, 0x631562f1f7a3a700ull, 0x50b555a482edb887ull, 0x3fbe532fcf33c9b7ull, 0x897f1231b0b11e8dull, 0xba406a33052b8d0eull, 0x1c31e6025bebaf00ull, 0x1b5acf7c6b3a2a41ull, + 0xdc69bc95894f455eull, 0x0f5b4794879e53dfull, 0x05238a17040b9f19ull, 0x0b9ed9ee63d82972ull, 0xa379e13eee8b0eb7ull, 0x0a75f0260ea17b27ull, 0x964f7c4d2cfc88b9ull, 0x8c65f784a47824ccull, + 0x55e521a2a346bd68ull, 0x566d04a22d56fa8full, 0x7595b733b70030aaull, 0xc7e6d1d5af426108ull, 0x9f75b8bc9d27aa23ull, 0x9363b556eedd83bcull, 0x305bb3c5e763bd12ull, 0xf998dc80a32b23f3ull, + 0xc9eab03b3fbb0ca4ull, 0x36c7156c238fcce9ull, 0x576b9d41066425bdull, 0x11d6c4e5cb55f9c9ull, 0x35ca8337b91469acull, 0x243d6ee5c95910b5ull, 0x0fce828d660c4426ull, 0xbf1d2b2a057e58efull + } +}; + +// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (268435455). +static const uint32_t IGNEUM_DS_HEAD[16] = { + 0xffc3cd94u, 0x5920ccd8u, 0x392f44bbu, 0x5e57f67au, 0x2f2bc2a9u, 0x620b0e36u, 0xbdc09014u, 0x436654bfu, + 0x311e0b48u, 0x1abd93adu, 0x59cc7ce8u, 0xee5247b2u, 0x86171fe8u, 0x6d874751u, 0xc9f7728fu, 0x7c2a435du +}; +static const uint32_t IGNEUM_DS_LAST_INDEX = 268435455u; +static const uint32_t IGNEUM_DS_LAST = 0xa33ada72u; +// 64 sampled dataset words (index, value) computed on the Mac. +#define IGNEUM_DS_SAMPLES 64 +static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = { + 59471966u, 217795994u, 208353206u, 42483309u, 172547758u, 148076330u, 183853158u, 214389424u, 267488061u, 169781097u, 184093494u, 153880993u, 84977930u, 46426879u, 3093825u, 225364072u, 44593546u, 260713159u, 168250303u, 52384140u, 223401610u, 45554030u, 95410555u, 175039924u, 79171087u, 267580473u, 24168642u, 37981670u, 171551130u, 195559979u, 204611762u, 140997658u, 138925853u, 86637313u, 20736778u, 219665210u, 160430336u, 264654675u, 8013395u, 228945585u, 213884386u, 104419827u, 44185464u, 142737231u, 99284897u, 132475900u, 61861762u, 132056166u, 262388043u, 91878046u, 117353561u, 124768597u, 71352993u, 190698941u, 46055428u, 55281366u, 165145231u, 106810753u, 171985651u, 232085256u, 159510492u, 40072060u, 209107596u, 39023794u +}; +static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = { + 0xe8b73d94u, 0x337028b5u, 0xafe148c9u, 0xab99f7aeu, 0x434ea619u, 0xd85cb880u, 0x54764c7fu, 0x82c7e420u, 0xedf4cb9eu, 0x9884c959u, 0x223ee793u, 0x3a9ccf69u, 0x81da4fd2u, 0xd6ce8cb9u, 0xe3922dcau, 0x3e7e6bdeu, 0x382a3acau, 0x567e7f7fu, 0x25a0f084u, 0xbfeef128u, 0xe338abfbu, 0x7c3b5280u, 0x909bc5f1u, 0xd8b74b9cu, 0x8e31a22eu, 0x26b5f1d8u, 0x79122c00u, 0xcafc3340u, 0xd5e02ea3u, 0x1aee1afdu, 0xdb090d9au, 0xb049f435u, 0x4954d8bau, 0x03797ba0u, 0x196eefbdu, 0xd153412au, 0xbe5d2c4bu, 0xdaa14f0eu, 0x8e61ed07u, 0x9e9a64c6u, 0x2e29ff36u, 0x392a8589u, 0xb56a5912u, 0xfa6e8b57u, 0xd1a737cbu, 0xb0fa841au, 0xbe1c341fu, 0xe25be0f1u, 0xe937f543u, 0xebab2248u, 0x8e1b607au, 0x202a2fedu, 0x95e2819cu, 0x9c9652d4u, 0x32fedef0u, 0xdecfff82u, 0xcb5d43e5u, 0xb735806au, 0x8905939cu, 0xfbf8472du, 0xada74e5du, 0x7ebdeeeau, 0x0119f2b3u, 0xa9a376b8u +}; +// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^26 words. +static const uint32_t IGNEUM_CACHE_HEAD[16] = { + 0x355a86d2u, 0x7957db1cu, 0xd21772afu, 0x6fc1e09bu, 0xd55ce61du, 0x6e6a278bu, 0xd3f543ceu, 0x223d8e82u, + 0x143ab337u, 0x2e9f05bdu, 0x2eb389bfu, 0x0c6e449eu, 0x5cfa4222u, 0xba6560feu, 0x8e3e1aa4u, 0xdbcc1d53u +}; +static const uint32_t IGNEUM_CACHE_LAST[16] = { + 0x41190d91u, 0xbd277957u, 0x22ddbb49u, 0x6986f207u, 0xdf69a4d6u, 0x26401a3au, 0x818230fbu, 0xc417122du, + 0x3597b211u, 0xb553ce55u, 0xcf39cc0du, 0x3b7fc43au, 0x3fd43b00u, 0x67e1c80eu, 0xffa7ea7du, 0xca2960abu +}; +static const uint64_t IGNEUM_CACHE_FNV64 = 0x48c4f5bf24166b2eull; +// Hot table self-test (docs/plans/hot-table.md): hot[0..15], the last 16 words, and FNV-1a 64 over all IGNEUM_HOT_WORDS words. +static const uint32_t IGNEUM_HOT_HEAD[16] = { + 0x8068cc73u, 0x6036ebf9u, 0xb604cd25u, 0x8ffb840eu, 0xc54074a2u, 0x285c0695u, 0x77512425u, 0xc26a58a7u, + 0x72c88757u, 0xc10fca78u, 0x513825ddu, 0x30d6ccc8u, 0x9a05e7cfu, 0xb9533f50u, 0x4bac3ba0u, 0xa5c19528u +}; +static const uint32_t IGNEUM_HOT_LAST[16] = { + 0x7c6d7cebu, 0xe24fcd46u, 0xe5cce976u, 0x3ecffe59u, 0x94e98b66u, 0x3f6ef45cu, 0xa3715aa1u, 0xdbe35281u, + 0xba05d27bu, 0x00541963u, 0xe636f453u, 0xd3972366u, 0x529599b8u, 0x79ae3c8cu, 0x48436863u, 0x897a8bf3u +}; +static const uint64_t IGNEUM_HOT_FNV64 = 0x77ca4b9527104530ull; diff --git a/proto-cuda/packs-ca2-hot/hot64k8/vectors.json b/proto-cuda/packs-ca2-hot/hot64k8/vectors.json new file mode 100644 index 000000000..2ac1d60d2 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot64k8/vectors.json @@ -0,0 +1,39 @@ +{ + "seed": "igneum-genesis", + "day": "2026-10-03", + "dataset_mode": "memory-hard", + "dataset_log2_words": 28, + "mask": "0x0fffffff", + "lanes": 32, + "source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset", + "warps": [ + {"base_nonce": 0, "expected": [ + "0xe2e3466ed893b4f8", "0x8a8a8c07512484a5", "0x2eab76ad896647eb", "0x308c5fb089674ebd", "0x6851d37a31b0da79", "0xdf80d0cd9d2bbd54", "0x0bb440d7e0a0c83b", "0xde6fb92cd5441665", + "0xd2a4449d68eb862c", "0xf863dc5645b0081b", "0xb08cdd14d2362b87", "0x27be058ae53b67c9", "0x898f910ebf79bf14", "0x25de3eb6b1834550", "0xd5e9f52d4a43dc4b", "0x17a4229a36fbfe81", + "0xb16266ab869066e8", "0x036640126ec2de22", "0xc32cba038e36202f", "0x0b35fe0444f234d7", "0x907e8a28e7216ec1", "0x013b9d3e595a08d8", "0xf113d862e5d87a49", "0xee1f6d8873e2fd6c", + "0xbc8eeb89f8c2468c", "0xc893fa696ddb1348", "0xf775fe419ea65ea0", "0x59daf16ad92be258", "0xafc9d2775158e383", "0x7bca50ec84d8f4e3", "0xb60e23cc4c57ec8b", "0x14ea22b4a5bf58cf" + ]}, + {"base_nonce": 4096, "expected": [ + "0x7493ca83442f0349", "0x14d80b5c689132df", "0xb3aae72dd710e516", "0xf0206e95d8082ec9", "0xf7edf2b8c33ed76e", "0xaa99a579cd72c064", "0x13982b590565347f", "0x91ca2fab3e30f7d9", + "0x1a23a5a146439384", "0x6d7e9a1095396238", "0x35f22e5244fd6eea", "0x35b7e646dc856457", "0x12cb9a082f761df4", "0x2dab46f7cee1d833", "0xb6b3a052f5f46a2f", "0x1bb92a53c6282e16", + "0xc464cdf1a18e398c", "0xe79e62c2cd81d3eb", "0x7d269c15d28e6e34", "0x635d32a0c93e5e73", "0x043280c159109d85", "0x34e079fb8358db7e", "0xa8eaa16f25e34d81", "0x862a845e3d6eaea4", + "0x173c275d8a9b36ee", "0x3d4b4470f1a1e240", "0xabd1ee2cdc472377", "0x1a6bed28465dcc1e", "0xa30a7ce538dc5919", "0x359785e23a8baef5", "0x639a2cab894c1953", "0xdce43d7b8b0ddfae" + ]}, + {"base_nonce": 1000000, "expected": [ + "0x893e52b4424a3fc3", "0x631562f1f7a3a700", "0x50b555a482edb887", "0x3fbe532fcf33c9b7", "0x897f1231b0b11e8d", "0xba406a33052b8d0e", "0x1c31e6025bebaf00", "0x1b5acf7c6b3a2a41", + "0xdc69bc95894f455e", "0x0f5b4794879e53df", "0x05238a17040b9f19", "0x0b9ed9ee63d82972", "0xa379e13eee8b0eb7", "0x0a75f0260ea17b27", "0x964f7c4d2cfc88b9", "0x8c65f784a47824cc", + "0x55e521a2a346bd68", "0x566d04a22d56fa8f", "0x7595b733b70030aa", "0xc7e6d1d5af426108", "0x9f75b8bc9d27aa23", "0x9363b556eedd83bc", "0x305bb3c5e763bd12", "0xf998dc80a32b23f3", + "0xc9eab03b3fbb0ca4", "0x36c7156c238fcce9", "0x576b9d41066425bd", "0x11d6c4e5cb55f9c9", "0x35ca8337b91469ac", "0x243d6ee5c95910b5", "0x0fce828d660c4426", "0xbf1d2b2a057e58ef" + ]} + ], + "dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"], + "dataset_last_index": 268435455, + "dataset_last": "0xa33ada72", + "dataset_samples": [{"index": 59471966, "value": "0xe8b73d94"}, {"index": 217795994, "value": "0x337028b5"}, {"index": 208353206, "value": "0xafe148c9"}, {"index": 42483309, "value": "0xab99f7ae"}, {"index": 172547758, "value": "0x434ea619"}, {"index": 148076330, "value": "0xd85cb880"}, {"index": 183853158, "value": "0x54764c7f"}, {"index": 214389424, "value": "0x82c7e420"}, {"index": 267488061, "value": "0xedf4cb9e"}, {"index": 169781097, "value": "0x9884c959"}, {"index": 184093494, "value": "0x223ee793"}, {"index": 153880993, "value": "0x3a9ccf69"}, {"index": 84977930, "value": "0x81da4fd2"}, {"index": 46426879, "value": "0xd6ce8cb9"}, {"index": 3093825, "value": "0xe3922dca"}, {"index": 225364072, "value": "0x3e7e6bde"}, {"index": 44593546, "value": "0x382a3aca"}, {"index": 260713159, "value": "0x567e7f7f"}, {"index": 168250303, "value": "0x25a0f084"}, {"index": 52384140, "value": "0xbfeef128"}, {"index": 223401610, "value": "0xe338abfb"}, {"index": 45554030, "value": "0x7c3b5280"}, {"index": 95410555, "value": "0x909bc5f1"}, {"index": 175039924, "value": "0xd8b74b9c"}, {"index": 79171087, "value": "0x8e31a22e"}, {"index": 267580473, "value": "0x26b5f1d8"}, {"index": 24168642, "value": "0x79122c00"}, {"index": 37981670, "value": "0xcafc3340"}, {"index": 171551130, "value": "0xd5e02ea3"}, {"index": 195559979, "value": "0x1aee1afd"}, {"index": 204611762, "value": "0xdb090d9a"}, {"index": 140997658, "value": "0xb049f435"}, {"index": 138925853, "value": "0x4954d8ba"}, {"index": 86637313, "value": "0x03797ba0"}, {"index": 20736778, "value": "0x196eefbd"}, {"index": 219665210, "value": "0xd153412a"}, {"index": 160430336, "value": "0xbe5d2c4b"}, {"index": 264654675, "value": "0xdaa14f0e"}, {"index": 8013395, "value": "0x8e61ed07"}, {"index": 228945585, "value": "0x9e9a64c6"}, {"index": 213884386, "value": "0x2e29ff36"}, {"index": 104419827, "value": "0x392a8589"}, {"index": 44185464, "value": "0xb56a5912"}, {"index": 142737231, "value": "0xfa6e8b57"}, {"index": 99284897, "value": "0xd1a737cb"}, {"index": 132475900, "value": "0xb0fa841a"}, {"index": 61861762, "value": "0xbe1c341f"}, {"index": 132056166, "value": "0xe25be0f1"}, {"index": 262388043, "value": "0xe937f543"}, {"index": 91878046, "value": "0xebab2248"}, {"index": 117353561, "value": "0x8e1b607a"}, {"index": 124768597, "value": "0x202a2fed"}, {"index": 71352993, "value": "0x95e2819c"}, {"index": 190698941, "value": "0x9c9652d4"}, {"index": 46055428, "value": "0x32fedef0"}, {"index": 55281366, "value": "0xdecfff82"}, {"index": 165145231, "value": "0xcb5d43e5"}, {"index": 106810753, "value": "0xb735806a"}, {"index": 171985651, "value": "0x8905939c"}, {"index": 232085256, "value": "0xfbf8472d"}, {"index": 159510492, "value": "0xada74e5d"}, {"index": 40072060, "value": "0x7ebdeeea"}, {"index": 209107596, "value": "0x0119f2b3"}, {"index": 39023794, "value": "0xa9a376b8"}], + "cache_head": ["0x355a86d2", "0x7957db1c", "0xd21772af", "0x6fc1e09b", "0xd55ce61d", "0x6e6a278b", "0xd3f543ce", "0x223d8e82", "0x143ab337", "0x2e9f05bd", "0x2eb389bf", "0x0c6e449e", "0x5cfa4222", "0xba6560fe", "0x8e3e1aa4", "0xdbcc1d53"], + "cache_last_line": ["0x41190d91", "0xbd277957", "0x22ddbb49", "0x6986f207", "0xdf69a4d6", "0x26401a3a", "0x818230fb", "0xc417122d", "0x3597b211", "0xb553ce55", "0xcf39cc0d", "0x3b7fc43a", "0x3fd43b00", "0x67e1c80e", "0xffa7ea7d", "0xca2960ab"], + "cache_fnv1a64": "0x48c4f5bf24166b2e", + "hot_head": ["0x8068cc73", "0x6036ebf9", "0xb604cd25", "0x8ffb840e", "0xc54074a2", "0x285c0695", "0x77512425", "0xc26a58a7", "0x72c88757", "0xc10fca78", "0x513825dd", "0x30d6ccc8", "0x9a05e7cf", "0xb9533f50", "0x4bac3ba0", "0xa5c19528"], + "hot_last_line": ["0x7c6d7ceb", "0xe24fcd46", "0xe5cce976", "0x3ecffe59", "0x94e98b66", "0x3f6ef45c", "0xa3715aa1", "0xdbe35281", "0xba05d27b", "0x00541963", "0xe636f453", "0xd3972366", "0x529599b8", "0x79ae3c8c", "0x48436863", "0x897a8bf3"], + "hot_fnv1a64": "0x77ca4b9527104530" +} diff --git a/proto-cuda/packs-ca2-hot/hot96k4/kernel.cl b/proto-cuda/packs-ca2-hot/hot96k4/kernel.cl new file mode 100644 index 000000000..d9c5169c7 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4/kernel.cl @@ -0,0 +1,305 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal). +// Built from source at runtime by proto-opencl/host.c, which passes these defines: +// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit) +// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default) +// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32 +// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition +// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units; +// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32. +#ifndef IGNEUM_GROUP +#define IGNEUM_GROUP 32 +#endif +#ifndef IGNEUM_EXCHANGE +#define IGNEUM_EXCHANGE 0 +#endif +#ifdef __OPENCL_VERSION__ +#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1))) +#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n] +#if IGNEUM_EXCHANGE == 1 +#ifdef cl_khr_subgroups +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif +#ifdef cl_khr_subgroup_shuffle +#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable +#endif +#elif IGNEUM_EXCHANGE == 2 +#pragma OPENCL EXTENSION cl_intel_subgroups : enable +#endif +#else +// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros. +#include "emu_opencl.h" +#endif + +#if IGNEUM_EXCHANGE == 1 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#elif IGNEUM_EXCHANGE == 2 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#else +// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per +// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane +// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's +// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier. +#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; } +#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; } +#endif + +// Hot table (96 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01800000u +static inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32. +static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); } +// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x. +static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); } +static inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +static inline void mh_chacha_block(const uint* x, uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +static inline void mh_cache_segment(__global uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +static inline void mh_mixer(uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +static inline void mh_item(__global const uint* cache, uint t, uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + __global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item. +// The same constants as memhard.h in this pack (one emitter, three dialects). +__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) { + uint t = (uint)get_global_id(0); + if (t < nItems) { + uint s[16]; + mh_item(cache, t, s); + __global uint* d = ds + ((ulong)t * 16u); + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (docs/plans/hot-table.md): 96 MiB = 24576 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +static inline void ht_segment(__global uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one work-item per segment. +__kernel void igneum_hot_fill(__global uint* hot, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the +// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and +// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). +IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ ds[r2 & mask]; // 4 load + r5 = r5 ^ ds[r7 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl + r1 = mul_hi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = mul_hi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl + r5 = r5 * r0; // 19 mul + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl + r6 = mul_hi(r6, r2); // 22 mulhi + r6 = r6 ^ ds[r1 & mask]; // 23 load + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ hot[mul_hi(r0, HOT_WORDS)]; // 32 hot + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[mul_hi(r1, HOT_WORDS)]; // 34 hot + r0 = mul_hi(r0, r5); // 35 mulhi + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl + r7 = r7 ^ ds[r0 & mask]; // 37 load + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ ds[r7 & mask]; // 44 load + r2 = r2 * r3; // 45 mul + r1 = mul_hi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ hot[mul_hi(r2, HOT_WORDS)]; // 59 hot + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} + +#if IGNEUM_EXCHANGE != 0 +// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the +// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a +// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md. +IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) { + if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); } +} +#endif diff --git a/proto-cuda/packs-ca2-hot/hot96k4/kernel.cu b/proto-cuda/packs-ca2-hot/hot96k4/kernel.cu new file mode 100644 index 000000000..7fdd2dd49 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4/kernel.cu @@ -0,0 +1,179 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal). +// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC. +#include +#include +#include "program.h" +#include "memhard.h" + +// Hot table (96 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01800000u +__device__ __forceinline__ uint32_t splitmix32(uint32_t x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31. +__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } +// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x. +__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) { + uint32_t x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item. +// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host. +__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) { + uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x; + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) { + uint32_t t = blockIdx.x * blockDim.x + threadIdx.x; + if (t < nItems) { + uint32_t s[16]; + mh_item(cache, t, s); + uint32_t* d = ds + (size_t)t * 16u; + for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (ht_segment is in memhard.h): one thread per segment. +__global__ void igneum_hot_fill(uint32_t* hot, uint32_t nSegments) { + uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x; + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every +// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a +// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid. +__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot) { + uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x; + uint32_t nonce = baseNonce + gid; + uint32_t r0, r1, r2, r3, r4, r5, r6, r7; + { uint32_t x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint32_t x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint32_t x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint32_t x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint32_t x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint32_t x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint32_t x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint32_t x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint32_t it = 0u; it < 8u; ++it) { + uint32_t sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ ds[r2 & mask]; // 4 load + r5 = r5 ^ ds[r7 & mask]; // 5 load + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 6 shfl + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 8); // 7 shfl + r1 = __umulhi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = __umulhi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[__umulhi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 18 shfl + r5 = r5 * r0; // 19 mul + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 20 shfl + r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 16); // 21 shfl + r6 = __umulhi(r6, r2); // 22 mulhi + r6 = r6 ^ ds[r1 & mask]; // 23 load + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r6, 2); // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ hot[__umulhi(r0, HOT_WORDS)]; // 32 hot + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[__umulhi(r1, HOT_WORDS)]; // 34 hot + r0 = __umulhi(r0, r5); // 35 mulhi + r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // 36 shfl + r7 = r7 ^ ds[r0 & mask]; // 37 load + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 4); // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ ds[r7 & mask]; // 44 load + r2 = r2 * r3; // 45 mul + r1 = __umulhi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ hot[__umulhi(r2, HOT_WORDS)]; // 59 hot + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo; +} + +// Host-side launch wrappers. Declared in program.h, called from host.cu. +cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) { + if (nSegments == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nSegments + block - 1u) / block; + igneum_cache_fill<<>>(cache, nSegments); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) { + if (nItems == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nItems + block - 1u) / block; + igneum_build<<>>(ds, cache, nItems); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments) { + if (nSegments == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nSegments + block - 1u) / block; + igneum_hot_fill<<>>(hot, nSegments); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot, + uint32_t nonces, uint32_t blockWarps) { + if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue; + uint32_t block = 32u * blockWarps; + if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue; + igneum_hash<<>>(ds, out, baseNonce, mask, hot); + return cudaGetLastError(); +} + +cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) { + cudaFuncAttributes attr; + cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash); + if (e != cudaSuccess) return e; + *numRegs = attr.numRegs; + return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0); +} diff --git a/proto-cuda/packs-ca2-hot/hot96k4/kernel_bound.cl b/proto-cuda/packs-ca2-hot/hot96k4/kernel_bound.cl new file mode 100644 index 000000000..c87d81a05 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4/kernel_bound.cl @@ -0,0 +1,399 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal). +// Built from source at runtime by proto-opencl/host.c, which passes these defines: +// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit) +// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default) +// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32 +// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition +// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units; +// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32. +#ifndef IGNEUM_GROUP +#define IGNEUM_GROUP 32 +#endif +#ifndef IGNEUM_EXCHANGE +#define IGNEUM_EXCHANGE 0 +#endif +#ifdef __OPENCL_VERSION__ +#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1))) +#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n] +#if IGNEUM_EXCHANGE == 1 +#ifdef cl_khr_subgroups +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif +#ifdef cl_khr_subgroup_shuffle +#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable +#endif +#elif IGNEUM_EXCHANGE == 2 +#pragma OPENCL EXTENSION cl_intel_subgroups : enable +#endif +#else +// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros. +#include "emu_opencl.h" +#endif + +#if IGNEUM_EXCHANGE == 1 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#elif IGNEUM_EXCHANGE == 2 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#else +// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per +// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane +// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's +// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier. +#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; } +#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; } +#endif + +// Hot table (96 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01800000u +static inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32. +static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); } +// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x. +static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); } +static inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +static inline void mh_chacha_block(const uint* x, uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +static inline void mh_cache_segment(__global uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +static inline void mh_mixer(uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +static inline void mh_item(__global const uint* cache, uint t, uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + __global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item. +// The same constants as memhard.h in this pack (one emitter, three dialects). +__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) { + uint t = (uint)get_global_id(0); + if (t < nItems) { + uint s[16]; + mh_item(cache, t, s); + __global uint* d = ds + ((ulong)t * 16u); + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (docs/plans/hot-table.md): 96 MiB = 24576 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +static inline void ht_segment(__global uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one work-item per segment. +__kernel void igneum_hot_fill(__global uint* hot, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the +// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and +// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). +IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ ds[r2 & mask]; // 4 load + r5 = r5 ^ ds[r7 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl + r1 = mul_hi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = mul_hi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl + r5 = r5 * r0; // 19 mul + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl + r6 = mul_hi(r6, r2); // 22 mulhi + r6 = r6 ^ ds[r1 & mask]; // 23 load + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ hot[mul_hi(r0, HOT_WORDS)]; // 32 hot + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[mul_hi(r1, HOT_WORDS)]; // 34 hot + r0 = mul_hi(r0, r5); // 35 mulhi + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl + r7 = r7 ^ ds[r0 & mask]; // 37 load + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ ds[r7 & mask]; // 44 load + r2 = r2 * r3; // 45 mul + r1 = mul_hi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ hot[mul_hi(r2, HOT_WORDS)]; // 59 hot + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} + +#if IGNEUM_EXCHANGE != 0 +// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the +// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a +// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md. +IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) { + if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); } +} +#endif + +// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash. +IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7]; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; } + { uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; } + { uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; } + { uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; } + { uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; } + { uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; } + { uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; } + { uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ ds[r2 & mask]; // 4 load + r5 = r5 ^ ds[r7 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl + r1 = mul_hi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = mul_hi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl + r5 = r5 * r0; // 19 mul + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl + r6 = mul_hi(r6, r2); // 22 mulhi + r6 = r6 ^ ds[r1 & mask]; // 23 load + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ hot[mul_hi(r0, HOT_WORDS)]; // 32 hot + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[mul_hi(r1, HOT_WORDS)]; // 34 hot + r0 = mul_hi(r0, r5); // 35 mulhi + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl + r7 = r7 ^ ds[r0 & mask]; // 37 load + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ ds[r7 & mask]; // 44 load + r2 = r2 * r3; // 45 mul + r1 = mul_hi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ hot[mul_hi(r2, HOT_WORDS)]; // 59 hot + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot96k4/kernel_bound.cu b/proto-cuda/packs-ca2-hot/hot96k4/kernel_bound.cu new file mode 100644 index 000000000..e96dc2ce4 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4/kernel_bound.cu @@ -0,0 +1,125 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW. +// Host declarations (also in program_bound.h if present): +// struct IgneumInitWords { uint32_t w[8]; }; +// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, +// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps); +// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps); +#include +#include +#include "program.h" + +struct IgneumInitWords { uint32_t w[8]; }; + +// Hot table (96 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01800000u +__device__ __forceinline__ uint32_t splitmix32(uint32_t x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } +__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } + +__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw, const uint32_t* hot) { + uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x; + uint32_t nonce = baseNonce + gid; + uint32_t r0, r1, r2, r3, r4, r5, r6, r7; + { uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; } + { uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; } + { uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; } + { uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; } + { uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; } + { uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; } + { uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; } + { uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; } + + for (uint32_t it = 0u; it < 8u; ++it) { + uint32_t sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r2 = r1 * r1 + r2; // 1 mad + r2 = r3 * r2 + r2; // 2 mad + r3 = r3 ^ r5; // 3 xor + r7 = r7 ^ ds[r2 & mask]; // 4 load + r5 = r5 ^ ds[r7 & mask]; // 5 load + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 6 shfl + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 8); // 7 shfl + r1 = __umulhi(r1, r5); // 8 mulhi + r6 = rotr_var(r6, r3); // 9 rotr + r3 = r3 | r4; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r0 = __umulhi(r0, r4); // 12 mulhi + r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add + r0 = r0 ^ ds[r4 & mask]; // 14 load + r2 = r2 - r4; // 15 sub + r2 = r2 ^ hot[__umulhi(r0, HOT_WORDS)]; // 16 hot + r7 = r7 ^ ds[r2 & mask]; // 17 load + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 18 shfl + r5 = r5 * r0; // 19 mul + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 20 shfl + r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 16); // 21 shfl + r6 = __umulhi(r6, r2); // 22 mulhi + r6 = r6 ^ ds[r1 & mask]; // 23 load + r5 = r5 * r0; // 24 mul + r5 = rotl_imm(r5, 19u); // 25 rotl + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r6, 2); // 26 shfl + r0 = r0 ^ r5; // 27 xor + r0 = r0 ^ r4; // 28 xor + r3 = r3 - r0; // 29 sub + r5 = r5 * r1; // 30 mul + r7 = r7 ^ ds[r2 & mask]; // 31 load + r1 = r1 ^ hot[__umulhi(r0, HOT_WORDS)]; // 32 hot + r5 = r5 ^ r6; // 33 xor + r5 = r5 ^ hot[__umulhi(r1, HOT_WORDS)]; // 34 hot + r0 = __umulhi(r0, r5); // 35 mulhi + r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // 36 shfl + r7 = r7 ^ ds[r0 & mask]; // 37 load + r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 4); // 39 shfl + r2 = r2 ^ r5; // 40 xor + r3 = r6 * r3 + r3; // 41 mad + r6 = r6 - r7; // 42 sub + r7 = r7 ^ r0; // 43 xor + r1 = r1 ^ ds[r7 & mask]; // 44 load + r2 = r2 * r3; // 45 mul + r1 = __umulhi(r1, r5); // 46 mulhi + r4 = r4 - r3; // 47 sub + r2 = rotr_var(r2, r6); // 48 rotr + r3 = r3 ^ ds[r5 & mask]; // 49 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add + r0 = r0 * r2; // 51 mul + r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add + r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add + r7 = rotl_imm(r7, 14u); // 54 rotl + r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add + r6 = r6 ^ ds[r7 & mask]; // 56 load + r1 = rotr_var(r1, r5); // 57 rotr + r5 = r5 ^ ds[r4 & mask]; // 58 load + r6 = r6 ^ hot[__umulhi(r2, HOT_WORDS)]; // 59 hot + r3 = r5 * r0 + r3; // 60 mad + r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add + r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add + r5 = rotl_imm(r5, 19u); // 63 rotl + } + uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo; +} + +cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, + IgneumInitWords iw, const uint32_t* hot, uint32_t nonces, uint32_t blockWarps) { + if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue; + uint32_t block = 32u * blockWarps; + if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue; + igneum_hash_bound<<>>(ds, out, baseNonce, mask, iw, hot); + return cudaGetLastError(); +} + +cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) { + cudaFuncAttributes attr; + cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound); + if (e != cudaSuccess) return e; + *numRegs = attr.numRegs; + return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0); +} diff --git a/proto-cuda/packs-ca2-hot/hot96k4/memhard.h b/proto-cuda/packs-ca2-hot/hot96k4/memhard.h new file mode 100644 index 000000000..8691d33e2 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4/memhard.h @@ -0,0 +1,129 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against. +// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference). +// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C. +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif +#if defined(__CUDACC__) +#define IGNEUM_HD __host__ __device__ __forceinline__ +#elif defined(_MSC_VER) && !defined(__cplusplus) +#define IGNEUM_HD static __inline +#else +#define IGNEUM_HD static inline +#endif +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) { + for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint32_t r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) { + uint32_t prev[16]; uint32_t x[16]; uint32_t y[16]; + for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +IGNEUM_HD void mh_item(const uint32_t* cache, uint32_t t, uint32_t* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint32_t r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +IGNEUM_HD uint32_t mh_word(const uint32_t* cache, uint32_t w) { uint32_t s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Hot table (docs/plans/hot-table.md): 96 MiB = 24576 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +IGNEUM_HD void ht_segment(uint32_t* hot, uint32_t seg) { + uint32_t prev[16]; uint32_t x[16]; uint32_t y[16]; + for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + uint32_t* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} diff --git a/proto-cuda/packs-ca2-hot/hot96k4/memhard.metal b/proto-cuda/packs-ca2-hot/hot96k4/memhard.metal new file mode 100644 index 000000000..0acb2b7e1 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4/memhard.metal @@ -0,0 +1,131 @@ +#include +using namespace metal; +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +inline void mh_chacha_block(const thread uint* x, thread uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +inline void mh_cache_segment(device uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +inline void mh_mixer(thread uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +inline void mh_item(device const uint* cache, uint t, thread uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +inline uint mh_word(device const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// One thread per segment (2^16 threads). +kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) { + mh_cache_segment(cache, gid); +} +// One thread per 64-byte item (dataset words / 16 threads). +kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]], + uint gid [[thread_position_in_grid]]) { + uint s[16]; + mh_item(cache, gid, s); + device uint* d = dataset + gid * 16u; + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; +} + +// Hot table (docs/plans/hot-table.md): 96 MiB = 24576 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +inline void ht_segment(device uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + device uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one thread per segment (IGNEUM_HOT_SEGMENTS threads). +kernel void igneum_hot_fill(device uint* hot [[buffer(0)]], uint gid [[thread_position_in_grid]]) { + ht_segment(hot, gid); +} diff --git a/proto-cuda/packs-ca2-hot/hot96k4/program.h b/proto-cuda/packs-ca2-hot/hot96k4/program.h new file mode 100644 index 000000000..f34d91554 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4/program.h @@ -0,0 +1,70 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Program metadata for host.cu plus the launch wrappers defined in kernel.cu. +// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros. +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif +#ifndef IGNEUM_NO_CUDA +#include +#endif + +#define IGNEUM_SEED_STRING "igneum-genesis" +#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973" +#define IGNEUM_GENERATOR 2 +#define IGNEUM_PROGRAM_ATTEMPT 0 +#define IGNEUM_PROGRAM_ID 0x329724466da667ecull +#define IGNEUM_DAY_STRING "2026-10-03" +#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033" +#define IGNEUM_DAY0 0x3067619fu +#define IGNEUM_DAY1 0x3c269176u +#define IGNEUM_DATASET_LOG2 28 +#define IGNEUM_MASK 0x0fffffffu +#define IGNEUM_LANES 32 +#define IGNEUM_ITERATIONS 8 +#define IGNEUM_INSTR_COUNT 64 +#define IGNEUM_LOADS_PER_HASH 128 +#define IGNEUM_WIDE_LOADS_PER_HASH 0 +#define IGNEUM_OP_MIX "load=12 add=8 shfl=8 xor=6 mad=5 mul=5 mulhi=5 hot=4 sub=4 rotl=3 rotr=3 or=1" +// Class v3 construction (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): version 2 loads; the dataset item +// derivation applies the mixer IGNEUM_MIXER_MULT times per round (memhard.h), and the cache follows the growth rule. +#define IGNEUM_LOAD_CLASS "hot96k4" +#define IGNEUM_LOAD_SLOTS 16 +#define IGNEUM_LOAD_MIX { 100, 0, 0 } +#define IGNEUM_LOAD_WIDTH_COUNTS { 12, 0, 0 } // loads of 4, 16, 64 bytes per program +#define IGNEUM_BYTES_PER_HASH 384 +#define IGNEUM_FOLD_ROT 11 +#define IGNEUM_FOLD_MUL 0x9e3779b1u +// Hot table (hot-table experiment, 5 October 2026, docs/plans/hot-table.md): NOT the lottery hash. IGNEUM_HOT_SLOTS of the +// 16 load slots read H[mulhi(src, IGNEUM_HOT_WORDS)], H = IGNEUM_HOT_MB MiB of chained ChaCha12 lines (the cache chain) under +// KH = seed_words("igneum-hot/" || epoch seed bytes) and the tag "Igne" "umHT"; filled by igneum_hot_fill once per epoch. +#define IGNEUM_HOT_MB 96 +#define IGNEUM_HOT_WORDS 0x01800000u +#define IGNEUM_HOT_SEGMENTS 24576u +#define IGNEUM_HOT_SLOTS 4 // hot loads per program (32 per hash), replacing the dataset loads (12 of them) +#define IGNEUM_HOT_ADDED 0 +#define IGNEUM_HOT_KEY_INIT { 0x3a48bef5u, 0x6b54b1a1u, 0x9ff897c3u, 0x7d5d85e4u, 0xcd35379fu, 0x9d76bb86u, 0xbe2affb1u, 0x0f8f80b4u } +// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h) +#define IGNEUM_DATASET_MODE 1 + +#define IGNEUM_SEEDW_INIT { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u } +#define IGNEUM_KEY_INIT { 0x3067619fu, 0x3c269176u, 0x84a03b03u, 0xf8c63294u, 0xff977c5bu, 0xe60def3eu, 0x63630141u, 0xb8fbcb58u } +#define IGNEUM_CACHE_LOG2_WORDS 26 +#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6 +#define IGNEUM_CACHE_SEGMENTS 65536u +#define IGNEUM_ITEM_ROUNDS 8 +#define IGNEUM_MIX_ROT_INIT { 20u, 20u, 19u, 4u, 26u, 3u, 3u, 27u } +#define IGNEUM_MIX_MUL_INIT { 0x42146205u, 0x52cbe0fbu, 0x7ecf4a03u, 0x6728907fu, 0xd81d9751u, 0x132952c3u, 0xf60de277u, 0x05358035u, 0xbaf6499du, 0xe4db9667u, 0x3e98f45du, 0xd0004eddu, 0x2691630du, 0x9beb3bcfu, 0xab310379u, 0x99cfb423u } +#define IGNEUM_MIX_RC_INIT { 0xbab68293u, 0xcc162340u, 0x6ce151ccu, 0xe62b8997u, 0xc9c80297u, 0xf74a1654u, 0x3d704af5u, 0x3cf522b7u, 0x2b9cac04u, 0xa880ac10u, 0x13e5dd1du, 0x6fc3e233u, 0x2d83eeacu, 0x9006e8bfu, 0x2c4b5362u, 0x31b49ee2u } + +#ifndef IGNEUM_NO_CUDA +// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError(). +cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments); +cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems); +cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments); +cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot, + uint32_t nonces, uint32_t blockWarps); +cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps); +#endif diff --git a/proto-cuda/packs-ca2-hot/hot96k4/program.json b/proto-cuda/packs-ca2-hot/hot96k4/program.json new file mode 100644 index 000000000..c0d414cfe --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4/program.json @@ -0,0 +1,129 @@ +{ + "format": "igneum-program-pack-3", + "generator": 2, + "attempt": 0, + "program_id": "0x329724466da667ec", + "program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32", + "dataset_mode": "memory-hard", + "seed": "igneum-genesis", + "seed_bytes": "69676e65756d2d67656e65736973", + "seed_words": ["0x67a9a7be", "0x1a155b25", "0xfddfb732", "0x4b5af2e8", "0xc55caf33", "0xa27c13b7", "0x06628a48", "0x03852469"], + "seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32", + "generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried", + "lanes": 32, + "registers": 8, + "iterations": 8, + "instruction_count": 64, + "loads_per_hash": 128, + "load_class": "hot96k4", + "load_slots": 16, + "load_mix_percent_4_16_64": [100, 0, 0], + "load_width_counts_4_16_64": [12, 0, 0], + "bytes_per_hash": 384, + "wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots", + "hot_table": {"mb": 96, "words": 25165824, "segments": 24576, "slots": 4, "form": "replaced: k of the class's load slots read the table", "dataset_slots": 12, "hot_loads_per_hash": 32, "key": ["0x3a48bef5", "0x6b54b1a1", "0x9ff897c3", "0x7d5d85e4", "0xcd35379f", "0x9d76bb86", "0xbe2affb1", "0x0f8f80b4"], "key_derivation": "seed_words_from_bytes('igneum-hot/' || seed_bytes), the same for every attempt of the epoch", "tag": ["0x49676e65", "0x756d4854"], "chain": "the cache chain of dataset.cache with the hot key and tag: in_j = prev ^ (sigma || key || seg || j || tag); line_j = block(in_j); prev_0 = 0", "load": "dst = dst ^ hot[mulhi(src, words)] (the high 32 bits of the 64-bit product; the index lies in [0, words) for any size)", "slots_rule": "the first k load slots in the Fisher-Yates draw order after the scratch slots (a uniform k-subset); no extra draw, so with version 2 widths and no scratch the program is the version 2 program with k loads redirected", "acceptance_stand_in": "dataset_elem(mulhi(src, words), seed_words[2], seed_words[3])", "program_id": "the read-width id with 'hot/' || mb || k appended", "spec": "docs/plans/hot-table.md"}, + "op_mix": {"load": 12, "add": 8, "shfl": 8, "xor": 6, "mad": 5, "mul": 5, "mulhi": 5, "hot": 4, "sub": 4, "rotl": 3, "rotr": 3, "or": 1}, + "register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]", + "splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16", + "iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order", + "output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo", + "op_semantics": { + "add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)", + "sub": "dst = dst - src", + "mul": "dst = dst * src (low 32)", + "mulhi": "dst = high 32 bits of dst * src", + "xor": "dst = dst ^ src", + "or": "dst = dst | src", + "rotl": "dst = rotl(dst, rot), rot in 1..31", + "rotr": "dst = rotr(dst, src & 31)", + "mad": "dst = src * src2 + dst", + "shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp", + "load": "dst = dst ^ dataset[src & dataset.mask]", + "wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)", + "hot": "dst = dst ^ hot[mulhi(src, hot_table.words)] (hot-table experiment)" + }, + "dataset": { + "log2_words": 28, + "bytes": 1073741824, + "mask": "0x0fffffff", + "day": "2026-10-03", + "day_bytes": "6461792f323032362d31302d3033", + "day_words_from": "seed_words_from_bytes(day_bytes)", + "d0": "0x3067619f", + "d1": "0x3c269176", + "mode": "memory-hard", + "spec": "proto-metal/MEMHARD.md", + "key": ["0x3067619f", "0x3c269176", "0x84a03b03", "0xf8c63294", "0xff977c5b", "0xe60def3e", "0x63630141", "0xb8fbcb58"], + "key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]", + "cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"}, + "mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [20, 20, 19, 4, 26, 3, 3, 27], "mul": ["0x42146205", "0x52cbe0fb", "0x7ecf4a03", "0x6728907f", "0xd81d9751", "0x132952c3", "0xf60de277", "0x05358035", "0xbaf6499d", "0xe4db9667", "0x3e98f45d", "0xd0004edd", "0x2691630d", "0x9beb3bcf", "0xab310379", "0x99cfb423"], "rc": ["0xbab68293", "0xcc162340", "0x6ce151cc", "0xe62b8997", "0xc9c80297", "0xf74a1654", "0x3d704af5", "0x3cf522b7", "0x2b9cac04", "0xa880ac10", "0x13e5dd1d", "0x6fc3e233", "0x2d83eeac", "0x9006e8bf", "0x2c4b5362", "0x31b49ee2"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"}, + "item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: s = M_r(s); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then s = M_8(s); item(t) = s", + "word": "dataset[w] = item(w >> 4)[w & 15]" + }, + "instructions": [ + {"i": 0, "op": "mad", "dst": 2, "src": 3, "src2": 4, "imm": "0xbf7b174d", "imm2": "0x337b762e", "rot": 17, "bit": 2, "mask": 2, "width": 1}, + {"i": 1, "op": "mad", "dst": 2, "src": 1, "src2": 1, "imm": "0xdd04a5da", "imm2": "0x42da7657", "rot": 15, "bit": 30, "mask": 16, "width": 1}, + {"i": 2, "op": "mad", "dst": 2, "src": 3, "src2": 2, "imm": "0x734003fa", "imm2": "0x5bb67700", "rot": 3, "bit": 20, "mask": 1, "width": 1}, + {"i": 3, "op": "xor", "dst": 3, "src": 5, "src2": 5, "imm": "0xc55a1b1c", "imm2": "0xa19720f3", "rot": 7, "bit": 8, "mask": 1, "width": 1}, + {"i": 4, "op": "load", "dst": 7, "src": 2, "src2": 5, "imm": "0xad572dd7", "imm2": "0x9ceb3ea7", "rot": 18, "bit": 30, "mask": 2, "width": 1}, + {"i": 5, "op": "load", "dst": 5, "src": 7, "src2": 3, "imm": "0x769a53be", "imm2": "0x80f9067e", "rot": 12, "bit": 22, "mask": 1, "width": 1}, + {"i": 6, "op": "shfl", "dst": 1, "src": 4, "src2": 0, "imm": "0xd3613d88", "imm2": "0x262fb219", "rot": 10, "bit": 30, "mask": 8, "width": 1}, + {"i": 7, "op": "shfl", "dst": 7, "src": 3, "src2": 4, "imm": "0xce38e42f", "imm2": "0xb868b818", "rot": 11, "bit": 8, "mask": 8, "width": 1}, + {"i": 8, "op": "mulhi", "dst": 1, "src": 5, "src2": 2, "imm": "0xa5eebca5", "imm2": "0x5703a72b", "rot": 13, "bit": 13, "mask": 16, "width": 1}, + {"i": 9, "op": "rotr", "dst": 6, "src": 3, "src2": 4, "imm": "0x17a5a9c7", "imm2": "0xdcfb93a1", "rot": 20, "bit": 27, "mask": 2, "width": 1}, + {"i": 10, "op": "or", "dst": 3, "src": 4, "src2": 1, "imm": "0xccb7d785", "imm2": "0xc335364c", "rot": 14, "bit": 12, "mask": 4, "width": 1}, + {"i": 11, "op": "load", "dst": 4, "src": 3, "src2": 2, "imm": "0x88cb9af3", "imm2": "0x4e7dc10d", "rot": 24, "bit": 17, "mask": 4, "width": 1}, + {"i": 12, "op": "mulhi", "dst": 0, "src": 4, "src2": 4, "imm": "0x45374321", "imm2": "0x3cd91989", "rot": 11, "bit": 4, "mask": 2, "width": 1}, + {"i": 13, "op": "add", "dst": 5, "src": 1, "src2": 2, "imm": "0xc7934706", "imm2": "0xd3177981", "rot": 16, "bit": 30, "mask": 2, "width": 1}, + {"i": 14, "op": "load", "dst": 0, "src": 4, "src2": 3, "imm": "0xfd7f56bb", "imm2": "0x65e14f52", "rot": 13, "bit": 22, "mask": 2, "width": 1}, + {"i": 15, "op": "sub", "dst": 2, "src": 4, "src2": 4, "imm": "0x35a80b49", "imm2": "0x060f2d13", "rot": 16, "bit": 20, "mask": 16, "width": 1}, + {"i": 16, "op": "hot", "dst": 2, "src": 0, "src2": 2, "imm": "0xae0a32c2", "imm2": "0x4c2a4cfe", "rot": 8, "bit": 31, "mask": 16, "width": 1}, + {"i": 17, "op": "load", "dst": 7, "src": 2, "src2": 6, "imm": "0x82a84cc3", "imm2": "0x21a38d68", "rot": 15, "bit": 21, "mask": 2, "width": 1}, + {"i": 18, "op": "shfl", "dst": 7, "src": 3, "src2": 3, "imm": "0xa3818806", "imm2": "0x8f66b5c8", "rot": 14, "bit": 6, "mask": 4, "width": 1}, + {"i": 19, "op": "mul", "dst": 5, "src": 0, "src2": 1, "imm": "0xa00de107", "imm2": "0x77bfcaa5", "rot": 3, "bit": 10, "mask": 2, "width": 1}, + {"i": 20, "op": "shfl", "dst": 3, "src": 4, "src2": 7, "imm": "0x1d2b8cab", "imm2": "0x80b4f9a2", "rot": 14, "bit": 25, "mask": 2, "width": 1}, + {"i": 21, "op": "shfl", "dst": 2, "src": 4, "src2": 5, "imm": "0x3ac915d2", "imm2": "0x5fba7bc2", "rot": 16, "bit": 1, "mask": 16, "width": 1}, + {"i": 22, "op": "mulhi", "dst": 6, "src": 2, "src2": 0, "imm": "0xdc3ec8fd", "imm2": "0x599e2fa3", "rot": 22, "bit": 3, "mask": 2, "width": 1}, + {"i": 23, "op": "load", "dst": 6, "src": 1, "src2": 5, "imm": "0x2a6b16d5", "imm2": "0xd73e396f", "rot": 28, "bit": 29, "mask": 2, "width": 1}, + {"i": 24, "op": "mul", "dst": 5, "src": 0, "src2": 7, "imm": "0x376d0223", "imm2": "0xe1c2169a", "rot": 4, "bit": 16, "mask": 16, "width": 1}, + {"i": 25, "op": "rotl", "dst": 5, "src": 7, "src2": 3, "imm": "0x78ad8c60", "imm2": "0x6f5b77d5", "rot": 19, "bit": 11, "mask": 16, "width": 1}, + {"i": 26, "op": "shfl", "dst": 7, "src": 6, "src2": 5, "imm": "0x93915b9f", "imm2": "0x1e61fb6b", "rot": 28, "bit": 23, "mask": 2, "width": 1}, + {"i": 27, "op": "xor", "dst": 0, "src": 5, "src2": 7, "imm": "0x6378fe15", "imm2": "0x66c78f42", "rot": 12, "bit": 31, "mask": 8, "width": 1}, + {"i": 28, "op": "xor", "dst": 0, "src": 4, "src2": 7, "imm": "0x20a57fda", "imm2": "0x088c848e", "rot": 16, "bit": 13, "mask": 4, "width": 1}, + {"i": 29, "op": "sub", "dst": 3, "src": 0, "src2": 2, "imm": "0x49d95fd5", "imm2": "0x1a5f946a", "rot": 6, "bit": 12, "mask": 1, "width": 1}, + {"i": 30, "op": "mul", "dst": 5, "src": 1, "src2": 7, "imm": "0x0a816217", "imm2": "0x405c4f73", "rot": 13, "bit": 27, "mask": 4, "width": 1}, + {"i": 31, "op": "load", "dst": 7, "src": 2, "src2": 2, "imm": "0x09ed045e", "imm2": "0xd69c4715", "rot": 5, "bit": 9, "mask": 2, "width": 1}, + {"i": 32, "op": "hot", "dst": 1, "src": 0, "src2": 6, "imm": "0xeb79ea49", "imm2": "0xcc587f5a", "rot": 6, "bit": 8, "mask": 16, "width": 1}, + {"i": 33, "op": "xor", "dst": 5, "src": 6, "src2": 1, "imm": "0x3027401e", "imm2": "0x5f20c27e", "rot": 18, "bit": 9, "mask": 2, "width": 1}, + {"i": 34, "op": "hot", "dst": 5, "src": 1, "src2": 3, "imm": "0x0e1cab07", "imm2": "0x09356c5b", "rot": 19, "bit": 31, "mask": 1, "width": 1}, + {"i": 35, "op": "mulhi", "dst": 0, "src": 5, "src2": 2, "imm": "0x90e31357", "imm2": "0xabd32484", "rot": 26, "bit": 5, "mask": 8, "width": 1}, + {"i": 36, "op": "shfl", "dst": 5, "src": 2, "src2": 5, "imm": "0xee9a955f", "imm2": "0x31b3faed", "rot": 8, "bit": 24, "mask": 4, "width": 1}, + {"i": 37, "op": "load", "dst": 7, "src": 0, "src2": 7, "imm": "0x3ba2f832", "imm2": "0x1160dcd3", "rot": 4, "bit": 29, "mask": 1, "width": 1}, + {"i": 38, "op": "add", "dst": 3, "src": 1, "src2": 7, "imm": "0x75ba2fad", "imm2": "0x230c005c", "rot": 4, "bit": 27, "mask": 1, "width": 1}, + {"i": 39, "op": "shfl", "dst": 1, "src": 5, "src2": 3, "imm": "0xdbf37e75", "imm2": "0xb5ac1969", "rot": 30, "bit": 13, "mask": 4, "width": 1}, + {"i": 40, "op": "xor", "dst": 2, "src": 5, "src2": 5, "imm": "0x47f136c5", "imm2": "0x06ce9153", "rot": 19, "bit": 10, "mask": 2, "width": 1}, + {"i": 41, "op": "mad", "dst": 3, "src": 6, "src2": 3, "imm": "0xce13eff8", "imm2": "0x04cc1d55", "rot": 3, "bit": 1, "mask": 4, "width": 1}, + {"i": 42, "op": "sub", "dst": 6, "src": 7, "src2": 1, "imm": "0x6a65ab71", "imm2": "0x8fbc1bcd", "rot": 4, "bit": 1, "mask": 8, "width": 1}, + {"i": 43, "op": "xor", "dst": 7, "src": 0, "src2": 7, "imm": "0xdaeb4928", "imm2": "0xc0423027", "rot": 24, "bit": 11, "mask": 8, "width": 1}, + {"i": 44, "op": "load", "dst": 1, "src": 7, "src2": 7, "imm": "0x778f01c9", "imm2": "0x28cedcea", "rot": 12, "bit": 4, "mask": 16, "width": 1}, + {"i": 45, "op": "mul", "dst": 2, "src": 3, "src2": 2, "imm": "0xf4264f1b", "imm2": "0x0f627d56", "rot": 5, "bit": 28, "mask": 8, "width": 1}, + {"i": 46, "op": "mulhi", "dst": 1, "src": 5, "src2": 1, "imm": "0xffb2147a", "imm2": "0xccde9b05", "rot": 13, "bit": 9, "mask": 2, "width": 1}, + {"i": 47, "op": "sub", "dst": 4, "src": 3, "src2": 5, "imm": "0x73b36234", "imm2": "0x3f5d5997", "rot": 7, "bit": 18, "mask": 2, "width": 1}, + {"i": 48, "op": "rotr", "dst": 2, "src": 6, "src2": 3, "imm": "0x3a4d9aa9", "imm2": "0x212bec7b", "rot": 4, "bit": 29, "mask": 16, "width": 1}, + {"i": 49, "op": "load", "dst": 3, "src": 5, "src2": 2, "imm": "0x626f11df", "imm2": "0x56cd5bfd", "rot": 7, "bit": 1, "mask": 1, "width": 1}, + {"i": 50, "op": "add", "dst": 1, "src": 5, "src2": 6, "imm": "0x81b8bc2c", "imm2": "0x1907970c", "rot": 28, "bit": 7, "mask": 4, "width": 1}, + {"i": 51, "op": "mul", "dst": 0, "src": 2, "src2": 2, "imm": "0xa8848b30", "imm2": "0xef6ac348", "rot": 9, "bit": 15, "mask": 8, "width": 1}, + {"i": 52, "op": "add", "dst": 0, "src": 2, "src2": 0, "imm": "0x4f92b968", "imm2": "0x699fd448", "rot": 22, "bit": 6, "mask": 4, "width": 1}, + {"i": 53, "op": "add", "dst": 1, "src": 0, "src2": 2, "imm": "0x2bb965af", "imm2": "0x77b1520d", "rot": 2, "bit": 12, "mask": 8, "width": 1}, + {"i": 54, "op": "rotl", "dst": 7, "src": 1, "src2": 0, "imm": "0x553e678b", "imm2": "0x3cc8eae0", "rot": 14, "bit": 20, "mask": 2, "width": 1}, + {"i": 55, "op": "add", "dst": 3, "src": 7, "src2": 2, "imm": "0x7b0fe07a", "imm2": "0xa54c55a0", "rot": 10, "bit": 1, "mask": 1, "width": 1}, + {"i": 56, "op": "load", "dst": 6, "src": 7, "src2": 4, "imm": "0x01eba9aa", "imm2": "0x2758c0f7", "rot": 14, "bit": 15, "mask": 4, "width": 1}, + {"i": 57, "op": "rotr", "dst": 1, "src": 5, "src2": 2, "imm": "0x1f5267b3", "imm2": "0x236f5a27", "rot": 2, "bit": 31, "mask": 16, "width": 1}, + {"i": 58, "op": "load", "dst": 5, "src": 4, "src2": 3, "imm": "0xa9954a9b", "imm2": "0x6a54d4e8", "rot": 11, "bit": 10, "mask": 16, "width": 1}, + {"i": 59, "op": "hot", "dst": 6, "src": 2, "src2": 4, "imm": "0x9923ff88", "imm2": "0x9357254e", "rot": 16, "bit": 1, "mask": 16, "width": 1}, + {"i": 60, "op": "mad", "dst": 3, "src": 5, "src2": 0, "imm": "0xc0cc51a6", "imm2": "0x3fd7701b", "rot": 20, "bit": 1, "mask": 4, "width": 1}, + {"i": 61, "op": "add", "dst": 5, "src": 7, "src2": 7, "imm": "0xaf9dd72d", "imm2": "0xad7493e7", "rot": 7, "bit": 31, "mask": 16, "width": 1}, + {"i": 62, "op": "add", "dst": 4, "src": 6, "src2": 2, "imm": "0x89841d87", "imm2": "0x1e07c3d9", "rot": 6, "bit": 27, "mask": 1, "width": 1}, + {"i": 63, "op": "rotl", "dst": 5, "src": 4, "src2": 2, "imm": "0xae210f8d", "imm2": "0x8e499ba4", "rot": 19, "bit": 9, "mask": 1, "width": 1} + ] +} diff --git a/proto-cuda/packs-ca2-hot/hot96k4/program.metal b/proto-cuda/packs-ca2-hot/hot96k4/program.metal new file mode 100644 index 000000000..a52962534 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4/program.metal @@ -0,0 +1,112 @@ +#include +using namespace metal; + +#define MASK 0x0fffffffu +// Hot table (96 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01800000u +constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }; + +inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 +inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +kernel void igneum_hash(device const uint* dataset [[buffer(0)]], + device ulong* out [[buffer(1)]], + constant uint& baseNonce [[buffer(2)]], + device const uint* hot [[buffer(3)]], + uint gid [[thread_position_in_grid]]) { + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + { uint x = nonce ^ SEEDW[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ SEEDW[1]; } + { uint x = nonce ^ SEEDW[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ SEEDW[2]; } + { uint x = nonce ^ SEEDW[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ SEEDW[3]; } + { uint x = nonce ^ SEEDW[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ SEEDW[4]; } + { uint x = nonce ^ SEEDW[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ SEEDW[5]; } + { uint x = nonce ^ SEEDW[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ SEEDW[6]; } + { uint x = nonce ^ SEEDW[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ SEEDW[7]; } + { uint x = nonce ^ SEEDW[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ SEEDW[0]; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 + r2 = r1 * r1 + r2; // 1 + r2 = r3 * r2 + r2; // 2 + r3 = r3 ^ r5; // 3 + r7 = r7 ^ dataset[r2 & MASK]; // 4 + r5 = r5 ^ dataset[r7 & MASK]; // 5 + r1 = r1 ^ simd_shuffle_xor(r4, (ushort)8); // 6 + r7 = r7 ^ simd_shuffle_xor(r3, (ushort)8); // 7 + r1 = mulhi(r1, r5); // 8 + r6 = rotr_var(r6, r3); // 9 + r3 = r3 | r4; // 10 + r4 = r4 ^ dataset[r3 & MASK]; // 11 + r0 = mulhi(r0, r4); // 12 + r5 = r5 + r1 + select(0xc7934706u, 0xd3177981u, ((sel >> 30u) & 1u) != 0u); // 13 + r0 = r0 ^ dataset[r4 & MASK]; // 14 + r2 = r2 - r4; // 15 + r2 = r2 ^ hot[mulhi(r0, HOT_WORDS)]; // 16 + r7 = r7 ^ dataset[r2 & MASK]; // 17 + r7 = r7 ^ simd_shuffle_xor(r3, (ushort)4); // 18 + r5 = r5 * r0; // 19 + r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 20 + r2 = r2 ^ simd_shuffle_xor(r4, (ushort)16); // 21 + r6 = mulhi(r6, r2); // 22 + r6 = r6 ^ dataset[r1 & MASK]; // 23 + r5 = r5 * r0; // 24 + r5 = rotl_imm(r5, 19u); // 25 + r7 = r7 ^ simd_shuffle_xor(r6, (ushort)2); // 26 + r0 = r0 ^ r5; // 27 + r0 = r0 ^ r4; // 28 + r3 = r3 - r0; // 29 + r5 = r5 * r1; // 30 + r7 = r7 ^ dataset[r2 & MASK]; // 31 + r1 = r1 ^ hot[mulhi(r0, HOT_WORDS)]; // 32 + r5 = r5 ^ r6; // 33 + r5 = r5 ^ hot[mulhi(r1, HOT_WORDS)]; // 34 + r0 = mulhi(r0, r5); // 35 + r5 = r5 ^ simd_shuffle_xor(r2, (ushort)4); // 36 + r7 = r7 ^ dataset[r0 & MASK]; // 37 + r3 = r3 + r1 + select(0x75ba2fadu, 0x230c005cu, ((sel >> 27u) & 1u) != 0u); // 38 + r1 = r1 ^ simd_shuffle_xor(r5, (ushort)4); // 39 + r2 = r2 ^ r5; // 40 + r3 = r6 * r3 + r3; // 41 + r6 = r6 - r7; // 42 + r7 = r7 ^ r0; // 43 + r1 = r1 ^ dataset[r7 & MASK]; // 44 + r2 = r2 * r3; // 45 + r1 = mulhi(r1, r5); // 46 + r4 = r4 - r3; // 47 + r2 = rotr_var(r2, r6); // 48 + r3 = r3 ^ dataset[r5 & MASK]; // 49 + r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 50 + r0 = r0 * r2; // 51 + r0 = r0 + r2 + select(0x4f92b968u, 0x699fd448u, ((sel >> 6u) & 1u) != 0u); // 52 + r1 = r1 + r0 + select(0x2bb965afu, 0x77b1520du, ((sel >> 12u) & 1u) != 0u); // 53 + r7 = rotl_imm(r7, 14u); // 54 + r3 = r3 + r7 + select(0x7b0fe07au, 0xa54c55a0u, ((sel >> 1u) & 1u) != 0u); // 55 + r6 = r6 ^ dataset[r7 & MASK]; // 56 + r1 = rotr_var(r1, r5); // 57 + r5 = r5 ^ dataset[r4 & MASK]; // 58 + r6 = r6 ^ hot[mulhi(r2, HOT_WORDS)]; // 59 + r3 = r5 * r0 + r3; // 60 + r5 = r5 + r7 + select(0xaf9dd72du, 0xad7493e7u, ((sel >> 31u) & 1u) != 0u); // 61 + r4 = r4 + r6 + select(0x89841d87u, 0x1e07c3d9u, ((sel >> 27u) & 1u) != 0u); // 62 + r5 = rotl_imm(r5, 19u); // 63 + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot96k4/program_bound.metal b/proto-cuda/packs-ca2-hot/hot96k4/program_bound.metal new file mode 100644 index 000000000..987f291d4 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4/program_bound.metal @@ -0,0 +1,114 @@ +#include +using namespace metal; + +#define MASK 0x0fffffffu +// Hot table (96 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01800000u +constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }; + +inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 +inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW. +kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], + device ulong* out [[buffer(1)]], + constant uint& baseNonce [[buffer(2)]], + constant uint* initw [[buffer(3)]], + device const uint* hot [[buffer(4)]], + uint gid [[thread_position_in_grid]]) { + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + { uint x = nonce ^ initw[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ initw[1]; } + { uint x = nonce ^ initw[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ initw[2]; } + { uint x = nonce ^ initw[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ initw[3]; } + { uint x = nonce ^ initw[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ initw[4]; } + { uint x = nonce ^ initw[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ initw[5]; } + { uint x = nonce ^ initw[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ initw[6]; } + { uint x = nonce ^ initw[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ initw[7]; } + { uint x = nonce ^ initw[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ initw[0]; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 + r2 = r1 * r1 + r2; // 1 + r2 = r3 * r2 + r2; // 2 + r3 = r3 ^ r5; // 3 + r7 = r7 ^ dataset[r2 & MASK]; // 4 + r5 = r5 ^ dataset[r7 & MASK]; // 5 + r1 = r1 ^ simd_shuffle_xor(r4, (ushort)8); // 6 + r7 = r7 ^ simd_shuffle_xor(r3, (ushort)8); // 7 + r1 = mulhi(r1, r5); // 8 + r6 = rotr_var(r6, r3); // 9 + r3 = r3 | r4; // 10 + r4 = r4 ^ dataset[r3 & MASK]; // 11 + r0 = mulhi(r0, r4); // 12 + r5 = r5 + r1 + select(0xc7934706u, 0xd3177981u, ((sel >> 30u) & 1u) != 0u); // 13 + r0 = r0 ^ dataset[r4 & MASK]; // 14 + r2 = r2 - r4; // 15 + r2 = r2 ^ hot[mulhi(r0, HOT_WORDS)]; // 16 + r7 = r7 ^ dataset[r2 & MASK]; // 17 + r7 = r7 ^ simd_shuffle_xor(r3, (ushort)4); // 18 + r5 = r5 * r0; // 19 + r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 20 + r2 = r2 ^ simd_shuffle_xor(r4, (ushort)16); // 21 + r6 = mulhi(r6, r2); // 22 + r6 = r6 ^ dataset[r1 & MASK]; // 23 + r5 = r5 * r0; // 24 + r5 = rotl_imm(r5, 19u); // 25 + r7 = r7 ^ simd_shuffle_xor(r6, (ushort)2); // 26 + r0 = r0 ^ r5; // 27 + r0 = r0 ^ r4; // 28 + r3 = r3 - r0; // 29 + r5 = r5 * r1; // 30 + r7 = r7 ^ dataset[r2 & MASK]; // 31 + r1 = r1 ^ hot[mulhi(r0, HOT_WORDS)]; // 32 + r5 = r5 ^ r6; // 33 + r5 = r5 ^ hot[mulhi(r1, HOT_WORDS)]; // 34 + r0 = mulhi(r0, r5); // 35 + r5 = r5 ^ simd_shuffle_xor(r2, (ushort)4); // 36 + r7 = r7 ^ dataset[r0 & MASK]; // 37 + r3 = r3 + r1 + select(0x75ba2fadu, 0x230c005cu, ((sel >> 27u) & 1u) != 0u); // 38 + r1 = r1 ^ simd_shuffle_xor(r5, (ushort)4); // 39 + r2 = r2 ^ r5; // 40 + r3 = r6 * r3 + r3; // 41 + r6 = r6 - r7; // 42 + r7 = r7 ^ r0; // 43 + r1 = r1 ^ dataset[r7 & MASK]; // 44 + r2 = r2 * r3; // 45 + r1 = mulhi(r1, r5); // 46 + r4 = r4 - r3; // 47 + r2 = rotr_var(r2, r6); // 48 + r3 = r3 ^ dataset[r5 & MASK]; // 49 + r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 50 + r0 = r0 * r2; // 51 + r0 = r0 + r2 + select(0x4f92b968u, 0x699fd448u, ((sel >> 6u) & 1u) != 0u); // 52 + r1 = r1 + r0 + select(0x2bb965afu, 0x77b1520du, ((sel >> 12u) & 1u) != 0u); // 53 + r7 = rotl_imm(r7, 14u); // 54 + r3 = r3 + r7 + select(0x7b0fe07au, 0xa54c55a0u, ((sel >> 1u) & 1u) != 0u); // 55 + r6 = r6 ^ dataset[r7 & MASK]; // 56 + r1 = rotr_var(r1, r5); // 57 + r5 = r5 ^ dataset[r4 & MASK]; // 58 + r6 = r6 ^ hot[mulhi(r2, HOT_WORDS)]; // 59 + r3 = r5 * r0 + r3; // 60 + r5 = r5 + r7 + select(0xaf9dd72du, 0xad7493e7u, ((sel >> 31u) & 1u) != 0u); // 61 + r4 = r4 + r6 + select(0x89841d87u, 0x1e07c3d9u, ((sel >> 27u) & 1u) != 0u); // 62 + r5 = rotl_imm(r5, 19u); // 63 + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot96k4/vectors.h b/proto-cuda/packs-ca2-hot/hot96k4/vectors.h new file mode 100644 index 000000000..bc373868f --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4/vectors.h @@ -0,0 +1,67 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Expected outputs: igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif + +#define IGNEUM_VEC_WARPS 3 +static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u }; +static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = { + { // base nonce 0 + 0x95847de49065f429ull, 0x983294df132660c1ull, 0x8cf35b7bbf61e453ull, 0xc04d397f351fdd4eull, 0x2b10fc021f6aea84ull, 0xb5591d0d87fc5eafull, 0x018a85ab6e325a15ull, 0x4f219f18bf55f435ull, + 0xdb04938b70d8fc69ull, 0x08e0cb2cf40d64a7ull, 0xd9aa606a36518d0bull, 0x827a1cd00d7f0a23ull, 0x271e519dd470b9c8ull, 0xa6938f1bf1488ee0ull, 0x2034a82470a14babull, 0x6188094c1269c4e4ull, + 0x0ae1302647de77cdull, 0x02d79731ec72d8f4ull, 0xfa6b22726a655761ull, 0x32f8f69ac8329daeull, 0xa2ea5de7f8b1594dull, 0xc0bf04c7dca99c0cull, 0xe6a53e5a0f10c953ull, 0x5dd892fad086cf60ull, + 0xd5ba31cceb072706ull, 0x8c4aa1b7c0bb12b4ull, 0xf2b9b796ea86a26full, 0xa9cd01a077ec5875ull, 0x68b529a99fae547full, 0xbb06696372460726ull, 0x4c3af022d242a559ull, 0xfa8c698cfde1d954ull + }, + { // base nonce 4096 + 0xb6f2d3fe8b047d57ull, 0x0ea1076659d8b7acull, 0x928ea485293e9fcfull, 0x2d1ab7a9ee6bdee7ull, 0x65c166a88f62b77dull, 0xc6be153a19a235b4ull, 0x1a2431743b715fe6ull, 0xbff2dc53b8aa1cf3ull, + 0x6545051f03d6a9fbull, 0x4999c409c24702e9ull, 0x14e5cb8d974ac5d2ull, 0x5d1768adcf8aba7full, 0x77e5b90b24663744ull, 0x83f9e42857dce25full, 0x797e152bfcbee580ull, 0x296d508ed98506a6ull, + 0x72d97df5bfccb6baull, 0x6ee38ffa21d697eaull, 0x1bacd341bf7d8608ull, 0xe743262b80cb9b94ull, 0xb90860f68731652dull, 0x043af90cc6ba4da5ull, 0x645dc8ed808377e3ull, 0xb8c20121544642adull, + 0x198f6d1e47df2d4bull, 0xa4da94bbc33a3998ull, 0xca72308b15e4e7c9ull, 0xb353fd44e86e3bd9ull, 0x9f441f8c41048905ull, 0x12f9710d754b1cfdull, 0x748701e5a0fcbfd8ull, 0x9c7e52f6397721f6ull + }, + { // base nonce 1000000 + 0xa403641ce530eb70ull, 0x62e8a999234b6ab9ull, 0x0b0fd8859e9ec3e7ull, 0xfd1a15a4bfd174a5ull, 0x1fe9299599b4ed7full, 0xe8a3d2e2b28a5ebdull, 0x1863fd4b1aa24d81ull, 0x4352e226ee0590acull, + 0x00b10083a0285d9dull, 0x937dd627f2f14211ull, 0xb66a7a50fb0f4bb0ull, 0x3270883959c2d5c7ull, 0x8847671b7b801481ull, 0x3c544c0f61491058ull, 0xf7bf297e001b48a2ull, 0x3d81db650d5b98a3ull, + 0xdf3c607b790f674eull, 0x0c2d334cd2be3e50ull, 0x9875cc4b8de6ef79ull, 0x7f602f155bcad8a4ull, 0x5fbd6a1f21faf66dull, 0x7c8d54057a8e551cull, 0xb092150d9c6308cfull, 0xb6ffb83489430ecaull, + 0x9a8cbc6a7c90d711ull, 0xe062d7e57bc91d20ull, 0xb852d7ee3cf0e27eull, 0xec081a517363cc33ull, 0x03e9344a2db8860cull, 0xbda7f825624bde06ull, 0x4184375356b1e7faull, 0x2975a40effe3f796ull + } +}; + +// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (268435455). +static const uint32_t IGNEUM_DS_HEAD[16] = { + 0xffc3cd94u, 0x5920ccd8u, 0x392f44bbu, 0x5e57f67au, 0x2f2bc2a9u, 0x620b0e36u, 0xbdc09014u, 0x436654bfu, + 0x311e0b48u, 0x1abd93adu, 0x59cc7ce8u, 0xee5247b2u, 0x86171fe8u, 0x6d874751u, 0xc9f7728fu, 0x7c2a435du +}; +static const uint32_t IGNEUM_DS_LAST_INDEX = 268435455u; +static const uint32_t IGNEUM_DS_LAST = 0xa33ada72u; +// 64 sampled dataset words (index, value) computed on the Mac. +#define IGNEUM_DS_SAMPLES 64 +static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = { + 59471966u, 217795994u, 208353206u, 42483309u, 172547758u, 148076330u, 183853158u, 214389424u, 267488061u, 169781097u, 184093494u, 153880993u, 84977930u, 46426879u, 3093825u, 225364072u, 44593546u, 260713159u, 168250303u, 52384140u, 223401610u, 45554030u, 95410555u, 175039924u, 79171087u, 267580473u, 24168642u, 37981670u, 171551130u, 195559979u, 204611762u, 140997658u, 138925853u, 86637313u, 20736778u, 219665210u, 160430336u, 264654675u, 8013395u, 228945585u, 213884386u, 104419827u, 44185464u, 142737231u, 99284897u, 132475900u, 61861762u, 132056166u, 262388043u, 91878046u, 117353561u, 124768597u, 71352993u, 190698941u, 46055428u, 55281366u, 165145231u, 106810753u, 171985651u, 232085256u, 159510492u, 40072060u, 209107596u, 39023794u +}; +static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = { + 0xe8b73d94u, 0x337028b5u, 0xafe148c9u, 0xab99f7aeu, 0x434ea619u, 0xd85cb880u, 0x54764c7fu, 0x82c7e420u, 0xedf4cb9eu, 0x9884c959u, 0x223ee793u, 0x3a9ccf69u, 0x81da4fd2u, 0xd6ce8cb9u, 0xe3922dcau, 0x3e7e6bdeu, 0x382a3acau, 0x567e7f7fu, 0x25a0f084u, 0xbfeef128u, 0xe338abfbu, 0x7c3b5280u, 0x909bc5f1u, 0xd8b74b9cu, 0x8e31a22eu, 0x26b5f1d8u, 0x79122c00u, 0xcafc3340u, 0xd5e02ea3u, 0x1aee1afdu, 0xdb090d9au, 0xb049f435u, 0x4954d8bau, 0x03797ba0u, 0x196eefbdu, 0xd153412au, 0xbe5d2c4bu, 0xdaa14f0eu, 0x8e61ed07u, 0x9e9a64c6u, 0x2e29ff36u, 0x392a8589u, 0xb56a5912u, 0xfa6e8b57u, 0xd1a737cbu, 0xb0fa841au, 0xbe1c341fu, 0xe25be0f1u, 0xe937f543u, 0xebab2248u, 0x8e1b607au, 0x202a2fedu, 0x95e2819cu, 0x9c9652d4u, 0x32fedef0u, 0xdecfff82u, 0xcb5d43e5u, 0xb735806au, 0x8905939cu, 0xfbf8472du, 0xada74e5du, 0x7ebdeeeau, 0x0119f2b3u, 0xa9a376b8u +}; +// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^26 words. +static const uint32_t IGNEUM_CACHE_HEAD[16] = { + 0x355a86d2u, 0x7957db1cu, 0xd21772afu, 0x6fc1e09bu, 0xd55ce61du, 0x6e6a278bu, 0xd3f543ceu, 0x223d8e82u, + 0x143ab337u, 0x2e9f05bdu, 0x2eb389bfu, 0x0c6e449eu, 0x5cfa4222u, 0xba6560feu, 0x8e3e1aa4u, 0xdbcc1d53u +}; +static const uint32_t IGNEUM_CACHE_LAST[16] = { + 0x41190d91u, 0xbd277957u, 0x22ddbb49u, 0x6986f207u, 0xdf69a4d6u, 0x26401a3au, 0x818230fbu, 0xc417122du, + 0x3597b211u, 0xb553ce55u, 0xcf39cc0du, 0x3b7fc43au, 0x3fd43b00u, 0x67e1c80eu, 0xffa7ea7du, 0xca2960abu +}; +static const uint64_t IGNEUM_CACHE_FNV64 = 0x48c4f5bf24166b2eull; +// Hot table self-test (docs/plans/hot-table.md): hot[0..15], the last 16 words, and FNV-1a 64 over all IGNEUM_HOT_WORDS words. +static const uint32_t IGNEUM_HOT_HEAD[16] = { + 0x8068cc73u, 0x6036ebf9u, 0xb604cd25u, 0x8ffb840eu, 0xc54074a2u, 0x285c0695u, 0x77512425u, 0xc26a58a7u, + 0x72c88757u, 0xc10fca78u, 0x513825ddu, 0x30d6ccc8u, 0x9a05e7cfu, 0xb9533f50u, 0x4bac3ba0u, 0xa5c19528u +}; +static const uint32_t IGNEUM_HOT_LAST[16] = { + 0xa3fd89f9u, 0xeb098388u, 0x19ee38d9u, 0x3700a229u, 0xbc0602a9u, 0x231d2a70u, 0xc23c57ceu, 0x2ff8dc6au, + 0x39683a2du, 0xbbe7d264u, 0xb8b28a43u, 0xb830041du, 0x86ae51e2u, 0xb3d22ccfu, 0x0e53335fu, 0xb56e328du +}; +static const uint64_t IGNEUM_HOT_FNV64 = 0x79bcf436c4e5bc47ull; diff --git a/proto-cuda/packs-ca2-hot/hot96k4/vectors.json b/proto-cuda/packs-ca2-hot/hot96k4/vectors.json new file mode 100644 index 000000000..4c138315f --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4/vectors.json @@ -0,0 +1,39 @@ +{ + "seed": "igneum-genesis", + "day": "2026-10-03", + "dataset_mode": "memory-hard", + "dataset_log2_words": 28, + "mask": "0x0fffffff", + "lanes": 32, + "source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset", + "warps": [ + {"base_nonce": 0, "expected": [ + "0x95847de49065f429", "0x983294df132660c1", "0x8cf35b7bbf61e453", "0xc04d397f351fdd4e", "0x2b10fc021f6aea84", "0xb5591d0d87fc5eaf", "0x018a85ab6e325a15", "0x4f219f18bf55f435", + "0xdb04938b70d8fc69", "0x08e0cb2cf40d64a7", "0xd9aa606a36518d0b", "0x827a1cd00d7f0a23", "0x271e519dd470b9c8", "0xa6938f1bf1488ee0", "0x2034a82470a14bab", "0x6188094c1269c4e4", + "0x0ae1302647de77cd", "0x02d79731ec72d8f4", "0xfa6b22726a655761", "0x32f8f69ac8329dae", "0xa2ea5de7f8b1594d", "0xc0bf04c7dca99c0c", "0xe6a53e5a0f10c953", "0x5dd892fad086cf60", + "0xd5ba31cceb072706", "0x8c4aa1b7c0bb12b4", "0xf2b9b796ea86a26f", "0xa9cd01a077ec5875", "0x68b529a99fae547f", "0xbb06696372460726", "0x4c3af022d242a559", "0xfa8c698cfde1d954" + ]}, + {"base_nonce": 4096, "expected": [ + "0xb6f2d3fe8b047d57", "0x0ea1076659d8b7ac", "0x928ea485293e9fcf", "0x2d1ab7a9ee6bdee7", "0x65c166a88f62b77d", "0xc6be153a19a235b4", "0x1a2431743b715fe6", "0xbff2dc53b8aa1cf3", + "0x6545051f03d6a9fb", "0x4999c409c24702e9", "0x14e5cb8d974ac5d2", "0x5d1768adcf8aba7f", "0x77e5b90b24663744", "0x83f9e42857dce25f", "0x797e152bfcbee580", "0x296d508ed98506a6", + "0x72d97df5bfccb6ba", "0x6ee38ffa21d697ea", "0x1bacd341bf7d8608", "0xe743262b80cb9b94", "0xb90860f68731652d", "0x043af90cc6ba4da5", "0x645dc8ed808377e3", "0xb8c20121544642ad", + "0x198f6d1e47df2d4b", "0xa4da94bbc33a3998", "0xca72308b15e4e7c9", "0xb353fd44e86e3bd9", "0x9f441f8c41048905", "0x12f9710d754b1cfd", "0x748701e5a0fcbfd8", "0x9c7e52f6397721f6" + ]}, + {"base_nonce": 1000000, "expected": [ + "0xa403641ce530eb70", "0x62e8a999234b6ab9", "0x0b0fd8859e9ec3e7", "0xfd1a15a4bfd174a5", "0x1fe9299599b4ed7f", "0xe8a3d2e2b28a5ebd", "0x1863fd4b1aa24d81", "0x4352e226ee0590ac", + "0x00b10083a0285d9d", "0x937dd627f2f14211", "0xb66a7a50fb0f4bb0", "0x3270883959c2d5c7", "0x8847671b7b801481", "0x3c544c0f61491058", "0xf7bf297e001b48a2", "0x3d81db650d5b98a3", + "0xdf3c607b790f674e", "0x0c2d334cd2be3e50", "0x9875cc4b8de6ef79", "0x7f602f155bcad8a4", "0x5fbd6a1f21faf66d", "0x7c8d54057a8e551c", "0xb092150d9c6308cf", "0xb6ffb83489430eca", + "0x9a8cbc6a7c90d711", "0xe062d7e57bc91d20", "0xb852d7ee3cf0e27e", "0xec081a517363cc33", "0x03e9344a2db8860c", "0xbda7f825624bde06", "0x4184375356b1e7fa", "0x2975a40effe3f796" + ]} + ], + "dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"], + "dataset_last_index": 268435455, + "dataset_last": "0xa33ada72", + "dataset_samples": [{"index": 59471966, "value": "0xe8b73d94"}, {"index": 217795994, "value": "0x337028b5"}, {"index": 208353206, "value": "0xafe148c9"}, {"index": 42483309, "value": "0xab99f7ae"}, {"index": 172547758, "value": "0x434ea619"}, {"index": 148076330, "value": "0xd85cb880"}, {"index": 183853158, "value": "0x54764c7f"}, {"index": 214389424, "value": "0x82c7e420"}, {"index": 267488061, "value": "0xedf4cb9e"}, {"index": 169781097, "value": "0x9884c959"}, {"index": 184093494, "value": "0x223ee793"}, {"index": 153880993, "value": "0x3a9ccf69"}, {"index": 84977930, "value": "0x81da4fd2"}, {"index": 46426879, "value": "0xd6ce8cb9"}, {"index": 3093825, "value": "0xe3922dca"}, {"index": 225364072, "value": "0x3e7e6bde"}, {"index": 44593546, "value": "0x382a3aca"}, {"index": 260713159, "value": "0x567e7f7f"}, {"index": 168250303, "value": "0x25a0f084"}, {"index": 52384140, "value": "0xbfeef128"}, {"index": 223401610, "value": "0xe338abfb"}, {"index": 45554030, "value": "0x7c3b5280"}, {"index": 95410555, "value": "0x909bc5f1"}, {"index": 175039924, "value": "0xd8b74b9c"}, {"index": 79171087, "value": "0x8e31a22e"}, {"index": 267580473, "value": "0x26b5f1d8"}, {"index": 24168642, "value": "0x79122c00"}, {"index": 37981670, "value": "0xcafc3340"}, {"index": 171551130, "value": "0xd5e02ea3"}, {"index": 195559979, "value": "0x1aee1afd"}, {"index": 204611762, "value": "0xdb090d9a"}, {"index": 140997658, "value": "0xb049f435"}, {"index": 138925853, "value": "0x4954d8ba"}, {"index": 86637313, "value": "0x03797ba0"}, {"index": 20736778, "value": "0x196eefbd"}, {"index": 219665210, "value": "0xd153412a"}, {"index": 160430336, "value": "0xbe5d2c4b"}, {"index": 264654675, "value": "0xdaa14f0e"}, {"index": 8013395, "value": "0x8e61ed07"}, {"index": 228945585, "value": "0x9e9a64c6"}, {"index": 213884386, "value": "0x2e29ff36"}, {"index": 104419827, "value": "0x392a8589"}, {"index": 44185464, "value": "0xb56a5912"}, {"index": 142737231, "value": "0xfa6e8b57"}, {"index": 99284897, "value": "0xd1a737cb"}, {"index": 132475900, "value": "0xb0fa841a"}, {"index": 61861762, "value": "0xbe1c341f"}, {"index": 132056166, "value": "0xe25be0f1"}, {"index": 262388043, "value": "0xe937f543"}, {"index": 91878046, "value": "0xebab2248"}, {"index": 117353561, "value": "0x8e1b607a"}, {"index": 124768597, "value": "0x202a2fed"}, {"index": 71352993, "value": "0x95e2819c"}, {"index": 190698941, "value": "0x9c9652d4"}, {"index": 46055428, "value": "0x32fedef0"}, {"index": 55281366, "value": "0xdecfff82"}, {"index": 165145231, "value": "0xcb5d43e5"}, {"index": 106810753, "value": "0xb735806a"}, {"index": 171985651, "value": "0x8905939c"}, {"index": 232085256, "value": "0xfbf8472d"}, {"index": 159510492, "value": "0xada74e5d"}, {"index": 40072060, "value": "0x7ebdeeea"}, {"index": 209107596, "value": "0x0119f2b3"}, {"index": 39023794, "value": "0xa9a376b8"}], + "cache_head": ["0x355a86d2", "0x7957db1c", "0xd21772af", "0x6fc1e09b", "0xd55ce61d", "0x6e6a278b", "0xd3f543ce", "0x223d8e82", "0x143ab337", "0x2e9f05bd", "0x2eb389bf", "0x0c6e449e", "0x5cfa4222", "0xba6560fe", "0x8e3e1aa4", "0xdbcc1d53"], + "cache_last_line": ["0x41190d91", "0xbd277957", "0x22ddbb49", "0x6986f207", "0xdf69a4d6", "0x26401a3a", "0x818230fb", "0xc417122d", "0x3597b211", "0xb553ce55", "0xcf39cc0d", "0x3b7fc43a", "0x3fd43b00", "0x67e1c80e", "0xffa7ea7d", "0xca2960ab"], + "cache_fnv1a64": "0x48c4f5bf24166b2e", + "hot_head": ["0x8068cc73", "0x6036ebf9", "0xb604cd25", "0x8ffb840e", "0xc54074a2", "0x285c0695", "0x77512425", "0xc26a58a7", "0x72c88757", "0xc10fca78", "0x513825dd", "0x30d6ccc8", "0x9a05e7cf", "0xb9533f50", "0x4bac3ba0", "0xa5c19528"], + "hot_last_line": ["0xa3fd89f9", "0xeb098388", "0x19ee38d9", "0x3700a229", "0xbc0602a9", "0x231d2a70", "0xc23c57ce", "0x2ff8dc6a", "0x39683a2d", "0xbbe7d264", "0xb8b28a43", "0xb830041d", "0x86ae51e2", "0xb3d22ccf", "0x0e53335f", "0xb56e328d"], + "hot_fnv1a64": "0x79bcf436c4e5bc47" +} diff --git a/proto-cuda/packs-ca2-hot/hot96k4a/kernel.cl b/proto-cuda/packs-ca2-hot/hot96k4a/kernel.cl new file mode 100644 index 000000000..2d3e2ace4 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4a/kernel.cl @@ -0,0 +1,305 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal). +// Built from source at runtime by proto-opencl/host.c, which passes these defines: +// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit) +// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default) +// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32 +// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition +// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units; +// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32. +#ifndef IGNEUM_GROUP +#define IGNEUM_GROUP 32 +#endif +#ifndef IGNEUM_EXCHANGE +#define IGNEUM_EXCHANGE 0 +#endif +#ifdef __OPENCL_VERSION__ +#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1))) +#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n] +#if IGNEUM_EXCHANGE == 1 +#ifdef cl_khr_subgroups +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif +#ifdef cl_khr_subgroup_shuffle +#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable +#endif +#elif IGNEUM_EXCHANGE == 2 +#pragma OPENCL EXTENSION cl_intel_subgroups : enable +#endif +#else +// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros. +#include "emu_opencl.h" +#endif + +#if IGNEUM_EXCHANGE == 1 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#elif IGNEUM_EXCHANGE == 2 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#else +// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per +// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane +// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's +// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier. +#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; } +#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; } +#endif + +// Hot table (96 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01800000u +static inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32. +static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); } +// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x. +static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); } +static inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +static inline void mh_chacha_block(const uint* x, uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +static inline void mh_cache_segment(__global uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +static inline void mh_mixer(uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +static inline void mh_item(__global const uint* cache, uint t, uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + __global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item. +// The same constants as memhard.h in this pack (one emitter, three dialects). +__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) { + uint t = (uint)get_global_id(0); + if (t < nItems) { + uint s[16]; + mh_item(cache, t, s); + __global uint* d = ds + ((ulong)t * 16u); + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (docs/plans/hot-table.md): 96 MiB = 24576 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +static inline void ht_segment(__global uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one work-item per segment. +__kernel void igneum_hot_fill(__global uint* hot, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the +// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and +// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). +IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r6 = r6 | r4; // 0 or + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 1u); r7 = r7 ^ t_; } // 1 shfl + r0 = r0 | r2; // 2 or + { uint t_; IGNEUM_SHFL_XOR(t_, r0, 16u); r3 = r3 ^ t_; } // 3 shfl + r7 = r7 ^ ds[r3 & mask]; // 4 load + r6 = r6 ^ ds[r0 & mask]; // 5 load + r1 = r1 * r4; // 6 mul + r0 = r0 - r1; // 7 sub + r3 = r3 ^ ds[r7 & mask]; // 8 load + r1 = r0 * r3 + r1; // 9 mad + r4 = r4 * r0; // 10 mul + r5 = r5 ^ ds[r4 & mask]; // 11 load + r1 = r1 ^ r7; // 12 xor + r1 = r1 ^ ds[r6 & mask]; // 13 load + r2 = r2 ^ ds[r1 & mask]; // 14 load + r3 = r3 * r1; // 15 mul + r6 = r6 ^ hot[mul_hi(r5, HOT_WORDS)]; // 16 hot + r0 = r0 ^ ds[r2 & mask]; // 17 load + r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x535b545fu : 0xc035a4e6u); // 18 add + r5 = r5 - r0; // 19 sub + r2 = r2 * r4; // 20 mul + r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x925d5e63u : 0x1ecdd2cfu); // 21 add + r3 = r3 ^ r7; // 22 xor + r7 = r7 ^ ds[r2 & mask]; // 23 load + r2 = mul_hi(r2, r5); // 24 mulhi + r5 = rotr_var(r5, r2); // 25 rotr + r3 = r2 * r7 + r3; // 26 mad + r2 = r2 - r0; // 27 sub + r6 = r1 * r5 + r6; // 28 mad + r2 = r2 + r3 + ((((sel >> 16u) & 1u) != 0u) ? 0x46b1b505u : 0xbdc6da76u); // 29 add + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r3 = r3 ^ t_; } // 30 shfl + r5 = r5 ^ ds[r7 & mask]; // 31 load + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 32 hot + r6 = r6 | r3; // 33 or + r3 = r3 ^ hot[mul_hi(r6, HOT_WORDS)]; // 34 hot + { uint t_; IGNEUM_SHFL_XOR(t_, r0, 2u); r4 = r4 ^ t_; } // 35 shfl + r5 = r5 ^ r2; // 36 xor + r3 = r3 ^ ds[r4 & mask]; // 37 load + r4 = r4 ^ ds[r3 & mask]; // 38 load + r1 = rotr_var(r1, r4); // 39 rotr + r3 = r3 + r6 + ((((sel >> 28u) & 1u) != 0u) ? 0x6328cb2cu : 0xbc3ff65fu); // 40 add + r5 = r5 ^ r1; // 41 xor + r5 = r5 | r0; // 42 or + r7 = r7 ^ r3; // 43 xor + r2 = r2 ^ ds[r5 & mask]; // 44 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r6 = r6 ^ t_; } // 45 shfl + r5 = r5 ^ r3; // 46 xor + r7 = r7 + r6 + ((((sel >> 22u) & 1u) != 0u) ? 0x9342df0du : 0x5af3bd5bu); // 47 add + r3 = r3 + r4 + ((((sel >> 23u) & 1u) != 0u) ? 0x485614dbu : 0x9ff95776u); // 48 add + r5 = r5 ^ ds[r3 & mask]; // 49 load + r4 = rotr_var(r4, r5); // 50 rotr + r0 = r1 * r7 + r0; // 51 mad + r0 = r0 * r3; // 52 mul + r5 = rotr_var(r5, r4); // 53 rotr + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r0 = r0 ^ t_; } // 54 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r7, 16u); r0 = r0 ^ t_; } // 55 shfl + r7 = r7 ^ ds[r4 & mask]; // 56 load + r7 = rotr_var(r7, r3); // 57 rotr + r0 = r0 ^ ds[r6 & mask]; // 58 load + r6 = r6 ^ hot[mul_hi(r2, HOT_WORDS)]; // 59 hot + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 1u); r3 = r3 ^ t_; } // 60 shfl + r7 = r7 - r5; // 61 sub + r1 = rotl_imm(r1, 4u); // 62 rotl + r4 = r4 ^ ds[r5 & mask]; // 63 load + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} + +#if IGNEUM_EXCHANGE != 0 +// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the +// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a +// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md. +IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) { + if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); } +} +#endif diff --git a/proto-cuda/packs-ca2-hot/hot96k4a/kernel.cu b/proto-cuda/packs-ca2-hot/hot96k4a/kernel.cu new file mode 100644 index 000000000..a3badefca --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4a/kernel.cu @@ -0,0 +1,179 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal). +// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC. +#include +#include +#include "program.h" +#include "memhard.h" + +// Hot table (96 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01800000u +__device__ __forceinline__ uint32_t splitmix32(uint32_t x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31. +__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } +// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x. +__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) { + uint32_t x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item. +// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host. +__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) { + uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x; + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) { + uint32_t t = blockIdx.x * blockDim.x + threadIdx.x; + if (t < nItems) { + uint32_t s[16]; + mh_item(cache, t, s); + uint32_t* d = ds + (size_t)t * 16u; + for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (ht_segment is in memhard.h): one thread per segment. +__global__ void igneum_hot_fill(uint32_t* hot, uint32_t nSegments) { + uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x; + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every +// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a +// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid. +__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot) { + uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x; + uint32_t nonce = baseNonce + gid; + uint32_t r0, r1, r2, r3, r4, r5, r6, r7; + { uint32_t x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint32_t x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint32_t x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint32_t x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint32_t x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint32_t x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint32_t x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint32_t x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint32_t it = 0u; it < 8u; ++it) { + uint32_t sel = r0; + r6 = r6 | r4; // 0 or + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r4, 1); // 1 shfl + r0 = r0 | r2; // 2 or + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r0, 16); // 3 shfl + r7 = r7 ^ ds[r3 & mask]; // 4 load + r6 = r6 ^ ds[r0 & mask]; // 5 load + r1 = r1 * r4; // 6 mul + r0 = r0 - r1; // 7 sub + r3 = r3 ^ ds[r7 & mask]; // 8 load + r1 = r0 * r3 + r1; // 9 mad + r4 = r4 * r0; // 10 mul + r5 = r5 ^ ds[r4 & mask]; // 11 load + r1 = r1 ^ r7; // 12 xor + r1 = r1 ^ ds[r6 & mask]; // 13 load + r2 = r2 ^ ds[r1 & mask]; // 14 load + r3 = r3 * r1; // 15 mul + r6 = r6 ^ hot[__umulhi(r5, HOT_WORDS)]; // 16 hot + r0 = r0 ^ ds[r2 & mask]; // 17 load + r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x535b545fu : 0xc035a4e6u); // 18 add + r5 = r5 - r0; // 19 sub + r2 = r2 * r4; // 20 mul + r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x925d5e63u : 0x1ecdd2cfu); // 21 add + r3 = r3 ^ r7; // 22 xor + r7 = r7 ^ ds[r2 & mask]; // 23 load + r2 = __umulhi(r2, r5); // 24 mulhi + r5 = rotr_var(r5, r2); // 25 rotr + r3 = r2 * r7 + r3; // 26 mad + r2 = r2 - r0; // 27 sub + r6 = r1 * r5 + r6; // 28 mad + r2 = r2 + r3 + ((((sel >> 16u) & 1u) != 0u) ? 0x46b1b505u : 0xbdc6da76u); // 29 add + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r2, 8); // 30 shfl + r5 = r5 ^ ds[r7 & mask]; // 31 load + r2 = r2 ^ hot[__umulhi(r0, HOT_WORDS)]; // 32 hot + r6 = r6 | r3; // 33 or + r3 = r3 ^ hot[__umulhi(r6, HOT_WORDS)]; // 34 hot + r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r0, 2); // 35 shfl + r5 = r5 ^ r2; // 36 xor + r3 = r3 ^ ds[r4 & mask]; // 37 load + r4 = r4 ^ ds[r3 & mask]; // 38 load + r1 = rotr_var(r1, r4); // 39 rotr + r3 = r3 + r6 + ((((sel >> 28u) & 1u) != 0u) ? 0x6328cb2cu : 0xbc3ff65fu); // 40 add + r5 = r5 ^ r1; // 41 xor + r5 = r5 | r0; // 42 or + r7 = r7 ^ r3; // 43 xor + r2 = r2 ^ ds[r5 & mask]; // 44 load + r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 45 shfl + r5 = r5 ^ r3; // 46 xor + r7 = r7 + r6 + ((((sel >> 22u) & 1u) != 0u) ? 0x9342df0du : 0x5af3bd5bu); // 47 add + r3 = r3 + r4 + ((((sel >> 23u) & 1u) != 0u) ? 0x485614dbu : 0x9ff95776u); // 48 add + r5 = r5 ^ ds[r3 & mask]; // 49 load + r4 = rotr_var(r4, r5); // 50 rotr + r0 = r1 * r7 + r0; // 51 mad + r0 = r0 * r3; // 52 mul + r5 = rotr_var(r5, r4); // 53 rotr + r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 1); // 54 shfl + r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r7, 16); // 55 shfl + r7 = r7 ^ ds[r4 & mask]; // 56 load + r7 = rotr_var(r7, r3); // 57 rotr + r0 = r0 ^ ds[r6 & mask]; // 58 load + r6 = r6 ^ hot[__umulhi(r2, HOT_WORDS)]; // 59 hot + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r2, 1); // 60 shfl + r7 = r7 - r5; // 61 sub + r1 = rotl_imm(r1, 4u); // 62 rotl + r4 = r4 ^ ds[r5 & mask]; // 63 load + } + uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo; +} + +// Host-side launch wrappers. Declared in program.h, called from host.cu. +cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) { + if (nSegments == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nSegments + block - 1u) / block; + igneum_cache_fill<<>>(cache, nSegments); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) { + if (nItems == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nItems + block - 1u) / block; + igneum_build<<>>(ds, cache, nItems); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments) { + if (nSegments == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nSegments + block - 1u) / block; + igneum_hot_fill<<>>(hot, nSegments); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot, + uint32_t nonces, uint32_t blockWarps) { + if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue; + uint32_t block = 32u * blockWarps; + if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue; + igneum_hash<<>>(ds, out, baseNonce, mask, hot); + return cudaGetLastError(); +} + +cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) { + cudaFuncAttributes attr; + cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash); + if (e != cudaSuccess) return e; + *numRegs = attr.numRegs; + return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0); +} diff --git a/proto-cuda/packs-ca2-hot/hot96k4a/kernel_bound.cl b/proto-cuda/packs-ca2-hot/hot96k4a/kernel_bound.cl new file mode 100644 index 000000000..be2b83a57 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4a/kernel_bound.cl @@ -0,0 +1,399 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal). +// Built from source at runtime by proto-opencl/host.c, which passes these defines: +// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit) +// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default) +// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32 +// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition +// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units; +// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32. +#ifndef IGNEUM_GROUP +#define IGNEUM_GROUP 32 +#endif +#ifndef IGNEUM_EXCHANGE +#define IGNEUM_EXCHANGE 0 +#endif +#ifdef __OPENCL_VERSION__ +#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1))) +#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n] +#if IGNEUM_EXCHANGE == 1 +#ifdef cl_khr_subgroups +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif +#ifdef cl_khr_subgroup_shuffle +#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable +#endif +#elif IGNEUM_EXCHANGE == 2 +#pragma OPENCL EXTENSION cl_intel_subgroups : enable +#endif +#else +// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros. +#include "emu_opencl.h" +#endif + +#if IGNEUM_EXCHANGE == 1 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#elif IGNEUM_EXCHANGE == 2 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#else +// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per +// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane +// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's +// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier. +#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; } +#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; } +#endif + +// Hot table (96 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01800000u +static inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32. +static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); } +// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x. +static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); } +static inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +static inline void mh_chacha_block(const uint* x, uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +static inline void mh_cache_segment(__global uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +static inline void mh_mixer(uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +static inline void mh_item(__global const uint* cache, uint t, uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + __global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item. +// The same constants as memhard.h in this pack (one emitter, three dialects). +__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) { + uint t = (uint)get_global_id(0); + if (t < nItems) { + uint s[16]; + mh_item(cache, t, s); + __global uint* d = ds + ((ulong)t * 16u); + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} +// Hot table (docs/plans/hot-table.md): 96 MiB = 24576 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +static inline void ht_segment(__global uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one work-item per segment. +__kernel void igneum_hot_fill(__global uint* hot, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) ht_segment(hot, seg); +} + +// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the +// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and +// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). +IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r6 = r6 | r4; // 0 or + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 1u); r7 = r7 ^ t_; } // 1 shfl + r0 = r0 | r2; // 2 or + { uint t_; IGNEUM_SHFL_XOR(t_, r0, 16u); r3 = r3 ^ t_; } // 3 shfl + r7 = r7 ^ ds[r3 & mask]; // 4 load + r6 = r6 ^ ds[r0 & mask]; // 5 load + r1 = r1 * r4; // 6 mul + r0 = r0 - r1; // 7 sub + r3 = r3 ^ ds[r7 & mask]; // 8 load + r1 = r0 * r3 + r1; // 9 mad + r4 = r4 * r0; // 10 mul + r5 = r5 ^ ds[r4 & mask]; // 11 load + r1 = r1 ^ r7; // 12 xor + r1 = r1 ^ ds[r6 & mask]; // 13 load + r2 = r2 ^ ds[r1 & mask]; // 14 load + r3 = r3 * r1; // 15 mul + r6 = r6 ^ hot[mul_hi(r5, HOT_WORDS)]; // 16 hot + r0 = r0 ^ ds[r2 & mask]; // 17 load + r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x535b545fu : 0xc035a4e6u); // 18 add + r5 = r5 - r0; // 19 sub + r2 = r2 * r4; // 20 mul + r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x925d5e63u : 0x1ecdd2cfu); // 21 add + r3 = r3 ^ r7; // 22 xor + r7 = r7 ^ ds[r2 & mask]; // 23 load + r2 = mul_hi(r2, r5); // 24 mulhi + r5 = rotr_var(r5, r2); // 25 rotr + r3 = r2 * r7 + r3; // 26 mad + r2 = r2 - r0; // 27 sub + r6 = r1 * r5 + r6; // 28 mad + r2 = r2 + r3 + ((((sel >> 16u) & 1u) != 0u) ? 0x46b1b505u : 0xbdc6da76u); // 29 add + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r3 = r3 ^ t_; } // 30 shfl + r5 = r5 ^ ds[r7 & mask]; // 31 load + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 32 hot + r6 = r6 | r3; // 33 or + r3 = r3 ^ hot[mul_hi(r6, HOT_WORDS)]; // 34 hot + { uint t_; IGNEUM_SHFL_XOR(t_, r0, 2u); r4 = r4 ^ t_; } // 35 shfl + r5 = r5 ^ r2; // 36 xor + r3 = r3 ^ ds[r4 & mask]; // 37 load + r4 = r4 ^ ds[r3 & mask]; // 38 load + r1 = rotr_var(r1, r4); // 39 rotr + r3 = r3 + r6 + ((((sel >> 28u) & 1u) != 0u) ? 0x6328cb2cu : 0xbc3ff65fu); // 40 add + r5 = r5 ^ r1; // 41 xor + r5 = r5 | r0; // 42 or + r7 = r7 ^ r3; // 43 xor + r2 = r2 ^ ds[r5 & mask]; // 44 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r6 = r6 ^ t_; } // 45 shfl + r5 = r5 ^ r3; // 46 xor + r7 = r7 + r6 + ((((sel >> 22u) & 1u) != 0u) ? 0x9342df0du : 0x5af3bd5bu); // 47 add + r3 = r3 + r4 + ((((sel >> 23u) & 1u) != 0u) ? 0x485614dbu : 0x9ff95776u); // 48 add + r5 = r5 ^ ds[r3 & mask]; // 49 load + r4 = rotr_var(r4, r5); // 50 rotr + r0 = r1 * r7 + r0; // 51 mad + r0 = r0 * r3; // 52 mul + r5 = rotr_var(r5, r4); // 53 rotr + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r0 = r0 ^ t_; } // 54 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r7, 16u); r0 = r0 ^ t_; } // 55 shfl + r7 = r7 ^ ds[r4 & mask]; // 56 load + r7 = rotr_var(r7, r3); // 57 rotr + r0 = r0 ^ ds[r6 & mask]; // 58 load + r6 = r6 ^ hot[mul_hi(r2, HOT_WORDS)]; // 59 hot + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 1u); r3 = r3 ^ t_; } // 60 shfl + r7 = r7 - r5; // 61 sub + r1 = rotl_imm(r1, 4u); // 62 rotl + r4 = r4 ^ ds[r5 & mask]; // 63 load + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} + +#if IGNEUM_EXCHANGE != 0 +// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the +// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a +// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md. +IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) { + if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); } +} +#endif + +// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash. +IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw, __global const uint* hot) { + uint gid = (uint)get_global_id(0); + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7]; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; } + { uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; } + { uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; } + { uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; } + { uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; } + { uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; } + { uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; } + { uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r6 = r6 | r4; // 0 or + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 1u); r7 = r7 ^ t_; } // 1 shfl + r0 = r0 | r2; // 2 or + { uint t_; IGNEUM_SHFL_XOR(t_, r0, 16u); r3 = r3 ^ t_; } // 3 shfl + r7 = r7 ^ ds[r3 & mask]; // 4 load + r6 = r6 ^ ds[r0 & mask]; // 5 load + r1 = r1 * r4; // 6 mul + r0 = r0 - r1; // 7 sub + r3 = r3 ^ ds[r7 & mask]; // 8 load + r1 = r0 * r3 + r1; // 9 mad + r4 = r4 * r0; // 10 mul + r5 = r5 ^ ds[r4 & mask]; // 11 load + r1 = r1 ^ r7; // 12 xor + r1 = r1 ^ ds[r6 & mask]; // 13 load + r2 = r2 ^ ds[r1 & mask]; // 14 load + r3 = r3 * r1; // 15 mul + r6 = r6 ^ hot[mul_hi(r5, HOT_WORDS)]; // 16 hot + r0 = r0 ^ ds[r2 & mask]; // 17 load + r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x535b545fu : 0xc035a4e6u); // 18 add + r5 = r5 - r0; // 19 sub + r2 = r2 * r4; // 20 mul + r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x925d5e63u : 0x1ecdd2cfu); // 21 add + r3 = r3 ^ r7; // 22 xor + r7 = r7 ^ ds[r2 & mask]; // 23 load + r2 = mul_hi(r2, r5); // 24 mulhi + r5 = rotr_var(r5, r2); // 25 rotr + r3 = r2 * r7 + r3; // 26 mad + r2 = r2 - r0; // 27 sub + r6 = r1 * r5 + r6; // 28 mad + r2 = r2 + r3 + ((((sel >> 16u) & 1u) != 0u) ? 0x46b1b505u : 0xbdc6da76u); // 29 add + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 8u); r3 = r3 ^ t_; } // 30 shfl + r5 = r5 ^ ds[r7 & mask]; // 31 load + r2 = r2 ^ hot[mul_hi(r0, HOT_WORDS)]; // 32 hot + r6 = r6 | r3; // 33 or + r3 = r3 ^ hot[mul_hi(r6, HOT_WORDS)]; // 34 hot + { uint t_; IGNEUM_SHFL_XOR(t_, r0, 2u); r4 = r4 ^ t_; } // 35 shfl + r5 = r5 ^ r2; // 36 xor + r3 = r3 ^ ds[r4 & mask]; // 37 load + r4 = r4 ^ ds[r3 & mask]; // 38 load + r1 = rotr_var(r1, r4); // 39 rotr + r3 = r3 + r6 + ((((sel >> 28u) & 1u) != 0u) ? 0x6328cb2cu : 0xbc3ff65fu); // 40 add + r5 = r5 ^ r1; // 41 xor + r5 = r5 | r0; // 42 or + r7 = r7 ^ r3; // 43 xor + r2 = r2 ^ ds[r5 & mask]; // 44 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r6 = r6 ^ t_; } // 45 shfl + r5 = r5 ^ r3; // 46 xor + r7 = r7 + r6 + ((((sel >> 22u) & 1u) != 0u) ? 0x9342df0du : 0x5af3bd5bu); // 47 add + r3 = r3 + r4 + ((((sel >> 23u) & 1u) != 0u) ? 0x485614dbu : 0x9ff95776u); // 48 add + r5 = r5 ^ ds[r3 & mask]; // 49 load + r4 = rotr_var(r4, r5); // 50 rotr + r0 = r1 * r7 + r0; // 51 mad + r0 = r0 * r3; // 52 mul + r5 = rotr_var(r5, r4); // 53 rotr + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r0 = r0 ^ t_; } // 54 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r7, 16u); r0 = r0 ^ t_; } // 55 shfl + r7 = r7 ^ ds[r4 & mask]; // 56 load + r7 = rotr_var(r7, r3); // 57 rotr + r0 = r0 ^ ds[r6 & mask]; // 58 load + r6 = r6 ^ hot[mul_hi(r2, HOT_WORDS)]; // 59 hot + { uint t_; IGNEUM_SHFL_XOR(t_, r2, 1u); r3 = r3 ^ t_; } // 60 shfl + r7 = r7 - r5; // 61 sub + r1 = rotl_imm(r1, 4u); // 62 rotl + r4 = r4 ^ ds[r5 & mask]; // 63 load + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot96k4a/kernel_bound.cu b/proto-cuda/packs-ca2-hot/hot96k4a/kernel_bound.cu new file mode 100644 index 000000000..dc4b0ed56 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4a/kernel_bound.cu @@ -0,0 +1,125 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW. +// Host declarations (also in program_bound.h if present): +// struct IgneumInitWords { uint32_t w[8]; }; +// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, +// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps); +// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps); +#include +#include +#include "program.h" + +struct IgneumInitWords { uint32_t w[8]; }; + +// Hot table (96 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01800000u +__device__ __forceinline__ uint32_t splitmix32(uint32_t x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } +__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } + +__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw, const uint32_t* hot) { + uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x; + uint32_t nonce = baseNonce + gid; + uint32_t r0, r1, r2, r3, r4, r5, r6, r7; + { uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; } + { uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; } + { uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; } + { uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; } + { uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; } + { uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; } + { uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; } + { uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; } + + for (uint32_t it = 0u; it < 8u; ++it) { + uint32_t sel = r0; + r6 = r6 | r4; // 0 or + r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r4, 1); // 1 shfl + r0 = r0 | r2; // 2 or + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r0, 16); // 3 shfl + r7 = r7 ^ ds[r3 & mask]; // 4 load + r6 = r6 ^ ds[r0 & mask]; // 5 load + r1 = r1 * r4; // 6 mul + r0 = r0 - r1; // 7 sub + r3 = r3 ^ ds[r7 & mask]; // 8 load + r1 = r0 * r3 + r1; // 9 mad + r4 = r4 * r0; // 10 mul + r5 = r5 ^ ds[r4 & mask]; // 11 load + r1 = r1 ^ r7; // 12 xor + r1 = r1 ^ ds[r6 & mask]; // 13 load + r2 = r2 ^ ds[r1 & mask]; // 14 load + r3 = r3 * r1; // 15 mul + r6 = r6 ^ hot[__umulhi(r5, HOT_WORDS)]; // 16 hot + r0 = r0 ^ ds[r2 & mask]; // 17 load + r0 = r0 + r7 + ((((sel >> 26u) & 1u) != 0u) ? 0x535b545fu : 0xc035a4e6u); // 18 add + r5 = r5 - r0; // 19 sub + r2 = r2 * r4; // 20 mul + r2 = r2 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x925d5e63u : 0x1ecdd2cfu); // 21 add + r3 = r3 ^ r7; // 22 xor + r7 = r7 ^ ds[r2 & mask]; // 23 load + r2 = __umulhi(r2, r5); // 24 mulhi + r5 = rotr_var(r5, r2); // 25 rotr + r3 = r2 * r7 + r3; // 26 mad + r2 = r2 - r0; // 27 sub + r6 = r1 * r5 + r6; // 28 mad + r2 = r2 + r3 + ((((sel >> 16u) & 1u) != 0u) ? 0x46b1b505u : 0xbdc6da76u); // 29 add + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r2, 8); // 30 shfl + r5 = r5 ^ ds[r7 & mask]; // 31 load + r2 = r2 ^ hot[__umulhi(r0, HOT_WORDS)]; // 32 hot + r6 = r6 | r3; // 33 or + r3 = r3 ^ hot[__umulhi(r6, HOT_WORDS)]; // 34 hot + r4 = r4 ^ __shfl_xor_sync(0xffffffffu, r0, 2); // 35 shfl + r5 = r5 ^ r2; // 36 xor + r3 = r3 ^ ds[r4 & mask]; // 37 load + r4 = r4 ^ ds[r3 & mask]; // 38 load + r1 = rotr_var(r1, r4); // 39 rotr + r3 = r3 + r6 + ((((sel >> 28u) & 1u) != 0u) ? 0x6328cb2cu : 0xbc3ff65fu); // 40 add + r5 = r5 ^ r1; // 41 xor + r5 = r5 | r0; // 42 or + r7 = r7 ^ r3; // 43 xor + r2 = r2 ^ ds[r5 & mask]; // 44 load + r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 45 shfl + r5 = r5 ^ r3; // 46 xor + r7 = r7 + r6 + ((((sel >> 22u) & 1u) != 0u) ? 0x9342df0du : 0x5af3bd5bu); // 47 add + r3 = r3 + r4 + ((((sel >> 23u) & 1u) != 0u) ? 0x485614dbu : 0x9ff95776u); // 48 add + r5 = r5 ^ ds[r3 & mask]; // 49 load + r4 = rotr_var(r4, r5); // 50 rotr + r0 = r1 * r7 + r0; // 51 mad + r0 = r0 * r3; // 52 mul + r5 = rotr_var(r5, r4); // 53 rotr + r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 1); // 54 shfl + r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r7, 16); // 55 shfl + r7 = r7 ^ ds[r4 & mask]; // 56 load + r7 = rotr_var(r7, r3); // 57 rotr + r0 = r0 ^ ds[r6 & mask]; // 58 load + r6 = r6 ^ hot[__umulhi(r2, HOT_WORDS)]; // 59 hot + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r2, 1); // 60 shfl + r7 = r7 - r5; // 61 sub + r1 = rotl_imm(r1, 4u); // 62 rotl + r4 = r4 ^ ds[r5 & mask]; // 63 load + } + uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo; +} + +cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, + IgneumInitWords iw, const uint32_t* hot, uint32_t nonces, uint32_t blockWarps) { + if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue; + uint32_t block = 32u * blockWarps; + if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue; + igneum_hash_bound<<>>(ds, out, baseNonce, mask, iw, hot); + return cudaGetLastError(); +} + +cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) { + cudaFuncAttributes attr; + cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound); + if (e != cudaSuccess) return e; + *numRegs = attr.numRegs; + return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0); +} diff --git a/proto-cuda/packs-ca2-hot/hot96k4a/memhard.h b/proto-cuda/packs-ca2-hot/hot96k4a/memhard.h new file mode 100644 index 000000000..8691d33e2 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4a/memhard.h @@ -0,0 +1,129 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against. +// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference). +// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C. +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif +#if defined(__CUDACC__) +#define IGNEUM_HD __host__ __device__ __forceinline__ +#elif defined(_MSC_VER) && !defined(__cplusplus) +#define IGNEUM_HD static __inline +#else +#define IGNEUM_HD static inline +#endif +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) { + for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint32_t r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) { + uint32_t prev[16]; uint32_t x[16]; uint32_t y[16]; + for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +IGNEUM_HD void mh_item(const uint32_t* cache, uint32_t t, uint32_t* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint32_t r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +IGNEUM_HD uint32_t mh_word(const uint32_t* cache, uint32_t w) { uint32_t s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Hot table (docs/plans/hot-table.md): 96 MiB = 24576 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +IGNEUM_HD void ht_segment(uint32_t* hot, uint32_t seg) { + uint32_t prev[16]; uint32_t x[16]; uint32_t y[16]; + for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + uint32_t* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} diff --git a/proto-cuda/packs-ca2-hot/hot96k4a/memhard.metal b/proto-cuda/packs-ca2-hot/hot96k4a/memhard.metal new file mode 100644 index 000000000..0acb2b7e1 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4a/memhard.metal @@ -0,0 +1,131 @@ +#include +using namespace metal; +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +inline void mh_chacha_block(const thread uint* x, thread uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +inline void mh_cache_segment(device uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +inline void mh_mixer(thread uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +inline void mh_item(device const uint* cache, uint t, thread uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +inline uint mh_word(device const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// One thread per segment (2^16 threads). +kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) { + mh_cache_segment(cache, gid); +} +// One thread per 64-byte item (dataset words / 16 threads). +kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]], + uint gid [[thread_position_in_grid]]) { + uint s[16]; + mh_item(cache, gid, s); + device uint* d = dataset + gid * 16u; + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; +} + +// Hot table (docs/plans/hot-table.md): 96 MiB = 24576 segments of 64 chained ChaCha12 lines under the hot key KH = seed_words("igneum-hot/" || epoch seed bytes), tag "Igne" "umHT". The cache chain with another key and tag. +inline void ht_segment(device uint* hot, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3a48bef5u ^ prev[4]; + x[5] = 0x6b54b1a1u ^ prev[5]; + x[6] = 0x9ff897c3u ^ prev[6]; + x[7] = 0x7d5d85e4u ^ prev[7]; + x[8] = 0xcd35379fu ^ prev[8]; + x[9] = 0x9d76bb86u ^ prev[9]; + x[10] = 0xbe2affb1u ^ prev[10]; + x[11] = 0x0f8f80b4u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4854u ^ prev[15]; + mh_chacha_block(x, y); + device uint* line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} +// Hot table: one thread per segment (IGNEUM_HOT_SEGMENTS threads). +kernel void igneum_hot_fill(device uint* hot [[buffer(0)]], uint gid [[thread_position_in_grid]]) { + ht_segment(hot, gid); +} diff --git a/proto-cuda/packs-ca2-hot/hot96k4a/program.h b/proto-cuda/packs-ca2-hot/hot96k4a/program.h new file mode 100644 index 000000000..3b39a0235 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4a/program.h @@ -0,0 +1,70 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Program metadata for host.cu plus the launch wrappers defined in kernel.cu. +// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros. +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif +#ifndef IGNEUM_NO_CUDA +#include +#endif + +#define IGNEUM_SEED_STRING "igneum-genesis" +#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973" +#define IGNEUM_GENERATOR 2 +#define IGNEUM_PROGRAM_ATTEMPT 0 +#define IGNEUM_PROGRAM_ID 0x2ff8e52c952a62c2ull +#define IGNEUM_DAY_STRING "2026-10-03" +#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033" +#define IGNEUM_DAY0 0x3067619fu +#define IGNEUM_DAY1 0x3c269176u +#define IGNEUM_DATASET_LOG2 28 +#define IGNEUM_MASK 0x0fffffffu +#define IGNEUM_LANES 32 +#define IGNEUM_ITERATIONS 8 +#define IGNEUM_INSTR_COUNT 64 +#define IGNEUM_LOADS_PER_HASH 160 +#define IGNEUM_WIDE_LOADS_PER_HASH 0 +#define IGNEUM_OP_MIX "load=16 shfl=8 add=6 xor=6 mul=5 rotr=5 hot=4 mad=4 or=4 sub=4 mulhi=1 rotl=1" +// Class v3 construction (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): version 2 loads; the dataset item +// derivation applies the mixer IGNEUM_MIXER_MULT times per round (memhard.h), and the cache follows the growth rule. +#define IGNEUM_LOAD_CLASS "hot96k4a" +#define IGNEUM_LOAD_SLOTS 20 +#define IGNEUM_LOAD_MIX { 100, 0, 0 } +#define IGNEUM_LOAD_WIDTH_COUNTS { 16, 0, 0 } // loads of 4, 16, 64 bytes per program +#define IGNEUM_BYTES_PER_HASH 512 +#define IGNEUM_FOLD_ROT 11 +#define IGNEUM_FOLD_MUL 0x9e3779b1u +// Hot table (hot-table experiment, 5 October 2026, docs/plans/hot-table.md): NOT the lottery hash. IGNEUM_HOT_SLOTS of the +// 16 load slots read H[mulhi(src, IGNEUM_HOT_WORDS)], H = IGNEUM_HOT_MB MiB of chained ChaCha12 lines (the cache chain) under +// KH = seed_words("igneum-hot/" || epoch seed bytes) and the tag "Igne" "umHT"; filled by igneum_hot_fill once per epoch. +#define IGNEUM_HOT_MB 96 +#define IGNEUM_HOT_WORDS 0x01800000u +#define IGNEUM_HOT_SEGMENTS 24576u +#define IGNEUM_HOT_SLOTS 4 // hot loads per program (32 per hash), added beside the dataset loads (16 of them) +#define IGNEUM_HOT_ADDED 1 +#define IGNEUM_HOT_KEY_INIT { 0x3a48bef5u, 0x6b54b1a1u, 0x9ff897c3u, 0x7d5d85e4u, 0xcd35379fu, 0x9d76bb86u, 0xbe2affb1u, 0x0f8f80b4u } +// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h) +#define IGNEUM_DATASET_MODE 1 + +#define IGNEUM_SEEDW_INIT { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u } +#define IGNEUM_KEY_INIT { 0x3067619fu, 0x3c269176u, 0x84a03b03u, 0xf8c63294u, 0xff977c5bu, 0xe60def3eu, 0x63630141u, 0xb8fbcb58u } +#define IGNEUM_CACHE_LOG2_WORDS 26 +#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6 +#define IGNEUM_CACHE_SEGMENTS 65536u +#define IGNEUM_ITEM_ROUNDS 8 +#define IGNEUM_MIX_ROT_INIT { 20u, 20u, 19u, 4u, 26u, 3u, 3u, 27u } +#define IGNEUM_MIX_MUL_INIT { 0x42146205u, 0x52cbe0fbu, 0x7ecf4a03u, 0x6728907fu, 0xd81d9751u, 0x132952c3u, 0xf60de277u, 0x05358035u, 0xbaf6499du, 0xe4db9667u, 0x3e98f45du, 0xd0004eddu, 0x2691630du, 0x9beb3bcfu, 0xab310379u, 0x99cfb423u } +#define IGNEUM_MIX_RC_INIT { 0xbab68293u, 0xcc162340u, 0x6ce151ccu, 0xe62b8997u, 0xc9c80297u, 0xf74a1654u, 0x3d704af5u, 0x3cf522b7u, 0x2b9cac04u, 0xa880ac10u, 0x13e5dd1du, 0x6fc3e233u, 0x2d83eeacu, 0x9006e8bfu, 0x2c4b5362u, 0x31b49ee2u } + +#ifndef IGNEUM_NO_CUDA +// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError(). +cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments); +cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems); +cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments); +cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, const uint32_t* hot, + uint32_t nonces, uint32_t blockWarps); +cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps); +#endif diff --git a/proto-cuda/packs-ca2-hot/hot96k4a/program.json b/proto-cuda/packs-ca2-hot/hot96k4a/program.json new file mode 100644 index 000000000..a11407837 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4a/program.json @@ -0,0 +1,129 @@ +{ + "format": "igneum-program-pack-3", + "generator": 2, + "attempt": 0, + "program_id": "0x2ff8e52c952a62c2", + "program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32", + "dataset_mode": "memory-hard", + "seed": "igneum-genesis", + "seed_bytes": "69676e65756d2d67656e65736973", + "seed_words": ["0x67a9a7be", "0x1a155b25", "0xfddfb732", "0x4b5af2e8", "0xc55caf33", "0xa27c13b7", "0x06628a48", "0x03852469"], + "seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32", + "generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried", + "lanes": 32, + "registers": 8, + "iterations": 8, + "instruction_count": 64, + "loads_per_hash": 160, + "load_class": "hot96k4a", + "load_slots": 20, + "load_mix_percent_4_16_64": [100, 0, 0], + "load_width_counts_4_16_64": [16, 0, 0], + "bytes_per_hash": 512, + "wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots", + "hot_table": {"mb": 96, "words": 25165824, "segments": 24576, "slots": 4, "form": "added: k load slots added beside the class's, the dataset loads unchanged", "dataset_slots": 16, "hot_loads_per_hash": 32, "key": ["0x3a48bef5", "0x6b54b1a1", "0x9ff897c3", "0x7d5d85e4", "0xcd35379f", "0x9d76bb86", "0xbe2affb1", "0x0f8f80b4"], "key_derivation": "seed_words_from_bytes('igneum-hot/' || seed_bytes), the same for every attempt of the epoch", "tag": ["0x49676e65", "0x756d4854"], "chain": "the cache chain of dataset.cache with the hot key and tag: in_j = prev ^ (sigma || key || seg || j || tag); line_j = block(in_j); prev_0 = 0", "load": "dst = dst ^ hot[mulhi(src, words)] (the high 32 bits of the 64-bit product; the index lies in [0, words) for any size)", "slots_rule": "the first k load slots in the Fisher-Yates draw order after the scratch slots (a uniform k-subset); no extra draw, so with version 2 widths and no scratch the program is the version 2 program with k loads redirected", "acceptance_stand_in": "dataset_elem(mulhi(src, words), seed_words[2], seed_words[3])", "program_id": "the read-width id with 'hot/' || mb || k appended", "spec": "docs/plans/hot-table.md"}, + "op_mix": {"load": 16, "shfl": 8, "add": 6, "xor": 6, "mul": 5, "rotr": 5, "hot": 4, "mad": 4, "or": 4, "sub": 4, "mulhi": 1, "rotl": 1}, + "register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]", + "splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16", + "iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order", + "output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo", + "op_semantics": { + "add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)", + "sub": "dst = dst - src", + "mul": "dst = dst * src (low 32)", + "mulhi": "dst = high 32 bits of dst * src", + "xor": "dst = dst ^ src", + "or": "dst = dst | src", + "rotl": "dst = rotl(dst, rot), rot in 1..31", + "rotr": "dst = rotr(dst, src & 31)", + "mad": "dst = src * src2 + dst", + "shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp", + "load": "dst = dst ^ dataset[src & dataset.mask]", + "wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)", + "hot": "dst = dst ^ hot[mulhi(src, hot_table.words)] (hot-table experiment)" + }, + "dataset": { + "log2_words": 28, + "bytes": 1073741824, + "mask": "0x0fffffff", + "day": "2026-10-03", + "day_bytes": "6461792f323032362d31302d3033", + "day_words_from": "seed_words_from_bytes(day_bytes)", + "d0": "0x3067619f", + "d1": "0x3c269176", + "mode": "memory-hard", + "spec": "proto-metal/MEMHARD.md", + "key": ["0x3067619f", "0x3c269176", "0x84a03b03", "0xf8c63294", "0xff977c5b", "0xe60def3e", "0x63630141", "0xb8fbcb58"], + "key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]", + "cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"}, + "mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [20, 20, 19, 4, 26, 3, 3, 27], "mul": ["0x42146205", "0x52cbe0fb", "0x7ecf4a03", "0x6728907f", "0xd81d9751", "0x132952c3", "0xf60de277", "0x05358035", "0xbaf6499d", "0xe4db9667", "0x3e98f45d", "0xd0004edd", "0x2691630d", "0x9beb3bcf", "0xab310379", "0x99cfb423"], "rc": ["0xbab68293", "0xcc162340", "0x6ce151cc", "0xe62b8997", "0xc9c80297", "0xf74a1654", "0x3d704af5", "0x3cf522b7", "0x2b9cac04", "0xa880ac10", "0x13e5dd1d", "0x6fc3e233", "0x2d83eeac", "0x9006e8bf", "0x2c4b5362", "0x31b49ee2"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"}, + "item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: s = M_r(s); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then s = M_8(s); item(t) = s", + "word": "dataset[w] = item(w >> 4)[w & 15]" + }, + "instructions": [ + {"i": 0, "op": "or", "dst": 6, "src": 4, "src2": 2, "imm": "0x535c5dd3", "imm2": "0xf8694615", "rot": 3, "bit": 1, "mask": 16, "width": 1}, + {"i": 1, "op": "shfl", "dst": 7, "src": 4, "src2": 6, "imm": "0x87bc9ee4", "imm2": "0xfdb0c856", "rot": 9, "bit": 25, "mask": 1, "width": 1}, + {"i": 2, "op": "or", "dst": 0, "src": 2, "src2": 4, "imm": "0xaf91f0c8", "imm2": "0xe2d5d1fa", "rot": 1, "bit": 24, "mask": 2, "width": 1}, + {"i": 3, "op": "shfl", "dst": 3, "src": 0, "src2": 0, "imm": "0x3044ba32", "imm2": "0x7b1a7ffe", "rot": 14, "bit": 10, "mask": 16, "width": 1}, + {"i": 4, "op": "load", "dst": 7, "src": 3, "src2": 6, "imm": "0x5d1ca2a2", "imm2": "0xe2481807", "rot": 24, "bit": 3, "mask": 1, "width": 1}, + {"i": 5, "op": "load", "dst": 6, "src": 0, "src2": 6, "imm": "0xaba3dbaa", "imm2": "0x987c017a", "rot": 16, "bit": 30, "mask": 1, "width": 1}, + {"i": 6, "op": "mul", "dst": 1, "src": 4, "src2": 6, "imm": "0x12a93d05", "imm2": "0x10761047", "rot": 17, "bit": 17, "mask": 4, "width": 1}, + {"i": 7, "op": "sub", "dst": 0, "src": 1, "src2": 0, "imm": "0x8884e389", "imm2": "0x31009b67", "rot": 5, "bit": 30, "mask": 1, "width": 1}, + {"i": 8, "op": "load", "dst": 3, "src": 7, "src2": 5, "imm": "0xfa6358b3", "imm2": "0xfea9038f", "rot": 17, "bit": 18, "mask": 2, "width": 1}, + {"i": 9, "op": "mad", "dst": 1, "src": 0, "src2": 3, "imm": "0xb154ae4f", "imm2": "0xe85f13f6", "rot": 6, "bit": 18, "mask": 16, "width": 1}, + {"i": 10, "op": "mul", "dst": 4, "src": 0, "src2": 4, "imm": "0x9edffbc3", "imm2": "0x643812d4", "rot": 24, "bit": 25, "mask": 4, "width": 1}, + {"i": 11, "op": "load", "dst": 5, "src": 4, "src2": 1, "imm": "0x163dbb1b", "imm2": "0x20c1f743", "rot": 27, "bit": 25, "mask": 16, "width": 1}, + {"i": 12, "op": "xor", "dst": 1, "src": 7, "src2": 4, "imm": "0x9cadb4f8", "imm2": "0xd9da785b", "rot": 8, "bit": 26, "mask": 4, "width": 1}, + {"i": 13, "op": "load", "dst": 1, "src": 6, "src2": 6, "imm": "0x1bb1b429", "imm2": "0x33d1e391", "rot": 20, "bit": 19, "mask": 16, "width": 1}, + {"i": 14, "op": "load", "dst": 2, "src": 1, "src2": 6, "imm": "0xa672cdd3", "imm2": "0x59a4829c", "rot": 22, "bit": 13, "mask": 16, "width": 1}, + {"i": 15, "op": "mul", "dst": 3, "src": 1, "src2": 4, "imm": "0xe3c4bf4d", "imm2": "0x028b4d37", "rot": 11, "bit": 28, "mask": 8, "width": 1}, + {"i": 16, "op": "hot", "dst": 6, "src": 5, "src2": 7, "imm": "0x71fbe6f2", "imm2": "0x61b6261e", "rot": 15, "bit": 9, "mask": 1, "width": 1}, + {"i": 17, "op": "load", "dst": 0, "src": 2, "src2": 5, "imm": "0x8e516fd3", "imm2": "0x2aa480c2", "rot": 19, "bit": 1, "mask": 8, "width": 1}, + {"i": 18, "op": "add", "dst": 0, "src": 7, "src2": 6, "imm": "0xc035a4e6", "imm2": "0x535b545f", "rot": 18, "bit": 26, "mask": 8, "width": 1}, + {"i": 19, "op": "sub", "dst": 5, "src": 0, "src2": 2, "imm": "0xde38f954", "imm2": "0xde1507bc", "rot": 31, "bit": 21, "mask": 4, "width": 1}, + {"i": 20, "op": "mul", "dst": 2, "src": 4, "src2": 1, "imm": "0x3f64b7c9", "imm2": "0x1f07754a", "rot": 23, "bit": 28, "mask": 4, "width": 1}, + {"i": 21, "op": "add", "dst": 2, "src": 1, "src2": 1, "imm": "0x1ecdd2cf", "imm2": "0x925d5e63", "rot": 6, "bit": 6, "mask": 8, "width": 1}, + {"i": 22, "op": "xor", "dst": 3, "src": 7, "src2": 3, "imm": "0x63c7f533", "imm2": "0xd312164d", "rot": 4, "bit": 14, "mask": 2, "width": 1}, + {"i": 23, "op": "load", "dst": 7, "src": 2, "src2": 5, "imm": "0x6d3ddc5a", "imm2": "0x433d8c2a", "rot": 28, "bit": 7, "mask": 2, "width": 1}, + {"i": 24, "op": "mulhi", "dst": 2, "src": 5, "src2": 0, "imm": "0xc7e9887a", "imm2": "0x19ec898f", "rot": 14, "bit": 9, "mask": 1, "width": 1}, + {"i": 25, "op": "rotr", "dst": 5, "src": 2, "src2": 3, "imm": "0x97acf9f2", "imm2": "0xc7fcfc8f", "rot": 8, "bit": 25, "mask": 1, "width": 1}, + {"i": 26, "op": "mad", "dst": 3, "src": 2, "src2": 7, "imm": "0x400383b6", "imm2": "0xe9bab735", "rot": 16, "bit": 18, "mask": 1, "width": 1}, + {"i": 27, "op": "sub", "dst": 2, "src": 0, "src2": 7, "imm": "0x8ea6cd8d", "imm2": "0x55b67f9f", "rot": 19, "bit": 8, "mask": 8, "width": 1}, + {"i": 28, "op": "mad", "dst": 6, "src": 1, "src2": 5, "imm": "0x17723e5a", "imm2": "0x00779664", "rot": 19, "bit": 2, "mask": 4, "width": 1}, + {"i": 29, "op": "add", "dst": 2, "src": 3, "src2": 4, "imm": "0xbdc6da76", "imm2": "0x46b1b505", "rot": 28, "bit": 16, "mask": 16, "width": 1}, + {"i": 30, "op": "shfl", "dst": 3, "src": 2, "src2": 3, "imm": "0xe586702c", "imm2": "0xd83a1462", "rot": 24, "bit": 6, "mask": 8, "width": 1}, + {"i": 31, "op": "load", "dst": 5, "src": 7, "src2": 1, "imm": "0x1cddad42", "imm2": "0xb0c6b0d0", "rot": 25, "bit": 25, "mask": 8, "width": 1}, + {"i": 32, "op": "hot", "dst": 2, "src": 0, "src2": 0, "imm": "0x2abcfdc6", "imm2": "0xfa4cc809", "rot": 22, "bit": 24, "mask": 2, "width": 1}, + {"i": 33, "op": "or", "dst": 6, "src": 3, "src2": 1, "imm": "0x61c9a38d", "imm2": "0x86597500", "rot": 15, "bit": 8, "mask": 2, "width": 1}, + {"i": 34, "op": "hot", "dst": 3, "src": 6, "src2": 7, "imm": "0xc37723fa", "imm2": "0xf3b024da", "rot": 16, "bit": 27, "mask": 16, "width": 1}, + {"i": 35, "op": "shfl", "dst": 4, "src": 0, "src2": 5, "imm": "0x84cad367", "imm2": "0xcc7972c4", "rot": 21, "bit": 17, "mask": 2, "width": 1}, + {"i": 36, "op": "xor", "dst": 5, "src": 2, "src2": 0, "imm": "0xc24a7d70", "imm2": "0x6591d24c", "rot": 21, "bit": 24, "mask": 1, "width": 1}, + {"i": 37, "op": "load", "dst": 3, "src": 4, "src2": 5, "imm": "0xd404cfe8", "imm2": "0x144ca538", "rot": 10, "bit": 11, "mask": 2, "width": 1}, + {"i": 38, "op": "load", "dst": 4, "src": 3, "src2": 3, "imm": "0x04b0080c", "imm2": "0xb938c290", "rot": 10, "bit": 1, "mask": 8, "width": 1}, + {"i": 39, "op": "rotr", "dst": 1, "src": 4, "src2": 5, "imm": "0x9fe93344", "imm2": "0xff7296e4", "rot": 29, "bit": 14, "mask": 1, "width": 1}, + {"i": 40, "op": "add", "dst": 3, "src": 6, "src2": 2, "imm": "0xbc3ff65f", "imm2": "0x6328cb2c", "rot": 4, "bit": 28, "mask": 16, "width": 1}, + {"i": 41, "op": "xor", "dst": 5, "src": 1, "src2": 1, "imm": "0xdcf4a02e", "imm2": "0x5ee3a976", "rot": 17, "bit": 0, "mask": 8, "width": 1}, + {"i": 42, "op": "or", "dst": 5, "src": 0, "src2": 1, "imm": "0x44846c7a", "imm2": "0x67338877", "rot": 27, "bit": 18, "mask": 1, "width": 1}, + {"i": 43, "op": "xor", "dst": 7, "src": 3, "src2": 3, "imm": "0x1b053acf", "imm2": "0x32e2d23d", "rot": 31, "bit": 23, "mask": 8, "width": 1}, + {"i": 44, "op": "load", "dst": 2, "src": 5, "src2": 4, "imm": "0xf8662282", "imm2": "0x10bb9e30", "rot": 8, "bit": 6, "mask": 2, "width": 1}, + {"i": 45, "op": "shfl", "dst": 6, "src": 4, "src2": 4, "imm": "0x87d9ef84", "imm2": "0x49087d74", "rot": 1, "bit": 2, "mask": 8, "width": 1}, + {"i": 46, "op": "xor", "dst": 5, "src": 3, "src2": 1, "imm": "0x74a59b7d", "imm2": "0xc4766ff1", "rot": 30, "bit": 3, "mask": 16, "width": 1}, + {"i": 47, "op": "add", "dst": 7, "src": 6, "src2": 2, "imm": "0x5af3bd5b", "imm2": "0x9342df0d", "rot": 26, "bit": 22, "mask": 2, "width": 1}, + {"i": 48, "op": "add", "dst": 3, "src": 4, "src2": 5, "imm": "0x9ff95776", "imm2": "0x485614db", "rot": 27, "bit": 23, "mask": 1, "width": 1}, + {"i": 49, "op": "load", "dst": 5, "src": 3, "src2": 1, "imm": "0xd022a812", "imm2": "0xfbe147d6", "rot": 15, "bit": 29, "mask": 4, "width": 1}, + {"i": 50, "op": "rotr", "dst": 4, "src": 5, "src2": 7, "imm": "0x7e099325", "imm2": "0x40d55d10", "rot": 29, "bit": 15, "mask": 16, "width": 1}, + {"i": 51, "op": "mad", "dst": 0, "src": 1, "src2": 7, "imm": "0xd8ab8843", "imm2": "0x7f842b90", "rot": 16, "bit": 9, "mask": 1, "width": 1}, + {"i": 52, "op": "mul", "dst": 0, "src": 3, "src2": 6, "imm": "0x4c30250b", "imm2": "0x9ee1681f", "rot": 20, "bit": 3, "mask": 8, "width": 1}, + {"i": 53, "op": "rotr", "dst": 5, "src": 4, "src2": 4, "imm": "0xc1c15026", "imm2": "0x588915e7", "rot": 1, "bit": 29, "mask": 16, "width": 1}, + {"i": 54, "op": "shfl", "dst": 0, "src": 6, "src2": 4, "imm": "0x636a9dc4", "imm2": "0xac023d9b", "rot": 22, "bit": 29, "mask": 1, "width": 1}, + {"i": 55, "op": "shfl", "dst": 0, "src": 7, "src2": 1, "imm": "0x5c933bd0", "imm2": "0x2baec8c9", "rot": 25, "bit": 27, "mask": 16, "width": 1}, + {"i": 56, "op": "load", "dst": 7, "src": 4, "src2": 7, "imm": "0xf62515d5", "imm2": "0x4a164f9f", "rot": 28, "bit": 30, "mask": 2, "width": 1}, + {"i": 57, "op": "rotr", "dst": 7, "src": 3, "src2": 7, "imm": "0x15ef64da", "imm2": "0x7149e3c9", "rot": 28, "bit": 30, "mask": 2, "width": 1}, + {"i": 58, "op": "load", "dst": 0, "src": 6, "src2": 2, "imm": "0xcde7100c", "imm2": "0x040fc0cf", "rot": 8, "bit": 28, "mask": 2, "width": 1}, + {"i": 59, "op": "hot", "dst": 6, "src": 2, "src2": 1, "imm": "0xcf8b48be", "imm2": "0x14cb1d3f", "rot": 16, "bit": 19, "mask": 1, "width": 1}, + {"i": 60, "op": "shfl", "dst": 3, "src": 2, "src2": 1, "imm": "0xae1529ee", "imm2": "0x732bc114", "rot": 4, "bit": 23, "mask": 1, "width": 1}, + {"i": 61, "op": "sub", "dst": 7, "src": 5, "src2": 7, "imm": "0x0a8f8168", "imm2": "0xf1c9b0ed", "rot": 29, "bit": 2, "mask": 8, "width": 1}, + {"i": 62, "op": "rotl", "dst": 1, "src": 2, "src2": 3, "imm": "0x9b7914dd", "imm2": "0xd14b33a3", "rot": 4, "bit": 16, "mask": 16, "width": 1}, + {"i": 63, "op": "load", "dst": 4, "src": 5, "src2": 1, "imm": "0xbbaa8e24", "imm2": "0xd15ec6a6", "rot": 24, "bit": 2, "mask": 2, "width": 1} + ] +} diff --git a/proto-cuda/packs-ca2-hot/hot96k4a/program.metal b/proto-cuda/packs-ca2-hot/hot96k4a/program.metal new file mode 100644 index 000000000..ea4b66cbe --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4a/program.metal @@ -0,0 +1,112 @@ +#include +using namespace metal; + +#define MASK 0x0fffffffu +// Hot table (96 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01800000u +constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }; + +inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 +inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +kernel void igneum_hash(device const uint* dataset [[buffer(0)]], + device ulong* out [[buffer(1)]], + constant uint& baseNonce [[buffer(2)]], + device const uint* hot [[buffer(3)]], + uint gid [[thread_position_in_grid]]) { + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + { uint x = nonce ^ SEEDW[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ SEEDW[1]; } + { uint x = nonce ^ SEEDW[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ SEEDW[2]; } + { uint x = nonce ^ SEEDW[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ SEEDW[3]; } + { uint x = nonce ^ SEEDW[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ SEEDW[4]; } + { uint x = nonce ^ SEEDW[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ SEEDW[5]; } + { uint x = nonce ^ SEEDW[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ SEEDW[6]; } + { uint x = nonce ^ SEEDW[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ SEEDW[7]; } + { uint x = nonce ^ SEEDW[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ SEEDW[0]; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r6 = r6 | r4; // 0 + r7 = r7 ^ simd_shuffle_xor(r4, (ushort)1); // 1 + r0 = r0 | r2; // 2 + r3 = r3 ^ simd_shuffle_xor(r0, (ushort)16); // 3 + r7 = r7 ^ dataset[r3 & MASK]; // 4 + r6 = r6 ^ dataset[r0 & MASK]; // 5 + r1 = r1 * r4; // 6 + r0 = r0 - r1; // 7 + r3 = r3 ^ dataset[r7 & MASK]; // 8 + r1 = r0 * r3 + r1; // 9 + r4 = r4 * r0; // 10 + r5 = r5 ^ dataset[r4 & MASK]; // 11 + r1 = r1 ^ r7; // 12 + r1 = r1 ^ dataset[r6 & MASK]; // 13 + r2 = r2 ^ dataset[r1 & MASK]; // 14 + r3 = r3 * r1; // 15 + r6 = r6 ^ hot[mulhi(r5, HOT_WORDS)]; // 16 + r0 = r0 ^ dataset[r2 & MASK]; // 17 + r0 = r0 + r7 + select(0xc035a4e6u, 0x535b545fu, ((sel >> 26u) & 1u) != 0u); // 18 + r5 = r5 - r0; // 19 + r2 = r2 * r4; // 20 + r2 = r2 + r1 + select(0x1ecdd2cfu, 0x925d5e63u, ((sel >> 6u) & 1u) != 0u); // 21 + r3 = r3 ^ r7; // 22 + r7 = r7 ^ dataset[r2 & MASK]; // 23 + r2 = mulhi(r2, r5); // 24 + r5 = rotr_var(r5, r2); // 25 + r3 = r2 * r7 + r3; // 26 + r2 = r2 - r0; // 27 + r6 = r1 * r5 + r6; // 28 + r2 = r2 + r3 + select(0xbdc6da76u, 0x46b1b505u, ((sel >> 16u) & 1u) != 0u); // 29 + r3 = r3 ^ simd_shuffle_xor(r2, (ushort)8); // 30 + r5 = r5 ^ dataset[r7 & MASK]; // 31 + r2 = r2 ^ hot[mulhi(r0, HOT_WORDS)]; // 32 + r6 = r6 | r3; // 33 + r3 = r3 ^ hot[mulhi(r6, HOT_WORDS)]; // 34 + r4 = r4 ^ simd_shuffle_xor(r0, (ushort)2); // 35 + r5 = r5 ^ r2; // 36 + r3 = r3 ^ dataset[r4 & MASK]; // 37 + r4 = r4 ^ dataset[r3 & MASK]; // 38 + r1 = rotr_var(r1, r4); // 39 + r3 = r3 + r6 + select(0xbc3ff65fu, 0x6328cb2cu, ((sel >> 28u) & 1u) != 0u); // 40 + r5 = r5 ^ r1; // 41 + r5 = r5 | r0; // 42 + r7 = r7 ^ r3; // 43 + r2 = r2 ^ dataset[r5 & MASK]; // 44 + r6 = r6 ^ simd_shuffle_xor(r4, (ushort)8); // 45 + r5 = r5 ^ r3; // 46 + r7 = r7 + r6 + select(0x5af3bd5bu, 0x9342df0du, ((sel >> 22u) & 1u) != 0u); // 47 + r3 = r3 + r4 + select(0x9ff95776u, 0x485614dbu, ((sel >> 23u) & 1u) != 0u); // 48 + r5 = r5 ^ dataset[r3 & MASK]; // 49 + r4 = rotr_var(r4, r5); // 50 + r0 = r1 * r7 + r0; // 51 + r0 = r0 * r3; // 52 + r5 = rotr_var(r5, r4); // 53 + r0 = r0 ^ simd_shuffle_xor(r6, (ushort)1); // 54 + r0 = r0 ^ simd_shuffle_xor(r7, (ushort)16); // 55 + r7 = r7 ^ dataset[r4 & MASK]; // 56 + r7 = rotr_var(r7, r3); // 57 + r0 = r0 ^ dataset[r6 & MASK]; // 58 + r6 = r6 ^ hot[mulhi(r2, HOT_WORDS)]; // 59 + r3 = r3 ^ simd_shuffle_xor(r2, (ushort)1); // 60 + r7 = r7 - r5; // 61 + r1 = rotl_imm(r1, 4u); // 62 + r4 = r4 ^ dataset[r5 & MASK]; // 63 + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot96k4a/program_bound.metal b/proto-cuda/packs-ca2-hot/hot96k4a/program_bound.metal new file mode 100644 index 000000000..ca471ed1b --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4a/program_bound.metal @@ -0,0 +1,114 @@ +#include +using namespace metal; + +#define MASK 0x0fffffffu +// Hot table (96 MiB, 4 of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS 0x01800000u +constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }; + +inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 +inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW. +kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], + device ulong* out [[buffer(1)]], + constant uint& baseNonce [[buffer(2)]], + constant uint* initw [[buffer(3)]], + device const uint* hot [[buffer(4)]], + uint gid [[thread_position_in_grid]]) { + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + { uint x = nonce ^ initw[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ initw[1]; } + { uint x = nonce ^ initw[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ initw[2]; } + { uint x = nonce ^ initw[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ initw[3]; } + { uint x = nonce ^ initw[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ initw[4]; } + { uint x = nonce ^ initw[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ initw[5]; } + { uint x = nonce ^ initw[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ initw[6]; } + { uint x = nonce ^ initw[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ initw[7]; } + { uint x = nonce ^ initw[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ initw[0]; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r6 = r6 | r4; // 0 + r7 = r7 ^ simd_shuffle_xor(r4, (ushort)1); // 1 + r0 = r0 | r2; // 2 + r3 = r3 ^ simd_shuffle_xor(r0, (ushort)16); // 3 + r7 = r7 ^ dataset[r3 & MASK]; // 4 + r6 = r6 ^ dataset[r0 & MASK]; // 5 + r1 = r1 * r4; // 6 + r0 = r0 - r1; // 7 + r3 = r3 ^ dataset[r7 & MASK]; // 8 + r1 = r0 * r3 + r1; // 9 + r4 = r4 * r0; // 10 + r5 = r5 ^ dataset[r4 & MASK]; // 11 + r1 = r1 ^ r7; // 12 + r1 = r1 ^ dataset[r6 & MASK]; // 13 + r2 = r2 ^ dataset[r1 & MASK]; // 14 + r3 = r3 * r1; // 15 + r6 = r6 ^ hot[mulhi(r5, HOT_WORDS)]; // 16 + r0 = r0 ^ dataset[r2 & MASK]; // 17 + r0 = r0 + r7 + select(0xc035a4e6u, 0x535b545fu, ((sel >> 26u) & 1u) != 0u); // 18 + r5 = r5 - r0; // 19 + r2 = r2 * r4; // 20 + r2 = r2 + r1 + select(0x1ecdd2cfu, 0x925d5e63u, ((sel >> 6u) & 1u) != 0u); // 21 + r3 = r3 ^ r7; // 22 + r7 = r7 ^ dataset[r2 & MASK]; // 23 + r2 = mulhi(r2, r5); // 24 + r5 = rotr_var(r5, r2); // 25 + r3 = r2 * r7 + r3; // 26 + r2 = r2 - r0; // 27 + r6 = r1 * r5 + r6; // 28 + r2 = r2 + r3 + select(0xbdc6da76u, 0x46b1b505u, ((sel >> 16u) & 1u) != 0u); // 29 + r3 = r3 ^ simd_shuffle_xor(r2, (ushort)8); // 30 + r5 = r5 ^ dataset[r7 & MASK]; // 31 + r2 = r2 ^ hot[mulhi(r0, HOT_WORDS)]; // 32 + r6 = r6 | r3; // 33 + r3 = r3 ^ hot[mulhi(r6, HOT_WORDS)]; // 34 + r4 = r4 ^ simd_shuffle_xor(r0, (ushort)2); // 35 + r5 = r5 ^ r2; // 36 + r3 = r3 ^ dataset[r4 & MASK]; // 37 + r4 = r4 ^ dataset[r3 & MASK]; // 38 + r1 = rotr_var(r1, r4); // 39 + r3 = r3 + r6 + select(0xbc3ff65fu, 0x6328cb2cu, ((sel >> 28u) & 1u) != 0u); // 40 + r5 = r5 ^ r1; // 41 + r5 = r5 | r0; // 42 + r7 = r7 ^ r3; // 43 + r2 = r2 ^ dataset[r5 & MASK]; // 44 + r6 = r6 ^ simd_shuffle_xor(r4, (ushort)8); // 45 + r5 = r5 ^ r3; // 46 + r7 = r7 + r6 + select(0x5af3bd5bu, 0x9342df0du, ((sel >> 22u) & 1u) != 0u); // 47 + r3 = r3 + r4 + select(0x9ff95776u, 0x485614dbu, ((sel >> 23u) & 1u) != 0u); // 48 + r5 = r5 ^ dataset[r3 & MASK]; // 49 + r4 = rotr_var(r4, r5); // 50 + r0 = r1 * r7 + r0; // 51 + r0 = r0 * r3; // 52 + r5 = rotr_var(r5, r4); // 53 + r0 = r0 ^ simd_shuffle_xor(r6, (ushort)1); // 54 + r0 = r0 ^ simd_shuffle_xor(r7, (ushort)16); // 55 + r7 = r7 ^ dataset[r4 & MASK]; // 56 + r7 = rotr_var(r7, r3); // 57 + r0 = r0 ^ dataset[r6 & MASK]; // 58 + r6 = r6 ^ hot[mulhi(r2, HOT_WORDS)]; // 59 + r3 = r3 ^ simd_shuffle_xor(r2, (ushort)1); // 60 + r7 = r7 - r5; // 61 + r1 = rotl_imm(r1, 4u); // 62 + r4 = r4 ^ dataset[r5 & MASK]; // 63 + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; +} diff --git a/proto-cuda/packs-ca2-hot/hot96k4a/vectors.h b/proto-cuda/packs-ca2-hot/hot96k4a/vectors.h new file mode 100644 index 000000000..18043ce08 --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4a/vectors.h @@ -0,0 +1,67 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Expected outputs: igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif + +#define IGNEUM_VEC_WARPS 3 +static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u }; +static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = { + { // base nonce 0 + 0xac19a15a875e6812ull, 0x57168da17e6d150bull, 0x40c24720e9782044ull, 0xdf31f7cca113338full, 0x1618866493c5bfb0ull, 0x6861843c8d4a1f9bull, 0x29f7222ca93b80b1ull, 0x380d16aea7ae1650ull, + 0x98e3b6ceb14d2029ull, 0xcd122ff3f3be095eull, 0xdfb361635ff84a9dull, 0xfc629557772b798cull, 0x2fac241df900a2a9ull, 0x15faaa85a53e234bull, 0xd7bcefa5b756274cull, 0xd6cfcdd9e64c9249ull, + 0x282fd82f83742f7cull, 0xfa86a0ee5c1553b5ull, 0x6febc7de42ed462dull, 0xcb28e0a1d59eeff9ull, 0x22896f7fe558b0ecull, 0x60a1f5ac4923b969ull, 0xbc52ad3abdcc5d51ull, 0xda76c8af7d2033beull, + 0x3a1d693159dad97full, 0x805dc813a7134b9bull, 0x5c4eecbc102e791full, 0x98f37407e3179ed8ull, 0x14a153370bb568f0ull, 0xc7c894ae380bae8full, 0x1fbd334958f05643ull, 0xf2d73b90c808fb39ull + }, + { // base nonce 4096 + 0x0c2989ad4df0e1cdull, 0x2d811ac3bb9e3f2full, 0x3cd0261958cc4404ull, 0x56f58aa7bb55cf62ull, 0x7c7ac451133e880full, 0x6fdaf187728cf641ull, 0x8bdf8d2b267158abull, 0x0d32af7cf3f776c5ull, + 0x70ae1ef3d1340be9ull, 0xca552a81c9458e68ull, 0x5c4e20ac46185947ull, 0x07b1c3f6fc4b3c3full, 0x8813092af249b0b8ull, 0x899470f1d3037960ull, 0xe32a6efe4ddfa953ull, 0x648c829bf0ab273full, + 0x1ff2f41db69683a0ull, 0x952b3549f6f743feull, 0xec54e41fd55dc8a7ull, 0xbea7a1a76e91146eull, 0x987b74d2ba2a529bull, 0x32b4114016d4a0e3ull, 0x95594a4164c4bc82ull, 0x8838ad8f5f8d4676ull, + 0x1fdf5118172bea81ull, 0xc0131630391256e0ull, 0x5581825ad7122ad7ull, 0x9a57d84307fc233dull, 0x7af8a1edaa80d692ull, 0xed126baaadab9becull, 0xe1f8cfb85ce5daa4ull, 0xba0fe02f1064a99cull + }, + { // base nonce 1000000 + 0x36e6ef3010df901dull, 0x85afd9fb1c568a06ull, 0xdde0164b224940c4ull, 0xaac7f87e8cb35df6ull, 0x882790cd2a1cf324ull, 0x95561dc7d227e745ull, 0x31990d0a651bf406ull, 0x0eb7c7c0efc14728ull, + 0xd2bdff62f5f88969ull, 0x8851f4b00a2607feull, 0x22930374c4de8c28ull, 0xc5addc6e021248c4ull, 0xf4d96093392f213dull, 0xbebe6d4007cc5351ull, 0xa8de4c82393572bbull, 0x4c11f846bdc055aaull, + 0xdf15ec1a6791fd28ull, 0x04cf5fe7b1c2a5c0ull, 0xa9491d11303b606eull, 0x5187e22028d3e7faull, 0x260f3ae9febc0da7ull, 0xd9676cd179690bd5ull, 0x982ba405a390bea4ull, 0x6e789a577618bbf6ull, + 0xac0c75a9eb8adb76ull, 0x6a76cb5b924f825full, 0x6849407291e20d93ull, 0x6c172e401c2f7327ull, 0xb961c7981cce34ebull, 0x95e3e701554d0979ull, 0xe427f16e4073f0baull, 0xfaa821d98d70e4c5ull + } +}; + +// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (268435455). +static const uint32_t IGNEUM_DS_HEAD[16] = { + 0xffc3cd94u, 0x5920ccd8u, 0x392f44bbu, 0x5e57f67au, 0x2f2bc2a9u, 0x620b0e36u, 0xbdc09014u, 0x436654bfu, + 0x311e0b48u, 0x1abd93adu, 0x59cc7ce8u, 0xee5247b2u, 0x86171fe8u, 0x6d874751u, 0xc9f7728fu, 0x7c2a435du +}; +static const uint32_t IGNEUM_DS_LAST_INDEX = 268435455u; +static const uint32_t IGNEUM_DS_LAST = 0xa33ada72u; +// 64 sampled dataset words (index, value) computed on the Mac. +#define IGNEUM_DS_SAMPLES 64 +static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = { + 59471966u, 217795994u, 208353206u, 42483309u, 172547758u, 148076330u, 183853158u, 214389424u, 267488061u, 169781097u, 184093494u, 153880993u, 84977930u, 46426879u, 3093825u, 225364072u, 44593546u, 260713159u, 168250303u, 52384140u, 223401610u, 45554030u, 95410555u, 175039924u, 79171087u, 267580473u, 24168642u, 37981670u, 171551130u, 195559979u, 204611762u, 140997658u, 138925853u, 86637313u, 20736778u, 219665210u, 160430336u, 264654675u, 8013395u, 228945585u, 213884386u, 104419827u, 44185464u, 142737231u, 99284897u, 132475900u, 61861762u, 132056166u, 262388043u, 91878046u, 117353561u, 124768597u, 71352993u, 190698941u, 46055428u, 55281366u, 165145231u, 106810753u, 171985651u, 232085256u, 159510492u, 40072060u, 209107596u, 39023794u +}; +static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = { + 0xe8b73d94u, 0x337028b5u, 0xafe148c9u, 0xab99f7aeu, 0x434ea619u, 0xd85cb880u, 0x54764c7fu, 0x82c7e420u, 0xedf4cb9eu, 0x9884c959u, 0x223ee793u, 0x3a9ccf69u, 0x81da4fd2u, 0xd6ce8cb9u, 0xe3922dcau, 0x3e7e6bdeu, 0x382a3acau, 0x567e7f7fu, 0x25a0f084u, 0xbfeef128u, 0xe338abfbu, 0x7c3b5280u, 0x909bc5f1u, 0xd8b74b9cu, 0x8e31a22eu, 0x26b5f1d8u, 0x79122c00u, 0xcafc3340u, 0xd5e02ea3u, 0x1aee1afdu, 0xdb090d9au, 0xb049f435u, 0x4954d8bau, 0x03797ba0u, 0x196eefbdu, 0xd153412au, 0xbe5d2c4bu, 0xdaa14f0eu, 0x8e61ed07u, 0x9e9a64c6u, 0x2e29ff36u, 0x392a8589u, 0xb56a5912u, 0xfa6e8b57u, 0xd1a737cbu, 0xb0fa841au, 0xbe1c341fu, 0xe25be0f1u, 0xe937f543u, 0xebab2248u, 0x8e1b607au, 0x202a2fedu, 0x95e2819cu, 0x9c9652d4u, 0x32fedef0u, 0xdecfff82u, 0xcb5d43e5u, 0xb735806au, 0x8905939cu, 0xfbf8472du, 0xada74e5du, 0x7ebdeeeau, 0x0119f2b3u, 0xa9a376b8u +}; +// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^26 words. +static const uint32_t IGNEUM_CACHE_HEAD[16] = { + 0x355a86d2u, 0x7957db1cu, 0xd21772afu, 0x6fc1e09bu, 0xd55ce61du, 0x6e6a278bu, 0xd3f543ceu, 0x223d8e82u, + 0x143ab337u, 0x2e9f05bdu, 0x2eb389bfu, 0x0c6e449eu, 0x5cfa4222u, 0xba6560feu, 0x8e3e1aa4u, 0xdbcc1d53u +}; +static const uint32_t IGNEUM_CACHE_LAST[16] = { + 0x41190d91u, 0xbd277957u, 0x22ddbb49u, 0x6986f207u, 0xdf69a4d6u, 0x26401a3au, 0x818230fbu, 0xc417122du, + 0x3597b211u, 0xb553ce55u, 0xcf39cc0du, 0x3b7fc43au, 0x3fd43b00u, 0x67e1c80eu, 0xffa7ea7du, 0xca2960abu +}; +static const uint64_t IGNEUM_CACHE_FNV64 = 0x48c4f5bf24166b2eull; +// Hot table self-test (docs/plans/hot-table.md): hot[0..15], the last 16 words, and FNV-1a 64 over all IGNEUM_HOT_WORDS words. +static const uint32_t IGNEUM_HOT_HEAD[16] = { + 0x8068cc73u, 0x6036ebf9u, 0xb604cd25u, 0x8ffb840eu, 0xc54074a2u, 0x285c0695u, 0x77512425u, 0xc26a58a7u, + 0x72c88757u, 0xc10fca78u, 0x513825ddu, 0x30d6ccc8u, 0x9a05e7cfu, 0xb9533f50u, 0x4bac3ba0u, 0xa5c19528u +}; +static const uint32_t IGNEUM_HOT_LAST[16] = { + 0xa3fd89f9u, 0xeb098388u, 0x19ee38d9u, 0x3700a229u, 0xbc0602a9u, 0x231d2a70u, 0xc23c57ceu, 0x2ff8dc6au, + 0x39683a2du, 0xbbe7d264u, 0xb8b28a43u, 0xb830041du, 0x86ae51e2u, 0xb3d22ccfu, 0x0e53335fu, 0xb56e328du +}; +static const uint64_t IGNEUM_HOT_FNV64 = 0x79bcf436c4e5bc47ull; diff --git a/proto-cuda/packs-ca2-hot/hot96k4a/vectors.json b/proto-cuda/packs-ca2-hot/hot96k4a/vectors.json new file mode 100644 index 000000000..b651c8e9f --- /dev/null +++ b/proto-cuda/packs-ca2-hot/hot96k4a/vectors.json @@ -0,0 +1,39 @@ +{ + "seed": "igneum-genesis", + "day": "2026-10-03", + "dataset_mode": "memory-hard", + "dataset_log2_words": 28, + "mask": "0x0fffffff", + "lanes": 32, + "source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset", + "warps": [ + {"base_nonce": 0, "expected": [ + "0xac19a15a875e6812", "0x57168da17e6d150b", "0x40c24720e9782044", "0xdf31f7cca113338f", "0x1618866493c5bfb0", "0x6861843c8d4a1f9b", "0x29f7222ca93b80b1", "0x380d16aea7ae1650", + "0x98e3b6ceb14d2029", "0xcd122ff3f3be095e", "0xdfb361635ff84a9d", "0xfc629557772b798c", "0x2fac241df900a2a9", "0x15faaa85a53e234b", "0xd7bcefa5b756274c", "0xd6cfcdd9e64c9249", + "0x282fd82f83742f7c", "0xfa86a0ee5c1553b5", "0x6febc7de42ed462d", "0xcb28e0a1d59eeff9", "0x22896f7fe558b0ec", "0x60a1f5ac4923b969", "0xbc52ad3abdcc5d51", "0xda76c8af7d2033be", + "0x3a1d693159dad97f", "0x805dc813a7134b9b", "0x5c4eecbc102e791f", "0x98f37407e3179ed8", "0x14a153370bb568f0", "0xc7c894ae380bae8f", "0x1fbd334958f05643", "0xf2d73b90c808fb39" + ]}, + {"base_nonce": 4096, "expected": [ + "0x0c2989ad4df0e1cd", "0x2d811ac3bb9e3f2f", "0x3cd0261958cc4404", "0x56f58aa7bb55cf62", "0x7c7ac451133e880f", "0x6fdaf187728cf641", "0x8bdf8d2b267158ab", "0x0d32af7cf3f776c5", + "0x70ae1ef3d1340be9", "0xca552a81c9458e68", "0x5c4e20ac46185947", "0x07b1c3f6fc4b3c3f", "0x8813092af249b0b8", "0x899470f1d3037960", "0xe32a6efe4ddfa953", "0x648c829bf0ab273f", + "0x1ff2f41db69683a0", "0x952b3549f6f743fe", "0xec54e41fd55dc8a7", "0xbea7a1a76e91146e", "0x987b74d2ba2a529b", "0x32b4114016d4a0e3", "0x95594a4164c4bc82", "0x8838ad8f5f8d4676", + "0x1fdf5118172bea81", "0xc0131630391256e0", "0x5581825ad7122ad7", "0x9a57d84307fc233d", "0x7af8a1edaa80d692", "0xed126baaadab9bec", "0xe1f8cfb85ce5daa4", "0xba0fe02f1064a99c" + ]}, + {"base_nonce": 1000000, "expected": [ + "0x36e6ef3010df901d", "0x85afd9fb1c568a06", "0xdde0164b224940c4", "0xaac7f87e8cb35df6", "0x882790cd2a1cf324", "0x95561dc7d227e745", "0x31990d0a651bf406", "0x0eb7c7c0efc14728", + "0xd2bdff62f5f88969", "0x8851f4b00a2607fe", "0x22930374c4de8c28", "0xc5addc6e021248c4", "0xf4d96093392f213d", "0xbebe6d4007cc5351", "0xa8de4c82393572bb", "0x4c11f846bdc055aa", + "0xdf15ec1a6791fd28", "0x04cf5fe7b1c2a5c0", "0xa9491d11303b606e", "0x5187e22028d3e7fa", "0x260f3ae9febc0da7", "0xd9676cd179690bd5", "0x982ba405a390bea4", "0x6e789a577618bbf6", + "0xac0c75a9eb8adb76", "0x6a76cb5b924f825f", "0x6849407291e20d93", "0x6c172e401c2f7327", "0xb961c7981cce34eb", "0x95e3e701554d0979", "0xe427f16e4073f0ba", "0xfaa821d98d70e4c5" + ]} + ], + "dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"], + "dataset_last_index": 268435455, + "dataset_last": "0xa33ada72", + "dataset_samples": [{"index": 59471966, "value": "0xe8b73d94"}, {"index": 217795994, "value": "0x337028b5"}, {"index": 208353206, "value": "0xafe148c9"}, {"index": 42483309, "value": "0xab99f7ae"}, {"index": 172547758, "value": "0x434ea619"}, {"index": 148076330, "value": "0xd85cb880"}, {"index": 183853158, "value": "0x54764c7f"}, {"index": 214389424, "value": "0x82c7e420"}, {"index": 267488061, "value": "0xedf4cb9e"}, {"index": 169781097, "value": "0x9884c959"}, {"index": 184093494, "value": "0x223ee793"}, {"index": 153880993, "value": "0x3a9ccf69"}, {"index": 84977930, "value": "0x81da4fd2"}, {"index": 46426879, "value": "0xd6ce8cb9"}, {"index": 3093825, "value": "0xe3922dca"}, {"index": 225364072, "value": "0x3e7e6bde"}, {"index": 44593546, "value": "0x382a3aca"}, {"index": 260713159, "value": "0x567e7f7f"}, {"index": 168250303, "value": "0x25a0f084"}, {"index": 52384140, "value": "0xbfeef128"}, {"index": 223401610, "value": "0xe338abfb"}, {"index": 45554030, "value": "0x7c3b5280"}, {"index": 95410555, "value": "0x909bc5f1"}, {"index": 175039924, "value": "0xd8b74b9c"}, {"index": 79171087, "value": "0x8e31a22e"}, {"index": 267580473, "value": "0x26b5f1d8"}, {"index": 24168642, "value": "0x79122c00"}, {"index": 37981670, "value": "0xcafc3340"}, {"index": 171551130, "value": "0xd5e02ea3"}, {"index": 195559979, "value": "0x1aee1afd"}, {"index": 204611762, "value": "0xdb090d9a"}, {"index": 140997658, "value": "0xb049f435"}, {"index": 138925853, "value": "0x4954d8ba"}, {"index": 86637313, "value": "0x03797ba0"}, {"index": 20736778, "value": "0x196eefbd"}, {"index": 219665210, "value": "0xd153412a"}, {"index": 160430336, "value": "0xbe5d2c4b"}, {"index": 264654675, "value": "0xdaa14f0e"}, {"index": 8013395, "value": "0x8e61ed07"}, {"index": 228945585, "value": "0x9e9a64c6"}, {"index": 213884386, "value": "0x2e29ff36"}, {"index": 104419827, "value": "0x392a8589"}, {"index": 44185464, "value": "0xb56a5912"}, {"index": 142737231, "value": "0xfa6e8b57"}, {"index": 99284897, "value": "0xd1a737cb"}, {"index": 132475900, "value": "0xb0fa841a"}, {"index": 61861762, "value": "0xbe1c341f"}, {"index": 132056166, "value": "0xe25be0f1"}, {"index": 262388043, "value": "0xe937f543"}, {"index": 91878046, "value": "0xebab2248"}, {"index": 117353561, "value": "0x8e1b607a"}, {"index": 124768597, "value": "0x202a2fed"}, {"index": 71352993, "value": "0x95e2819c"}, {"index": 190698941, "value": "0x9c9652d4"}, {"index": 46055428, "value": "0x32fedef0"}, {"index": 55281366, "value": "0xdecfff82"}, {"index": 165145231, "value": "0xcb5d43e5"}, {"index": 106810753, "value": "0xb735806a"}, {"index": 171985651, "value": "0x8905939c"}, {"index": 232085256, "value": "0xfbf8472d"}, {"index": 159510492, "value": "0xada74e5d"}, {"index": 40072060, "value": "0x7ebdeeea"}, {"index": 209107596, "value": "0x0119f2b3"}, {"index": 39023794, "value": "0xa9a376b8"}], + "cache_head": ["0x355a86d2", "0x7957db1c", "0xd21772af", "0x6fc1e09b", "0xd55ce61d", "0x6e6a278b", "0xd3f543ce", "0x223d8e82", "0x143ab337", "0x2e9f05bd", "0x2eb389bf", "0x0c6e449e", "0x5cfa4222", "0xba6560fe", "0x8e3e1aa4", "0xdbcc1d53"], + "cache_last_line": ["0x41190d91", "0xbd277957", "0x22ddbb49", "0x6986f207", "0xdf69a4d6", "0x26401a3a", "0x818230fb", "0xc417122d", "0x3597b211", "0xb553ce55", "0xcf39cc0d", "0x3b7fc43a", "0x3fd43b00", "0x67e1c80e", "0xffa7ea7d", "0xca2960ab"], + "cache_fnv1a64": "0x48c4f5bf24166b2e", + "hot_head": ["0x8068cc73", "0x6036ebf9", "0xb604cd25", "0x8ffb840e", "0xc54074a2", "0x285c0695", "0x77512425", "0xc26a58a7", "0x72c88757", "0xc10fca78", "0x513825dd", "0x30d6ccc8", "0x9a05e7cf", "0xb9533f50", "0x4bac3ba0", "0xa5c19528"], + "hot_last_line": ["0xa3fd89f9", "0xeb098388", "0x19ee38d9", "0x3700a229", "0xbc0602a9", "0x231d2a70", "0xc23c57ce", "0x2ff8dc6a", "0x39683a2d", "0xbbe7d264", "0xb8b28a43", "0xb830041d", "0x86ae51e2", "0xb3d22ccf", "0x0e53335f", "0xb56e328d"], + "hot_fnv1a64": "0x79bcf436c4e5bc47" +} diff --git a/proto-metal/packbench.swift b/proto-metal/packbench.swift index b7eda2f3c..a38012db8 100644 --- a/proto-metal/packbench.swift +++ b/proto-metal/packbench.swift @@ -63,6 +63,12 @@ let scratchWordsPerLane = Int(defineU32("IGNEUM_SCRATCH_WORDS_PER_LANE") ?? 8192 let className = defineStr("IGNEUM_LOAD_CLASS") ?? "v2" // Counter ASIC 2.0 (5 October 2026): the mixer multiplier of the item derivation, 1 when absent (version 2), 4 under class v3 let mixerMult = Int(defineU32("IGNEUM_MIXER_MULT") ?? 1) +// hot-table experiment (docs/plans/hot-table.md): the epoch table, filled on the device from memhard.metal's igneum_hot_fill +let hotMb = Int(defineU32("IGNEUM_HOT_MB") ?? 0) +let hotWords = Int(defineU32("IGNEUM_HOT_WORDS") ?? 0) +let hotSegments = Int(defineU32("IGNEUM_HOT_SEGMENTS") ?? 0) +let hotSlots = Int(defineU32("IGNEUM_HOT_SLOTS") ?? 0) +if hotMb > 0 && (hotWords != hotMb << 18 || hotSegments != hotMb * 256) { fail("program.h hot table sizes disagree") } let seedString = defineStr("IGNEUM_SEED_STRING") ?? "?" let programId = defineStr("IGNEUM_PROGRAM_ID") ?? "" @@ -77,6 +83,7 @@ let dsHead = (vj["dataset_head"] as! [String]).map(hex32) let dsLastIndex = UInt32((vj["dataset_last_index"] as! NSNumber).uint64Value) let dsLast = hex32(vj["dataset_last"] as! String) let cacheFnvWant = hex64(vj["cache_fnv1a64"] as! String) +let hotFnvWant: UInt64? = (vj["hot_fnv1a64"] as? String).map(hex64) guard let device = MTLCreateSystemDefaultDevice(), let queue = device.makeCommandQueue() else { fail("no Metal device") } let words = 1 << datasetLog2 @@ -93,6 +100,11 @@ guard let fillFn = mhLib.makeFunction(name: "igneum_cache_fill"), let buildFn = let fillPipe = try! device.makeComputePipelineState(function: fillFn) let buildPipe = try! device.makeComputePipelineState(function: buildFn) let hashPipe = try! device.makeComputePipelineState(function: hashFn) +var hotPipe: MTLComputePipelineState? = nil +if hotMb > 0 { + guard let hotFn = mhLib.makeFunction(name: "igneum_hot_fill") else { fail("program.h says IGNEUM_HOT_MB \(hotMb) but memhard.metal has no igneum_hot_fill") } + hotPipe = try! device.makeComputePipelineState(function: hotFn) +} let compileMs = nowMs() - t0 if hashPipe.threadExecutionWidth != 32 { print("WARNING: threadExecutionWidth \(hashPipe.threadExecutionWidth), not 32") } @@ -134,6 +146,22 @@ func fnv1a64(_ p: UnsafeRawPointer, _ n: Int) -> UInt64 { let cacheCopy = blit(cache, 0, cacheWords * 4) let cacheFnv = fnv1a64(cacheCopy.contents(), cacheWords * 4) let cacheOk = cacheFnv == cacheFnvWant +// hot table: fill, then fingerprint against vectors.json +var hot: MTLBuffer? = nil +var hotGpu = 0.0, hotWall = 0.0 +var hotOk = true +var hotFnv: UInt64 = 0 +if hotMb > 0 { + guard let h = device.makeBuffer(length: hotWords * 4, options: .storageModePrivate) else { fail("hot table alloc of \(hotMb) MiB") } + hot = h + (hotWall, hotGpu) = run { enc in + enc.setComputePipelineState(hotPipe!); enc.setBuffer(h, offset: 0, index: 0) + enc.dispatchThreadgroups(MTLSize(width: hotSegments / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1)) + } + let hotCopy = blit(h, 0, hotWords * 4) + hotFnv = fnv1a64(hotCopy.contents(), hotWords * 4) + if let want = hotFnvWant { hotOk = hotFnv == want } +} let headCopy = blit(dataset, 0, 64) let headPtr = headCopy.contents().bindMemory(to: UInt32.self, capacity: 16) var dsOk = (0..<16).allSatisfy { headPtr[$0] == dsHead[$0] } @@ -155,13 +183,16 @@ func encodeHash(_ enc: MTLComputeCommandEncoder, out: MTLBuffer, base: UInt32, n enc.setBuffer(dataset, offset: 0, index: 0) enc.setBuffer(out, offset: 0, index: 1) var b = base; enc.setBytes(&b, length: 4, index: 2) + // the hot table is buffer 3 (the scratch triple moves up by one when both are present) + let next = hot == nil ? 3 : 4 + if let h = hot { enc.setBuffer(h, offset: 0, index: 3) } if persistent { let units = nonces / 32 let nw = min(warpsN, units) if units % nw != 0 { fail("nonces \(nonces) is not a multiple of 32 x \(nw) warps") } - enc.setBuffer(scratch!, offset: 0, index: 3) - var g = UInt32(units); enc.setBytes(&g, length: 4, index: 4) - var s = salt; enc.setBytes(&s, length: 4, index: 5) + enc.setBuffer(scratch!, offset: 0, index: next) + var g = UInt32(units); enc.setBytes(&g, length: 4, index: next + 1) + var s = salt; enc.setBytes(&s, length: 4, index: next + 2) salt = salt &+ UInt32(units) let threads = nw * 32 let tg = min(group, threads) @@ -205,7 +236,8 @@ let packName = (opts.pack as NSString).lastPathComponent print("pack \(packName) seed \"\(seedString)\" id \(programId) class \(className): loads/hash \(loadsPerHash), dataset bytes/hash \(bytesPerHash), scratch ops/hash \(scratchOps * 8), mixer x\(mixerMult), cache 2^\(cacheLog2) words") print("device \(device.name); compile \(String(format: "%.0f", compileMs)) ms; cache fill \(String(format: "%.1f", cacheGpu)) ms GPU (\(String(format: "%.1f", cacheWall)) wall); dataset build \(String(format: "%.1f", buildGpu)) ms GPU (\(String(format: "%.1f", buildWall)) wall)") print("cache FNV-1a 64 \(String(format: "%016llx", cacheFnv)) \(cacheOk ? "PASS" : "FAIL"); dataset head and last \(dsOk ? "PASS" : "FAIL"); vectors standalone \(vecPass)/\(vecBases.count), in batch \(batchVecPass)/\(batchVecN)") +if hotMb > 0 { print("hot table \(hotMb) MiB (\(hotSlots) of 16 load slots): fill \(String(format: "%.2f", hotGpu)) ms GPU (\(String(format: "%.2f", hotWall)) wall), FNV-1a 64 \(String(format: "%016llx", hotFnv)) \(hotFnvWant == nil ? "(not in vectors.json)" : (hotOk ? "PASS" : "FAIL"))") } print("warm-up batch \(nonces) hashes: \(String(format: "%.1f", warmGpu)) ms GPU, \(String(format: "%.1f", warmWall)) ms wall") -let overall = cacheOk && dsOk && vecPass == vecBases.count && batchVecPass == batchVecN -print("RESULT pack=\(packName) class=\(className) device=\(device.name.replacingOccurrences(of: " ", with: "_")) group=\(opts.group) warps=\(warpsN) arena_mib=\(persistent ? warpsN * 32 * scratchWordsPerLane * 4 / 1048576 : 0) nonces=\(nonces) batches=\(opts.batches) vectors=\(vecPass)/\(vecBases.count) batch_vectors=\(batchVecPass)/\(batchVecN) cache=\(cacheOk ? "PASS" : "FAIL") dataset=\(dsOk ? "PASS" : "FAIL") fingerprint=\(String(format: "%016llx", fingerprint)) mhs_gpu=\(String(format: "%.3f", mhsGpu)) mhs_wall=\(String(format: "%.3f", mhsWall)) loads=\(loadsPerHash) bytes=\(bytesPerHash) scratch_ops=\(scratchOps * 8) overall=\(overall ? "PASS" : "FAIL")") +let overall = cacheOk && dsOk && hotOk && vecPass == vecBases.count && batchVecPass == batchVecN +print("RESULT pack=\(packName) class=\(className) device=\(device.name.replacingOccurrences(of: " ", with: "_")) group=\(opts.group) warps=\(warpsN) arena_mib=\(persistent ? warpsN * 32 * scratchWordsPerLane * 4 / 1048576 : 0) hot_mib=\(hotMb) hot_slots=\(hotSlots) hot_fill_ms=\(String(format: "%.2f", hotGpu)) hot=\(hotMb > 0 ? (hotOk ? "PASS" : "FAIL") : "none") nonces=\(nonces) batches=\(opts.batches) vectors=\(vecPass)/\(vecBases.count) batch_vectors=\(batchVecPass)/\(batchVecN) cache=\(cacheOk ? "PASS" : "FAIL") dataset=\(dsOk ? "PASS" : "FAIL") fingerprint=\(String(format: "%016llx", fingerprint)) mhs_gpu=\(String(format: "%.3f", mhsGpu)) mhs_wall=\(String(format: "%.3f", mhsWall)) loads=\(loadsPerHash) bytes=\(bytesPerHash) scratch_ops=\(scratchOps * 8) overall=\(overall ? "PASS" : "FAIL")") exit(overall ? 0 : 1) diff --git a/proto-opencl/host.c b/proto-opencl/host.c index 8423180ad..8c87584a6 100644 --- a/proto-opencl/host.c +++ b/proto-opencl/host.c @@ -1049,6 +1049,12 @@ typedef struct { cl_program prog; cl_kernel kHashBound, kCacheFill, kBuild; cl_mem cache, ds; + /* hot-table experiment (5 October 2026, docs/plans/hot-table.md): the epoch's table, filled on the device by the + * pack's igneum_hot_fill, the argument after the init words; hotWords 0 for a pack without one */ + cl_kernel kHotFill; + cl_mem hot; + uint32_t hotWords, hotSegments, hotMb, hotSlots; + double hotMs; double buildMs, cacheMs, datasetMs, checkMs; char check[1024]; /* the self-test verdict (packfile.h), one line */ int checked; @@ -1089,6 +1095,8 @@ static void releasePair(ServePair* p) { if (!p) return; if (p->ds) { clReleaseMemObject(p->ds); ++gMemReleased; } if (p->cache) { clReleaseMemObject(p->cache); ++gMemReleased; } + if (p->hot) { clReleaseMemObject(p->hot); ++gMemReleased; } + if (p->kHotFill) clReleaseKernel(p->kHotFill); if (p->kHashBound) clReleaseKernel(p->kHashBound); if (p->kCacheFill) clReleaseKernel(p->kCacheFill); if (p->kBuild) clReleaseKernel(p->kBuild); @@ -1124,6 +1132,8 @@ static int pairSelfTest(Device* dv, const DeviceInfo* di, cl_command_queue q, Se PfPack pk; char perr[256]; uint32_t cacheHead[16], cacheLast[16], dsHead[16], dsLast = 0; + uint32_t hotHead[16], hotLast[16]; + uint64_t hotFnv = 0; uint32_t samples[PF_MAX_SAMPLES]; uint64_t vec[PF_MAX_WARPS * 32]; uint64_t fnv; @@ -1131,6 +1141,7 @@ static int pairSelfTest(Device* dv, const DeviceInfo* di, cl_command_queue q, Se cl_mem out = NULL, init = NULL; cl_int e = 0; int w, i, ok; + cl_uint extra = 5; /* the first argument after the five of igneum_hash_bound: the hot table, then the scratch triple */ size_t local = (size_t)dv->groupSize, g = (size_t)dv->groupSize; /* one work-group; its first 32 lanes are the warp */ double tb = wallMs(); (void)di; @@ -1152,6 +1163,16 @@ static int pairSelfTest(Device* dv, const DeviceInfo* di, cl_command_queue q, Se if (e == CL_SUCCESS && pk.dsLastIndex < words) e = clEnqueueReadBuffer(q, p->ds, CL_TRUE, (size_t)pk.dsLastIndex * 4u, 4, &dsLast, 0, NULL, NULL); for (i = 0; i < pk.nSamples && e == CL_SUCCESS; ++i) { samples[i] = 0; if (pk.sampleIdx[i] < words) e = clEnqueueReadBuffer(q, p->ds, CL_TRUE, (size_t)pk.sampleIdx[i] * 4u, 4, &samples[i], 0, NULL, NULL); } if (e != CL_SUCCESS) { snprintf(err, errCap, "self-test: dataset read-back (%s)", clErrName(e)); return 0; } + if (p->hot) { + uint32_t* hw = (uint32_t*)malloc((size_t)p->hotWords * 4u); + if (!hw) { snprintf(err, errCap, "self-test: no host memory for the hot table read-back"); return 0; } + e = clEnqueueReadBuffer(q, p->hot, CL_TRUE, 0, (size_t)p->hotWords * 4u, hw, 0, NULL, NULL); + if (e != CL_SUCCESS) { free(hw); snprintf(err, errCap, "self-test: hot table read-back (%s)", clErrName(e)); return 0; } + memcpy(hotHead, hw, 64); + memcpy(hotLast, hw + p->hotWords - 16u, 64); + hotFnv = pf_fnv1a64(hw, (size_t)p->hotWords * 4u); + free(hw); + } out = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, g * sizeof(uint64_t), NULL, &e); if (e == CL_SUCCESS) { ++gMemCreated; init = clCreateBuffer(dv->ctx, CL_MEM_READ_ONLY | CL_MEM_COPY_HOST_PTR, 32, pk.seedw, &e); } if (e == CL_SUCCESS) ++gMemCreated; @@ -1166,14 +1187,15 @@ static int pairSelfTest(Device* dv, const DeviceInfo* di, cl_command_queue q, Se if (e == CL_SUCCESS) e = clSetKernelArg(p->kHashBound, 2, sizeof(cl_uint), &base); if (e == CL_SUCCESS) e = clSetKernelArg(p->kHashBound, 3, sizeof(cl_uint), &mask); if (e == CL_SUCCESS) e = clSetKernelArg(p->kHashBound, 4, sizeof(cl_mem), &init); - if (e == CL_SUCCESS && pk.persistent) e = setScratchArgs(p->kHashBound, 5, 1u); + if (e == CL_SUCCESS && p->hot) { e = clSetKernelArg(p->kHashBound, 5, sizeof(cl_mem), &p->hot); extra = 6; } + if (e == CL_SUCCESS && pk.persistent) e = setScratchArgs(p->kHashBound, extra, 1u); if (e == CL_SUCCESS) e = clEnqueueNDRangeKernel(q, p->kHashBound, 1, NULL, &g, &local, 0, NULL, NULL); if (e == CL_SUCCESS) e = clEnqueueReadBuffer(q, out, CL_TRUE, 0, 32 * sizeof(uint64_t), &vec[w * 32], 0, NULL, NULL); } if (out) { clReleaseMemObject(out); ++gMemReleased; } if (init) { clReleaseMemObject(init); ++gMemReleased; } if (e != CL_SUCCESS) { snprintf(err, errCap, "self-test: vector warp (%s)", clErrName(e)); return 0; } - ok = pf_selftest(&pk, cacheHead, cacheLast, fnv, dsHead, dsLast, samples, vec, p->check, sizeof(p->check)); + ok = pf_selftest(&pk, cacheHead, cacheLast, fnv, dsHead, dsLast, samples, vec, p->hot ? hotHead : NULL, p->hot ? hotLast : NULL, hotFnv, p->check, sizeof(p->check)); p->checked = 1; p->checkMs = wallMs() - tb; if (!ok) { snprintf(err, errCap, "%s", p->check); return 0; } @@ -1211,9 +1233,37 @@ static int pairBuffers(Device* dv, const DeviceInfo* di, cl_command_queue q, Ser if (e == CL_SUCCESS) e = clFinish(q); if (e != CL_SUCCESS) { snprintf(err, errCap, "dataset build (%s)", clErrName(e)); return 0; } p->datasetMs = wallMs() - tb; + if (p->kHotFill && p->hotWords) { + /* hot-table experiment: the epoch's table from the pack's own fill kernel (one work-item per segment) */ + cl_uint nSegH = p->hotSegments; + tb = wallMs(); + p->hot = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, (size_t)p->hotWords * 4u, NULL, &e); + if (e != CL_SUCCESS) { snprintf(err, errCap, "clCreateBuffer hot table (%s)", clErrName(e)); return 0; } + ++gMemCreated; + local = kernelMaxLocal(dv, p->kHotFill, di, 256); + g = ((nSegH + local - 1) / local) * local; + e = clSetKernelArg(p->kHotFill, 0, sizeof(cl_mem), &p->hot); + if (e == CL_SUCCESS) e = clSetKernelArg(p->kHotFill, 1, sizeof(cl_uint), &nSegH); + if (e == CL_SUCCESS) e = clEnqueueNDRangeKernel(q, p->kHotFill, 1, NULL, &g, &local, 0, NULL, NULL); + if (e == CL_SUCCESS) e = clFinish(q); + if (e != CL_SUCCESS) { snprintf(err, errCap, "hot table fill (%s)", clErrName(e)); return 0; } + p->hotMs = wallMs() - tb; + } return pairSelfTest(dv, di, q, p, words, cacheWords, packDir, err, errCap); } +/* The hot-table kernel and sizes of a pair whose program came from a pack directory (none when the pack has no hot table). */ +static int pairHotKernel(ServePair* p, const char* packDir, char* err, size_t errCap) { + PfPack pk; + char perr[256]; + cl_int e = 0; + if (!packDir || !packDir[0] || !pf_load(packDir, &pk, perr, sizeof(perr)) || !pk.hotMb) return 1; + p->kHotFill = clCreateKernel(p->prog, "igneum_hot_fill", &e); + if (e != CL_SUCCESS) { snprintf(err, errCap, "clCreateKernel igneum_hot_fill (%s): the pack says IGNEUM_HOT_MB %u but its kernel source has no hot fill", clErrName(e), pk.hotMb); return 0; } + p->hotWords = pk.hotWords; p->hotSegments = pk.hotSegments; p->hotMb = pk.hotMb; p->hotSlots = pk.hotSlots; + return 1; +} + /* Builds the pair for a prepare request. Runs on its own thread with its own command queue. */ static void prepareRun(PrepareTask* t) { cl_int err = 0; @@ -1257,6 +1307,7 @@ static void prepareRun(PrepareTask* t) { if (err != CL_SUCCESS) { prepareFail(t, "clCreateKernel igneum_cache_fill", err); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; } p->kBuild = clCreateKernel(p->prog, "igneum_build", &err); if (err != CL_SUCCESS) { prepareFail(t, "clCreateKernel igneum_build", err); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; } + if (!pairHotKernel(p, t->packDir, t->error, sizeof(t->error))) { releasePair(p); t->doneAt = wallMs(); t->done = 1; return; } q = clCreateCommandQueue(t->dv->ctx, t->di->device, 0, &err); if (err != CL_SUCCESS) { prepareFail(t, "clCreateCommandQueue (prepare)", err); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; } if (!pairBuffers(t->dv, t->di, q, p, t->words, t->cacheWords, t->segments, t->packDir, t->error, sizeof(t->error))) { clReleaseCommandQueue(q); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; } @@ -1295,8 +1346,8 @@ static int runBenchPack(Device* dv, const DeviceInfo* di, const Options* o) { uint64_t* hOut; ServePair* cur; char perr[512], devName[256]; - double t0 = wallMs(), sum = 0, warmMs; - uint64_t fp; + double t0 = wallMs(), sum = 0, warmMs = 0; + uint64_t fp = 0; int b, k; cl_uint warps = (cl_uint)(o->warps > 0 ? o->warps : 2048), units = nonces / 32u; if (!dv->kHashBound) { printf("FAIL: the kernel source has no igneum_hash_bound\n"); return 2; } @@ -1314,8 +1365,9 @@ static int runBenchPack(Device* dv, const DeviceInfo* di, const Options* o) { cur->kHashBound = dv->kHashBound; cur->kCacheFill = dv->kCacheFill; cur->kBuild = dv->kBuild; cur->prog = dv->prog; dv->kHashBound = dv->kCacheFill = dv->kBuild = NULL; dv->prog = NULL; memcpy(cur->sw, gPack.seedw, 32); memcpy(cur->kw, gPack.keyw, 32); + if (!pairHotKernel(cur, o->packDir, perr, sizeof(perr))) { printf("FAIL: pack %s: %s\n", o->packDir, perr); return 1; } if (!pairBuffers(dv, di, dv->q, cur, words, gServeCacheWords, gServeSegments, o->packDir, perr, sizeof(perr))) { printf("FAIL: pack %s: %s\n", o->packDir, perr); return 1; } - printf("pack %s: cache %.0f dataset %.0f check %.0f ms (%.0f ms in all); %s\n", o->packDir, cur->cacheMs, cur->datasetMs, cur->checkMs, wallMs() - t0, cur->check); + printf("pack %s: cache %.0f dataset %.0f hot %.0f check %.0f ms (%.0f ms in all); %s\n", o->packDir, cur->cacheMs, cur->datasetMs, cur->hotMs, cur->checkMs, wallMs() - t0, cur->check); printKernelInfo(di, cur->kHashBound, "igneum_hash_bound", (int)groupSize, ""); dOut = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, (size_t)nonces * sizeof(uint64_t), NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer out"); dInit = clCreateBuffer(dv->ctx, CL_MEM_READ_ONLY | CL_MEM_COPY_HOST_PTR, 32, cur->sw, &err); CL_CHECK_ERR(err, "clCreateBuffer init words"); @@ -1332,7 +1384,8 @@ static int runBenchPack(Device* dv, const DeviceInfo* di, const Options* o) { CL_CHECK(clSetKernelArg(cur->kHashBound, 2, sizeof(cl_uint), &base)); CL_CHECK(clSetKernelArg(cur->kHashBound, 3, sizeof(cl_uint), &mask)); CL_CHECK(clSetKernelArg(cur->kHashBound, 4, sizeof(cl_mem), &dInit)); - if (gPack.persistent) CL_CHECK(setScratchArgs(cur->kHashBound, 5, units)); + if (cur->hot) CL_CHECK(clSetKernelArg(cur->kHashBound, 5, sizeof(cl_mem), &cur->hot)); + if (gPack.persistent) CL_CHECK(setScratchArgs(cur->kHashBound, cur->hot ? 6 : 5, units)); { double w0 = wallMs(); CL_CHECK(clEnqueueNDRangeKernel(dv->q, cur->kHashBound, 1, NULL, &g, &groupSize, 0, NULL, &ev)); @@ -1348,9 +1401,9 @@ static int runBenchPack(Device* dv, const DeviceInfo* di, const Options* o) { } else sum += ms; } printf("warm-up dispatch (base 0): %.2f ms; %d timed dispatches of 2^%d nonces: mean %.2f ms\n", warmMs, o->batches, o->batchLog2, sum / o->batches); - printf("RESULT pack=%s class=%s device=%s platform=%s group=%d warps=%u arena_mib=%llu nonces=%u batches=%d check=%s fingerprint=%016llx mhs=%.3f loads=%u bytes=%u scratch_ops=%u time=%s\n", + printf("RESULT pack=%s class=%s device=%s platform=%s group=%d warps=%u arena_mib=%llu hot_mib=%u hot_slots=%u hot_fill_ms=%.2f nonces=%u batches=%d check=%s fingerprint=%016llx mhs=%.3f loads=%u bytes=%u scratch_ops=%u time=%s\n", o->packDir, gPack.loadClass, devName, strcmp(di->platformName, "Apple") == 0 ? "Apple" : "other", (int)groupSize, gPack.persistent ? warps : 0u, - gPack.persistent ? (unsigned long long)(((size_t)warps * 32u * gPack.scratchWordsPerLane * 4u) >> 20) : 0ull, nonces, o->batches, + gPack.persistent ? (unsigned long long)(((size_t)warps * 32u * gPack.scratchWordsPerLane * 4u) >> 20) : 0ull, cur->hotMb, cur->hotSlots, cur->hotMs, nonces, o->batches, cur->checked ? "PASS" : "skipped", (unsigned long long)fp, (double)nonces * (double)o->batches / (sum / 1000.0) / 1e6, gPack.loadsPerHash, gPack.bytesPerHash, gPack.scratchOps * 8u, o->timeWall ? "wall" : "event"); free(hOut); @@ -1395,6 +1448,7 @@ static int runServe(Device* dv, const DeviceInfo* di, const Options* o) { strncpy(cur->epochHex, gPack.epochHex, 64); cur->epochHex[64] = 0; strncpy(cur->dayHex, gPack.dayHex, sizeof(cur->dayHex) - 1); strncpy(cur->programClass, gPack.programClass, sizeof(cur->programClass) - 1); strncpy(cur->eraHex, gPack.eraHex, sizeof(cur->eraHex) - 1); + if (!pairHotKernel(cur, o->packDir, perr, sizeof(perr))) { printf("error 0 pack %s: %s\n", o->packDir, perr); fflush(stdout); return 1; } if (!pairBuffers(dv, di, dv->q, cur, words, gServeCacheWords, gServeSegments, o->packDir, perr, sizeof(perr))) { printf("error 0 pack %s: %s\n", o->packDir, perr); fflush(stdout); return 1; } printf("info first pack %s: cache %.0f dataset %.0f check %.0f ms (%.0f ms in all); %s\n", o->packDir, cur->cacheMs, cur->datasetMs, cur->checkMs, wallMs() - t0, cur->check); } else { @@ -1607,6 +1661,7 @@ static int runServe(Device* dv, const DeviceInfo* di, const Options* o) { SERVE_CHECK(jobId, clSetKernelArg(cur->kHashBound, 2, sizeof(cl_uint), &baseNonce)); SERVE_CHECK(jobId, clSetKernelArg(cur->kHashBound, 3, sizeof(cl_uint), &maskArg)); SERVE_CHECK(jobId, clSetKernelArg(cur->kHashBound, 4, sizeof(cl_mem), &dInit)); + if (cur->hot) SERVE_CHECK(jobId, clSetKernelArg(cur->kHashBound, 5, sizeof(cl_mem), &cur->hot)); ev = NULL; if (gFaultTestChunk >= 0 && (long)chunksSeen >= gFaultTestChunk) { /* IGNEUM_FAULT_TEST=N (test only): from chunk N on, behave like a runtime that answers every call with diff --git a/relay/playbooks/ca2-hot-5090-bench.ps1 b/relay/playbooks/ca2-hot-5090-bench.ps1 new file mode 100644 index 000000000..48f869f94 --- /dev/null +++ b/relay/playbooks/ca2-hot-5090-bench.ps1 @@ -0,0 +1,66 @@ +# Igneum run job: the hot-table experiment (Counter ASIC 2.0 layer 5, docs/plans/hot-table.md) on PC 2's RTX 5090 (machine 1ccfe586), 5 October 2026 (the probe at the hot sizes 32, 64 and 96 MiB, then the five hot packs). +# Published as a plain `run` job (NOT --stop-miners): the installed app keeps every other card mining; this script switches +# off ONLY the NVIDIA card in the app through POST api/cards, waits for its worker to stop, runs the fetched +# igneum-worker-cuda.exe (--memprobe, then --bench on every pack of the fetched packs folder), and switches the card back +# on with the settings it had. Every result line starts with RESULT so `node tools/jobs.mjs ` shows them. +$ErrorActionPreference = 'Continue' +function Say([string] $m) { Write-Host ("[" + (Get-Date -Format 'HH:mm:ss') + "] " + $m) } +$jobs = Split-Path $env:IGNEUM_JOB_DIR +$fetched = Join-Path $jobs 'fetch-ca2-hot-20261005' +$exe = Join-Path $fetched 'igneum-worker-cuda.exe' +$packs = Join-Path $fetched 'packs-ca2-hot' +if (-not (Test-Path $exe)) { Write-Output "RESULT error worker missing at $exe (the fetch job runs first)"; exit 2 } +if (-not (Test-Path $packs)) { Write-Output "RESULT error packs missing at $packs"; exit 2 } +$inst = @("$env:LOCALAPPDATA\Programs\Igneum Miner", "$env:ProgramFiles\Igneum Miner") | Where-Object { Test-Path (Join-Path $_ 'igneum-worker-cuda.exe') } | Select-Object -First 1 +if (-not $inst) { Write-Output 'RESULT error no installed igneum-worker-cuda.exe (the NVRTC DLLs come from there)'; exit 2 } +Get-ChildItem $inst -Filter 'nvrtc*.dll' | Copy-Item -Destination $fetched -Force +Write-Output "RESULT worker $exe sha256 $((Get-FileHash -Algorithm SHA256 $exe).Hash.ToLower()) with $((Get-ChildItem $fetched -Filter 'nvrtc*.dll').Count) NVRTC DLL(s) from $inst" + +# the app: switch off the NVIDIA card only, remember its settings +$appDir = $env:IGNEUM_APP_DIR +if (-not $appDir) { $appDir = Join-Path $env:LOCALAPPDATA 'igneum\app' } +$urlFile = Join-Path $appDir 'app.url' +$url = $null +if (Test-Path $urlFile) { $url = (Get-Content -LiteralPath $urlFile -Raw).Trim() } +$card = $null +if ($url) { + try { + $st = Invoke-RestMethod -Uri ($url + 'api/state') -Method GET -TimeoutSec 10 + $card = $st.mining.cards | Where-Object { $_.vendor -eq 'nvidia' } | Select-Object -First 1 + if (-not $card) { $card = $st.cards | Where-Object { $_.vendor -eq 'nvidia' } | Select-Object -First 1 } + } catch { Say ("api/state: " + $_.Exception.Message) } +} +if ($card) { + Write-Output ("RESULT card " + $card.key + " enabled=" + $card.enabled + " identities=" + $card.identities + " power_pct=" + $card.power_pct + " state=" + $card.state) + $body = @{ cards = @(@{ key = $card.key; enabled = $false; identities = [int]$card.identities; power_pct = [int]$card.power_pct }) } | ConvertTo-Json -Depth 5 + try { Invoke-RestMethod -Uri ($url + 'api/cards') -Method POST -Body $body -ContentType 'application/json' -TimeoutSec 10 | Out-Null; Say "card off requested" } catch { Say ("api/cards off: " + $_.Exception.Message) } + $t = 0 + while ($t -lt 90) { + Start-Sleep -Seconds 5; $t += 5 + try { $st = Invoke-RestMethod -Uri ($url + 'api/state') -Method GET -TimeoutSec 10; $c2 = $st.mining.cards | Where-Object { $_.key -eq $card.key }; if (-not $c2) { $c2 = $st.cards | Where-Object { $_.key -eq $card.key } }; if ($c2 -and $c2.state -eq 'off' -and $c2.pid -eq 0) { break } } catch { } + } + Write-Output ("RESULT card-off after " + $t + " s") + Start-Sleep -Seconds 5 +} else { Write-Output 'RESULT card none-found (the app is not running or has no NVIDIA card); measuring with whatever else runs on the GPU' } + +& nvidia-smi --query-gpu=name,driver_version,power.limit,clocks.sm,clocks.mem,memory.used,temperature.gpu --format=csv,noheader 2>&1 | ForEach-Object { "RESULT gpu-before $_" } +# the probe at the hot sizes (and 1024 MiB again, the same session): the random-read ceiling inside the 96 MiB L2 +foreach ($mib in @(32, 64, 96, 1024)) { + Write-Output "RESULT memprobe $mib start $(Get-Date -Format HH:mm:ss)" + & $exe --memprobe --probe-mib $mib 2>&1 | ForEach-Object { "RESULT $_" } +} +# the eight hot packs (five replaced, three added, docs/plans/hot-table.md 2.3): --bench builds the pack, fills the hot table on the device from the epoch seed, self-tests it +# (cache, dataset, hot table head, last line and FNV, 96 vector lanes) and times 5 dispatches of 2^24 nonces +foreach ($pk in @('hot32k4', 'hot64k4', 'hot96k4', 'hot64k2', 'hot64k8', 'hot32k4a', 'hot64k4a', 'hot96k4a')) { + $d = Join-Path $packs $pk + Write-Output "RESULT bench $pk start $(Get-Date -Format HH:mm:ss)" + & $exe --bench --pack $d --batches 5 --batch-log2 24 --block-warps 1 2>&1 | ForEach-Object { "RESULT $_" } + & $exe --bench --pack $d --batches 5 --batch-log2 24 --block-warps 8 2>&1 | Where-Object { $_ -match '^RESULT|error|FAIL' } | ForEach-Object { "RESULT $_" } +} +& nvidia-smi --query-gpu=power.draw,clocks.sm,clocks.mem,memory.used,temperature.gpu --format=csv,noheader 2>&1 | ForEach-Object { "RESULT gpu-after $_" } + +if ($card) { + $body = @{ cards = @(@{ key = $card.key; enabled = [bool]$card.enabled; identities = [int]$card.identities; power_pct = [int]$card.power_pct }) } | ConvertTo-Json -Depth 5 + try { Invoke-RestMethod -Uri ($url + 'api/cards') -Method POST -Body $body -ContentType 'application/json' -TimeoutSec 10 | Out-Null; Write-Output ("RESULT card restored enabled=" + $card.enabled) } catch { Write-Output ("RESULT error card restore: " + $_.Exception.Message) } +} +exit 0 diff --git a/relay/playbooks/ca2-hot-9070-bench.ps1 b/relay/playbooks/ca2-hot-9070-bench.ps1 new file mode 100644 index 000000000..f16bd665c --- /dev/null +++ b/relay/playbooks/ca2-hot-9070-bench.ps1 @@ -0,0 +1,85 @@ +# Igneum run job: the hot-table experiment (Counter ASIC 2.0 layer 5, docs/plans/hot-table.md) on PC 1's RX 9070 XT on the eGPU (machine ae432dc7), 5 October 2026 (the probe at the hot sizes 32, 64 and 96 MiB, then the five hot packs). +# Published as a plain `run` job (NOT --stop-miners): the installed app keeps every other card mining; this script switches +# off ONLY the NVIDIA card in the app through POST api/cards, waits for its worker to stop, runs the fetched +# igneum-worker-opencl.exe (--memprobe, then --bench on every pack of the fetched packs folder), and switches the card back +# on with the settings it had. Every result line starts with RESULT so `node tools/jobs.mjs ` shows them. +$ErrorActionPreference = 'Continue' +function Say([string] $m) { Write-Host ("[" + (Get-Date -Format 'HH:mm:ss') + "] " + $m) } +$jobs = Split-Path $env:IGNEUM_JOB_DIR +$fetched = Join-Path $jobs 'fetch-ca2-hot-20261005' +$exe = Join-Path $fetched 'igneum-worker-opencl.exe' +$packs = Join-Path $fetched 'packs-ca2-hot' +if (-not (Test-Path $exe)) { Write-Output "RESULT error worker missing at $exe (the fetch job runs first)"; exit 2 } +if (-not (Test-Path $packs)) { Write-Output "RESULT error packs missing at $packs"; exit 2 } +Write-Output "RESULT worker $exe sha256 $((Get-FileHash -Algorithm SHA256 $exe).Hash.ToLower())" +# the card's OpenCL device index on the current (3683.0) platform, from the worker's own list (the older platform's duplicate is marked dup) +$list = & $exe --list 2>&1 +$list | ForEach-Object { "RESULT list $_" } +$dev = $null +$cardName = 'gfx1201' +foreach ($l in $list) { if ($l -match '^\s*\[(\d+)\].*gfx1201' -and $l -notmatch 'dup') { $dev = [int]$Matches[1]; break } } +# 5 October 2026, 20:40 UTC: the Sonnet eGPU box dropped off PC 1's USB4 link (gfx1201 absent at 20:45:34Z). Gate G1's +# AMD vendor is then the integrated gfx1036 (coordinator's ruling): bit-exactness only (vectors, hot table head, last +# line and FNV, the 2^24 fingerprint at base 0 equal to the 5090's and the Mac's); its probe and hash rate say nothing +# about the 9070 XT (2 MB of cache, approximate) and are skipped as owed. If gfx1201 is back at run time, it runs as planned. +if ($null -eq $dev) { + foreach ($l in $list) { if ($l -match '^\s*\[(\d+)\].*gfx1036' -and $l -notmatch 'dup') { $dev = [int]$Matches[1]; $cardName = 'gfx1036'; break } } +} +if ($null -eq $dev) { Write-Output 'RESULT error neither gfx1201 nor gfx1036 in --list'; exit 2 } +Write-Output "RESULT amd-device $dev $cardName" +Write-Output "RESULT device $dev" + +# the app: switch off the NVIDIA card only, remember its settings +$appDir = $env:IGNEUM_APP_DIR +if (-not $appDir) { $appDir = Join-Path $env:LOCALAPPDATA 'igneum\app' } +$urlFile = Join-Path $appDir 'app.url' +$url = $null +if (Test-Path $urlFile) { $url = (Get-Content -LiteralPath $urlFile -Raw).Trim() } +$card = $null +if ($url) { + try { + $st = Invoke-RestMethod -Uri ($url + 'api/state') -Method GET -TimeoutSec 10 + $card = $st.mining.cards | Where-Object { ($_.vendor -eq 'amd' -and $_.key -match $cardName) } | Select-Object -First 1 + if (-not $card) { $card = $st.cards | Where-Object { ($_.vendor -eq 'amd' -and $_.key -match $cardName) } | Select-Object -First 1 } + } catch { Say ("api/state: " + $_.Exception.Message) } +} +if ($card) { + Write-Output ("RESULT card " + $card.key + " enabled=" + $card.enabled + " identities=" + $card.identities + " power_pct=" + $card.power_pct + " state=" + $card.state) + $body = @{ cards = @(@{ key = $card.key; enabled = $false; identities = [int]$card.identities; power_pct = [int]$card.power_pct }) } | ConvertTo-Json -Depth 5 + try { Invoke-RestMethod -Uri ($url + 'api/cards') -Method POST -Body $body -ContentType 'application/json' -TimeoutSec 10 | Out-Null; Say "card off requested" } catch { Say ("api/cards off: " + $_.Exception.Message) } + $t = 0 + while ($t -lt 90) { + Start-Sleep -Seconds 5; $t += 5 + try { $st = Invoke-RestMethod -Uri ($url + 'api/state') -Method GET -TimeoutSec 10; $c2 = $st.mining.cards | Where-Object { $_.key -eq $card.key }; if (-not $c2) { $c2 = $st.cards | Where-Object { $_.key -eq $card.key } }; if ($c2 -and $c2.state -eq 'off' -and $c2.pid -eq 0) { break } } catch { } + } + Write-Output ("RESULT card-off after " + $t + " s") + Start-Sleep -Seconds 5 +} else { Write-Output 'RESULT card none-found (the app is not running or has no NVIDIA card); measuring with whatever else runs on the GPU' } + +# the probe ran in run-readwidth-9070-20261005 (bench-log, 5 October 2026); this second run is the bench only +# the probe at the hot sizes (64 MiB exists from the 9070 XT entry: 9.17 G loads/s; 32 and 96 MiB are new; 1024 MiB again, the same session) +if ($cardName -eq 'gfx1201') { + foreach ($mib in @(32, 64, 96, 1024)) { + Write-Output "RESULT memprobe $mib start $(Get-Date -Format HH:mm:ss)" + & $exe --memprobe --probe-mib $mib --device $dev 2>&1 | ForEach-Object { "RESULT $_" } + } +} else { Write-Output 'RESULT memprobe skipped (gfx1036: its cache says nothing about the 9070 XT; the 32, 96 and 1024 MiB rows stay owed)' } +# the eight hot packs (five replaced, three added, docs/plans/hot-table.md 2.3): --bench-pack builds the pack, fills the hot table on the device from the epoch seed, self-tests it +# (cache, dataset, hot table head, last line and FNV, 96 vector lanes) and times 5 dispatches of 2^24 nonces +foreach ($pk in @('hot32k4', 'hot64k4', 'hot96k4', 'hot64k2', 'hot64k8', 'hot32k4a', 'hot64k4a', 'hot96k4a')) { + $d = Join-Path $packs $pk + Write-Output "RESULT bench $pk start $(Get-Date -Format HH:mm:ss)" + if ($cardName -eq 'gfx1201') { + & $exe --bench-pack --pack $d --batches 5 --batch-log2 24 --device $dev 2>&1 | ForEach-Object { "RESULT $_" } + & $exe --bench-pack --pack $d --batches 5 --batch-log2 24 --device $dev --group-warps 8 2>&1 | Where-Object { $_ -match '^RESULT|error|FAIL' } | ForEach-Object { "RESULT $_" } + } else { + # bit-exactness only: one timed dispatch after the warm-up, the same 2^24 nonces at base 0 as the 5090 and the Mac (about 6 s each at 3 MH/s); the mhs= field is not a number + & $exe --bench-pack --pack $d --batches 1 --batch-log2 24 --device $dev 2>&1 | ForEach-Object { "RESULT $_" } + } +} + +if ($card) { + $body = @{ cards = @(@{ key = $card.key; enabled = [bool]$card.enabled; identities = [int]$card.identities; power_pct = [int]$card.power_pct }) } | ConvertTo-Json -Depth 5 + try { Invoke-RestMethod -Uri ($url + 'api/cards') -Method POST -Body $body -ContentType 'application/json' -TimeoutSec 10 | Out-Null; Write-Output ("RESULT card restored enabled=" + $card.enabled) } catch { Write-Output ("RESULT error card restore: " + $_.Exception.Message) } +} +exit 0