igneum/proto-metal/main.swift
igneum-labs ff263245ac Igneum: design docs, Metal lottery-hash prototype, CUDA test pack, finality simulation
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
2026-10-03 15:06:01 +00:00

1592 lines
79 KiB
Swift

// igneum-bench: first prototype of Igneum's random-program GPU proof-of-work.
// One file. Build: swiftc -O -o igneum-bench main.swift -framework Metal
// Metal shaders are compiled at runtime from generated source (no Xcode needed).
import Foundation
import Metal
// MARK: - Options
struct Options {
var seed = "igneum-genesis"
var day = "2026-10-03"
var hours = 2 // number of epochs (seeds) run in sequence; default 2 so verification covers 2 seeds
var batchLog2 = 22 // nonces per batch
var batches = 4 // timed batches
var datasetLog2 = 28 // 2^28 uint32 = 1 GiB
var verifyWarps = 3
var dumpDir: String? = nil
var exportPack: String? = nil // write a CUDA program pack for --seed into this directory and exit
// Hardening tests (added 3 October 2026). Any of these runs instead of the bench.
var fuzz: Int? = nil // --fuzz N: N random programs, GPU vs CPU on 4 random warps each
var fuzzSeed = "igneum-fuzz-2026-10-03"
var edge = false // --edge: hand-built edge-case programs
var stats = false // --stats: output distribution sanity checks on 2^20 nonces
var determinism = false // --determinism: 5 identical GPU runs + double compile
var memcheck = false // --memcheck: static mask check + 4 MiB run with wrapping nonces
var anyTest: Bool { fuzz != nil || edge || stats || determinism || memcheck }
}
func parseArgs() -> Options {
var o = Options()
var args = Array(CommandLine.arguments.dropFirst())
func take() -> String { args.isEmpty ? "" : args.removeFirst() }
while !args.isEmpty {
let a = take()
switch a {
case "--seed": o.seed = take()
case "--day": o.day = take()
case "--hours": o.hours = Int(take()) ?? o.hours
case "--batch-log2": o.batchLog2 = Int(take()) ?? o.batchLog2
case "--batches": o.batches = Int(take()) ?? o.batches
case "--dataset-log2": o.datasetLog2 = Int(take()) ?? o.datasetLog2
case "--verify-warps": o.verifyWarps = Int(take()) ?? o.verifyWarps
case "--dump": o.dumpDir = take()
case "--export-pack": o.exportPack = take()
case "--fuzz": o.fuzz = Int(take()) ?? 200
case "--fuzz-seed": o.fuzzSeed = take()
case "--edge": o.edge = true
case "--stats": o.stats = true
case "--determinism": o.determinism = true
case "--memcheck": o.memcheck = true
case "-h", "--help":
print("""
igneum-bench [--seed <string>] [--hours N] [--batch-log2 22] [--batches 4]
[--dataset-log2 28] [--verify-warps 3] [--dump <dir>] [--day <string>]
[--export-pack <dir>] write the CUDA program pack for --seed, then exit
hardening tests (run instead of the bench; several may be combined; exit 0 only if all pass):
[--fuzz N [--fuzz-seed <string>]] N random programs, GPU vs CPU, 4 random warps each,
dataset size drawn from 64 MiB, 256 MiB, 1 GiB
[--edge] hand-built edge-case programs, GPU vs CPU
[--stats] output distribution sanity checks on 2^20 nonces, 3 seeds
[--determinism] 5 identical GPU runs of 2^20 nonces, double compile, dataset fill check
[--memcheck] static dataset-index mask check, 4 MiB run with wrapping nonces
""")
exit(0)
default:
print("unknown argument \(a)"); exit(2)
}
}
return o
}
// MARK: - Integer helpers (CPU side, must match MSL bit for bit)
@inline(__always) func rotl32(_ x: UInt32, _ n: UInt32) -> UInt32 {
let n = n & 31
return n == 0 ? x : (x << n) | (x >> (32 - n))
}
@inline(__always) func rotr32(_ x: UInt32, _ n: UInt32) -> UInt32 {
let n = n & 31
return n == 0 ? x : (x >> n) | (x << (32 - n))
}
@inline(__always) func mulhi32(_ a: UInt32, _ b: UInt32) -> UInt32 {
UInt32(truncatingIfNeeded: (UInt64(a) &* UInt64(b)) >> 32)
}
@inline(__always) func splitmix32(_ v: UInt32) -> UInt32 {
var x = v
x ^= x >> 16; x &*= 0x7feb352d
x ^= x >> 15; x &*= 0x846ca68b
x ^= x >> 16
return x
}
// Dataset element, closed form of (daySeed, index). Same formula is emitted into the MSL.
@inline(__always) func datasetElem(_ i: UInt32, _ d0: UInt32, _ d1: UInt32) -> UInt32 {
var x = i ^ d0
x &*= 0x9E3779B1; x ^= x >> 15
x &+= d1
x &*= 0x85EBCA77; x ^= x >> 13
x &*= 0xC2B2AE3D; x ^= x >> 16
return x
}
// 32-byte seed (8 x uint32) from a string: FNV-1a 64 with four salts, each finalised.
func seedWords(_ s: String) -> [UInt32] {
var words = [UInt32]()
for salt in 0..<4 {
var h: UInt64 = 0xcbf29ce484222325 ^ (UInt64(salt) &* 0x9E3779B97F4A7C15)
for b in s.utf8 { h ^= UInt64(b); h &*= 0x100000001b3 }
h ^= h >> 33; h &*= 0xff51afd7ed558ccd; h ^= h >> 33
words.append(UInt32(truncatingIfNeeded: h))
words.append(UInt32(truncatingIfNeeded: h >> 32))
}
return words
}
struct SplitMix64 {
var s: UInt64
mutating func next() -> UInt64 {
s &+= 0x9E3779B97F4A7C15
var z = s
z = (z ^ (z >> 30)) &* 0xBF58476D1CE4E5B9
z = (z ^ (z >> 27)) &* 0x94D049BB133111EB
return z ^ (z >> 31)
}
mutating func below(_ n: Int) -> Int { Int(next() % UInt64(n)) }
}
// MARK: - Program
enum Op: String { case add, sub, mul, mulhi, xor, or, rotl, rotr, mad, shfl, load }
struct Instr {
var op: Op
var dst: Int
var a: Int // source register, never equal to dst
var b: Int // second source (mad only)
var imm: UInt32 // add immediate A
var imm2: UInt32 // add immediate B
var rot: UInt32 // rotl amount 1..31
var bit: Int // selector bit of r0 for add
var mask: Int // shuffle xor mask: 1,2,4,8,16
}
struct Program {
let seedString: String
let seed: [UInt32]
let instrs: [Instr]
static let iterations = 8
static let count = 64
var loadsPerHash: Int { instrs.filter { $0.op == .load }.count * Program.iterations }
var histogram: [(String, Int)] {
var d = [String: Int]()
for i in instrs { d[i.op.rawValue, default: 0] += 1 }
return d.sorted { $0.1 != $1.1 ? $0.1 > $1.1 : $0.0 < $1.0 } // count desc, then name, so output is deterministic
}
}
// Weights sum to 100. Loads are 25 percent so the kernel leans on memory.
let opWeights: [(Op, Int)] = [(.load, 25), (.add, 12), (.xor, 10), (.mul, 8), (.mad, 8), (.shfl, 8),
(.rotl, 7), (.sub, 6), (.mulhi, 6), (.rotr, 6), (.or, 4)]
func generateProgram(seedString: String) -> Program {
let sw = seedWords(seedString)
var rng = SplitMix64(s: (UInt64(sw[0]) | (UInt64(sw[1]) << 32)) ^ ((UInt64(sw[2]) | (UInt64(sw[3]) << 32)) &* 0x9E3779B97F4A7C15))
var instrs = [Instr]()
for _ in 0..<Program.count {
var roll = rng.below(100)
var op = Op.add
for (o, w) in opWeights { if roll < w { op = o; break }; roll -= w }
let dst = rng.below(8)
var a = rng.below(7); if a >= dst { a += 1 }
let b = rng.below(8)
let imm = UInt32(truncatingIfNeeded: rng.next())
let imm2 = UInt32(truncatingIfNeeded: rng.next())
let rot = UInt32(1 + rng.below(31))
let bit = rng.below(32)
let mask = 1 << rng.below(5)
instrs.append(Instr(op: op, dst: dst, a: a, b: b, imm: imm, imm2: imm2, rot: rot, bit: bit, mask: mask))
}
return Program(seedString: seedString, seed: sw, instrs: instrs)
}
// MARK: - MSL generation
func hex(_ v: UInt32) -> String { String(format: "0x%08xu", v) }
func generateMSL(_ p: Program, datasetLog2: Int) -> String {
let mask = UInt32((1 << datasetLog2) - 1)
var s = """
#include <metal_stdlib>
using namespace metal;
#define MASK \(hex(mask))
constant uint SEEDW[8] = { \(p.seed.map(hex).joined(separator: ", ")) };
inline uint splitmix32(uint x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
inline uint ds_elem(uint i, uint d0, uint d1) {
uint x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
device ulong* out [[buffer(1)]],
constant uint& baseNonce [[buffer(2)]],
uint gid [[thread_position_in_grid]]) {
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
"""
for i in 0..<8 {
s += " { uint x = nonce ^ SEEDW[\(i)]; x += 0x9e3779b9u * \(i + 1)u; x = splitmix32(x); r\(i) = x ^ SEEDW[\((i + 1) & 7)]; }\n"
}
s += "\n for (uint it = 0u; it < \(Program.iterations)u; ++it) {\n uint sel = r0;\n"
for (k, ins) in p.instrs.enumerated() {
let d = "r\(ins.dst)", a = "r\(ins.a)", b = "r\(ins.b)"
var line: String
switch ins.op {
case .add: line = "\(d) = \(d) + \(a) + select(\(hex(ins.imm)), \(hex(ins.imm2)), ((sel >> \(ins.bit)u) & 1u) != 0u);"
case .sub: line = "\(d) = \(d) - \(a);"
case .mul: line = "\(d) = \(d) * \(a);"
case .mulhi: line = "\(d) = mulhi(\(d), \(a));"
case .xor: line = "\(d) = \(d) ^ \(a);"
case .or: line = "\(d) = \(d) | \(a);"
case .rotl: line = "\(d) = rotl_imm(\(d), \(ins.rot)u);"
case .rotr: line = "\(d) = rotr_var(\(d), \(a));"
case .mad: line = "\(d) = \(a) * \(b) + \(d);"
case .shfl: line = "\(d) = \(d) ^ simd_shuffle_xor(\(a), (ushort)\(ins.mask));"
case .load: line = "\(d) = \(d) ^ dataset[\(a) & MASK];"
}
s += " \(line) // \(k)\n"
}
s += """
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}
"""
return s
}
let fillMSL = """
#include <metal_stdlib>
using namespace metal;
inline uint ds_elem(uint i, uint d0, uint d1) {
uint x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
kernel void igneum_fill(device uint* dataset [[buffer(0)]],
constant uint2& day [[buffer(1)]],
uint gid [[thread_position_in_grid]]) {
dataset[gid] = ds_elem(gid, day.x, day.y);
}
"""
// MARK: - CPU reference interpreter for one 32-lane warp
func cpuWarp(_ p: Program, baseNonce: UInt32, day: (UInt32, UInt32), mask: UInt32) -> [UInt64] {
cpuWarpTraced(p, baseNonce: baseNonce, day: day, mask: mask, trace: nil)
}
// Same interpreter with an optional hook. When `trace` is set it is called before every instruction with
// (iteration, instruction index, the 8 registers of lane 0). The edge-case tests use it to prove that the
// operand values they were built to produce really occurred. The bench passes nil.
func cpuWarpTraced(_ p: Program, baseNonce: UInt32, day: (UInt32, UInt32), mask: UInt32,
trace: ((Int, Int, [UInt32]) -> Void)?) -> [UInt64] {
let lanes = 32
var r = [UInt32](repeating: 0, count: lanes * 8) // r[lane*8 + reg]
for lane in 0..<lanes {
let nonce = baseNonce &+ UInt32(lane)
for i in 0..<8 {
var x = nonce ^ p.seed[i]
x &+= 0x9e3779b9 &* UInt32(i + 1)
x = splitmix32(x)
r[lane * 8 + i] = x ^ p.seed[(i + 1) & 7]
}
}
var tmp = [UInt32](repeating: 0, count: lanes)
for it in 0..<Program.iterations {
for lane in 0..<lanes { tmp[lane] = r[lane * 8] } // sel = r0 at the top of the iteration
let sel = tmp
for (k, ins) in p.instrs.enumerated() {
if let t = trace { t(it, k, Array(r[0..<8])) }
switch ins.op {
case .shfl:
for lane in 0..<lanes { tmp[lane] = r[lane * 8 + ins.a] }
for lane in 0..<lanes { r[lane * 8 + ins.dst] ^= tmp[lane ^ ins.mask] }
default:
for lane in 0..<lanes {
let base = lane * 8
let d = r[base + ins.dst], a = r[base + ins.a]
var v: UInt32
switch ins.op {
case .add:
let s = (sel[lane] >> UInt32(ins.bit)) & 1
v = d &+ a &+ (s != 0 ? ins.imm2 : ins.imm)
case .sub: v = d &- a
case .mul: v = d &* a
case .mulhi: v = mulhi32(d, a)
case .xor: v = d ^ a
case .or: v = d | a
case .rotl: v = rotl32(d, ins.rot)
case .rotr: v = rotr32(d, a)
case .mad: v = (a &* r[base + ins.b]) &+ d
case .load: v = d ^ datasetElem(a & mask, day.0, day.1)
case .shfl: v = d // unreachable
}
r[base + ins.dst] = v
}
}
}
}
var out = [UInt64](repeating: 0, count: lanes)
for lane in 0..<lanes {
let b = lane * 8
let lo = r[b] ^ rotl32(r[b + 1], 7) ^ rotl32(r[b + 2], 14) ^ rotl32(r[b + 3], 21)
let hi = r[b + 4] ^ rotl32(r[b + 5], 9) ^ rotl32(r[b + 6], 18) ^ rotl32(r[b + 7], 27)
out[lane] = (UInt64(hi) << 32) | UInt64(lo)
}
return out
}
// MARK: - Program pack export (CUDA twin of the Metal kernel)
//
// Writes, for one seed: program.json, vectors.json, kernel.cu, program.h, vectors.h, program.metal.
// The CUDA kernel is emitted from the same Instr list as the MSL above, line for line.
// Differences by design: the dataset mask is a kernel argument (so the host can sweep dataset
// sizes with one ahead-of-time compile), seeds are inlined as literals, and the host launch
// wrappers live in kernel.cu so host.cu never declares a __global__ across translation units.
let packVectorBases: [UInt32] = [0, 4096, 1000000]
func hex64(_ v: UInt64) -> String { String(format: "0x%016llxull", v) }
func jhex(_ v: UInt32) -> String { String(format: "\"0x%08x\"", v) }
func jhex64(_ v: UInt64) -> String { String(format: "\"0x%016llx\"", v) }
func jstr(_ s: String) -> String {
var o = "\""
for c in s.unicodeScalars {
switch c {
case "\"": o += "\\\""
case "\\": o += "\\\\"
case "\n": o += "\\n"
default: o.unicodeScalars.append(c)
}
}
return o + "\""
}
func generateCUDA(_ p: Program) -> String {
var s = """
// Generated by proto-metal/igneum-bench --export-pack for seed "\(p.seedString)". Do not edit by hand.
// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal).
// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC.
#include <cuda_runtime.h>
#include <cstdint>
#include "program.h"
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31.
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x.
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) {
uint32_t x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
// dataset[i] = ds_elem(i, d0, d1) for i < n. Same closed form as the Metal igneum_fill kernel.
__global__ void igneum_fill(uint32_t* ds, uint32_t n, uint32_t d0, uint32_t d1) {
uint32_t i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < n) ds[i] = ds_elem(i, d0, d1);
}
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask) {
uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;
uint32_t nonce = baseNonce + gid;
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
"""
for i in 0..<8 {
let addc = 0x9e3779b9 &* UInt32(i + 1)
s += " { uint32_t x = nonce ^ \(hex(p.seed[i])); x += \(hex(addc)); x = splitmix32(x); r\(i) = x ^ \(hex(p.seed[(i + 1) & 7])); } // SEEDW[\(i)], 0x9e3779b9u * \(i + 1)u, SEEDW[\((i + 1) & 7)]\n"
}
s += "\n for (uint32_t it = 0u; it < \(Program.iterations)u; ++it) {\n uint32_t sel = r0;\n"
for (k, ins) in p.instrs.enumerated() {
let d = "r\(ins.dst)", a = "r\(ins.a)", b = "r\(ins.b)"
var line: String
switch ins.op {
// Metal select(A, B, c) returns c ? B : A, so the true branch is imm2 here as well.
case .add: line = "\(d) = \(d) + \(a) + ((((sel >> \(ins.bit)u) & 1u) != 0u) ? \(hex(ins.imm2)) : \(hex(ins.imm)));"
case .sub: line = "\(d) = \(d) - \(a);"
case .mul: line = "\(d) = \(d) * \(a);"
case .mulhi: line = "\(d) = __umulhi(\(d), \(a));"
case .xor: line = "\(d) = \(d) ^ \(a);"
case .or: line = "\(d) = \(d) | \(a);"
case .rotl: line = "\(d) = rotl_imm(\(d), \(ins.rot)u);"
case .rotr: line = "\(d) = rotr_var(\(d), \(a));"
case .mad: line = "\(d) = \(a) * \(b) + \(d);"
case .shfl: line = "\(d) = \(d) ^ __shfl_xor_sync(0xffffffffu, \(a), \(ins.mask));"
case .load: line = "\(d) = \(d) ^ ds[\(a) & mask];"
}
s += " \(line) // \(k) \(ins.op.rawValue)\n"
}
s += """
}
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
}
// Host-side launch wrappers. Declared in program.h, called from host.cu.
cudaError_t igneum_launch_fill(uint32_t* ds, uint32_t nWords, uint32_t d0, uint32_t d1) {
if (nWords == 0u) return cudaErrorInvalidValue;
uint32_t block = 256u;
uint32_t grid = (nWords + block - 1u) / block;
igneum_fill<<<grid, block>>>(ds, nWords, d0, d1);
return cudaGetLastError();
}
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
uint32_t nonces, uint32_t blockWarps) {
if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;
uint32_t block = 32u * blockWarps;
if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;
igneum_hash<<<nonces / block, block>>>(ds, out, baseNonce, mask);
return cudaGetLastError();
}
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
cudaFuncAttributes attr;
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash);
if (e != cudaSuccess) return e;
*numRegs = attr.numRegs;
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0);
}
"""
return s
}
func generateProgramHeader(_ p: Program, dayString: String, day: (UInt32, UInt32), datasetLog2: Int) -> String {
let mask = UInt32((1 << datasetLog2) - 1)
let mix = p.histogram.map { "\($0.0)=\($0.1)" }.joined(separator: " ")
return """
// Generated by proto-metal/igneum-bench --export-pack for seed "\(p.seedString)". Do not edit by hand.
// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.
#pragma once
#include <cuda_runtime.h>
#include <cstdint>
#define IGNEUM_SEED_STRING \(jstr(p.seedString))
#define IGNEUM_DAY_STRING \(jstr(dayString))
#define IGNEUM_DAY0 \(hex(day.0))
#define IGNEUM_DAY1 \(hex(day.1))
#define IGNEUM_DATASET_LOG2 \(datasetLog2)
#define IGNEUM_MASK \(hex(mask))
#define IGNEUM_LANES 32
#define IGNEUM_ITERATIONS \(Program.iterations)
#define IGNEUM_INSTR_COUNT \(Program.count)
#define IGNEUM_LOADS_PER_HASH \(p.loadsPerHash)
#define IGNEUM_OP_MIX \(jstr(mix))
#define IGNEUM_SEEDW_INIT { \(p.seed.map(hex).joined(separator: ", ")) }
// Defined in kernel.cu. Both launch on the default stream and return cudaGetLastError().
cudaError_t igneum_launch_fill(uint32_t* ds, uint32_t nWords, uint32_t d0, uint32_t d1);
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
uint32_t nonces, uint32_t blockWarps);
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
"""
}
func generateVectorsHeader(_ p: Program, bases: [UInt32], outs: [[UInt64]], head: [UInt32], last: UInt32, mask: UInt32, source: String) -> String {
var s = """
// Generated by proto-metal/igneum-bench --export-pack for seed "\(p.seedString)". Do not edit by hand.
// Expected outputs: \(source)
#pragma once
#include <cstdint>
#define IGNEUM_VEC_WARPS \(bases.count)
static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { \(bases.map { "\($0)u" }.joined(separator: ", ")) };
static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {
"""
for (i, o) in outs.enumerated() {
s += " { // base nonce \(bases[i])\n"
for row in 0..<4 {
s += " " + (0..<8).map { hex64(o[row * 8 + $0]) }.joined(separator: ", ") + (row == 3 ? "\n" : ",\n")
}
s += i == outs.count - 1 ? " }\n" : " },\n"
}
s += """
};
// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (\(mask)).
static const uint32_t IGNEUM_DS_HEAD[16] = {
\((0..<8).map { hex(head[$0]) }.joined(separator: ", ")),
\((8..<16).map { hex(head[$0]) }.joined(separator: ", "))
};
static const uint32_t IGNEUM_DS_LAST_INDEX = \(mask)u;
static const uint32_t IGNEUM_DS_LAST = \(hex(last));
"""
return s
}
func generateProgramJSON(_ p: Program, dayString: String, day: (UInt32, UInt32), datasetLog2: Int) -> String {
let mask = UInt32((1 << datasetLog2) - 1)
var s = "{\n"
s += " \"format\": \"igneum-program-pack-1\",\n"
s += " \"seed\": \(jstr(p.seedString)),\n"
s += " \"seed_words\": [\(p.seed.map(jhex).joined(separator: ", "))],\n"
s += " \"seed_derivation\": \"FNV-1a 64 over UTF-8 of seed, basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32\",\n"
s += " \"lanes\": 32,\n"
s += " \"registers\": 8,\n"
s += " \"iterations\": \(Program.iterations),\n"
s += " \"instruction_count\": \(Program.count),\n"
s += " \"loads_per_hash\": \(p.loadsPerHash),\n"
s += " \"op_mix\": {\(p.histogram.map { "\(jstr($0.0)): \($0.1)" }.joined(separator: ", "))},\n"
s += " \"register_init\": \"for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]\",\n"
s += " \"splitmix32\": \"x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16\",\n"
s += " \"iteration\": \"sel = r0 sampled once at the top of each iteration, then all instructions in order\",\n"
s += " \"output\": \"lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo\",\n"
s += " \"op_semantics\": {\n"
s += " \"add\": \"dst = dst + src + (bit `bit` of sel ? imm2 : imm)\",\n"
s += " \"sub\": \"dst = dst - src\",\n"
s += " \"mul\": \"dst = dst * src (low 32)\",\n"
s += " \"mulhi\": \"dst = high 32 bits of dst * src\",\n"
s += " \"xor\": \"dst = dst ^ src\",\n"
s += " \"or\": \"dst = dst | src\",\n"
s += " \"rotl\": \"dst = rotl(dst, rot), rot in 1..31\",\n"
s += " \"rotr\": \"dst = rotr(dst, src & 31)\",\n"
s += " \"mad\": \"dst = src * src2 + dst\",\n"
s += " \"shfl\": \"dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp\",\n"
s += " \"load\": \"dst = dst ^ dataset[src & dataset.mask]\"\n"
s += " },\n"
s += " \"dataset\": {\n"
s += " \"log2_words\": \(datasetLog2),\n"
s += " \"bytes\": \(UInt64(1) << UInt64(datasetLog2 + 2)),\n"
s += " \"mask\": \(jhex(mask)),\n"
s += " \"day\": \(jstr(dayString)),\n"
s += " \"day_words_from\": \(jstr("day/" + dayString)),\n"
s += " \"d0\": \(jhex(day.0)),\n"
s += " \"d1\": \(jhex(day.1)),\n"
s += " \"formula\": \"x = i ^ d0; x *= 0x9E3779B1; x ^= x>>15; x += d1; x *= 0x85EBCA77; x ^= x>>13; x *= 0xC2B2AE3D; x ^= x>>16 (all mod 2^32)\"\n"
s += " },\n"
s += " \"instructions\": [\n"
for (k, ins) in p.instrs.enumerated() {
s += " {\"i\": \(k), \"op\": \(jstr(ins.op.rawValue)), \"dst\": \(ins.dst), \"src\": \(ins.a), \"src2\": \(ins.b), \"imm\": \(jhex(ins.imm)), \"imm2\": \(jhex(ins.imm2)), \"rot\": \(ins.rot), \"bit\": \(ins.bit), \"mask\": \(ins.mask)}"
s += k == p.instrs.count - 1 ? "\n" : ",\n"
}
s += " ]\n}\n"
return s
}
func generateVectorsJSON(_ p: Program, dayString: String, datasetLog2: Int, bases: [UInt32], outs: [[UInt64]], head: [UInt32], last: UInt32, mask: UInt32, source: String) -> String {
var s = "{\n"
s += " \"seed\": \(jstr(p.seedString)),\n"
s += " \"day\": \(jstr(dayString)),\n"
s += " \"dataset_log2_words\": \(datasetLog2),\n"
s += " \"mask\": \(jhex(mask)),\n"
s += " \"lanes\": 32,\n"
s += " \"source\": \(jstr(source)),\n"
s += " \"warps\": [\n"
for (i, o) in outs.enumerated() {
s += " {\"base_nonce\": \(bases[i]), \"expected\": [\n"
for row in 0..<4 {
s += " " + (0..<8).map { jhex64(o[row * 8 + $0]) }.joined(separator: ", ") + (row == 3 ? "\n" : ",\n")
}
s += i == outs.count - 1 ? " ]}\n" : " ]},\n"
}
s += " ],\n"
s += " \"dataset_head\": [\(head.map(jhex).joined(separator: ", "))],\n"
s += " \"dataset_last_index\": \(mask),\n"
s += " \"dataset_last\": \(jhex(last))\n"
s += "}\n"
return s
}
// Runs the Metal kernel for each base nonce (one 32-thread threadgroup each) and compares with `expected`.
func metalCrossCheck(_ p: Program, datasetLog2: Int, day: (UInt32, UInt32), bases: [UInt32], expected: [[UInt64]]) -> (ok: Bool, detail: String) {
guard let device = MTLCreateSystemDefaultDevice(), let queue = device.makeCommandQueue() else { return (false, "no Metal device") }
let words = 1 << datasetLog2
guard let dataset = device.makeBuffer(length: words * 4, options: .storageModePrivate),
let outBuf = device.makeBuffer(length: 32 * 8, options: .storageModeShared) else { return (false, "buffer allocation failed") }
do {
let flib = try device.makeLibrary(source: fillMSL, options: MTLCompileOptions())
let fpipe = try device.makeComputePipelineState(function: flib.makeFunction(name: "igneum_fill")!)
let hlib = try device.makeLibrary(source: generateMSL(p, datasetLog2: datasetLog2), options: MTLCompileOptions())
let hpipe = try device.makeComputePipelineState(function: hlib.makeFunction(name: "igneum_hash")!)
let cb = queue.makeCommandBuffer()!
let enc = cb.makeComputeCommandEncoder()!
enc.setComputePipelineState(fpipe)
enc.setBuffer(dataset, offset: 0, index: 0)
var d = (day.0, day.1)
enc.setBytes(&d, length: 8, index: 1)
enc.dispatchThreadgroups(MTLSize(width: words / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
enc.endEncoding()
cb.commit(); cb.waitUntilCompleted()
if let e = cb.error { return (false, "fill error \(e)") }
var bad = [String]()
for (i, base) in bases.enumerated() {
let cb2 = queue.makeCommandBuffer()!
let e2 = cb2.makeComputeCommandEncoder()!
e2.setComputePipelineState(hpipe)
e2.setBuffer(dataset, offset: 0, index: 0)
e2.setBuffer(outBuf, offset: 0, index: 1)
var b = base
e2.setBytes(&b, length: 4, index: 2)
e2.dispatchThreadgroups(MTLSize(width: 1, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 32, height: 1, depth: 1))
e2.endEncoding()
cb2.commit(); cb2.waitUntilCompleted()
if let e = cb2.error { return (false, "hash error \(e)") }
let ptr = outBuf.contents().bindMemory(to: UInt64.self, capacity: 32)
let got = (0..<32).map { ptr[$0] }
if got != expected[i] { bad.append("base \(base)") }
}
return (bad.isEmpty, bad.isEmpty ? "Metal GPU cross-check PASS \(bases.count)/\(bases.count) warps" : "Metal GPU cross-check FAIL: \(bad.joined(separator: ", "))")
} catch {
return (false, "Metal compile error \(error)")
}
}
func exportPack(_ opts: Options) -> Never {
let dir = opts.exportPack!
let program = generateProgram(seedString: opts.seed)
let dayWords = seedWords("day/" + opts.day)
let day = (dayWords[0], dayWords[1])
let mask = UInt32((1 << opts.datasetLog2) - 1)
print("igneum-bench --export-pack \(dir)")
print("seed \"\(opts.seed)\", day \"\(opts.day)\", dataset 2^\(opts.datasetLog2) words, loads/hash \(program.loadsPerHash)")
print("op mix: " + program.histogram.map { "\($0.0)=\($0.1)" }.joined(separator: " "))
let outs = packVectorBases.map { cpuWarp(program, baseNonce: $0, day: day, mask: mask) }
let head = (0..<16).map { datasetElem(UInt32($0), day.0, day.1) }
let last = datasetElem(mask, day.0, day.1)
var source = "proto-metal CPU interpreter (cpuWarp) on Apple M5 Max"
let check = metalCrossCheck(program, datasetLog2: opts.datasetLog2, day: day, bases: packVectorBases, expected: outs)
print(check.detail)
source += "; " + check.detail
if !check.ok { print("FAIL: vectors do not match the Metal GPU, pack not written"); exit(1) }
let files: [(String, String)] = [
("program.json", generateProgramJSON(program, dayString: opts.day, day: day, datasetLog2: opts.datasetLog2)),
("vectors.json", generateVectorsJSON(program, dayString: opts.day, datasetLog2: opts.datasetLog2, bases: packVectorBases, outs: outs, head: head, last: last, mask: mask, source: source)),
("kernel.cu", generateCUDA(program)),
("program.h", generateProgramHeader(program, dayString: opts.day, day: day, datasetLog2: opts.datasetLog2)),
("vectors.h", generateVectorsHeader(program, bases: packVectorBases, outs: outs, head: head, last: last, mask: mask, source: source)),
("program.metal", generateMSL(program, datasetLog2: opts.datasetLog2)),
]
do {
try FileManager.default.createDirectory(atPath: dir, withIntermediateDirectories: true)
for (name, text) in files {
try text.write(toFile: "\(dir)/\(name)", atomically: true, encoding: .utf8)
print("wrote \(dir)/\(name) (\(text.utf8.count) bytes)")
}
} catch {
print("FAIL: write error \(error)"); exit(1)
}
for (i, b) in packVectorBases.enumerated() {
print("vector warp base \(b): lane0 \(String(format: "%016llx", outs[i][0])) lane31 \(String(format: "%016llx", outs[i][31]))")
}
print("OVERALL: PASS (pack written)")
exit(0)
}
// MARK: - Timing
@inline(__always) func nowNs() -> UInt64 { clock_gettime_nsec_np(CLOCK_UPTIME_RAW) }
func ms(_ a: UInt64, _ b: UInt64) -> Double { Double(b - a) / 1e6 }
func fmt(_ v: Double, _ digits: Int = 2) -> String { String(format: "%.\(digits)f", v) }
// MARK: - GPU context
final class GPU {
let device: MTLDevice
let queue: MTLCommandQueue
init() {
guard let d = MTLCreateSystemDefaultDevice(), let q = d.makeCommandQueue() else {
print("FAIL: no Metal device"); exit(1)
}
device = d; queue = q
}
}
struct EpochResult {
var seed: String
var libraryMs: Double
var pipelineMs: Double
var hashesPerSecWall: Double
var hashesPerSecGPU: Double
var gbpsWall: Double
var gbpsGPU: Double
var loadsPerHash: Int
var verify: [(warp: Int, pass: Bool, ms: Double, repMs: Double)]
var allPass: Bool { verify.allSatisfy { $0.pass } }
}
func runEpoch(gpu: GPU, opts: Options, seedString: String, dataset: MTLBuffer, day: (UInt32, UInt32)) -> EpochResult {
let program = generateProgram(seedString: seedString)
let msl = generateMSL(program, datasetLog2: opts.datasetLog2)
if let dir = opts.dumpDir {
try? FileManager.default.createDirectory(atPath: dir, withIntermediateDirectories: true)
let safe = seedString.replacingOccurrences(of: "/", with: "_")
try? msl.write(toFile: "\(dir)/program-\(safe).metal", atomically: true, encoding: .utf8)
}
print("\n=== epoch seed \"\(seedString)\" ===")
print("program: \(Program.count) instructions x \(Program.iterations) iterations, loads/hash = \(program.loadsPerHash)")
print("op mix: " + program.histogram.map { "\($0.0)=\($0.1)" }.joined(separator: " "))
// Runtime compile
let t0 = nowNs()
let library: MTLLibrary
do {
let copts = MTLCompileOptions()
library = try gpu.device.makeLibrary(source: msl, options: copts)
} catch {
print("FAIL: Metal compile error:\n\(error)")
exit(1)
}
let t1 = nowNs()
guard let fn = library.makeFunction(name: "igneum_hash") else { print("FAIL: no igneum_hash"); exit(1) }
let pipeline: MTLComputePipelineState
do { pipeline = try gpu.device.makeComputePipelineState(function: fn) } catch {
print("FAIL: pipeline error: \(error)"); exit(1)
}
let t2 = nowNs()
let libMs = ms(t0, t1), pipeMs = ms(t1, t2)
print("compile: library \(fmt(libMs)) ms, pipeline \(fmt(pipeMs)) ms, total \(fmt(libMs + pipeMs)) ms")
print("threadExecutionWidth = \(pipeline.threadExecutionWidth), maxTotalThreadsPerThreadgroup = \(pipeline.maxTotalThreadsPerThreadgroup)")
if pipeline.threadExecutionWidth != 32 {
print("WARNING: threadExecutionWidth is not 32; the one-warp-per-threadgroup assumption does not hold on this device")
}
// Buffers
let n = 1 << opts.batchLog2
let groups = n / 32
guard let outBuf = gpu.device.makeBuffer(length: n * 8, options: .storageModeShared) else { print("FAIL: out buffer"); exit(1) }
func encodeBatch(_ cb: MTLCommandBuffer, base: UInt32) {
let enc = cb.makeComputeCommandEncoder()!
enc.setComputePipelineState(pipeline)
enc.setBuffer(dataset, offset: 0, index: 0)
enc.setBuffer(outBuf, offset: 0, index: 1)
var b = base
enc.setBytes(&b, length: 4, index: 2)
enc.dispatchThreadgroups(MTLSize(width: groups, height: 1, depth: 1),
threadsPerThreadgroup: MTLSize(width: 32, height: 1, depth: 1))
enc.endEncoding()
}
// Batch 0: warm-up and verification source (baseNonce 0)
do {
let cb = gpu.queue.makeCommandBuffer()!
encodeBatch(cb, base: 0)
let w0 = nowNs()
cb.commit(); cb.waitUntilCompleted()
let w1 = nowNs()
if let e = cb.error { print("FAIL: batch 0 error \(e)"); exit(1) }
print("warm-up batch: \(n) hashes in \(fmt(ms(w0, w1))) ms wall, \(fmt((cb.gpuEndTime - cb.gpuStartTime) * 1000)) ms GPU")
}
// Pick warps to verify from batch 0
var warps = [0, groups / 2 + 1, groups - 1]
var vr = SplitMix64(s: UInt64(program.seed[4]) | (UInt64(program.seed[5]) << 32))
while warps.count < opts.verifyWarps { warps.append(vr.below(groups)) }
warps = Array(warps.prefix(max(opts.verifyWarps, 1)))
let outPtr = outBuf.contents().bindMemory(to: UInt64.self, capacity: n)
var gpuOutputs = [[UInt64]]()
for w in warps { gpuOutputs.append((0..<32).map { outPtr[w * 32 + $0] }) }
// Timed batches
var cbs = [MTLCommandBuffer]()
for b in 0..<opts.batches {
let cb = gpu.queue.makeCommandBuffer()!
encodeBatch(cb, base: UInt32(truncatingIfNeeded: (b + 1) * n))
cbs.append(cb)
}
let s0 = nowNs()
for cb in cbs { cb.commit() }
cbs.last!.waitUntilCompleted()
let s1 = nowNs()
for cb in cbs { if let e = cb.error { print("FAIL: batch error \(e)"); exit(1) } }
let gpuSeconds = cbs.reduce(0.0) { $0 + ($1.gpuEndTime - $1.gpuStartTime) }
let wallSeconds = Double(s1 - s0) / 1e9
let totalHashes = Double(n * opts.batches)
let hpsWall = totalHashes / wallSeconds
let hpsGPU = totalHashes / gpuSeconds
let bytesPerHash = Double(program.loadsPerHash * 4)
let gbpsWall = hpsWall * bytesPerHash / 1e9
let gbpsGPU = hpsGPU * bytesPerHash / 1e9
print("timed: \(opts.batches) batches x \(n) hashes = \(Int(totalHashes)) hashes")
print(" wall \(fmt(wallSeconds * 1000)) ms -> \(fmt(hpsWall / 1e6, 3)) Mhash/s, \(fmt(gbpsWall)) GB/s useful (loads x 4 B)")
print(" GPU \(fmt(gpuSeconds * 1000)) ms -> \(fmt(hpsGPU / 1e6, 3)) Mhash/s, \(fmt(gbpsGPU)) GB/s useful (loads x 4 B)")
// CPU verification
let mask = UInt32((1 << opts.datasetLog2) - 1)
var verify = [(warp: Int, pass: Bool, ms: Double, repMs: Double)]()
for (i, w) in warps.enumerated() {
let base = UInt32(w * 32)
let c0 = nowNs()
let cpu = cpuWarp(program, baseNonce: base, day: day, mask: mask)
let c1 = nowNs()
// repeated runs for a steadier figure
let reps = 20
let r0 = nowNs()
var sink: UInt64 = 0
for _ in 0..<reps { sink ^= cpuWarp(program, baseNonce: base, day: day, mask: mask)[0] }
let r1 = nowNs()
let pass = cpu == gpuOutputs[i] && sink != 1
let single = ms(c0, c1), rep = ms(r0, r1) / Double(reps)
verify.append((w, pass, single, rep))
var detail = ""
if !pass {
let bad = (0..<32).filter { cpu[$0] != gpuOutputs[i][$0] }
detail = " mismatched lanes: \(bad) first: cpu=\(String(format: "%016llx", cpu[bad.first ?? 0])) gpu=\(String(format: "%016llx", gpuOutputs[i][bad.first ?? 0]))"
}
print("verify warp \(w) (nonces \(base)..\(base + 31)): \(pass ? "PASS" : "FAIL") cpu \(fmt(single, 3)) ms single, \(fmt(rep, 3)) ms avg of \(reps)\(detail)")
}
return EpochResult(seed: seedString, libraryMs: libMs, pipelineMs: pipeMs,
hashesPerSecWall: hpsWall, hashesPerSecGPU: hpsGPU, gbpsWall: gbpsWall, gbpsGPU: gbpsGPU,
loadsPerHash: program.loadsPerHash, verify: verify)
}
// MARK: - Hardening tests: shared helpers
//
// Added 3 October 2026. Everything below reuses generateProgram, generateMSL, fillMSL and cpuWarp
// unchanged; the helpers only wrap compile, fill and dispatch so the tests can run many programs
// and many warps cheaply. Nothing in the bench path (runEpoch) or the pack exporter calls these.
struct CompiledHash {
let pipeline: MTLComputePipelineState
let libraryMs: Double
let pipelineMs: Double
var totalMs: Double { libraryMs + pipelineMs }
}
struct IgneumError: Error, CustomStringConvertible {
let description: String
init(_ s: String) { description = s }
}
func compileHash(_ gpu: GPU, msl: String) throws -> CompiledHash {
let t0 = nowNs()
let lib = try gpu.device.makeLibrary(source: msl, options: MTLCompileOptions())
let t1 = nowNs()
guard let fn = lib.makeFunction(name: "igneum_hash") else { throw IgneumError("no igneum_hash function in library") }
let pipe = try gpu.device.makeComputePipelineState(function: fn)
let t2 = nowNs()
return CompiledHash(pipeline: pipe, libraryMs: ms(t0, t1), pipelineMs: ms(t1, t2))
}
// Allocates a private 2^log2-word dataset and fills it on the GPU with the closed form for `day`.
func makeDataset(_ gpu: GPU, log2: Int, day: (UInt32, UInt32)) -> MTLBuffer {
let words = 1 << log2
guard let buf = gpu.device.makeBuffer(length: words * 4, options: .storageModePrivate) else {
print("FAIL: cannot allocate 2^\(log2) word dataset"); exit(1)
}
do {
let lib = try gpu.device.makeLibrary(source: fillMSL, options: MTLCompileOptions())
let pipe = try gpu.device.makeComputePipelineState(function: lib.makeFunction(name: "igneum_fill")!)
let cb = gpu.queue.makeCommandBuffer()!
let enc = cb.makeComputeCommandEncoder()!
enc.setComputePipelineState(pipe)
enc.setBuffer(buf, offset: 0, index: 0)
var d = (day.0, day.1)
enc.setBytes(&d, length: 8, index: 1)
enc.dispatchThreadgroups(MTLSize(width: words / 256, height: 1, depth: 1),
threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
enc.endEncoding()
cb.commit(); cb.waitUntilCompleted()
if let e = cb.error { print("FAIL: fill error \(e)"); exit(1) }
} catch {
print("FAIL: fill kernel \(error)"); exit(1)
}
return buf
}
// One 32-thread threadgroup per base nonce, all dispatched from one encoder. Warp i lands at byte offset
// i * 256 of the output buffer, which is pre-filled with a sentinel so an unwritten lane is visible.
func gpuWarps(_ gpu: GPU, _ k: CompiledHash, dataset: MTLBuffer, bases: [UInt32]) -> [[UInt64]]? {
guard !bases.isEmpty, let outBuf = gpu.device.makeBuffer(length: bases.count * 256, options: .storageModeShared) else { return nil }
memset(outBuf.contents(), 0xAA, outBuf.length)
let cb = gpu.queue.makeCommandBuffer()!
let enc = cb.makeComputeCommandEncoder()!
enc.setComputePipelineState(k.pipeline)
enc.setBuffer(dataset, offset: 0, index: 0)
for (i, base) in bases.enumerated() {
enc.setBuffer(outBuf, offset: i * 256, index: 1)
var b = base
enc.setBytes(&b, length: 4, index: 2)
enc.dispatchThreadgroups(MTLSize(width: 1, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 32, height: 1, depth: 1))
}
enc.endEncoding()
cb.commit(); cb.waitUntilCompleted()
if cb.error != nil { return nil }
let p = outBuf.contents().bindMemory(to: UInt64.self, capacity: bases.count * 32)
return (0..<bases.count).map { i in (0..<32).map { p[i * 32 + $0] } }
}
// `count` consecutive nonces from `base` (count a multiple of 32) into `out`. Returns GPU time in ms.
func gpuRange(_ gpu: GPU, _ k: CompiledHash, dataset: MTLBuffer, base: UInt32, count: Int, out: MTLBuffer) -> Double? {
let cb = gpu.queue.makeCommandBuffer()!
let enc = cb.makeComputeCommandEncoder()!
enc.setComputePipelineState(k.pipeline)
enc.setBuffer(dataset, offset: 0, index: 0)
enc.setBuffer(out, offset: 0, index: 1)
var b = base
enc.setBytes(&b, length: 4, index: 2)
enc.dispatchThreadgroups(MTLSize(width: count / 32, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 32, height: 1, depth: 1))
enc.endEncoding()
cb.commit(); cb.waitUntilCompleted()
if cb.error != nil { return nil }
return (cb.gpuEndTime - cb.gpuStartTime) * 1000
}
func h64(_ v: UInt64) -> String { String(format: "%016llx", v) }
func pad(_ s: String, _ n: Int) -> String { s.count >= n ? s : s + String(repeating: " ", count: n - s.count) }
func describeProgram(_ p: Program) -> String {
var s = " program seed \"\(p.seedString)\" words [\(p.seed.map(hex).joined(separator: ", "))], \(p.instrs.count) instructions\n"
for (k, i) in p.instrs.enumerated() {
s += " \(pad(String(k), 3)) \(pad(i.op.rawValue, 5)) dst=r\(i.dst) src=r\(i.a) src2=r\(i.b) imm=\(hex(i.imm)) imm2=\(hex(i.imm2)) rot=\(i.rot) bit=\(i.bit) mask=\(i.mask)\n"
}
return s
}
func fnv64(_ ptr: UnsafeRawPointer, _ count: Int) -> UInt64 {
var h: UInt64 = 0xcbf29ce484222325
let b = ptr.bindMemory(to: UInt8.self, capacity: count)
for i in 0..<count { h ^= UInt64(b[i]); h &*= 0x100000001b3 }
return h
}
func regexCount(_ pattern: String, in text: String) -> Int {
let re = try! NSRegularExpression(pattern: pattern)
return re.numberOfMatches(in: text, range: NSRange(text.startIndex..., in: text))
}
// Static check: every dataset access in the generated MSL is `dataset[rN & MASK]`, and the identifier
// `dataset` appears nowhere else except the kernel parameter.
func maskCheckMSL(_ msl: String) -> (ok: Bool, detail: String) {
let total = regexCount("dataset\\[", in: msl)
let masked = regexCount("dataset\\[r[0-7] & MASK\\]", in: msl)
let words = regexCount("\\bdataset\\b", in: msl)
let ok = total == masked && words == total + 1
return (ok, "MSL: \(total) dataset[ accesses, \(masked) of the form dataset[rN & MASK], identifier appears \(words) times (expected \(total + 1))")
}
// Same for the CUDA twin: hash accesses are `ds[rN & mask]`; the fill kernel's one write is guarded by `if (i < n)`.
func maskCheckCUDA(_ cu: String) -> (ok: Bool, detail: String) {
let total = regexCount("\\bds\\[", in: cu)
let masked = regexCount("\\bds\\[r[0-7] & mask\\]", in: cu)
let fill = regexCount("if \\(i < n\\) ds\\[i\\] = ds_elem", in: cu)
let ok = total == masked + fill && fill == 1
return (ok, "CUDA: \(total) ds[ accesses, \(masked) of the form ds[rN & mask], \(fill) guarded fill write")
}
// MARK: - --fuzz
func runFuzz(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
let n = max(opts.fuzz ?? 200, 1)
let master = opts.fuzzSeed
print("\n=== fuzz: \(n) random programs, master seed \"\(master)\", 4 random warps each ===")
let sizes = [24, 26, 28]
let d0 = nowNs()
var datasets = [Int: MTLBuffer]()
for s in sizes { datasets[s] = makeDataset(gpu, log2: s, day: day) }
print("datasets " + sizes.map { "2^\($0) (\((1 << $0) * 4 / (1 << 20)) MiB)" }.joined(separator: ", ") + " filled in \(fmt(ms(d0, nowNs()), 1)) ms")
let mw = seedWords("fuzz/" + master)
var rng = SplitMix64(s: UInt64(mw[0]) | (UInt64(mw[1]) << 32))
var pass = 0, fail = 0, compileFail = 0, staticFail = 0, contractFail = 0
var perSize = [Int: (pass: Int, fail: Int)]()
var compileMs = [Double]()
var cpuNs: UInt64 = 0, gpuNs: UInt64 = 0
var warps = 0
var opCount = [String: Int]()
var loadsMin = Int.max, loadsMax = 0
let t0 = nowNs()
for i in 0..<n {
let seedString = "\(master)/\(i)/\(h64(rng.next()))"
let log2 = sizes[rng.below(sizes.count)]
let bases = (0..<4).map { _ in UInt32(truncatingIfNeeded: rng.next()) }
let program = generateProgram(seedString: seedString)
for ins in program.instrs {
opCount[ins.op.rawValue, default: 0] += 1
// Generator contract, relied on by the MSL emitter: rotl amount 1..31, shuffle mask a power of two <= 16,
// source register never the destination.
if ins.rot < 1 || ins.rot > 31 || ![1, 2, 4, 8, 16].contains(ins.mask) || ins.a == ins.dst || ins.dst > 7 || ins.a > 7 || ins.b > 7 {
contractFail += 1
print("CONTRACT FAIL seed \"\(seedString)\": \(ins)")
}
}
loadsMin = min(loadsMin, program.loadsPerHash); loadsMax = max(loadsMax, program.loadsPerHash)
let msl = generateMSL(program, datasetLog2: log2)
let sc = maskCheckMSL(msl)
if !sc.ok { staticFail += 1; print("STATIC MASK FAIL seed \"\(seedString)\": \(sc.detail)") }
let k: CompiledHash
do { k = try compileHash(gpu, msl: msl) } catch {
compileFail += 1
print("COMPILE FAIL seed \"\(seedString)\" dataset 2^\(log2):\n\(error)\n\(describeProgram(program))")
continue
}
compileMs.append(k.totalMs)
let g0 = nowNs()
guard let gpuOut = gpuWarps(gpu, k, dataset: datasets[log2]!, bases: bases) else {
fail += 1; print("GPU RUN FAIL seed \"\(seedString)\" dataset 2^\(log2)"); continue
}
let g1 = nowNs()
let mask = UInt32((1 << log2) - 1)
var ok = true
for (w, base) in bases.enumerated() {
let cpu = cpuWarp(program, baseNonce: base, day: day, mask: mask)
warps += 1
if cpu != gpuOut[w] {
ok = false
let bad = (0..<32).filter { cpu[$0] != gpuOut[w][$0] }
print("MISMATCH seed \"\(seedString)\" dataset 2^\(log2) warp \(w) base nonce \(base) (\(hex(base))) lanes \(bad)")
for l in bad { print(" lane \(l) nonce \(base &+ UInt32(l)): gpu \(h64(gpuOut[w][l])) cpu \(h64(cpu[l]))") }
print(describeProgram(program))
}
}
let g2 = nowNs()
gpuNs += g1 - g0; cpuNs += g2 - g1
if ok { pass += 1 } else { fail += 1 }
var ps = perSize[log2] ?? (0, 0)
if ok { ps.pass += 1 } else { ps.fail += 1 }
perSize[log2] = ps
if (i + 1) % 100 == 0 || i + 1 == n {
print(" \(i + 1)/\(n): pass \(pass) fail \(fail) compile-fail \(compileFail), \(fmt(Double(nowNs() - t0) / 1e9, 1)) s elapsed")
}
}
let total = Double(nowNs() - t0) / 1e9
let cAvg = compileMs.isEmpty ? 0 : compileMs.reduce(0, +) / Double(compileMs.count)
print("\n| Dataset | Programs | Pass | Fail |")
print("|---|---|---|---|")
for s in sizes {
let ps = perSize[s] ?? (0, 0)
print("| 2^\(s) words (\((1 << s) * 4 / (1 << 20)) MiB) | \(ps.pass + ps.fail) | \(ps.pass) | \(ps.fail) |")
}
print("| all | \(pass + fail) | \(pass) | \(fail) |")
print("programs \(n): pass \(pass), mismatch \(fail), compile failures \(compileFail), static mask failures \(staticFail), generator contract failures \(contractFail)")
print("warps compared \(warps) (\(warps * 32) hashes), loads/hash range \(loadsMin)..\(loadsMax)")
print("op totals over all programs: " + opCount.sorted { $0.value != $1.value ? $0.value > $1.value : $0.key < $1.key }.map { "\($0.key)=\($0.value)" }.joined(separator: " "))
print("compile ms (library+pipeline): min \(fmt(compileMs.min() ?? 0, 1)) avg \(fmt(cAvg, 1)) max \(fmt(compileMs.max() ?? 0, 1)); GPU dispatch total \(fmt(Double(gpuNs) / 1e6, 1)) ms; CPU interpreter total \(fmt(Double(cpuNs) / 1e6, 1)) ms; wall \(fmt(total, 1)) s")
let ok = fail == 0 && compileFail == 0 && staticFail == 0 && contractFail == 0 && pass == n
print("FUZZ: \(ok ? "PASS" : "FAIL")")
return ok
}
// MARK: - --edge
struct EdgeCase {
let name: String
let instrs: [Instr]
// (instruction index, what must hold, check on lane-0 registers as they are just before that instruction)
let pre: [(Int, String, ([UInt32]) -> Bool)]
let informational: Bool // reported but not counted: exercises something the generator never emits
}
// Instruction builder for hand-made programs. imm2 = imm so the `add` is a constant regardless of the selector bit.
func I(_ op: Op, _ dst: Int, _ a: Int, b: Int = 0, imm: UInt32 = 0, rot: UInt32 = 1, bit: Int = 0, mask: Int = 1) -> Instr {
Instr(op: op, dst: dst, a: a, b: b, imm: imm, imm2: imm, rot: rot, bit: bit, mask: mask)
}
func zero(_ r: Int) -> Instr { I(.sub, r, r) } // r = r - r = 0 (src == dst, never generated, legal MSL)
func set(_ r: Int, _ v: UInt32) -> [Instr] { [zero(r), I(.add, r, 7, imm: v)] } // needs r7 == 0
func edgeCases(mask: UInt32) -> [EdgeCase] {
let M = mask
var c = [EdgeCase]()
c.append(EdgeCase(name: "rotl immediate by 1 and by 31",
instrs: [I(.rotl, 1, 0, rot: 1), I(.rotl, 2, 0, rot: 31), I(.xor, 3, 1), I(.xor, 4, 2), I(.rotl, 5, 0, rot: 1), I(.rotl, 6, 0, rot: 31)],
pre: [], informational: false))
c.append(EdgeCase(name: "rotr by register == 0",
instrs: [zero(7), I(.rotr, 3, 7), I(.xor, 4, 3)],
pre: [(1, "r7 == 0", { $0[7] == 0 })], informational: false))
c.append(EdgeCase(name: "rotr by register == 32 (32 mod 32 = 0)",
instrs: [zero(7)] + (set(1, 32) + [I(.rotr, 3, 1), I(.xor, 4, 3)]),
pre: [(3, "r1 == 32", { $0[1] == 32 })], informational: false))
c.append(EdgeCase(name: "rotr by register == 0xFFFFFFE0 (-32, 0 mod 32)",
instrs: [zero(7)] + (set(1, 0xFFFFFFE0) + [I(.rotr, 3, 1), I(.xor, 4, 3)]),
pre: [(3, "r1 == 0xFFFFFFE0", { $0[1] == 0xFFFFFFE0 })], informational: false))
var rr: [Instr] = [zero(7)]
rr += set(1, 31); rr += [I(.rotr, 3, 1), I(.add, 1, 7, imm: 32), I(.rotr, 4, 1)]
rr += set(2, 1); rr += [I(.rotr, 5, 2), I(.xor, 6, 5)]
c.append(EdgeCase(name: "rotr by register == 31 and == 63 and == 1",
instrs: rr,
pre: [(3, "r1 == 31", { $0[1] == 31 }), (5, "r1 == 63", { $0[1] == 63 }), (8, "r2 == 1", { $0[2] == 1 })], informational: false))
var mh: [Instr] = [zero(7)]
mh += set(1, 0xFFFFFFFF); mh += set(2, 0xFFFFFFFF)
mh += [I(.mulhi, 1, 2), I(.xor, 3, 1)]
mh += set(4, 0x80000000); mh += set(5, 2)
mh += [I(.mulhi, 4, 5), I(.xor, 3, 4), I(.mulhi, 6, 7), I(.xor, 0, 6)]
c.append(EdgeCase(name: "mulhi 0xFFFFFFFF x 0xFFFFFFFF, 0x80000000 x 2, x 0",
instrs: mh,
pre: [(5, "r1 == r2 == 0xFFFFFFFF", { $0[1] == 0xFFFFFFFF && $0[2] == 0xFFFFFFFF }),
(6, "mulhi result r1 == 0xFFFFFFFE", { $0[1] == 0xFFFFFFFE }),
(11, "r4 == 0x80000000, r5 == 2", { $0[4] == 0x80000000 && $0[5] == 2 }),
(12, "mulhi result r4 == 1", { $0[4] == 1 }),
(13, "r7 == 0", { $0[7] == 0 }),
(14, "mulhi by 0 gives r6 == 0", { $0[6] == 0 })], informational: false))
c.append(EdgeCase(name: "shfl_xor every mask 1..16 in sequence (generator uses only 1,2,4,8,16)",
instrs: (1...16).map { I(.shfl, $0 % 8, ($0 + 1) % 8, mask: $0) },
pre: [], informational: false))
var l0: [Instr] = [zero(7), I(.load, 3, 7)]
l0 += set(1, M &+ 1); l0 += [I(.load, 4, 1), I(.xor, 5, 4)]
c.append(EdgeCase(name: "load at index 0 (register 0, and register MASK+1 which masks to 0)",
instrs: l0,
pre: [(1, "r7 & MASK == 0", { $0[7] & M == 0 }),
(4, "r1 == MASK+1, so unmasked index is out of range and masked index is 0", { $0[1] == M &+ 1 && ($0[1] & M) == 0 })],
informational: false))
var lm: [Instr] = [zero(7)]
lm += set(1, M); lm += [I(.load, 3, 1)]
lm += set(2, 0xFFFFFFFF); lm += [I(.load, 4, 2), I(.xor, 5, 4)]
c.append(EdgeCase(name: "load at index MASK (register MASK, and register 0xFFFFFFFF which masks to MASK)",
instrs: lm,
pre: [(3, "r1 == MASK", { $0[1] == M }),
(6, "r2 == 0xFFFFFFFF, masked index == MASK", { $0[2] == 0xFFFFFFFF && ($0[2] & M) == M })],
informational: false))
c.append(EdgeCase(name: "add wraparound 0xFFFFFFFF + 1",
instrs: [zero(7)] + (set(1, 0xFFFFFFFF) + [I(.add, 1, 7, imm: 1), I(.xor, 2, 1)]),
pre: [(3, "r1 == 0xFFFFFFFF", { $0[1] == 0xFFFFFFFF }), (4, "r1 == 0 after add", { $0[1] == 0 })], informational: false))
c.append(EdgeCase(name: "sub wraparound 0 - 1",
instrs: [zero(7), zero(1)] + (set(2, 1) + [I(.sub, 1, 2), I(.xor, 3, 1)]),
pre: [(4, "r1 == 0, r2 == 1", { $0[1] == 0 && $0[2] == 1 }), (5, "r1 == 0xFFFFFFFF after sub", { $0[1] == 0xFFFFFFFF })], informational: false))
var mm: [Instr] = [zero(7)]
mm += set(1, 0xFFFFFFFF); mm += set(2, 0xFFFFFFFF)
mm += [I(.mul, 1, 2), I(.xor, 3, 1)]
mm += set(4, 0xFFFFFFFF); mm += set(5, 0xFFFFFFFF); mm += set(6, 5)
mm += [I(.mad, 6, 4, b: 5), I(.xor, 0, 6)]
c.append(EdgeCase(name: "mul and mad wraparound 0xFFFFFFFF x 0xFFFFFFFF",
instrs: mm,
pre: [(5, "r1 == r2 == 0xFFFFFFFF", { $0[1] == 0xFFFFFFFF && $0[2] == 0xFFFFFFFF }),
(6, "mul low result r1 == 1", { $0[1] == 1 }),
(13, "r4 == r5 == 0xFFFFFFFF, r6 == 5", { $0[4] == 0xFFFFFFFF && $0[5] == 0xFFFFFFFF && $0[6] == 5 }),
(14, "mad result r6 == 6", { $0[6] == 6 })], informational: false))
let z = generateProgram(seedString: "edge/zero-loads")
c.append(EdgeCase(name: "generated program with every load replaced by xor (zero loads)",
instrs: z.instrs.map { ins in var m = ins; if m.op == .load { m.op = .xor }; return m },
pre: [], informational: false))
c.append(EdgeCase(name: "64 loads and nothing else",
instrs: (0..<64).map { I(.load, $0 % 8, ($0 + 3) % 8) },
pre: [], informational: false))
c.append(EdgeCase(name: "rotl immediate by 0 (outside the generator's 1..31 contract; MSL shifts by 32)",
instrs: [I(.rotl, 1, 0, rot: 0), I(.xor, 2, 1)],
pre: [], informational: true))
return c
}
func runEdge(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
let log2 = opts.datasetLog2
let mask = UInt32((1 << log2) - 1)
print("\n=== edge cases, dataset 2^\(log2) words, MASK \(hex(mask)) ===")
let dataset = makeDataset(gpu, log2: log2, day: day)
let bases: [UInt32] = [0, 1 << 20, 0x7FFFFFF0, 0xFFFFFFE0]
print("warps: base nonces " + bases.map { hex($0) }.joined(separator: ", ") + " (the last two straddle 2^31 and wrap past 2^32)")
var allOk = true
var rows = [String]()
for ec in edgeCases(mask: mask) {
let program = Program(seedString: "edge/\(ec.name)", seed: seedWords("edge/\(ec.name)"), instrs: ec.instrs)
let msl = generateMSL(program, datasetLog2: log2)
var status = "", detail = ""
var ok = true
// Preconditions, checked on lane 0 of every warp in every iteration.
var preOk = true
var preNotes = [String]()
if !ec.pre.isEmpty {
for base in bases {
var hits = [Int: Int]()
var misses = [Int: Int]()
_ = cpuWarpTraced(program, baseNonce: base, day: day, mask: mask) { _, k, regs in
for (idx, _, check) in ec.pre where idx == k {
if check(regs) { hits[idx, default: 0] += 1 } else { misses[idx, default: 0] += 1 }
}
}
for (idx, what, _) in ec.pre {
if (misses[idx] ?? 0) > 0 || (hits[idx] ?? 0) != Program.iterations {
preOk = false
preNotes.append("base \(hex(base)) instr \(idx) '\(what)' held \(hits[idx] ?? 0)/\(Program.iterations) iterations")
}
}
}
if preOk { preNotes = ec.pre.map { "instr \($0.0): \($0.1)" } }
}
do {
let k = try compileHash(gpu, msl: msl)
guard let gpuOut = gpuWarps(gpu, k, dataset: dataset, bases: bases) else { throw IgneumError("GPU run failed") }
var badLanes = 0
var first = ""
for (w, base) in bases.enumerated() {
let cpu = cpuWarp(program, baseNonce: base, day: day, mask: mask)
for l in 0..<32 where cpu[l] != gpuOut[w][l] {
badLanes += 1
if first.isEmpty { first = "first: base \(hex(base)) lane \(l) gpu \(h64(gpuOut[w][l])) cpu \(h64(cpu[l]))" }
}
}
ok = badLanes == 0 && preOk
status = badLanes == 0 ? "GPU == CPU 128/128 lanes" : "MISMATCH \(badLanes)/128 lanes, \(first)"
detail = "compile \(fmt(k.totalMs, 1)) ms"
if badLanes > 0 { print(describeProgram(program)) }
} catch {
ok = false
status = "COMPILE FAIL: \(error)"
}
let pre = ec.pre.isEmpty ? "none needed" : (preOk ? "held (all 8 iterations, lane 0, 4 warps)" : "NOT HELD")
let verdict = ec.informational ? (ok ? "info: agrees" : "info: differs") : (ok ? "PASS" : "FAIL")
if !ec.informational && !ok { allOk = false }
print("\(verdict): \(ec.name)")
print(" \(ec.instrs.count) instructions, loads/hash \(program.loadsPerHash), \(status), \(detail)")
for n in preNotes { print(" precondition \(n)") }
rows.append("| \(ec.name) | \(ec.instrs.count) | \(program.loadsPerHash) | \(pre) | \(status) | \(verdict) |")
}
print("\n| Case | Instrs | Loads/hash | Preconditions | GPU vs CPU | Result |")
print("|---|---|---|---|---|---|")
for r in rows { print(r) }
print("EDGE: \(allOk ? "PASS" : "FAIL")")
return allOk
}
// MARK: - --stats
func popcount64(_ v: UInt64) -> Int { v.nonzeroBitCount }
func runStats(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
let log2 = opts.datasetLog2
let mask = UInt32((1 << log2) - 1)
let n = 1 << 20
print("\n=== output statistics, 2^20 consecutive nonces per seed, dataset 2^\(log2) words ===")
print("This is a sanity check for obvious structural bias. It is not a proof of cryptographic strength.")
let dataset = makeDataset(gpu, log2: log2, day: day)
guard let outBuf = gpu.device.makeBuffer(length: n * 8, options: .storageModeShared) else { print("FAIL: out buffer"); return false }
let seeds = [opts.seed, "\(opts.seed)/stats1", "\(opts.seed)/stats2"]
var allOk = true
var rows = [String]()
for seedString in seeds {
let program = generateProgram(seedString: seedString)
let k: CompiledHash
do { k = try compileHash(gpu, msl: generateMSL(program, datasetLog2: log2)) } catch { print("FAIL: compile \(error)"); return false }
memset(outBuf.contents(), 0, n * 8)
guard let gms = gpuRange(gpu, k, dataset: dataset, base: 0, count: n, out: outBuf) else { print("FAIL: GPU run"); return false }
let p = outBuf.contents().bindMemory(to: UInt64.self, capacity: n)
let outs = (0..<n).map { p[$0] }
// Spot check 2 warps against the CPU so the statistics are known to describe the verified function.
var spot = true
for w in [0, (n / 32) - 1] {
let cpu = cpuWarp(program, baseNonce: UInt32(w * 32), day: day, mask: mask)
if cpu != Array(outs[(w * 32)..<(w * 32 + 32)]) { spot = false }
}
// (a) bit frequency per output bit position
var ones = [Int](repeating: 0, count: 64)
for v in outs { var x = v; var b = 0; while x != 0 { if x & 1 == 1 { ones[b] += 1 }; x >>= 1; b += 1 } }
let expected = Double(n) / 2, sigma = (Double(n) * 0.25).squareRoot()
var maxDev = 0.0, maxBit = 0
for b in 0..<64 { let d = abs(Double(ones[b]) - expected); if d > maxDev { maxDev = d; maxBit = b } }
let maxZ = maxDev / sigma
let minFreq = Double(ones.min()!) / Double(n), maxFreq = Double(ones.max()!) / Double(n)
// (c) chi-square over 65536 buckets for each 16-bit window of the output
var chiRows = [String]()
var chiWorstZ = 0.0
for shift in [0, 16, 32, 48] {
var buckets = [Int](repeating: 0, count: 65536)
for v in outs { buckets[Int((v >> UInt64(shift)) & 0xFFFF)] += 1 }
let e = Double(n) / 65536
var chi = 0.0
for c in buckets { let d = Double(c) - e; chi += d * d / e }
let df = 65535.0
let z = (chi - df) / (2 * df).squareRoot()
chiWorstZ = max(chiWorstZ, abs(z))
chiRows.append("bits \(shift)..\(shift + 15): chi2 \(fmt(chi, 0)) (df 65535, z \(fmt(z, 2)))")
}
// (d) duplicates
let sorted = outs.sorted()
var dups = 0
for i in 1..<n where sorted[i] == sorted[i - 1] { dups += 1 }
// (b) avalanche: 1000 random nonces, flip bit (i mod 32), count changed output bits. Run on the GPU.
let sw = seedWords("avalanche/" + seedString)
var rng = SplitMix64(s: UInt64(sw[0]) | (UInt64(sw[1]) << 32))
let trials = 1000
var bases = [UInt32]()
var flipped = [Int]()
for t in 0..<trials {
let nonce = UInt32(truncatingIfNeeded: rng.next())
let bit = t % 32
bases.append(nonce); bases.append(nonce ^ (1 << UInt32(bit)))
flipped.append(bit)
}
guard let av = gpuWarps(gpu, k, dataset: dataset, bases: bases) else { print("FAIL: avalanche GPU run"); return false }
var diffs = [Int]()
var perBitSum = [Int](repeating: 0, count: 32), perBitN = [Int](repeating: 0, count: 32)
var minDiff = 64, maxDiff = 0
for t in 0..<trials {
let d = popcount64(av[2 * t][0] ^ av[2 * t + 1][0])
diffs.append(d)
perBitSum[flipped[t]] += d; perBitN[flipped[t]] += 1
minDiff = min(minDiff, d); maxDiff = max(maxDiff, d)
}
let mean = Double(diffs.reduce(0, +)) / Double(trials)
let variance = diffs.reduce(0.0) { $0 + (Double($1) - mean) * (Double($1) - mean) } / Double(trials - 1)
let std = variance.squareRoot()
var perBitMin = 64.0, perBitMax = 0.0
for b in 0..<32 where perBitN[b] > 0 { let m = Double(perBitSum[b]) / Double(perBitN[b]); perBitMin = min(perBitMin, m); perBitMax = max(perBitMax, m) }
// Expected for an ideal function: mean 32, std 4 (binomial 64 x 0.5). Standard error of the mean over 1000 trials is 0.13.
let avOk = abs(mean - 32) < 0.6 && std > 3.3 && std < 4.7
let freqOk = maxZ < 4.5
let chiOk = chiWorstZ < 4.5
let dupOk = dups == 0
let ok = avOk && freqOk && chiOk && dupOk && spot
if !ok { allOk = false }
print("\nseed \"\(seedString)\": loads/hash \(program.loadsPerHash), GPU \(fmt(gms, 1)) ms for 2^20 hashes, CPU spot check 2 warps \(spot ? "PASS" : "FAIL")")
print(" (a) bit frequency: min \(fmt(minFreq, 4)) max \(fmt(maxFreq, 4)); largest deviation \(fmt(maxDev, 0)) counts at bit \(maxBit) = \(fmt(maxZ, 2)) sigma (sigma \(fmt(sigma, 0)), 64 bits, expect max under about 3.5)")
print(" (b) avalanche over \(trials) single-bit nonce flips: mean \(fmt(mean, 2)) std \(fmt(std, 2)) min \(minDiff) max \(maxDiff) of 64 bits (expect mean 32, std 4); per-input-bit mean range \(fmt(perBitMin, 1))..\(fmt(perBitMax, 1))")
for r in chiRows { print(" (c) \(r)") }
print(" (d) duplicate 64-bit outputs among 2^20: \(dups) (expected about 3e-8)")
print(" verdict: \(ok ? "no obvious bias" : "SUSPECT")")
rows.append("| \(seedString) | \(program.loadsPerHash) | \(fmt(minFreq, 4))..\(fmt(maxFreq, 4)) | \(fmt(maxZ, 2)) | \(fmt(mean, 2)) | \(fmt(std, 2)) | \(fmt(chiWorstZ, 2)) | \(dups) | \(ok ? "uniform-looking" : "SUSPECT") |")
}
print("\n| Seed | Loads/hash | Bit freq min..max | Max bit z | Avalanche mean | Avalanche std | Worst chi2 z (4 windows) | Dups | Verdict |")
print("|---|---|---|---|---|---|---|---|---|")
for r in rows { print(r) }
print("STATS: \(allOk ? "PASS (no obvious structural bias; not a security proof)" : "FAIL (something looks biased)")")
return allOk
}
// MARK: - --determinism
func runDeterminism(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
let log2 = opts.datasetLog2
let mask = UInt32((1 << log2) - 1)
let n = 1 << 20
print("\n=== determinism, seed \"\(opts.seed)\", 2^20 nonces from base 0, dataset 2^\(log2) words ===")
let dataset = makeDataset(gpu, log2: log2, day: day)
guard let outBuf = gpu.device.makeBuffer(length: n * 8, options: .storageModeShared) else { print("FAIL: out buffer"); return false }
var ok = true
// Generator and emitter determinism: two independent generations give identical MSL text.
let p1 = generateProgram(seedString: opts.seed), p2 = generateProgram(seedString: opts.seed)
let msl1 = generateMSL(p1, datasetLog2: log2), msl2 = generateMSL(p2, datasetLog2: log2)
let sameSource = msl1 == msl2
print("generator: two generations of the program give identical MSL source: \(sameSource ? "yes" : "NO") (\(msl1.utf8.count) bytes)")
if !sameSource { ok = false }
// Two separate compiles of the same source.
let k1: CompiledHash, k2: CompiledHash
do { k1 = try compileHash(gpu, msl: msl1); k2 = try compileHash(gpu, msl: msl2) } catch { print("FAIL: compile \(error)"); return false }
print("compiled twice: \(fmt(k1.totalMs, 1)) ms and \(fmt(k2.totalMs, 1)) ms")
func runOnce(_ k: CompiledHash) -> (fp: UInt64, sentinels: Int, ms: Double)? {
memset(outBuf.contents(), 0xAA, n * 8)
guard let gms = gpuRange(gpu, k, dataset: dataset, base: 0, count: n, out: outBuf) else { return nil }
let p = outBuf.contents().bindMemory(to: UInt64.self, capacity: n)
var s = 0
for i in 0..<n where p[i] == 0xAAAAAAAAAAAAAAAA { s += 1 }
return (fnv64(outBuf.contents(), n * 8), s, gms)
}
var reference = [UInt64]()
var fps = [UInt64]()
for run in 0..<5 {
guard let r = runOnce(k1) else { print("FAIL: GPU run \(run)"); return false }
let p = outBuf.contents().bindMemory(to: UInt64.self, capacity: n)
if run == 0 { reference = (0..<n).map { p[$0] } }
var differ = 0
for i in 0..<n where p[i] != reference[i] { differ += 1 }
fps.append(r.fp)
let same = differ == 0 && r.sentinels == 0
if !same { ok = false }
print("run \(run + 1)/5 (compile 1): fingerprint \(h64(r.fp)), \(differ) of \(n) outputs differ from run 1, \(r.sentinels) unwritten lanes, GPU \(fmt(r.ms, 1)) ms: \(same ? "identical" : "DIFFERENT")")
}
guard let r2 = runOnce(k2) else { print("FAIL: GPU run on compile 2"); return false }
let p = outBuf.contents().bindMemory(to: UInt64.self, capacity: n)
var differ2 = 0
for i in 0..<n where p[i] != reference[i] { differ2 += 1 }
if differ2 != 0 || r2.sentinels != 0 { ok = false }
print("run on compile 2: fingerprint \(h64(r2.fp)), \(differ2) outputs differ from compile 1 run 1, \(r2.sentinels) unwritten lanes: \(differ2 == 0 ? "identical" : "DIFFERENT")")
// CPU reference on the first and last warp and 6 others, so the fingerprint is tied to the verified function.
var cpuBad = 0
var vr = SplitMix64(s: 0x1234_5678_9abc_def0)
var warps = [0, n / 32 - 1]
while warps.count < 8 { warps.append(vr.below(n / 32)) }
for w in warps {
let cpu = cpuWarp(p1, baseNonce: UInt32(w * 32), day: day, mask: mask)
if cpu != Array(reference[(w * 32)..<(w * 32 + 32)]) { cpuBad += 1 }
}
if cpuBad != 0 { ok = false }
print("CPU interpreter on \(warps.count) warps of the reference run: \(cpuBad == 0 ? "all match" : "\(cpuBad) MISMATCH")")
// Dataset fill determinism and GPU-vs-CPU agreement of the dataset itself: fill a second buffer, blit both
// to shared memory, fingerprint, and compare sampled words (including 0 and MASK) with datasetElem.
let words = 1 << log2
let dataset2 = makeDataset(gpu, log2: log2, day: day)
var fillFps = [UInt64]()
var sampleBad = 0
if let shared = gpu.device.makeBuffer(length: words * 4, options: .storageModeShared) {
for (idx, ds) in [dataset, dataset2].enumerated() {
let cb = gpu.queue.makeCommandBuffer()!
let blit = cb.makeBlitCommandEncoder()!
blit.copy(from: ds, sourceOffset: 0, to: shared, destinationOffset: 0, size: words * 4)
blit.endEncoding()
cb.commit(); cb.waitUntilCompleted()
fillFps.append(fnv64(shared.contents(), words * 4))
if idx == 0 {
let dp = shared.contents().bindMemory(to: UInt32.self, capacity: words)
var sr = SplitMix64(s: 0xfeed_beef)
var idxs: [UInt32] = [0, 1, mask - 1, mask]
while idxs.count < 4096 { idxs.append(UInt32(sr.below(words))) }
for i in idxs where dp[Int(i)] != datasetElem(i, day.0, day.1) { sampleBad += 1 }
}
}
let sameFill = fillFps[0] == fillFps[1]
if !sameFill || sampleBad != 0 { ok = false }
print("dataset fill: two fills fingerprint \(h64(fillFps[0])) and \(h64(fillFps[1])): \(sameFill ? "identical" : "DIFFERENT"); 4096 sampled words (incl. 0, 1, MASK-1, MASK) vs CPU datasetElem: \(sampleBad == 0 ? "all match" : "\(sampleBad) MISMATCH")")
} else {
print("dataset fill check skipped: could not allocate a shared copy")
}
print("DETERMINISM: \(ok ? "PASS" : "FAIL")")
return ok
}
// MARK: - --memcheck
func runMemcheck(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
print("\n=== memcheck, seed \"\(opts.seed)\" ===")
var ok = true
let program = generateProgram(seedString: opts.seed)
// Static: every dataset index in the generated sources is masked. Checked at three dataset sizes because
// the MASK literal changes with size.
for log2 in [20, 24, 28] {
let msl = generateMSL(program, datasetLog2: log2)
let r = maskCheckMSL(msl)
if !r.ok { ok = false }
print("static 2^\(log2): \(r.ok ? "PASS" : "FAIL") \(r.detail)")
}
let cu = maskCheckCUDA(generateCUDA(program))
if !cu.ok { ok = false }
print("static CUDA twin: \(cu.ok ? "PASS" : "FAIL") \(cu.detail)")
print("program has \(program.instrs.filter { $0.op == .load }.count) load instructions (\(program.loadsPerHash) loads/hash)")
// Dynamic: a 4 MiB dataset with nonces at the top of the 32-bit range (they wrap to 0 inside the batch),
// one full batch of 2^20 nonces plus 4 warps verified against the CPU. Metal does not bounds-check device
// buffers, so "no crash" is weak evidence by itself; the static check above is the real guarantee.
let log2 = 20
let mask = UInt32((1 << log2) - 1)
let dataset = makeDataset(gpu, log2: log2, day: day)
let k: CompiledHash
do { k = try compileHash(gpu, msl: generateMSL(program, datasetLog2: log2)) } catch { print("FAIL: compile \(error)"); return false }
let n = 1 << 20
guard let outBuf = gpu.device.makeBuffer(length: n * 8, options: .storageModeShared) else { print("FAIL: out buffer"); return false }
for base: UInt32 in [0xFFF00000, 0xFFFFFFE0, 0x80000000, 0] {
if let gms = gpuRange(gpu, k, dataset: dataset, base: base, count: n, out: outBuf) {
print("dynamic 4 MiB: 2^20 nonces from base \(hex(base)) (last nonce \(hex(base &+ UInt32(n - 1)))): completed, GPU \(fmt(gms, 1)) ms")
} else { ok = false; print("dynamic 4 MiB: base \(hex(base)): GPU ERROR") }
}
let bases: [UInt32] = [0xFFFFFFE0, 0xFFFFFFFF, 0x80000000, 0xFFF00000]
guard let gpuOut = gpuWarps(gpu, k, dataset: dataset, bases: bases) else { print("FAIL: GPU warps"); return false }
var overMask = 0, loads = 0
for (w, base) in bases.enumerated() {
let cpu = cpuWarpTraced(program, baseNonce: base, day: day, mask: mask) { _, kk, regs in
let ins = program.instrs[kk]
if ins.op == .load { loads += 1; if regs[ins.a] > mask { overMask += 1 } }
}
let match = cpu == gpuOut[w]
if !match { ok = false }
print("verify warp base \(hex(base)): GPU vs CPU \(match ? "PASS" : "FAIL")")
}
print("in those 4 warps (lane 0, all iterations) \(overMask) of \(loads) load indices were above MASK before masking, so the mask was exercised")
print("MEMCHECK: \(ok ? "PASS" : "FAIL")")
return ok
}
// MARK: - Test dispatcher
func runTests(_ opts: Options) -> Never {
let gpu = GPU()
print("igneum-bench hardening tests")
print("GPU: \(gpu.device.name) (maxBufferLength \(gpu.device.maxBufferLength / (1 << 20)) MiB, unified memory \(gpu.device.hasUnifiedMemory)), day \"\(opts.day)\"")
let dayWords = seedWords("day/" + opts.day)
let day = (dayWords[0], dayWords[1])
var results = [(String, Bool)]()
let t0 = nowNs()
if opts.fuzz != nil { results.append(("fuzz", runFuzz(opts, gpu: gpu, day: day))) }
if opts.edge { results.append(("edge", runEdge(opts, gpu: gpu, day: day))) }
if opts.stats { results.append(("stats", runStats(opts, gpu: gpu, day: day))) }
if opts.determinism { results.append(("determinism", runDeterminism(opts, gpu: gpu, day: day))) }
if opts.memcheck { results.append(("memcheck", runMemcheck(opts, gpu: gpu, day: day))) }
print("\n=== tests summary (\(fmt(Double(nowNs() - t0) / 1e9, 1)) s) ===")
for (name, ok) in results { print("\(pad(name, 12)) \(ok ? "PASS" : "FAIL")") }
let all = results.allSatisfy { $0.1 }
print("OVERALL: \(all ? "PASS" : "FAIL")")
exit(all ? 0 : 1)
}
// MARK: - Main
let opts = parseArgs()
if opts.exportPack != nil { exportPack(opts) }
if opts.anyTest { runTests(opts) }
let gpu = GPU()
print("igneum-bench")
print("GPU: \(gpu.device.name) (maxBufferLength \(gpu.device.maxBufferLength / (1 << 20)) MiB, unified memory \(gpu.device.hasUnifiedMemory))")
print("dataset: 2^\(opts.datasetLog2) uint32 = \(fmt(Double(1 << opts.datasetLog2) * 4 / Double(1 << 20), 0)) MiB, day \"\(opts.day)\"")
let dayWords = seedWords("day/" + opts.day)
let day = (dayWords[0], dayWords[1])
let datasetWords = 1 << opts.datasetLog2
guard let dataset = gpu.device.makeBuffer(length: datasetWords * 4, options: .storageModePrivate) else {
print("FAIL: cannot allocate dataset buffer"); exit(1)
}
// Fill the dataset on the GPU, timed.
var fillMsWall = 0.0, fillMsGPU = 0.0
do {
let f0 = nowNs()
let lib = try gpu.device.makeLibrary(source: fillMSL, options: MTLCompileOptions())
let pipe = try gpu.device.makeComputePipelineState(function: lib.makeFunction(name: "igneum_fill")!)
let f1 = nowNs()
let cb = gpu.queue.makeCommandBuffer()!
let enc = cb.makeComputeCommandEncoder()!
enc.setComputePipelineState(pipe)
enc.setBuffer(dataset, offset: 0, index: 0)
var d = (day.0, day.1)
enc.setBytes(&d, length: 8, index: 1)
let tg = 256
enc.dispatchThreadgroups(MTLSize(width: datasetWords / tg, height: 1, depth: 1),
threadsPerThreadgroup: MTLSize(width: tg, height: 1, depth: 1))
enc.endEncoding()
let f2 = nowNs()
cb.commit(); cb.waitUntilCompleted()
let f3 = nowNs()
if let e = cb.error { print("FAIL: fill error \(e)"); exit(1) }
fillMsWall = ms(f2, f3)
fillMsGPU = (cb.gpuEndTime - cb.gpuStartTime) * 1000
let gib = Double(datasetWords * 4) / Double(1 << 30)
print("dataset fill: compile \(fmt(ms(f0, f1))) ms; fill \(fmt(fillMsWall)) ms wall, \(fmt(fillMsGPU)) ms GPU -> \(fmt(gib / (fillMsGPU / 1000))) GB/s write (GPU time)")
} catch {
print("FAIL: fill kernel: \(error)"); exit(1)
}
var results = [EpochResult]()
for epoch in 0..<max(opts.hours, 1) {
let seedString = epoch == 0 ? opts.seed : "\(opts.seed)/epoch\(epoch)"
results.append(runEpoch(gpu: gpu, opts: opts, seedString: seedString, dataset: dataset, day: day))
}
// Summary table
print("\n=== summary (\(gpu.device.name), dataset 2^\(opts.datasetLog2) words, batch 2^\(opts.batchLog2) x \(opts.batches)) ===")
print("| seed | compile ms (lib+pipe) | Mhash/s (wall) | GB/s useful (wall) | loads/hash | CPU verify ms/warp (avg) | verify |")
print("|---|---|---|---|---|---|---|")
for r in results {
let avg = r.verify.map { $0.repMs }.reduce(0, +) / Double(max(r.verify.count, 1))
print("| \(r.seed) | \(fmt(r.libraryMs + r.pipelineMs, 1)) | \(fmt(r.hashesPerSecWall / 1e6, 3)) | \(fmt(r.gbpsWall)) | \(r.loadsPerHash) | \(fmt(avg, 3)) | \(r.allPass ? "PASS" : "FAIL") (\(r.verify.count) warps) |")
}
let overall = results.allSatisfy { $0.allPass }
print("dataset fill: \(fmt(fillMsGPU)) ms GPU time for \(datasetWords * 4 / (1 << 20)) MiB")
print("OVERALL: \(overall ? "PASS" : "FAIL")")
exit(overall ? 0 : 1)