// igneum-bench: first prototype of Igneum's random-program GPU proof-of-work. // One file. Build: swiftc -O -o igneum-bench main.swift -framework Metal // Metal shaders are compiled at runtime from generated source (no Xcode needed). import Foundation import Metal // MARK: - Options struct Options { var seed = "igneum-genesis" var day = "2026-10-03" var hours = 2 // number of epochs (seeds) run in sequence; default 2 so verification covers 2 seeds var batchLog2 = 22 // nonces per batch var batches = 4 // timed batches var datasetLog2 = 28 // 2^28 uint32 = 1 GiB var verifyWarps = 3 var dumpDir: String? = nil var exportPack: String? = nil // write a CUDA program pack for --seed into this directory and exit // Hardening tests (added 3 October 2026). Any of these runs instead of the bench. var fuzz: Int? = nil // --fuzz N: N random programs, GPU vs CPU on 4 random warps each var fuzzSeed = "igneum-fuzz-2026-10-03" var edge = false // --edge: hand-built edge-case programs var stats = false // --stats: output distribution sanity checks on 2^20 nonces var determinism = false // --determinism: 5 identical GPU runs + double compile var memcheck = false // --memcheck: static mask check + 4 MiB run with wrapping nonces var anyTest: Bool { fuzz != nil || edge || stats || determinism || memcheck } } func parseArgs() -> Options { var o = Options() var args = Array(CommandLine.arguments.dropFirst()) func take() -> String { args.isEmpty ? "" : args.removeFirst() } while !args.isEmpty { let a = take() switch a { case "--seed": o.seed = take() case "--day": o.day = take() case "--hours": o.hours = Int(take()) ?? o.hours case "--batch-log2": o.batchLog2 = Int(take()) ?? o.batchLog2 case "--batches": o.batches = Int(take()) ?? o.batches case "--dataset-log2": o.datasetLog2 = Int(take()) ?? o.datasetLog2 case "--verify-warps": o.verifyWarps = Int(take()) ?? o.verifyWarps case "--dump": o.dumpDir = take() case "--export-pack": o.exportPack = take() case "--fuzz": o.fuzz = Int(take()) ?? 200 case "--fuzz-seed": o.fuzzSeed = take() case "--edge": o.edge = true case "--stats": o.stats = true case "--determinism": o.determinism = true case "--memcheck": o.memcheck = true case "-h", "--help": print(""" igneum-bench [--seed ] [--hours N] [--batch-log2 22] [--batches 4] [--dataset-log2 28] [--verify-warps 3] [--dump ] [--day ] [--export-pack ] write the CUDA program pack for --seed, then exit hardening tests (run instead of the bench; several may be combined; exit 0 only if all pass): [--fuzz N [--fuzz-seed ]] N random programs, GPU vs CPU, 4 random warps each, dataset size drawn from 64 MiB, 256 MiB, 1 GiB [--edge] hand-built edge-case programs, GPU vs CPU [--stats] output distribution sanity checks on 2^20 nonces, 3 seeds [--determinism] 5 identical GPU runs of 2^20 nonces, double compile, dataset fill check [--memcheck] static dataset-index mask check, 4 MiB run with wrapping nonces """) exit(0) default: print("unknown argument \(a)"); exit(2) } } return o } // MARK: - Integer helpers (CPU side, must match MSL bit for bit) @inline(__always) func rotl32(_ x: UInt32, _ n: UInt32) -> UInt32 { let n = n & 31 return n == 0 ? x : (x << n) | (x >> (32 - n)) } @inline(__always) func rotr32(_ x: UInt32, _ n: UInt32) -> UInt32 { let n = n & 31 return n == 0 ? x : (x >> n) | (x << (32 - n)) } @inline(__always) func mulhi32(_ a: UInt32, _ b: UInt32) -> UInt32 { UInt32(truncatingIfNeeded: (UInt64(a) &* UInt64(b)) >> 32) } @inline(__always) func splitmix32(_ v: UInt32) -> UInt32 { var x = v x ^= x >> 16; x &*= 0x7feb352d x ^= x >> 15; x &*= 0x846ca68b x ^= x >> 16 return x } // Dataset element, closed form of (daySeed, index). Same formula is emitted into the MSL. @inline(__always) func datasetElem(_ i: UInt32, _ d0: UInt32, _ d1: UInt32) -> UInt32 { var x = i ^ d0 x &*= 0x9E3779B1; x ^= x >> 15 x &+= d1 x &*= 0x85EBCA77; x ^= x >> 13 x &*= 0xC2B2AE3D; x ^= x >> 16 return x } // 32-byte seed (8 x uint32) from a string: FNV-1a 64 with four salts, each finalised. func seedWords(_ s: String) -> [UInt32] { var words = [UInt32]() for salt in 0..<4 { var h: UInt64 = 0xcbf29ce484222325 ^ (UInt64(salt) &* 0x9E3779B97F4A7C15) for b in s.utf8 { h ^= UInt64(b); h &*= 0x100000001b3 } h ^= h >> 33; h &*= 0xff51afd7ed558ccd; h ^= h >> 33 words.append(UInt32(truncatingIfNeeded: h)) words.append(UInt32(truncatingIfNeeded: h >> 32)) } return words } struct SplitMix64 { var s: UInt64 mutating func next() -> UInt64 { s &+= 0x9E3779B97F4A7C15 var z = s z = (z ^ (z >> 30)) &* 0xBF58476D1CE4E5B9 z = (z ^ (z >> 27)) &* 0x94D049BB133111EB return z ^ (z >> 31) } mutating func below(_ n: Int) -> Int { Int(next() % UInt64(n)) } } // MARK: - Program enum Op: String { case add, sub, mul, mulhi, xor, or, rotl, rotr, mad, shfl, load } struct Instr { var op: Op var dst: Int var a: Int // source register, never equal to dst var b: Int // second source (mad only) var imm: UInt32 // add immediate A var imm2: UInt32 // add immediate B var rot: UInt32 // rotl amount 1..31 var bit: Int // selector bit of r0 for add var mask: Int // shuffle xor mask: 1,2,4,8,16 } struct Program { let seedString: String let seed: [UInt32] let instrs: [Instr] static let iterations = 8 static let count = 64 var loadsPerHash: Int { instrs.filter { $0.op == .load }.count * Program.iterations } var histogram: [(String, Int)] { var d = [String: Int]() for i in instrs { d[i.op.rawValue, default: 0] += 1 } return d.sorted { $0.1 != $1.1 ? $0.1 > $1.1 : $0.0 < $1.0 } // count desc, then name, so output is deterministic } } // Weights sum to 100. Loads are 25 percent so the kernel leans on memory. let opWeights: [(Op, Int)] = [(.load, 25), (.add, 12), (.xor, 10), (.mul, 8), (.mad, 8), (.shfl, 8), (.rotl, 7), (.sub, 6), (.mulhi, 6), (.rotr, 6), (.or, 4)] func generateProgram(seedString: String) -> Program { let sw = seedWords(seedString) var rng = SplitMix64(s: (UInt64(sw[0]) | (UInt64(sw[1]) << 32)) ^ ((UInt64(sw[2]) | (UInt64(sw[3]) << 32)) &* 0x9E3779B97F4A7C15)) var instrs = [Instr]() for _ in 0..= dst { a += 1 } let b = rng.below(8) let imm = UInt32(truncatingIfNeeded: rng.next()) let imm2 = UInt32(truncatingIfNeeded: rng.next()) let rot = UInt32(1 + rng.below(31)) let bit = rng.below(32) let mask = 1 << rng.below(5) instrs.append(Instr(op: op, dst: dst, a: a, b: b, imm: imm, imm2: imm2, rot: rot, bit: bit, mask: mask)) } return Program(seedString: seedString, seed: sw, instrs: instrs) } // MARK: - MSL generation func hex(_ v: UInt32) -> String { String(format: "0x%08xu", v) } func generateMSL(_ p: Program, datasetLog2: Int) -> String { let mask = UInt32((1 << datasetLog2) - 1) var s = """ #include using namespace metal; #define MASK \(hex(mask)) constant uint SEEDW[8] = { \(p.seed.map(hex).joined(separator: ", ")) }; inline uint splitmix32(uint x) { x ^= x >> 16; x *= 0x7feb352du; x ^= x >> 15; x *= 0x846ca68bu; x ^= x >> 16; return x; } inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } inline uint ds_elem(uint i, uint d0, uint d1) { uint x = i ^ d0; x *= 0x9E3779B1u; x ^= x >> 15; x += d1; x *= 0x85EBCA77u; x ^= x >> 13; x *= 0xC2B2AE3Du; x ^= x >> 16; return x; } kernel void igneum_hash(device const uint* dataset [[buffer(0)]], device ulong* out [[buffer(1)]], constant uint& baseNonce [[buffer(2)]], uint gid [[thread_position_in_grid]]) { uint nonce = baseNonce + gid; uint r0, r1, r2, r3, r4, r5, r6, r7; """ for i in 0..<8 { s += " { uint x = nonce ^ SEEDW[\(i)]; x += 0x9e3779b9u * \(i + 1)u; x = splitmix32(x); r\(i) = x ^ SEEDW[\((i + 1) & 7)]; }\n" } s += "\n for (uint it = 0u; it < \(Program.iterations)u; ++it) {\n uint sel = r0;\n" for (k, ins) in p.instrs.enumerated() { let d = "r\(ins.dst)", a = "r\(ins.a)", b = "r\(ins.b)" var line: String switch ins.op { case .add: line = "\(d) = \(d) + \(a) + select(\(hex(ins.imm)), \(hex(ins.imm2)), ((sel >> \(ins.bit)u) & 1u) != 0u);" case .sub: line = "\(d) = \(d) - \(a);" case .mul: line = "\(d) = \(d) * \(a);" case .mulhi: line = "\(d) = mulhi(\(d), \(a));" case .xor: line = "\(d) = \(d) ^ \(a);" case .or: line = "\(d) = \(d) | \(a);" case .rotl: line = "\(d) = rotl_imm(\(d), \(ins.rot)u);" case .rotr: line = "\(d) = rotr_var(\(d), \(a));" case .mad: line = "\(d) = \(a) * \(b) + \(d);" case .shfl: line = "\(d) = \(d) ^ simd_shuffle_xor(\(a), (ushort)\(ins.mask));" case .load: line = "\(d) = \(d) ^ dataset[\(a) & MASK];" } s += " \(line) // \(k)\n" } s += """ } uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); out[gid] = ((ulong)hi << 32) | (ulong)lo; } """ return s } let fillMSL = """ #include using namespace metal; inline uint ds_elem(uint i, uint d0, uint d1) { uint x = i ^ d0; x *= 0x9E3779B1u; x ^= x >> 15; x += d1; x *= 0x85EBCA77u; x ^= x >> 13; x *= 0xC2B2AE3Du; x ^= x >> 16; return x; } kernel void igneum_fill(device uint* dataset [[buffer(0)]], constant uint2& day [[buffer(1)]], uint gid [[thread_position_in_grid]]) { dataset[gid] = ds_elem(gid, day.x, day.y); } """ // MARK: - CPU reference interpreter for one 32-lane warp func cpuWarp(_ p: Program, baseNonce: UInt32, day: (UInt32, UInt32), mask: UInt32) -> [UInt64] { cpuWarpTraced(p, baseNonce: baseNonce, day: day, mask: mask, trace: nil) } // Same interpreter with an optional hook. When `trace` is set it is called before every instruction with // (iteration, instruction index, the 8 registers of lane 0). The edge-case tests use it to prove that the // operand values they were built to produce really occurred. The bench passes nil. func cpuWarpTraced(_ p: Program, baseNonce: UInt32, day: (UInt32, UInt32), mask: UInt32, trace: ((Int, Int, [UInt32]) -> Void)?) -> [UInt64] { let lanes = 32 var r = [UInt32](repeating: 0, count: lanes * 8) // r[lane*8 + reg] for lane in 0..> UInt32(ins.bit)) & 1 v = d &+ a &+ (s != 0 ? ins.imm2 : ins.imm) case .sub: v = d &- a case .mul: v = d &* a case .mulhi: v = mulhi32(d, a) case .xor: v = d ^ a case .or: v = d | a case .rotl: v = rotl32(d, ins.rot) case .rotr: v = rotr32(d, a) case .mad: v = (a &* r[base + ins.b]) &+ d case .load: v = d ^ datasetElem(a & mask, day.0, day.1) case .shfl: v = d // unreachable } r[base + ins.dst] = v } } } } var out = [UInt64](repeating: 0, count: lanes) for lane in 0.. String { String(format: "0x%016llxull", v) } func jhex(_ v: UInt32) -> String { String(format: "\"0x%08x\"", v) } func jhex64(_ v: UInt64) -> String { String(format: "\"0x%016llx\"", v) } func jstr(_ s: String) -> String { var o = "\"" for c in s.unicodeScalars { switch c { case "\"": o += "\\\"" case "\\": o += "\\\\" case "\n": o += "\\n" default: o.unicodeScalars.append(c) } } return o + "\"" } func generateCUDA(_ p: Program) -> String { var s = """ // Generated by proto-metal/igneum-bench --export-pack for seed "\(p.seedString)". Do not edit by hand. // Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal). // Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC. #include #include #include "program.h" __device__ __forceinline__ uint32_t splitmix32(uint32_t x) { x ^= x >> 16; x *= 0x7feb352du; x ^= x >> 15; x *= 0x846ca68bu; x ^= x >> 16; return x; } // n is a literal in 1..31 at every call site, so both shift amounts are in 1..31. __device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x. __device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } __device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) { uint32_t x = i ^ d0; x *= 0x9E3779B1u; x ^= x >> 15; x += d1; x *= 0x85EBCA77u; x ^= x >> 13; x *= 0xC2B2AE3Du; x ^= x >> 16; return x; } // dataset[i] = ds_elem(i, d0, d1) for i < n. Same closed form as the Metal igneum_fill kernel. __global__ void igneum_fill(uint32_t* ds, uint32_t n, uint32_t d0, uint32_t d1) { uint32_t i = blockIdx.x * blockDim.x + threadIdx.x; if (i < n) ds[i] = ds_elem(i, d0, d1); } // One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every // __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a // 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid. __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask) { uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x; uint32_t nonce = baseNonce + gid; uint32_t r0, r1, r2, r3, r4, r5, r6, r7; """ for i in 0..<8 { let addc = 0x9e3779b9 &* UInt32(i + 1) s += " { uint32_t x = nonce ^ \(hex(p.seed[i])); x += \(hex(addc)); x = splitmix32(x); r\(i) = x ^ \(hex(p.seed[(i + 1) & 7])); } // SEEDW[\(i)], 0x9e3779b9u * \(i + 1)u, SEEDW[\((i + 1) & 7)]\n" } s += "\n for (uint32_t it = 0u; it < \(Program.iterations)u; ++it) {\n uint32_t sel = r0;\n" for (k, ins) in p.instrs.enumerated() { let d = "r\(ins.dst)", a = "r\(ins.a)", b = "r\(ins.b)" var line: String switch ins.op { // Metal select(A, B, c) returns c ? B : A, so the true branch is imm2 here as well. case .add: line = "\(d) = \(d) + \(a) + ((((sel >> \(ins.bit)u) & 1u) != 0u) ? \(hex(ins.imm2)) : \(hex(ins.imm)));" case .sub: line = "\(d) = \(d) - \(a);" case .mul: line = "\(d) = \(d) * \(a);" case .mulhi: line = "\(d) = __umulhi(\(d), \(a));" case .xor: line = "\(d) = \(d) ^ \(a);" case .or: line = "\(d) = \(d) | \(a);" case .rotl: line = "\(d) = rotl_imm(\(d), \(ins.rot)u);" case .rotr: line = "\(d) = rotr_var(\(d), \(a));" case .mad: line = "\(d) = \(a) * \(b) + \(d);" case .shfl: line = "\(d) = \(d) ^ __shfl_xor_sync(0xffffffffu, \(a), \(ins.mask));" case .load: line = "\(d) = \(d) ^ ds[\(a) & mask];" } s += " \(line) // \(k) \(ins.op.rawValue)\n" } s += """ } uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo; } // Host-side launch wrappers. Declared in program.h, called from host.cu. cudaError_t igneum_launch_fill(uint32_t* ds, uint32_t nWords, uint32_t d0, uint32_t d1) { if (nWords == 0u) return cudaErrorInvalidValue; uint32_t block = 256u; uint32_t grid = (nWords + block - 1u) / block; igneum_fill<<>>(ds, nWords, d0, d1); return cudaGetLastError(); } cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, uint32_t nonces, uint32_t blockWarps) { if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue; uint32_t block = 32u * blockWarps; if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue; igneum_hash<<>>(ds, out, baseNonce, mask); return cudaGetLastError(); } cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) { cudaFuncAttributes attr; cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash); if (e != cudaSuccess) return e; *numRegs = attr.numRegs; return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0); } """ return s } func generateProgramHeader(_ p: Program, dayString: String, day: (UInt32, UInt32), datasetLog2: Int) -> String { let mask = UInt32((1 << datasetLog2) - 1) let mix = p.histogram.map { "\($0.0)=\($0.1)" }.joined(separator: " ") return """ // Generated by proto-metal/igneum-bench --export-pack for seed "\(p.seedString)". Do not edit by hand. // Program metadata for host.cu plus the launch wrappers defined in kernel.cu. #pragma once #include #include #define IGNEUM_SEED_STRING \(jstr(p.seedString)) #define IGNEUM_DAY_STRING \(jstr(dayString)) #define IGNEUM_DAY0 \(hex(day.0)) #define IGNEUM_DAY1 \(hex(day.1)) #define IGNEUM_DATASET_LOG2 \(datasetLog2) #define IGNEUM_MASK \(hex(mask)) #define IGNEUM_LANES 32 #define IGNEUM_ITERATIONS \(Program.iterations) #define IGNEUM_INSTR_COUNT \(Program.count) #define IGNEUM_LOADS_PER_HASH \(p.loadsPerHash) #define IGNEUM_OP_MIX \(jstr(mix)) #define IGNEUM_SEEDW_INIT { \(p.seed.map(hex).joined(separator: ", ")) } // Defined in kernel.cu. Both launch on the default stream and return cudaGetLastError(). cudaError_t igneum_launch_fill(uint32_t* ds, uint32_t nWords, uint32_t d0, uint32_t d1); cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, uint32_t nonces, uint32_t blockWarps); cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps); """ } func generateVectorsHeader(_ p: Program, bases: [UInt32], outs: [[UInt64]], head: [UInt32], last: UInt32, mask: UInt32, source: String) -> String { var s = """ // Generated by proto-metal/igneum-bench --export-pack for seed "\(p.seedString)". Do not edit by hand. // Expected outputs: \(source) #pragma once #include #define IGNEUM_VEC_WARPS \(bases.count) static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { \(bases.map { "\($0)u" }.joined(separator: ", ")) }; static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = { """ for (i, o) in outs.enumerated() { s += " { // base nonce \(bases[i])\n" for row in 0..<4 { s += " " + (0..<8).map { hex64(o[row * 8 + $0]) }.joined(separator: ", ") + (row == 3 ? "\n" : ",\n") } s += i == outs.count - 1 ? " }\n" : " },\n" } s += """ }; // Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (\(mask)). static const uint32_t IGNEUM_DS_HEAD[16] = { \((0..<8).map { hex(head[$0]) }.joined(separator: ", ")), \((8..<16).map { hex(head[$0]) }.joined(separator: ", ")) }; static const uint32_t IGNEUM_DS_LAST_INDEX = \(mask)u; static const uint32_t IGNEUM_DS_LAST = \(hex(last)); """ return s } func generateProgramJSON(_ p: Program, dayString: String, day: (UInt32, UInt32), datasetLog2: Int) -> String { let mask = UInt32((1 << datasetLog2) - 1) var s = "{\n" s += " \"format\": \"igneum-program-pack-1\",\n" s += " \"seed\": \(jstr(p.seedString)),\n" s += " \"seed_words\": [\(p.seed.map(jhex).joined(separator: ", "))],\n" s += " \"seed_derivation\": \"FNV-1a 64 over UTF-8 of seed, basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32\",\n" s += " \"lanes\": 32,\n" s += " \"registers\": 8,\n" s += " \"iterations\": \(Program.iterations),\n" s += " \"instruction_count\": \(Program.count),\n" s += " \"loads_per_hash\": \(p.loadsPerHash),\n" s += " \"op_mix\": {\(p.histogram.map { "\(jstr($0.0)): \($0.1)" }.joined(separator: ", "))},\n" s += " \"register_init\": \"for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]\",\n" s += " \"splitmix32\": \"x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16\",\n" s += " \"iteration\": \"sel = r0 sampled once at the top of each iteration, then all instructions in order\",\n" s += " \"output\": \"lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo\",\n" s += " \"op_semantics\": {\n" s += " \"add\": \"dst = dst + src + (bit `bit` of sel ? imm2 : imm)\",\n" s += " \"sub\": \"dst = dst - src\",\n" s += " \"mul\": \"dst = dst * src (low 32)\",\n" s += " \"mulhi\": \"dst = high 32 bits of dst * src\",\n" s += " \"xor\": \"dst = dst ^ src\",\n" s += " \"or\": \"dst = dst | src\",\n" s += " \"rotl\": \"dst = rotl(dst, rot), rot in 1..31\",\n" s += " \"rotr\": \"dst = rotr(dst, src & 31)\",\n" s += " \"mad\": \"dst = src * src2 + dst\",\n" s += " \"shfl\": \"dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp\",\n" s += " \"load\": \"dst = dst ^ dataset[src & dataset.mask]\"\n" s += " },\n" s += " \"dataset\": {\n" s += " \"log2_words\": \(datasetLog2),\n" s += " \"bytes\": \(UInt64(1) << UInt64(datasetLog2 + 2)),\n" s += " \"mask\": \(jhex(mask)),\n" s += " \"day\": \(jstr(dayString)),\n" s += " \"day_words_from\": \(jstr("day/" + dayString)),\n" s += " \"d0\": \(jhex(day.0)),\n" s += " \"d1\": \(jhex(day.1)),\n" s += " \"formula\": \"x = i ^ d0; x *= 0x9E3779B1; x ^= x>>15; x += d1; x *= 0x85EBCA77; x ^= x>>13; x *= 0xC2B2AE3D; x ^= x>>16 (all mod 2^32)\"\n" s += " },\n" s += " \"instructions\": [\n" for (k, ins) in p.instrs.enumerated() { s += " {\"i\": \(k), \"op\": \(jstr(ins.op.rawValue)), \"dst\": \(ins.dst), \"src\": \(ins.a), \"src2\": \(ins.b), \"imm\": \(jhex(ins.imm)), \"imm2\": \(jhex(ins.imm2)), \"rot\": \(ins.rot), \"bit\": \(ins.bit), \"mask\": \(ins.mask)}" s += k == p.instrs.count - 1 ? "\n" : ",\n" } s += " ]\n}\n" return s } func generateVectorsJSON(_ p: Program, dayString: String, datasetLog2: Int, bases: [UInt32], outs: [[UInt64]], head: [UInt32], last: UInt32, mask: UInt32, source: String) -> String { var s = "{\n" s += " \"seed\": \(jstr(p.seedString)),\n" s += " \"day\": \(jstr(dayString)),\n" s += " \"dataset_log2_words\": \(datasetLog2),\n" s += " \"mask\": \(jhex(mask)),\n" s += " \"lanes\": 32,\n" s += " \"source\": \(jstr(source)),\n" s += " \"warps\": [\n" for (i, o) in outs.enumerated() { s += " {\"base_nonce\": \(bases[i]), \"expected\": [\n" for row in 0..<4 { s += " " + (0..<8).map { jhex64(o[row * 8 + $0]) }.joined(separator: ", ") + (row == 3 ? "\n" : ",\n") } s += i == outs.count - 1 ? " ]}\n" : " ]},\n" } s += " ],\n" s += " \"dataset_head\": [\(head.map(jhex).joined(separator: ", "))],\n" s += " \"dataset_last_index\": \(mask),\n" s += " \"dataset_last\": \(jhex(last))\n" s += "}\n" return s } // Runs the Metal kernel for each base nonce (one 32-thread threadgroup each) and compares with `expected`. func metalCrossCheck(_ p: Program, datasetLog2: Int, day: (UInt32, UInt32), bases: [UInt32], expected: [[UInt64]]) -> (ok: Bool, detail: String) { guard let device = MTLCreateSystemDefaultDevice(), let queue = device.makeCommandQueue() else { return (false, "no Metal device") } let words = 1 << datasetLog2 guard let dataset = device.makeBuffer(length: words * 4, options: .storageModePrivate), let outBuf = device.makeBuffer(length: 32 * 8, options: .storageModeShared) else { return (false, "buffer allocation failed") } do { let flib = try device.makeLibrary(source: fillMSL, options: MTLCompileOptions()) let fpipe = try device.makeComputePipelineState(function: flib.makeFunction(name: "igneum_fill")!) let hlib = try device.makeLibrary(source: generateMSL(p, datasetLog2: datasetLog2), options: MTLCompileOptions()) let hpipe = try device.makeComputePipelineState(function: hlib.makeFunction(name: "igneum_hash")!) let cb = queue.makeCommandBuffer()! let enc = cb.makeComputeCommandEncoder()! enc.setComputePipelineState(fpipe) enc.setBuffer(dataset, offset: 0, index: 0) var d = (day.0, day.1) enc.setBytes(&d, length: 8, index: 1) enc.dispatchThreadgroups(MTLSize(width: words / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1)) enc.endEncoding() cb.commit(); cb.waitUntilCompleted() if let e = cb.error { return (false, "fill error \(e)") } var bad = [String]() for (i, base) in bases.enumerated() { let cb2 = queue.makeCommandBuffer()! let e2 = cb2.makeComputeCommandEncoder()! e2.setComputePipelineState(hpipe) e2.setBuffer(dataset, offset: 0, index: 0) e2.setBuffer(outBuf, offset: 0, index: 1) var b = base e2.setBytes(&b, length: 4, index: 2) e2.dispatchThreadgroups(MTLSize(width: 1, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 32, height: 1, depth: 1)) e2.endEncoding() cb2.commit(); cb2.waitUntilCompleted() if let e = cb2.error { return (false, "hash error \(e)") } let ptr = outBuf.contents().bindMemory(to: UInt64.self, capacity: 32) let got = (0..<32).map { ptr[$0] } if got != expected[i] { bad.append("base \(base)") } } return (bad.isEmpty, bad.isEmpty ? "Metal GPU cross-check PASS \(bases.count)/\(bases.count) warps" : "Metal GPU cross-check FAIL: \(bad.joined(separator: ", "))") } catch { return (false, "Metal compile error \(error)") } } func exportPack(_ opts: Options) -> Never { let dir = opts.exportPack! let program = generateProgram(seedString: opts.seed) let dayWords = seedWords("day/" + opts.day) let day = (dayWords[0], dayWords[1]) let mask = UInt32((1 << opts.datasetLog2) - 1) print("igneum-bench --export-pack \(dir)") print("seed \"\(opts.seed)\", day \"\(opts.day)\", dataset 2^\(opts.datasetLog2) words, loads/hash \(program.loadsPerHash)") print("op mix: " + program.histogram.map { "\($0.0)=\($0.1)" }.joined(separator: " ")) let outs = packVectorBases.map { cpuWarp(program, baseNonce: $0, day: day, mask: mask) } let head = (0..<16).map { datasetElem(UInt32($0), day.0, day.1) } let last = datasetElem(mask, day.0, day.1) var source = "proto-metal CPU interpreter (cpuWarp) on Apple M5 Max" let check = metalCrossCheck(program, datasetLog2: opts.datasetLog2, day: day, bases: packVectorBases, expected: outs) print(check.detail) source += "; " + check.detail if !check.ok { print("FAIL: vectors do not match the Metal GPU, pack not written"); exit(1) } let files: [(String, String)] = [ ("program.json", generateProgramJSON(program, dayString: opts.day, day: day, datasetLog2: opts.datasetLog2)), ("vectors.json", generateVectorsJSON(program, dayString: opts.day, datasetLog2: opts.datasetLog2, bases: packVectorBases, outs: outs, head: head, last: last, mask: mask, source: source)), ("kernel.cu", generateCUDA(program)), ("program.h", generateProgramHeader(program, dayString: opts.day, day: day, datasetLog2: opts.datasetLog2)), ("vectors.h", generateVectorsHeader(program, bases: packVectorBases, outs: outs, head: head, last: last, mask: mask, source: source)), ("program.metal", generateMSL(program, datasetLog2: opts.datasetLog2)), ] do { try FileManager.default.createDirectory(atPath: dir, withIntermediateDirectories: true) for (name, text) in files { try text.write(toFile: "\(dir)/\(name)", atomically: true, encoding: .utf8) print("wrote \(dir)/\(name) (\(text.utf8.count) bytes)") } } catch { print("FAIL: write error \(error)"); exit(1) } for (i, b) in packVectorBases.enumerated() { print("vector warp base \(b): lane0 \(String(format: "%016llx", outs[i][0])) lane31 \(String(format: "%016llx", outs[i][31]))") } print("OVERALL: PASS (pack written)") exit(0) } // MARK: - Timing @inline(__always) func nowNs() -> UInt64 { clock_gettime_nsec_np(CLOCK_UPTIME_RAW) } func ms(_ a: UInt64, _ b: UInt64) -> Double { Double(b - a) / 1e6 } func fmt(_ v: Double, _ digits: Int = 2) -> String { String(format: "%.\(digits)f", v) } // MARK: - GPU context final class GPU { let device: MTLDevice let queue: MTLCommandQueue init() { guard let d = MTLCreateSystemDefaultDevice(), let q = d.makeCommandQueue() else { print("FAIL: no Metal device"); exit(1) } device = d; queue = q } } struct EpochResult { var seed: String var libraryMs: Double var pipelineMs: Double var hashesPerSecWall: Double var hashesPerSecGPU: Double var gbpsWall: Double var gbpsGPU: Double var loadsPerHash: Int var verify: [(warp: Int, pass: Bool, ms: Double, repMs: Double)] var allPass: Bool { verify.allSatisfy { $0.pass } } } func runEpoch(gpu: GPU, opts: Options, seedString: String, dataset: MTLBuffer, day: (UInt32, UInt32)) -> EpochResult { let program = generateProgram(seedString: seedString) let msl = generateMSL(program, datasetLog2: opts.datasetLog2) if let dir = opts.dumpDir { try? FileManager.default.createDirectory(atPath: dir, withIntermediateDirectories: true) let safe = seedString.replacingOccurrences(of: "/", with: "_") try? msl.write(toFile: "\(dir)/program-\(safe).metal", atomically: true, encoding: .utf8) } print("\n=== epoch seed \"\(seedString)\" ===") print("program: \(Program.count) instructions x \(Program.iterations) iterations, loads/hash = \(program.loadsPerHash)") print("op mix: " + program.histogram.map { "\($0.0)=\($0.1)" }.joined(separator: " ")) // Runtime compile let t0 = nowNs() let library: MTLLibrary do { let copts = MTLCompileOptions() library = try gpu.device.makeLibrary(source: msl, options: copts) } catch { print("FAIL: Metal compile error:\n\(error)") exit(1) } let t1 = nowNs() guard let fn = library.makeFunction(name: "igneum_hash") else { print("FAIL: no igneum_hash"); exit(1) } let pipeline: MTLComputePipelineState do { pipeline = try gpu.device.makeComputePipelineState(function: fn) } catch { print("FAIL: pipeline error: \(error)"); exit(1) } let t2 = nowNs() let libMs = ms(t0, t1), pipeMs = ms(t1, t2) print("compile: library \(fmt(libMs)) ms, pipeline \(fmt(pipeMs)) ms, total \(fmt(libMs + pipeMs)) ms") print("threadExecutionWidth = \(pipeline.threadExecutionWidth), maxTotalThreadsPerThreadgroup = \(pipeline.maxTotalThreadsPerThreadgroup)") if pipeline.threadExecutionWidth != 32 { print("WARNING: threadExecutionWidth is not 32; the one-warp-per-threadgroup assumption does not hold on this device") } // Buffers let n = 1 << opts.batchLog2 let groups = n / 32 guard let outBuf = gpu.device.makeBuffer(length: n * 8, options: .storageModeShared) else { print("FAIL: out buffer"); exit(1) } func encodeBatch(_ cb: MTLCommandBuffer, base: UInt32) { let enc = cb.makeComputeCommandEncoder()! enc.setComputePipelineState(pipeline) enc.setBuffer(dataset, offset: 0, index: 0) enc.setBuffer(outBuf, offset: 0, index: 1) var b = base enc.setBytes(&b, length: 4, index: 2) enc.dispatchThreadgroups(MTLSize(width: groups, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 32, height: 1, depth: 1)) enc.endEncoding() } // Batch 0: warm-up and verification source (baseNonce 0) do { let cb = gpu.queue.makeCommandBuffer()! encodeBatch(cb, base: 0) let w0 = nowNs() cb.commit(); cb.waitUntilCompleted() let w1 = nowNs() if let e = cb.error { print("FAIL: batch 0 error \(e)"); exit(1) } print("warm-up batch: \(n) hashes in \(fmt(ms(w0, w1))) ms wall, \(fmt((cb.gpuEndTime - cb.gpuStartTime) * 1000)) ms GPU") } // Pick warps to verify from batch 0 var warps = [0, groups / 2 + 1, groups - 1] var vr = SplitMix64(s: UInt64(program.seed[4]) | (UInt64(program.seed[5]) << 32)) while warps.count < opts.verifyWarps { warps.append(vr.below(groups)) } warps = Array(warps.prefix(max(opts.verifyWarps, 1))) let outPtr = outBuf.contents().bindMemory(to: UInt64.self, capacity: n) var gpuOutputs = [[UInt64]]() for w in warps { gpuOutputs.append((0..<32).map { outPtr[w * 32 + $0] }) } // Timed batches var cbs = [MTLCommandBuffer]() for b in 0.. \(fmt(hpsWall / 1e6, 3)) Mhash/s, \(fmt(gbpsWall)) GB/s useful (loads x 4 B)") print(" GPU \(fmt(gpuSeconds * 1000)) ms -> \(fmt(hpsGPU / 1e6, 3)) Mhash/s, \(fmt(gbpsGPU)) GB/s useful (loads x 4 B)") // CPU verification let mask = UInt32((1 << opts.datasetLog2) - 1) var verify = [(warp: Int, pass: Bool, ms: Double, repMs: Double)]() for (i, w) in warps.enumerated() { let base = UInt32(w * 32) let c0 = nowNs() let cpu = cpuWarp(program, baseNonce: base, day: day, mask: mask) let c1 = nowNs() // repeated runs for a steadier figure let reps = 20 let r0 = nowNs() var sink: UInt64 = 0 for _ in 0.. CompiledHash { let t0 = nowNs() let lib = try gpu.device.makeLibrary(source: msl, options: MTLCompileOptions()) let t1 = nowNs() guard let fn = lib.makeFunction(name: "igneum_hash") else { throw IgneumError("no igneum_hash function in library") } let pipe = try gpu.device.makeComputePipelineState(function: fn) let t2 = nowNs() return CompiledHash(pipeline: pipe, libraryMs: ms(t0, t1), pipelineMs: ms(t1, t2)) } // Allocates a private 2^log2-word dataset and fills it on the GPU with the closed form for `day`. func makeDataset(_ gpu: GPU, log2: Int, day: (UInt32, UInt32)) -> MTLBuffer { let words = 1 << log2 guard let buf = gpu.device.makeBuffer(length: words * 4, options: .storageModePrivate) else { print("FAIL: cannot allocate 2^\(log2) word dataset"); exit(1) } do { let lib = try gpu.device.makeLibrary(source: fillMSL, options: MTLCompileOptions()) let pipe = try gpu.device.makeComputePipelineState(function: lib.makeFunction(name: "igneum_fill")!) let cb = gpu.queue.makeCommandBuffer()! let enc = cb.makeComputeCommandEncoder()! enc.setComputePipelineState(pipe) enc.setBuffer(buf, offset: 0, index: 0) var d = (day.0, day.1) enc.setBytes(&d, length: 8, index: 1) enc.dispatchThreadgroups(MTLSize(width: words / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1)) enc.endEncoding() cb.commit(); cb.waitUntilCompleted() if let e = cb.error { print("FAIL: fill error \(e)"); exit(1) } } catch { print("FAIL: fill kernel \(error)"); exit(1) } return buf } // One 32-thread threadgroup per base nonce, all dispatched from one encoder. Warp i lands at byte offset // i * 256 of the output buffer, which is pre-filled with a sentinel so an unwritten lane is visible. func gpuWarps(_ gpu: GPU, _ k: CompiledHash, dataset: MTLBuffer, bases: [UInt32]) -> [[UInt64]]? { guard !bases.isEmpty, let outBuf = gpu.device.makeBuffer(length: bases.count * 256, options: .storageModeShared) else { return nil } memset(outBuf.contents(), 0xAA, outBuf.length) let cb = gpu.queue.makeCommandBuffer()! let enc = cb.makeComputeCommandEncoder()! enc.setComputePipelineState(k.pipeline) enc.setBuffer(dataset, offset: 0, index: 0) for (i, base) in bases.enumerated() { enc.setBuffer(outBuf, offset: i * 256, index: 1) var b = base enc.setBytes(&b, length: 4, index: 2) enc.dispatchThreadgroups(MTLSize(width: 1, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 32, height: 1, depth: 1)) } enc.endEncoding() cb.commit(); cb.waitUntilCompleted() if cb.error != nil { return nil } let p = outBuf.contents().bindMemory(to: UInt64.self, capacity: bases.count * 32) return (0.. Double? { let cb = gpu.queue.makeCommandBuffer()! let enc = cb.makeComputeCommandEncoder()! enc.setComputePipelineState(k.pipeline) enc.setBuffer(dataset, offset: 0, index: 0) enc.setBuffer(out, offset: 0, index: 1) var b = base enc.setBytes(&b, length: 4, index: 2) enc.dispatchThreadgroups(MTLSize(width: count / 32, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 32, height: 1, depth: 1)) enc.endEncoding() cb.commit(); cb.waitUntilCompleted() if cb.error != nil { return nil } return (cb.gpuEndTime - cb.gpuStartTime) * 1000 } func h64(_ v: UInt64) -> String { String(format: "%016llx", v) } func pad(_ s: String, _ n: Int) -> String { s.count >= n ? s : s + String(repeating: " ", count: n - s.count) } func describeProgram(_ p: Program) -> String { var s = " program seed \"\(p.seedString)\" words [\(p.seed.map(hex).joined(separator: ", "))], \(p.instrs.count) instructions\n" for (k, i) in p.instrs.enumerated() { s += " \(pad(String(k), 3)) \(pad(i.op.rawValue, 5)) dst=r\(i.dst) src=r\(i.a) src2=r\(i.b) imm=\(hex(i.imm)) imm2=\(hex(i.imm2)) rot=\(i.rot) bit=\(i.bit) mask=\(i.mask)\n" } return s } func fnv64(_ ptr: UnsafeRawPointer, _ count: Int) -> UInt64 { var h: UInt64 = 0xcbf29ce484222325 let b = ptr.bindMemory(to: UInt8.self, capacity: count) for i in 0.. Int { let re = try! NSRegularExpression(pattern: pattern) return re.numberOfMatches(in: text, range: NSRange(text.startIndex..., in: text)) } // Static check: every dataset access in the generated MSL is `dataset[rN & MASK]`, and the identifier // `dataset` appears nowhere else except the kernel parameter. func maskCheckMSL(_ msl: String) -> (ok: Bool, detail: String) { let total = regexCount("dataset\\[", in: msl) let masked = regexCount("dataset\\[r[0-7] & MASK\\]", in: msl) let words = regexCount("\\bdataset\\b", in: msl) let ok = total == masked && words == total + 1 return (ok, "MSL: \(total) dataset[ accesses, \(masked) of the form dataset[rN & MASK], identifier appears \(words) times (expected \(total + 1))") } // Same for the CUDA twin: hash accesses are `ds[rN & mask]`; the fill kernel's one write is guarded by `if (i < n)`. func maskCheckCUDA(_ cu: String) -> (ok: Bool, detail: String) { let total = regexCount("\\bds\\[", in: cu) let masked = regexCount("\\bds\\[r[0-7] & mask\\]", in: cu) let fill = regexCount("if \\(i < n\\) ds\\[i\\] = ds_elem", in: cu) let ok = total == masked + fill && fill == 1 return (ok, "CUDA: \(total) ds[ accesses, \(masked) of the form ds[rN & mask], \(fill) guarded fill write") } // MARK: - --fuzz func runFuzz(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool { let n = max(opts.fuzz ?? 200, 1) let master = opts.fuzzSeed print("\n=== fuzz: \(n) random programs, master seed \"\(master)\", 4 random warps each ===") let sizes = [24, 26, 28] let d0 = nowNs() var datasets = [Int: MTLBuffer]() for s in sizes { datasets[s] = makeDataset(gpu, log2: s, day: day) } print("datasets " + sizes.map { "2^\($0) (\((1 << $0) * 4 / (1 << 20)) MiB)" }.joined(separator: ", ") + " filled in \(fmt(ms(d0, nowNs()), 1)) ms") let mw = seedWords("fuzz/" + master) var rng = SplitMix64(s: UInt64(mw[0]) | (UInt64(mw[1]) << 32)) var pass = 0, fail = 0, compileFail = 0, staticFail = 0, contractFail = 0 var perSize = [Int: (pass: Int, fail: Int)]() var compileMs = [Double]() var cpuNs: UInt64 = 0, gpuNs: UInt64 = 0 var warps = 0 var opCount = [String: Int]() var loadsMin = Int.max, loadsMax = 0 let t0 = nowNs() for i in 0.. 31 || ![1, 2, 4, 8, 16].contains(ins.mask) || ins.a == ins.dst || ins.dst > 7 || ins.a > 7 || ins.b > 7 { contractFail += 1 print("CONTRACT FAIL seed \"\(seedString)\": \(ins)") } } loadsMin = min(loadsMin, program.loadsPerHash); loadsMax = max(loadsMax, program.loadsPerHash) let msl = generateMSL(program, datasetLog2: log2) let sc = maskCheckMSL(msl) if !sc.ok { staticFail += 1; print("STATIC MASK FAIL seed \"\(seedString)\": \(sc.detail)") } let k: CompiledHash do { k = try compileHash(gpu, msl: msl) } catch { compileFail += 1 print("COMPILE FAIL seed \"\(seedString)\" dataset 2^\(log2):\n\(error)\n\(describeProgram(program))") continue } compileMs.append(k.totalMs) let g0 = nowNs() guard let gpuOut = gpuWarps(gpu, k, dataset: datasets[log2]!, bases: bases) else { fail += 1; print("GPU RUN FAIL seed \"\(seedString)\" dataset 2^\(log2)"); continue } let g1 = nowNs() let mask = UInt32((1 << log2) - 1) var ok = true for (w, base) in bases.enumerated() { let cpu = cpuWarp(program, baseNonce: base, day: day, mask: mask) warps += 1 if cpu != gpuOut[w] { ok = false let bad = (0..<32).filter { cpu[$0] != gpuOut[w][$0] } print("MISMATCH seed \"\(seedString)\" dataset 2^\(log2) warp \(w) base nonce \(base) (\(hex(base))) lanes \(bad)") for l in bad { print(" lane \(l) nonce \(base &+ UInt32(l)): gpu \(h64(gpuOut[w][l])) cpu \(h64(cpu[l]))") } print(describeProgram(program)) } } let g2 = nowNs() gpuNs += g1 - g0; cpuNs += g2 - g1 if ok { pass += 1 } else { fail += 1 } var ps = perSize[log2] ?? (0, 0) if ok { ps.pass += 1 } else { ps.fail += 1 } perSize[log2] = ps if (i + 1) % 100 == 0 || i + 1 == n { print(" \(i + 1)/\(n): pass \(pass) fail \(fail) compile-fail \(compileFail), \(fmt(Double(nowNs() - t0) / 1e9, 1)) s elapsed") } } let total = Double(nowNs() - t0) / 1e9 let cAvg = compileMs.isEmpty ? 0 : compileMs.reduce(0, +) / Double(compileMs.count) print("\n| Dataset | Programs | Pass | Fail |") print("|---|---|---|---|") for s in sizes { let ps = perSize[s] ?? (0, 0) print("| 2^\(s) words (\((1 << s) * 4 / (1 << 20)) MiB) | \(ps.pass + ps.fail) | \(ps.pass) | \(ps.fail) |") } print("| all | \(pass + fail) | \(pass) | \(fail) |") print("programs \(n): pass \(pass), mismatch \(fail), compile failures \(compileFail), static mask failures \(staticFail), generator contract failures \(contractFail)") print("warps compared \(warps) (\(warps * 32) hashes), loads/hash range \(loadsMin)..\(loadsMax)") print("op totals over all programs: " + opCount.sorted { $0.value != $1.value ? $0.value > $1.value : $0.key < $1.key }.map { "\($0.key)=\($0.value)" }.joined(separator: " ")) print("compile ms (library+pipeline): min \(fmt(compileMs.min() ?? 0, 1)) avg \(fmt(cAvg, 1)) max \(fmt(compileMs.max() ?? 0, 1)); GPU dispatch total \(fmt(Double(gpuNs) / 1e6, 1)) ms; CPU interpreter total \(fmt(Double(cpuNs) / 1e6, 1)) ms; wall \(fmt(total, 1)) s") let ok = fail == 0 && compileFail == 0 && staticFail == 0 && contractFail == 0 && pass == n print("FUZZ: \(ok ? "PASS" : "FAIL")") return ok } // MARK: - --edge struct EdgeCase { let name: String let instrs: [Instr] // (instruction index, what must hold, check on lane-0 registers as they are just before that instruction) let pre: [(Int, String, ([UInt32]) -> Bool)] let informational: Bool // reported but not counted: exercises something the generator never emits } // Instruction builder for hand-made programs. imm2 = imm so the `add` is a constant regardless of the selector bit. func I(_ op: Op, _ dst: Int, _ a: Int, b: Int = 0, imm: UInt32 = 0, rot: UInt32 = 1, bit: Int = 0, mask: Int = 1) -> Instr { Instr(op: op, dst: dst, a: a, b: b, imm: imm, imm2: imm, rot: rot, bit: bit, mask: mask) } func zero(_ r: Int) -> Instr { I(.sub, r, r) } // r = r - r = 0 (src == dst, never generated, legal MSL) func set(_ r: Int, _ v: UInt32) -> [Instr] { [zero(r), I(.add, r, 7, imm: v)] } // needs r7 == 0 func edgeCases(mask: UInt32) -> [EdgeCase] { let M = mask var c = [EdgeCase]() c.append(EdgeCase(name: "rotl immediate by 1 and by 31", instrs: [I(.rotl, 1, 0, rot: 1), I(.rotl, 2, 0, rot: 31), I(.xor, 3, 1), I(.xor, 4, 2), I(.rotl, 5, 0, rot: 1), I(.rotl, 6, 0, rot: 31)], pre: [], informational: false)) c.append(EdgeCase(name: "rotr by register == 0", instrs: [zero(7), I(.rotr, 3, 7), I(.xor, 4, 3)], pre: [(1, "r7 == 0", { $0[7] == 0 })], informational: false)) c.append(EdgeCase(name: "rotr by register == 32 (32 mod 32 = 0)", instrs: [zero(7)] + (set(1, 32) + [I(.rotr, 3, 1), I(.xor, 4, 3)]), pre: [(3, "r1 == 32", { $0[1] == 32 })], informational: false)) c.append(EdgeCase(name: "rotr by register == 0xFFFFFFE0 (-32, 0 mod 32)", instrs: [zero(7)] + (set(1, 0xFFFFFFE0) + [I(.rotr, 3, 1), I(.xor, 4, 3)]), pre: [(3, "r1 == 0xFFFFFFE0", { $0[1] == 0xFFFFFFE0 })], informational: false)) var rr: [Instr] = [zero(7)] rr += set(1, 31); rr += [I(.rotr, 3, 1), I(.add, 1, 7, imm: 32), I(.rotr, 4, 1)] rr += set(2, 1); rr += [I(.rotr, 5, 2), I(.xor, 6, 5)] c.append(EdgeCase(name: "rotr by register == 31 and == 63 and == 1", instrs: rr, pre: [(3, "r1 == 31", { $0[1] == 31 }), (5, "r1 == 63", { $0[1] == 63 }), (8, "r2 == 1", { $0[2] == 1 })], informational: false)) var mh: [Instr] = [zero(7)] mh += set(1, 0xFFFFFFFF); mh += set(2, 0xFFFFFFFF) mh += [I(.mulhi, 1, 2), I(.xor, 3, 1)] mh += set(4, 0x80000000); mh += set(5, 2) mh += [I(.mulhi, 4, 5), I(.xor, 3, 4), I(.mulhi, 6, 7), I(.xor, 0, 6)] c.append(EdgeCase(name: "mulhi 0xFFFFFFFF x 0xFFFFFFFF, 0x80000000 x 2, x 0", instrs: mh, pre: [(5, "r1 == r2 == 0xFFFFFFFF", { $0[1] == 0xFFFFFFFF && $0[2] == 0xFFFFFFFF }), (6, "mulhi result r1 == 0xFFFFFFFE", { $0[1] == 0xFFFFFFFE }), (11, "r4 == 0x80000000, r5 == 2", { $0[4] == 0x80000000 && $0[5] == 2 }), (12, "mulhi result r4 == 1", { $0[4] == 1 }), (13, "r7 == 0", { $0[7] == 0 }), (14, "mulhi by 0 gives r6 == 0", { $0[6] == 0 })], informational: false)) c.append(EdgeCase(name: "shfl_xor every mask 1..16 in sequence (generator uses only 1,2,4,8,16)", instrs: (1...16).map { I(.shfl, $0 % 8, ($0 + 1) % 8, mask: $0) }, pre: [], informational: false)) var l0: [Instr] = [zero(7), I(.load, 3, 7)] l0 += set(1, M &+ 1); l0 += [I(.load, 4, 1), I(.xor, 5, 4)] c.append(EdgeCase(name: "load at index 0 (register 0, and register MASK+1 which masks to 0)", instrs: l0, pre: [(1, "r7 & MASK == 0", { $0[7] & M == 0 }), (4, "r1 == MASK+1, so unmasked index is out of range and masked index is 0", { $0[1] == M &+ 1 && ($0[1] & M) == 0 })], informational: false)) var lm: [Instr] = [zero(7)] lm += set(1, M); lm += [I(.load, 3, 1)] lm += set(2, 0xFFFFFFFF); lm += [I(.load, 4, 2), I(.xor, 5, 4)] c.append(EdgeCase(name: "load at index MASK (register MASK, and register 0xFFFFFFFF which masks to MASK)", instrs: lm, pre: [(3, "r1 == MASK", { $0[1] == M }), (6, "r2 == 0xFFFFFFFF, masked index == MASK", { $0[2] == 0xFFFFFFFF && ($0[2] & M) == M })], informational: false)) c.append(EdgeCase(name: "add wraparound 0xFFFFFFFF + 1", instrs: [zero(7)] + (set(1, 0xFFFFFFFF) + [I(.add, 1, 7, imm: 1), I(.xor, 2, 1)]), pre: [(3, "r1 == 0xFFFFFFFF", { $0[1] == 0xFFFFFFFF }), (4, "r1 == 0 after add", { $0[1] == 0 })], informational: false)) c.append(EdgeCase(name: "sub wraparound 0 - 1", instrs: [zero(7), zero(1)] + (set(2, 1) + [I(.sub, 1, 2), I(.xor, 3, 1)]), pre: [(4, "r1 == 0, r2 == 1", { $0[1] == 0 && $0[2] == 1 }), (5, "r1 == 0xFFFFFFFF after sub", { $0[1] == 0xFFFFFFFF })], informational: false)) var mm: [Instr] = [zero(7)] mm += set(1, 0xFFFFFFFF); mm += set(2, 0xFFFFFFFF) mm += [I(.mul, 1, 2), I(.xor, 3, 1)] mm += set(4, 0xFFFFFFFF); mm += set(5, 0xFFFFFFFF); mm += set(6, 5) mm += [I(.mad, 6, 4, b: 5), I(.xor, 0, 6)] c.append(EdgeCase(name: "mul and mad wraparound 0xFFFFFFFF x 0xFFFFFFFF", instrs: mm, pre: [(5, "r1 == r2 == 0xFFFFFFFF", { $0[1] == 0xFFFFFFFF && $0[2] == 0xFFFFFFFF }), (6, "mul low result r1 == 1", { $0[1] == 1 }), (13, "r4 == r5 == 0xFFFFFFFF, r6 == 5", { $0[4] == 0xFFFFFFFF && $0[5] == 0xFFFFFFFF && $0[6] == 5 }), (14, "mad result r6 == 6", { $0[6] == 6 })], informational: false)) let z = generateProgram(seedString: "edge/zero-loads") c.append(EdgeCase(name: "generated program with every load replaced by xor (zero loads)", instrs: z.instrs.map { ins in var m = ins; if m.op == .load { m.op = .xor }; return m }, pre: [], informational: false)) c.append(EdgeCase(name: "64 loads and nothing else", instrs: (0..<64).map { I(.load, $0 % 8, ($0 + 3) % 8) }, pre: [], informational: false)) c.append(EdgeCase(name: "rotl immediate by 0 (outside the generator's 1..31 contract; MSL shifts by 32)", instrs: [I(.rotl, 1, 0, rot: 0), I(.xor, 2, 1)], pre: [], informational: true)) return c } func runEdge(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool { let log2 = opts.datasetLog2 let mask = UInt32((1 << log2) - 1) print("\n=== edge cases, dataset 2^\(log2) words, MASK \(hex(mask)) ===") let dataset = makeDataset(gpu, log2: log2, day: day) let bases: [UInt32] = [0, 1 << 20, 0x7FFFFFF0, 0xFFFFFFE0] print("warps: base nonces " + bases.map { hex($0) }.joined(separator: ", ") + " (the last two straddle 2^31 and wrap past 2^32)") var allOk = true var rows = [String]() for ec in edgeCases(mask: mask) { let program = Program(seedString: "edge/\(ec.name)", seed: seedWords("edge/\(ec.name)"), instrs: ec.instrs) let msl = generateMSL(program, datasetLog2: log2) var status = "", detail = "" var ok = true // Preconditions, checked on lane 0 of every warp in every iteration. var preOk = true var preNotes = [String]() if !ec.pre.isEmpty { for base in bases { var hits = [Int: Int]() var misses = [Int: Int]() _ = cpuWarpTraced(program, baseNonce: base, day: day, mask: mask) { _, k, regs in for (idx, _, check) in ec.pre where idx == k { if check(regs) { hits[idx, default: 0] += 1 } else { misses[idx, default: 0] += 1 } } } for (idx, what, _) in ec.pre { if (misses[idx] ?? 0) > 0 || (hits[idx] ?? 0) != Program.iterations { preOk = false preNotes.append("base \(hex(base)) instr \(idx) '\(what)' held \(hits[idx] ?? 0)/\(Program.iterations) iterations") } } } if preOk { preNotes = ec.pre.map { "instr \($0.0): \($0.1)" } } } do { let k = try compileHash(gpu, msl: msl) guard let gpuOut = gpuWarps(gpu, k, dataset: dataset, bases: bases) else { throw IgneumError("GPU run failed") } var badLanes = 0 var first = "" for (w, base) in bases.enumerated() { let cpu = cpuWarp(program, baseNonce: base, day: day, mask: mask) for l in 0..<32 where cpu[l] != gpuOut[w][l] { badLanes += 1 if first.isEmpty { first = "first: base \(hex(base)) lane \(l) gpu \(h64(gpuOut[w][l])) cpu \(h64(cpu[l]))" } } } ok = badLanes == 0 && preOk status = badLanes == 0 ? "GPU == CPU 128/128 lanes" : "MISMATCH \(badLanes)/128 lanes, \(first)" detail = "compile \(fmt(k.totalMs, 1)) ms" if badLanes > 0 { print(describeProgram(program)) } } catch { ok = false status = "COMPILE FAIL: \(error)" } let pre = ec.pre.isEmpty ? "none needed" : (preOk ? "held (all 8 iterations, lane 0, 4 warps)" : "NOT HELD") let verdict = ec.informational ? (ok ? "info: agrees" : "info: differs") : (ok ? "PASS" : "FAIL") if !ec.informational && !ok { allOk = false } print("\(verdict): \(ec.name)") print(" \(ec.instrs.count) instructions, loads/hash \(program.loadsPerHash), \(status), \(detail)") for n in preNotes { print(" precondition \(n)") } rows.append("| \(ec.name) | \(ec.instrs.count) | \(program.loadsPerHash) | \(pre) | \(status) | \(verdict) |") } print("\n| Case | Instrs | Loads/hash | Preconditions | GPU vs CPU | Result |") print("|---|---|---|---|---|---|") for r in rows { print(r) } print("EDGE: \(allOk ? "PASS" : "FAIL")") return allOk } // MARK: - --stats func popcount64(_ v: UInt64) -> Int { v.nonzeroBitCount } func runStats(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool { let log2 = opts.datasetLog2 let mask = UInt32((1 << log2) - 1) let n = 1 << 20 print("\n=== output statistics, 2^20 consecutive nonces per seed, dataset 2^\(log2) words ===") print("This is a sanity check for obvious structural bias. It is not a proof of cryptographic strength.") let dataset = makeDataset(gpu, log2: log2, day: day) guard let outBuf = gpu.device.makeBuffer(length: n * 8, options: .storageModeShared) else { print("FAIL: out buffer"); return false } let seeds = [opts.seed, "\(opts.seed)/stats1", "\(opts.seed)/stats2"] var allOk = true var rows = [String]() for seedString in seeds { let program = generateProgram(seedString: seedString) let k: CompiledHash do { k = try compileHash(gpu, msl: generateMSL(program, datasetLog2: log2)) } catch { print("FAIL: compile \(error)"); return false } memset(outBuf.contents(), 0, n * 8) guard let gms = gpuRange(gpu, k, dataset: dataset, base: 0, count: n, out: outBuf) else { print("FAIL: GPU run"); return false } let p = outBuf.contents().bindMemory(to: UInt64.self, capacity: n) let outs = (0..>= 1; b += 1 } } let expected = Double(n) / 2, sigma = (Double(n) * 0.25).squareRoot() var maxDev = 0.0, maxBit = 0 for b in 0..<64 { let d = abs(Double(ones[b]) - expected); if d > maxDev { maxDev = d; maxBit = b } } let maxZ = maxDev / sigma let minFreq = Double(ones.min()!) / Double(n), maxFreq = Double(ones.max()!) / Double(n) // (c) chi-square over 65536 buckets for each 16-bit window of the output var chiRows = [String]() var chiWorstZ = 0.0 for shift in [0, 16, 32, 48] { var buckets = [Int](repeating: 0, count: 65536) for v in outs { buckets[Int((v >> UInt64(shift)) & 0xFFFF)] += 1 } let e = Double(n) / 65536 var chi = 0.0 for c in buckets { let d = Double(c) - e; chi += d * d / e } let df = 65535.0 let z = (chi - df) / (2 * df).squareRoot() chiWorstZ = max(chiWorstZ, abs(z)) chiRows.append("bits \(shift)..\(shift + 15): chi2 \(fmt(chi, 0)) (df 65535, z \(fmt(z, 2)))") } // (d) duplicates let sorted = outs.sorted() var dups = 0 for i in 1.. 0 { let m = Double(perBitSum[b]) / Double(perBitN[b]); perBitMin = min(perBitMin, m); perBitMax = max(perBitMax, m) } // Expected for an ideal function: mean 32, std 4 (binomial 64 x 0.5). Standard error of the mean over 1000 trials is 0.13. let avOk = abs(mean - 32) < 0.6 && std > 3.3 && std < 4.7 let freqOk = maxZ < 4.5 let chiOk = chiWorstZ < 4.5 let dupOk = dups == 0 let ok = avOk && freqOk && chiOk && dupOk && spot if !ok { allOk = false } print("\nseed \"\(seedString)\": loads/hash \(program.loadsPerHash), GPU \(fmt(gms, 1)) ms for 2^20 hashes, CPU spot check 2 warps \(spot ? "PASS" : "FAIL")") print(" (a) bit frequency: min \(fmt(minFreq, 4)) max \(fmt(maxFreq, 4)); largest deviation \(fmt(maxDev, 0)) counts at bit \(maxBit) = \(fmt(maxZ, 2)) sigma (sigma \(fmt(sigma, 0)), 64 bits, expect max under about 3.5)") print(" (b) avalanche over \(trials) single-bit nonce flips: mean \(fmt(mean, 2)) std \(fmt(std, 2)) min \(minDiff) max \(maxDiff) of 64 bits (expect mean 32, std 4); per-input-bit mean range \(fmt(perBitMin, 1))..\(fmt(perBitMax, 1))") for r in chiRows { print(" (c) \(r)") } print(" (d) duplicate 64-bit outputs among 2^20: \(dups) (expected about 3e-8)") print(" verdict: \(ok ? "no obvious bias" : "SUSPECT")") rows.append("| \(seedString) | \(program.loadsPerHash) | \(fmt(minFreq, 4))..\(fmt(maxFreq, 4)) | \(fmt(maxZ, 2)) | \(fmt(mean, 2)) | \(fmt(std, 2)) | \(fmt(chiWorstZ, 2)) | \(dups) | \(ok ? "uniform-looking" : "SUSPECT") |") } print("\n| Seed | Loads/hash | Bit freq min..max | Max bit z | Avalanche mean | Avalanche std | Worst chi2 z (4 windows) | Dups | Verdict |") print("|---|---|---|---|---|---|---|---|---|") for r in rows { print(r) } print("STATS: \(allOk ? "PASS (no obvious structural bias; not a security proof)" : "FAIL (something looks biased)")") return allOk } // MARK: - --determinism func runDeterminism(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool { let log2 = opts.datasetLog2 let mask = UInt32((1 << log2) - 1) let n = 1 << 20 print("\n=== determinism, seed \"\(opts.seed)\", 2^20 nonces from base 0, dataset 2^\(log2) words ===") let dataset = makeDataset(gpu, log2: log2, day: day) guard let outBuf = gpu.device.makeBuffer(length: n * 8, options: .storageModeShared) else { print("FAIL: out buffer"); return false } var ok = true // Generator and emitter determinism: two independent generations give identical MSL text. let p1 = generateProgram(seedString: opts.seed), p2 = generateProgram(seedString: opts.seed) let msl1 = generateMSL(p1, datasetLog2: log2), msl2 = generateMSL(p2, datasetLog2: log2) let sameSource = msl1 == msl2 print("generator: two generations of the program give identical MSL source: \(sameSource ? "yes" : "NO") (\(msl1.utf8.count) bytes)") if !sameSource { ok = false } // Two separate compiles of the same source. let k1: CompiledHash, k2: CompiledHash do { k1 = try compileHash(gpu, msl: msl1); k2 = try compileHash(gpu, msl: msl2) } catch { print("FAIL: compile \(error)"); return false } print("compiled twice: \(fmt(k1.totalMs, 1)) ms and \(fmt(k2.totalMs, 1)) ms") func runOnce(_ k: CompiledHash) -> (fp: UInt64, sentinels: Int, ms: Double)? { memset(outBuf.contents(), 0xAA, n * 8) guard let gms = gpuRange(gpu, k, dataset: dataset, base: 0, count: n, out: outBuf) else { return nil } let p = outBuf.contents().bindMemory(to: UInt64.self, capacity: n) var s = 0 for i in 0.. Bool { print("\n=== memcheck, seed \"\(opts.seed)\" ===") var ok = true let program = generateProgram(seedString: opts.seed) // Static: every dataset index in the generated sources is masked. Checked at three dataset sizes because // the MASK literal changes with size. for log2 in [20, 24, 28] { let msl = generateMSL(program, datasetLog2: log2) let r = maskCheckMSL(msl) if !r.ok { ok = false } print("static 2^\(log2): \(r.ok ? "PASS" : "FAIL") \(r.detail)") } let cu = maskCheckCUDA(generateCUDA(program)) if !cu.ok { ok = false } print("static CUDA twin: \(cu.ok ? "PASS" : "FAIL") \(cu.detail)") print("program has \(program.instrs.filter { $0.op == .load }.count) load instructions (\(program.loadsPerHash) loads/hash)") // Dynamic: a 4 MiB dataset with nonces at the top of the 32-bit range (they wrap to 0 inside the batch), // one full batch of 2^20 nonces plus 4 warps verified against the CPU. Metal does not bounds-check device // buffers, so "no crash" is weak evidence by itself; the static check above is the real guarantee. let log2 = 20 let mask = UInt32((1 << log2) - 1) let dataset = makeDataset(gpu, log2: log2, day: day) let k: CompiledHash do { k = try compileHash(gpu, msl: generateMSL(program, datasetLog2: log2)) } catch { print("FAIL: compile \(error)"); return false } let n = 1 << 20 guard let outBuf = gpu.device.makeBuffer(length: n * 8, options: .storageModeShared) else { print("FAIL: out buffer"); return false } for base: UInt32 in [0xFFF00000, 0xFFFFFFE0, 0x80000000, 0] { if let gms = gpuRange(gpu, k, dataset: dataset, base: base, count: n, out: outBuf) { print("dynamic 4 MiB: 2^20 nonces from base \(hex(base)) (last nonce \(hex(base &+ UInt32(n - 1)))): completed, GPU \(fmt(gms, 1)) ms") } else { ok = false; print("dynamic 4 MiB: base \(hex(base)): GPU ERROR") } } let bases: [UInt32] = [0xFFFFFFE0, 0xFFFFFFFF, 0x80000000, 0xFFF00000] guard let gpuOut = gpuWarps(gpu, k, dataset: dataset, bases: bases) else { print("FAIL: GPU warps"); return false } var overMask = 0, loads = 0 for (w, base) in bases.enumerated() { let cpu = cpuWarpTraced(program, baseNonce: base, day: day, mask: mask) { _, kk, regs in let ins = program.instrs[kk] if ins.op == .load { loads += 1; if regs[ins.a] > mask { overMask += 1 } } } let match = cpu == gpuOut[w] if !match { ok = false } print("verify warp base \(hex(base)): GPU vs CPU \(match ? "PASS" : "FAIL")") } print("in those 4 warps (lane 0, all iterations) \(overMask) of \(loads) load indices were above MASK before masking, so the mask was exercised") print("MEMCHECK: \(ok ? "PASS" : "FAIL")") return ok } // MARK: - Test dispatcher func runTests(_ opts: Options) -> Never { let gpu = GPU() print("igneum-bench hardening tests") print("GPU: \(gpu.device.name) (maxBufferLength \(gpu.device.maxBufferLength / (1 << 20)) MiB, unified memory \(gpu.device.hasUnifiedMemory)), day \"\(opts.day)\"") let dayWords = seedWords("day/" + opts.day) let day = (dayWords[0], dayWords[1]) var results = [(String, Bool)]() let t0 = nowNs() if opts.fuzz != nil { results.append(("fuzz", runFuzz(opts, gpu: gpu, day: day))) } if opts.edge { results.append(("edge", runEdge(opts, gpu: gpu, day: day))) } if opts.stats { results.append(("stats", runStats(opts, gpu: gpu, day: day))) } if opts.determinism { results.append(("determinism", runDeterminism(opts, gpu: gpu, day: day))) } if opts.memcheck { results.append(("memcheck", runMemcheck(opts, gpu: gpu, day: day))) } print("\n=== tests summary (\(fmt(Double(nowNs() - t0) / 1e9, 1)) s) ===") for (name, ok) in results { print("\(pad(name, 12)) \(ok ? "PASS" : "FAIL")") } let all = results.allSatisfy { $0.1 } print("OVERALL: \(all ? "PASS" : "FAIL")") exit(all ? 0 : 1) } // MARK: - Main let opts = parseArgs() if opts.exportPack != nil { exportPack(opts) } if opts.anyTest { runTests(opts) } let gpu = GPU() print("igneum-bench") print("GPU: \(gpu.device.name) (maxBufferLength \(gpu.device.maxBufferLength / (1 << 20)) MiB, unified memory \(gpu.device.hasUnifiedMemory))") print("dataset: 2^\(opts.datasetLog2) uint32 = \(fmt(Double(1 << opts.datasetLog2) * 4 / Double(1 << 20), 0)) MiB, day \"\(opts.day)\"") let dayWords = seedWords("day/" + opts.day) let day = (dayWords[0], dayWords[1]) let datasetWords = 1 << opts.datasetLog2 guard let dataset = gpu.device.makeBuffer(length: datasetWords * 4, options: .storageModePrivate) else { print("FAIL: cannot allocate dataset buffer"); exit(1) } // Fill the dataset on the GPU, timed. var fillMsWall = 0.0, fillMsGPU = 0.0 do { let f0 = nowNs() let lib = try gpu.device.makeLibrary(source: fillMSL, options: MTLCompileOptions()) let pipe = try gpu.device.makeComputePipelineState(function: lib.makeFunction(name: "igneum_fill")!) let f1 = nowNs() let cb = gpu.queue.makeCommandBuffer()! let enc = cb.makeComputeCommandEncoder()! enc.setComputePipelineState(pipe) enc.setBuffer(dataset, offset: 0, index: 0) var d = (day.0, day.1) enc.setBytes(&d, length: 8, index: 1) let tg = 256 enc.dispatchThreadgroups(MTLSize(width: datasetWords / tg, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: tg, height: 1, depth: 1)) enc.endEncoding() let f2 = nowNs() cb.commit(); cb.waitUntilCompleted() let f3 = nowNs() if let e = cb.error { print("FAIL: fill error \(e)"); exit(1) } fillMsWall = ms(f2, f3) fillMsGPU = (cb.gpuEndTime - cb.gpuStartTime) * 1000 let gib = Double(datasetWords * 4) / Double(1 << 30) print("dataset fill: compile \(fmt(ms(f0, f1))) ms; fill \(fmt(fillMsWall)) ms wall, \(fmt(fillMsGPU)) ms GPU -> \(fmt(gib / (fillMsGPU / 1000))) GB/s write (GPU time)") } catch { print("FAIL: fill kernel: \(error)"); exit(1) } var results = [EpochResult]() for epoch in 0..