1592 lines
79 KiB
Swift
1592 lines
79 KiB
Swift
// igneum-bench: first prototype of Igneum's random-program GPU proof-of-work.
|
|
// One file. Build: swiftc -O -o igneum-bench main.swift -framework Metal
|
|
// Metal shaders are compiled at runtime from generated source (no Xcode needed).
|
|
|
|
import Foundation
|
|
import Metal
|
|
|
|
// MARK: - Options
|
|
|
|
struct Options {
|
|
var seed = "igneum-genesis"
|
|
var day = "2026-10-03"
|
|
var hours = 2 // number of epochs (seeds) run in sequence; default 2 so verification covers 2 seeds
|
|
var batchLog2 = 22 // nonces per batch
|
|
var batches = 4 // timed batches
|
|
var datasetLog2 = 28 // 2^28 uint32 = 1 GiB
|
|
var verifyWarps = 3
|
|
var dumpDir: String? = nil
|
|
var exportPack: String? = nil // write a CUDA program pack for --seed into this directory and exit
|
|
// Hardening tests (added 3 October 2026). Any of these runs instead of the bench.
|
|
var fuzz: Int? = nil // --fuzz N: N random programs, GPU vs CPU on 4 random warps each
|
|
var fuzzSeed = "igneum-fuzz-2026-10-03"
|
|
var edge = false // --edge: hand-built edge-case programs
|
|
var stats = false // --stats: output distribution sanity checks on 2^20 nonces
|
|
var determinism = false // --determinism: 5 identical GPU runs + double compile
|
|
var memcheck = false // --memcheck: static mask check + 4 MiB run with wrapping nonces
|
|
var anyTest: Bool { fuzz != nil || edge || stats || determinism || memcheck }
|
|
}
|
|
|
|
func parseArgs() -> Options {
|
|
var o = Options()
|
|
var args = Array(CommandLine.arguments.dropFirst())
|
|
func take() -> String { args.isEmpty ? "" : args.removeFirst() }
|
|
while !args.isEmpty {
|
|
let a = take()
|
|
switch a {
|
|
case "--seed": o.seed = take()
|
|
case "--day": o.day = take()
|
|
case "--hours": o.hours = Int(take()) ?? o.hours
|
|
case "--batch-log2": o.batchLog2 = Int(take()) ?? o.batchLog2
|
|
case "--batches": o.batches = Int(take()) ?? o.batches
|
|
case "--dataset-log2": o.datasetLog2 = Int(take()) ?? o.datasetLog2
|
|
case "--verify-warps": o.verifyWarps = Int(take()) ?? o.verifyWarps
|
|
case "--dump": o.dumpDir = take()
|
|
case "--export-pack": o.exportPack = take()
|
|
case "--fuzz": o.fuzz = Int(take()) ?? 200
|
|
case "--fuzz-seed": o.fuzzSeed = take()
|
|
case "--edge": o.edge = true
|
|
case "--stats": o.stats = true
|
|
case "--determinism": o.determinism = true
|
|
case "--memcheck": o.memcheck = true
|
|
case "-h", "--help":
|
|
print("""
|
|
igneum-bench [--seed <string>] [--hours N] [--batch-log2 22] [--batches 4]
|
|
[--dataset-log2 28] [--verify-warps 3] [--dump <dir>] [--day <string>]
|
|
[--export-pack <dir>] write the CUDA program pack for --seed, then exit
|
|
hardening tests (run instead of the bench; several may be combined; exit 0 only if all pass):
|
|
[--fuzz N [--fuzz-seed <string>]] N random programs, GPU vs CPU, 4 random warps each,
|
|
dataset size drawn from 64 MiB, 256 MiB, 1 GiB
|
|
[--edge] hand-built edge-case programs, GPU vs CPU
|
|
[--stats] output distribution sanity checks on 2^20 nonces, 3 seeds
|
|
[--determinism] 5 identical GPU runs of 2^20 nonces, double compile, dataset fill check
|
|
[--memcheck] static dataset-index mask check, 4 MiB run with wrapping nonces
|
|
""")
|
|
exit(0)
|
|
default:
|
|
print("unknown argument \(a)"); exit(2)
|
|
}
|
|
}
|
|
return o
|
|
}
|
|
|
|
// MARK: - Integer helpers (CPU side, must match MSL bit for bit)
|
|
|
|
@inline(__always) func rotl32(_ x: UInt32, _ n: UInt32) -> UInt32 {
|
|
let n = n & 31
|
|
return n == 0 ? x : (x << n) | (x >> (32 - n))
|
|
}
|
|
@inline(__always) func rotr32(_ x: UInt32, _ n: UInt32) -> UInt32 {
|
|
let n = n & 31
|
|
return n == 0 ? x : (x >> n) | (x << (32 - n))
|
|
}
|
|
@inline(__always) func mulhi32(_ a: UInt32, _ b: UInt32) -> UInt32 {
|
|
UInt32(truncatingIfNeeded: (UInt64(a) &* UInt64(b)) >> 32)
|
|
}
|
|
@inline(__always) func splitmix32(_ v: UInt32) -> UInt32 {
|
|
var x = v
|
|
x ^= x >> 16; x &*= 0x7feb352d
|
|
x ^= x >> 15; x &*= 0x846ca68b
|
|
x ^= x >> 16
|
|
return x
|
|
}
|
|
// Dataset element, closed form of (daySeed, index). Same formula is emitted into the MSL.
|
|
@inline(__always) func datasetElem(_ i: UInt32, _ d0: UInt32, _ d1: UInt32) -> UInt32 {
|
|
var x = i ^ d0
|
|
x &*= 0x9E3779B1; x ^= x >> 15
|
|
x &+= d1
|
|
x &*= 0x85EBCA77; x ^= x >> 13
|
|
x &*= 0xC2B2AE3D; x ^= x >> 16
|
|
return x
|
|
}
|
|
|
|
// 32-byte seed (8 x uint32) from a string: FNV-1a 64 with four salts, each finalised.
|
|
func seedWords(_ s: String) -> [UInt32] {
|
|
var words = [UInt32]()
|
|
for salt in 0..<4 {
|
|
var h: UInt64 = 0xcbf29ce484222325 ^ (UInt64(salt) &* 0x9E3779B97F4A7C15)
|
|
for b in s.utf8 { h ^= UInt64(b); h &*= 0x100000001b3 }
|
|
h ^= h >> 33; h &*= 0xff51afd7ed558ccd; h ^= h >> 33
|
|
words.append(UInt32(truncatingIfNeeded: h))
|
|
words.append(UInt32(truncatingIfNeeded: h >> 32))
|
|
}
|
|
return words
|
|
}
|
|
|
|
struct SplitMix64 {
|
|
var s: UInt64
|
|
mutating func next() -> UInt64 {
|
|
s &+= 0x9E3779B97F4A7C15
|
|
var z = s
|
|
z = (z ^ (z >> 30)) &* 0xBF58476D1CE4E5B9
|
|
z = (z ^ (z >> 27)) &* 0x94D049BB133111EB
|
|
return z ^ (z >> 31)
|
|
}
|
|
mutating func below(_ n: Int) -> Int { Int(next() % UInt64(n)) }
|
|
}
|
|
|
|
// MARK: - Program
|
|
|
|
enum Op: String { case add, sub, mul, mulhi, xor, or, rotl, rotr, mad, shfl, load }
|
|
|
|
struct Instr {
|
|
var op: Op
|
|
var dst: Int
|
|
var a: Int // source register, never equal to dst
|
|
var b: Int // second source (mad only)
|
|
var imm: UInt32 // add immediate A
|
|
var imm2: UInt32 // add immediate B
|
|
var rot: UInt32 // rotl amount 1..31
|
|
var bit: Int // selector bit of r0 for add
|
|
var mask: Int // shuffle xor mask: 1,2,4,8,16
|
|
}
|
|
|
|
struct Program {
|
|
let seedString: String
|
|
let seed: [UInt32]
|
|
let instrs: [Instr]
|
|
static let iterations = 8
|
|
static let count = 64
|
|
var loadsPerHash: Int { instrs.filter { $0.op == .load }.count * Program.iterations }
|
|
var histogram: [(String, Int)] {
|
|
var d = [String: Int]()
|
|
for i in instrs { d[i.op.rawValue, default: 0] += 1 }
|
|
return d.sorted { $0.1 != $1.1 ? $0.1 > $1.1 : $0.0 < $1.0 } // count desc, then name, so output is deterministic
|
|
}
|
|
}
|
|
|
|
// Weights sum to 100. Loads are 25 percent so the kernel leans on memory.
|
|
let opWeights: [(Op, Int)] = [(.load, 25), (.add, 12), (.xor, 10), (.mul, 8), (.mad, 8), (.shfl, 8),
|
|
(.rotl, 7), (.sub, 6), (.mulhi, 6), (.rotr, 6), (.or, 4)]
|
|
|
|
func generateProgram(seedString: String) -> Program {
|
|
let sw = seedWords(seedString)
|
|
var rng = SplitMix64(s: (UInt64(sw[0]) | (UInt64(sw[1]) << 32)) ^ ((UInt64(sw[2]) | (UInt64(sw[3]) << 32)) &* 0x9E3779B97F4A7C15))
|
|
var instrs = [Instr]()
|
|
for _ in 0..<Program.count {
|
|
var roll = rng.below(100)
|
|
var op = Op.add
|
|
for (o, w) in opWeights { if roll < w { op = o; break }; roll -= w }
|
|
let dst = rng.below(8)
|
|
var a = rng.below(7); if a >= dst { a += 1 }
|
|
let b = rng.below(8)
|
|
let imm = UInt32(truncatingIfNeeded: rng.next())
|
|
let imm2 = UInt32(truncatingIfNeeded: rng.next())
|
|
let rot = UInt32(1 + rng.below(31))
|
|
let bit = rng.below(32)
|
|
let mask = 1 << rng.below(5)
|
|
instrs.append(Instr(op: op, dst: dst, a: a, b: b, imm: imm, imm2: imm2, rot: rot, bit: bit, mask: mask))
|
|
}
|
|
return Program(seedString: seedString, seed: sw, instrs: instrs)
|
|
}
|
|
|
|
// MARK: - MSL generation
|
|
|
|
func hex(_ v: UInt32) -> String { String(format: "0x%08xu", v) }
|
|
|
|
func generateMSL(_ p: Program, datasetLog2: Int) -> String {
|
|
let mask = UInt32((1 << datasetLog2) - 1)
|
|
var s = """
|
|
#include <metal_stdlib>
|
|
using namespace metal;
|
|
|
|
#define MASK \(hex(mask))
|
|
constant uint SEEDW[8] = { \(p.seed.map(hex).joined(separator: ", ")) };
|
|
|
|
inline uint splitmix32(uint x) {
|
|
x ^= x >> 16; x *= 0x7feb352du;
|
|
x ^= x >> 15; x *= 0x846ca68bu;
|
|
x ^= x >> 16;
|
|
return x;
|
|
}
|
|
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
|
|
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
|
inline uint ds_elem(uint i, uint d0, uint d1) {
|
|
uint x = i ^ d0;
|
|
x *= 0x9E3779B1u; x ^= x >> 15;
|
|
x += d1;
|
|
x *= 0x85EBCA77u; x ^= x >> 13;
|
|
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
|
return x;
|
|
}
|
|
|
|
kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
|
|
device ulong* out [[buffer(1)]],
|
|
constant uint& baseNonce [[buffer(2)]],
|
|
uint gid [[thread_position_in_grid]]) {
|
|
uint nonce = baseNonce + gid;
|
|
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
|
|
|
"""
|
|
for i in 0..<8 {
|
|
s += " { uint x = nonce ^ SEEDW[\(i)]; x += 0x9e3779b9u * \(i + 1)u; x = splitmix32(x); r\(i) = x ^ SEEDW[\((i + 1) & 7)]; }\n"
|
|
}
|
|
s += "\n for (uint it = 0u; it < \(Program.iterations)u; ++it) {\n uint sel = r0;\n"
|
|
for (k, ins) in p.instrs.enumerated() {
|
|
let d = "r\(ins.dst)", a = "r\(ins.a)", b = "r\(ins.b)"
|
|
var line: String
|
|
switch ins.op {
|
|
case .add: line = "\(d) = \(d) + \(a) + select(\(hex(ins.imm)), \(hex(ins.imm2)), ((sel >> \(ins.bit)u) & 1u) != 0u);"
|
|
case .sub: line = "\(d) = \(d) - \(a);"
|
|
case .mul: line = "\(d) = \(d) * \(a);"
|
|
case .mulhi: line = "\(d) = mulhi(\(d), \(a));"
|
|
case .xor: line = "\(d) = \(d) ^ \(a);"
|
|
case .or: line = "\(d) = \(d) | \(a);"
|
|
case .rotl: line = "\(d) = rotl_imm(\(d), \(ins.rot)u);"
|
|
case .rotr: line = "\(d) = rotr_var(\(d), \(a));"
|
|
case .mad: line = "\(d) = \(a) * \(b) + \(d);"
|
|
case .shfl: line = "\(d) = \(d) ^ simd_shuffle_xor(\(a), (ushort)\(ins.mask));"
|
|
case .load: line = "\(d) = \(d) ^ dataset[\(a) & MASK];"
|
|
}
|
|
s += " \(line) // \(k)\n"
|
|
}
|
|
s += """
|
|
}
|
|
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
|
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
|
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
|
}
|
|
|
|
"""
|
|
return s
|
|
}
|
|
|
|
let fillMSL = """
|
|
#include <metal_stdlib>
|
|
using namespace metal;
|
|
inline uint ds_elem(uint i, uint d0, uint d1) {
|
|
uint x = i ^ d0;
|
|
x *= 0x9E3779B1u; x ^= x >> 15;
|
|
x += d1;
|
|
x *= 0x85EBCA77u; x ^= x >> 13;
|
|
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
|
return x;
|
|
}
|
|
kernel void igneum_fill(device uint* dataset [[buffer(0)]],
|
|
constant uint2& day [[buffer(1)]],
|
|
uint gid [[thread_position_in_grid]]) {
|
|
dataset[gid] = ds_elem(gid, day.x, day.y);
|
|
}
|
|
"""
|
|
|
|
// MARK: - CPU reference interpreter for one 32-lane warp
|
|
|
|
func cpuWarp(_ p: Program, baseNonce: UInt32, day: (UInt32, UInt32), mask: UInt32) -> [UInt64] {
|
|
cpuWarpTraced(p, baseNonce: baseNonce, day: day, mask: mask, trace: nil)
|
|
}
|
|
|
|
// Same interpreter with an optional hook. When `trace` is set it is called before every instruction with
|
|
// (iteration, instruction index, the 8 registers of lane 0). The edge-case tests use it to prove that the
|
|
// operand values they were built to produce really occurred. The bench passes nil.
|
|
func cpuWarpTraced(_ p: Program, baseNonce: UInt32, day: (UInt32, UInt32), mask: UInt32,
|
|
trace: ((Int, Int, [UInt32]) -> Void)?) -> [UInt64] {
|
|
let lanes = 32
|
|
var r = [UInt32](repeating: 0, count: lanes * 8) // r[lane*8 + reg]
|
|
for lane in 0..<lanes {
|
|
let nonce = baseNonce &+ UInt32(lane)
|
|
for i in 0..<8 {
|
|
var x = nonce ^ p.seed[i]
|
|
x &+= 0x9e3779b9 &* UInt32(i + 1)
|
|
x = splitmix32(x)
|
|
r[lane * 8 + i] = x ^ p.seed[(i + 1) & 7]
|
|
}
|
|
}
|
|
var tmp = [UInt32](repeating: 0, count: lanes)
|
|
for it in 0..<Program.iterations {
|
|
for lane in 0..<lanes { tmp[lane] = r[lane * 8] } // sel = r0 at the top of the iteration
|
|
let sel = tmp
|
|
for (k, ins) in p.instrs.enumerated() {
|
|
if let t = trace { t(it, k, Array(r[0..<8])) }
|
|
switch ins.op {
|
|
case .shfl:
|
|
for lane in 0..<lanes { tmp[lane] = r[lane * 8 + ins.a] }
|
|
for lane in 0..<lanes { r[lane * 8 + ins.dst] ^= tmp[lane ^ ins.mask] }
|
|
default:
|
|
for lane in 0..<lanes {
|
|
let base = lane * 8
|
|
let d = r[base + ins.dst], a = r[base + ins.a]
|
|
var v: UInt32
|
|
switch ins.op {
|
|
case .add:
|
|
let s = (sel[lane] >> UInt32(ins.bit)) & 1
|
|
v = d &+ a &+ (s != 0 ? ins.imm2 : ins.imm)
|
|
case .sub: v = d &- a
|
|
case .mul: v = d &* a
|
|
case .mulhi: v = mulhi32(d, a)
|
|
case .xor: v = d ^ a
|
|
case .or: v = d | a
|
|
case .rotl: v = rotl32(d, ins.rot)
|
|
case .rotr: v = rotr32(d, a)
|
|
case .mad: v = (a &* r[base + ins.b]) &+ d
|
|
case .load: v = d ^ datasetElem(a & mask, day.0, day.1)
|
|
case .shfl: v = d // unreachable
|
|
}
|
|
r[base + ins.dst] = v
|
|
}
|
|
}
|
|
}
|
|
}
|
|
var out = [UInt64](repeating: 0, count: lanes)
|
|
for lane in 0..<lanes {
|
|
let b = lane * 8
|
|
let lo = r[b] ^ rotl32(r[b + 1], 7) ^ rotl32(r[b + 2], 14) ^ rotl32(r[b + 3], 21)
|
|
let hi = r[b + 4] ^ rotl32(r[b + 5], 9) ^ rotl32(r[b + 6], 18) ^ rotl32(r[b + 7], 27)
|
|
out[lane] = (UInt64(hi) << 32) | UInt64(lo)
|
|
}
|
|
return out
|
|
}
|
|
|
|
// MARK: - Program pack export (CUDA twin of the Metal kernel)
|
|
//
|
|
// Writes, for one seed: program.json, vectors.json, kernel.cu, program.h, vectors.h, program.metal.
|
|
// The CUDA kernel is emitted from the same Instr list as the MSL above, line for line.
|
|
// Differences by design: the dataset mask is a kernel argument (so the host can sweep dataset
|
|
// sizes with one ahead-of-time compile), seeds are inlined as literals, and the host launch
|
|
// wrappers live in kernel.cu so host.cu never declares a __global__ across translation units.
|
|
|
|
let packVectorBases: [UInt32] = [0, 4096, 1000000]
|
|
|
|
func hex64(_ v: UInt64) -> String { String(format: "0x%016llxull", v) }
|
|
func jhex(_ v: UInt32) -> String { String(format: "\"0x%08x\"", v) }
|
|
func jhex64(_ v: UInt64) -> String { String(format: "\"0x%016llx\"", v) }
|
|
func jstr(_ s: String) -> String {
|
|
var o = "\""
|
|
for c in s.unicodeScalars {
|
|
switch c {
|
|
case "\"": o += "\\\""
|
|
case "\\": o += "\\\\"
|
|
case "\n": o += "\\n"
|
|
default: o.unicodeScalars.append(c)
|
|
}
|
|
}
|
|
return o + "\""
|
|
}
|
|
|
|
func generateCUDA(_ p: Program) -> String {
|
|
var s = """
|
|
// Generated by proto-metal/igneum-bench --export-pack for seed "\(p.seedString)". Do not edit by hand.
|
|
// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal).
|
|
// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC.
|
|
#include <cuda_runtime.h>
|
|
#include <cstdint>
|
|
#include "program.h"
|
|
|
|
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
|
|
x ^= x >> 16; x *= 0x7feb352du;
|
|
x ^= x >> 15; x *= 0x846ca68bu;
|
|
x ^= x >> 16;
|
|
return x;
|
|
}
|
|
// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31.
|
|
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
|
|
// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x.
|
|
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
|
__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) {
|
|
uint32_t x = i ^ d0;
|
|
x *= 0x9E3779B1u; x ^= x >> 15;
|
|
x += d1;
|
|
x *= 0x85EBCA77u; x ^= x >> 13;
|
|
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
|
return x;
|
|
}
|
|
|
|
// dataset[i] = ds_elem(i, d0, d1) for i < n. Same closed form as the Metal igneum_fill kernel.
|
|
__global__ void igneum_fill(uint32_t* ds, uint32_t n, uint32_t d0, uint32_t d1) {
|
|
uint32_t i = blockIdx.x * blockDim.x + threadIdx.x;
|
|
if (i < n) ds[i] = ds_elem(i, d0, d1);
|
|
}
|
|
|
|
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
|
|
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
|
|
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
|
|
__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask) {
|
|
uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;
|
|
uint32_t nonce = baseNonce + gid;
|
|
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
|
|
|
|
"""
|
|
for i in 0..<8 {
|
|
let addc = 0x9e3779b9 &* UInt32(i + 1)
|
|
s += " { uint32_t x = nonce ^ \(hex(p.seed[i])); x += \(hex(addc)); x = splitmix32(x); r\(i) = x ^ \(hex(p.seed[(i + 1) & 7])); } // SEEDW[\(i)], 0x9e3779b9u * \(i + 1)u, SEEDW[\((i + 1) & 7)]\n"
|
|
}
|
|
s += "\n for (uint32_t it = 0u; it < \(Program.iterations)u; ++it) {\n uint32_t sel = r0;\n"
|
|
for (k, ins) in p.instrs.enumerated() {
|
|
let d = "r\(ins.dst)", a = "r\(ins.a)", b = "r\(ins.b)"
|
|
var line: String
|
|
switch ins.op {
|
|
// Metal select(A, B, c) returns c ? B : A, so the true branch is imm2 here as well.
|
|
case .add: line = "\(d) = \(d) + \(a) + ((((sel >> \(ins.bit)u) & 1u) != 0u) ? \(hex(ins.imm2)) : \(hex(ins.imm)));"
|
|
case .sub: line = "\(d) = \(d) - \(a);"
|
|
case .mul: line = "\(d) = \(d) * \(a);"
|
|
case .mulhi: line = "\(d) = __umulhi(\(d), \(a));"
|
|
case .xor: line = "\(d) = \(d) ^ \(a);"
|
|
case .or: line = "\(d) = \(d) | \(a);"
|
|
case .rotl: line = "\(d) = rotl_imm(\(d), \(ins.rot)u);"
|
|
case .rotr: line = "\(d) = rotr_var(\(d), \(a));"
|
|
case .mad: line = "\(d) = \(a) * \(b) + \(d);"
|
|
case .shfl: line = "\(d) = \(d) ^ __shfl_xor_sync(0xffffffffu, \(a), \(ins.mask));"
|
|
case .load: line = "\(d) = \(d) ^ ds[\(a) & mask];"
|
|
}
|
|
s += " \(line) // \(k) \(ins.op.rawValue)\n"
|
|
}
|
|
s += """
|
|
}
|
|
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
|
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
|
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
|
|
}
|
|
|
|
// Host-side launch wrappers. Declared in program.h, called from host.cu.
|
|
cudaError_t igneum_launch_fill(uint32_t* ds, uint32_t nWords, uint32_t d0, uint32_t d1) {
|
|
if (nWords == 0u) return cudaErrorInvalidValue;
|
|
uint32_t block = 256u;
|
|
uint32_t grid = (nWords + block - 1u) / block;
|
|
igneum_fill<<<grid, block>>>(ds, nWords, d0, d1);
|
|
return cudaGetLastError();
|
|
}
|
|
|
|
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
|
uint32_t nonces, uint32_t blockWarps) {
|
|
if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;
|
|
uint32_t block = 32u * blockWarps;
|
|
if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;
|
|
igneum_hash<<<nonces / block, block>>>(ds, out, baseNonce, mask);
|
|
return cudaGetLastError();
|
|
}
|
|
|
|
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
|
|
cudaFuncAttributes attr;
|
|
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash);
|
|
if (e != cudaSuccess) return e;
|
|
*numRegs = attr.numRegs;
|
|
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0);
|
|
}
|
|
|
|
"""
|
|
return s
|
|
}
|
|
|
|
func generateProgramHeader(_ p: Program, dayString: String, day: (UInt32, UInt32), datasetLog2: Int) -> String {
|
|
let mask = UInt32((1 << datasetLog2) - 1)
|
|
let mix = p.histogram.map { "\($0.0)=\($0.1)" }.joined(separator: " ")
|
|
return """
|
|
// Generated by proto-metal/igneum-bench --export-pack for seed "\(p.seedString)". Do not edit by hand.
|
|
// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.
|
|
#pragma once
|
|
#include <cuda_runtime.h>
|
|
#include <cstdint>
|
|
|
|
#define IGNEUM_SEED_STRING \(jstr(p.seedString))
|
|
#define IGNEUM_DAY_STRING \(jstr(dayString))
|
|
#define IGNEUM_DAY0 \(hex(day.0))
|
|
#define IGNEUM_DAY1 \(hex(day.1))
|
|
#define IGNEUM_DATASET_LOG2 \(datasetLog2)
|
|
#define IGNEUM_MASK \(hex(mask))
|
|
#define IGNEUM_LANES 32
|
|
#define IGNEUM_ITERATIONS \(Program.iterations)
|
|
#define IGNEUM_INSTR_COUNT \(Program.count)
|
|
#define IGNEUM_LOADS_PER_HASH \(p.loadsPerHash)
|
|
#define IGNEUM_OP_MIX \(jstr(mix))
|
|
|
|
#define IGNEUM_SEEDW_INIT { \(p.seed.map(hex).joined(separator: ", ")) }
|
|
|
|
// Defined in kernel.cu. Both launch on the default stream and return cudaGetLastError().
|
|
cudaError_t igneum_launch_fill(uint32_t* ds, uint32_t nWords, uint32_t d0, uint32_t d1);
|
|
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
|
uint32_t nonces, uint32_t blockWarps);
|
|
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
|
|
|
|
"""
|
|
}
|
|
|
|
func generateVectorsHeader(_ p: Program, bases: [UInt32], outs: [[UInt64]], head: [UInt32], last: UInt32, mask: UInt32, source: String) -> String {
|
|
var s = """
|
|
// Generated by proto-metal/igneum-bench --export-pack for seed "\(p.seedString)". Do not edit by hand.
|
|
// Expected outputs: \(source)
|
|
#pragma once
|
|
#include <cstdint>
|
|
|
|
#define IGNEUM_VEC_WARPS \(bases.count)
|
|
static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { \(bases.map { "\($0)u" }.joined(separator: ", ")) };
|
|
static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {
|
|
|
|
"""
|
|
for (i, o) in outs.enumerated() {
|
|
s += " { // base nonce \(bases[i])\n"
|
|
for row in 0..<4 {
|
|
s += " " + (0..<8).map { hex64(o[row * 8 + $0]) }.joined(separator: ", ") + (row == 3 ? "\n" : ",\n")
|
|
}
|
|
s += i == outs.count - 1 ? " }\n" : " },\n"
|
|
}
|
|
s += """
|
|
};
|
|
|
|
// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (\(mask)).
|
|
static const uint32_t IGNEUM_DS_HEAD[16] = {
|
|
\((0..<8).map { hex(head[$0]) }.joined(separator: ", ")),
|
|
\((8..<16).map { hex(head[$0]) }.joined(separator: ", "))
|
|
};
|
|
static const uint32_t IGNEUM_DS_LAST_INDEX = \(mask)u;
|
|
static const uint32_t IGNEUM_DS_LAST = \(hex(last));
|
|
|
|
"""
|
|
return s
|
|
}
|
|
|
|
func generateProgramJSON(_ p: Program, dayString: String, day: (UInt32, UInt32), datasetLog2: Int) -> String {
|
|
let mask = UInt32((1 << datasetLog2) - 1)
|
|
var s = "{\n"
|
|
s += " \"format\": \"igneum-program-pack-1\",\n"
|
|
s += " \"seed\": \(jstr(p.seedString)),\n"
|
|
s += " \"seed_words\": [\(p.seed.map(jhex).joined(separator: ", "))],\n"
|
|
s += " \"seed_derivation\": \"FNV-1a 64 over UTF-8 of seed, basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32\",\n"
|
|
s += " \"lanes\": 32,\n"
|
|
s += " \"registers\": 8,\n"
|
|
s += " \"iterations\": \(Program.iterations),\n"
|
|
s += " \"instruction_count\": \(Program.count),\n"
|
|
s += " \"loads_per_hash\": \(p.loadsPerHash),\n"
|
|
s += " \"op_mix\": {\(p.histogram.map { "\(jstr($0.0)): \($0.1)" }.joined(separator: ", "))},\n"
|
|
s += " \"register_init\": \"for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]\",\n"
|
|
s += " \"splitmix32\": \"x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16\",\n"
|
|
s += " \"iteration\": \"sel = r0 sampled once at the top of each iteration, then all instructions in order\",\n"
|
|
s += " \"output\": \"lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo\",\n"
|
|
s += " \"op_semantics\": {\n"
|
|
s += " \"add\": \"dst = dst + src + (bit `bit` of sel ? imm2 : imm)\",\n"
|
|
s += " \"sub\": \"dst = dst - src\",\n"
|
|
s += " \"mul\": \"dst = dst * src (low 32)\",\n"
|
|
s += " \"mulhi\": \"dst = high 32 bits of dst * src\",\n"
|
|
s += " \"xor\": \"dst = dst ^ src\",\n"
|
|
s += " \"or\": \"dst = dst | src\",\n"
|
|
s += " \"rotl\": \"dst = rotl(dst, rot), rot in 1..31\",\n"
|
|
s += " \"rotr\": \"dst = rotr(dst, src & 31)\",\n"
|
|
s += " \"mad\": \"dst = src * src2 + dst\",\n"
|
|
s += " \"shfl\": \"dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp\",\n"
|
|
s += " \"load\": \"dst = dst ^ dataset[src & dataset.mask]\"\n"
|
|
s += " },\n"
|
|
s += " \"dataset\": {\n"
|
|
s += " \"log2_words\": \(datasetLog2),\n"
|
|
s += " \"bytes\": \(UInt64(1) << UInt64(datasetLog2 + 2)),\n"
|
|
s += " \"mask\": \(jhex(mask)),\n"
|
|
s += " \"day\": \(jstr(dayString)),\n"
|
|
s += " \"day_words_from\": \(jstr("day/" + dayString)),\n"
|
|
s += " \"d0\": \(jhex(day.0)),\n"
|
|
s += " \"d1\": \(jhex(day.1)),\n"
|
|
s += " \"formula\": \"x = i ^ d0; x *= 0x9E3779B1; x ^= x>>15; x += d1; x *= 0x85EBCA77; x ^= x>>13; x *= 0xC2B2AE3D; x ^= x>>16 (all mod 2^32)\"\n"
|
|
s += " },\n"
|
|
s += " \"instructions\": [\n"
|
|
for (k, ins) in p.instrs.enumerated() {
|
|
s += " {\"i\": \(k), \"op\": \(jstr(ins.op.rawValue)), \"dst\": \(ins.dst), \"src\": \(ins.a), \"src2\": \(ins.b), \"imm\": \(jhex(ins.imm)), \"imm2\": \(jhex(ins.imm2)), \"rot\": \(ins.rot), \"bit\": \(ins.bit), \"mask\": \(ins.mask)}"
|
|
s += k == p.instrs.count - 1 ? "\n" : ",\n"
|
|
}
|
|
s += " ]\n}\n"
|
|
return s
|
|
}
|
|
|
|
func generateVectorsJSON(_ p: Program, dayString: String, datasetLog2: Int, bases: [UInt32], outs: [[UInt64]], head: [UInt32], last: UInt32, mask: UInt32, source: String) -> String {
|
|
var s = "{\n"
|
|
s += " \"seed\": \(jstr(p.seedString)),\n"
|
|
s += " \"day\": \(jstr(dayString)),\n"
|
|
s += " \"dataset_log2_words\": \(datasetLog2),\n"
|
|
s += " \"mask\": \(jhex(mask)),\n"
|
|
s += " \"lanes\": 32,\n"
|
|
s += " \"source\": \(jstr(source)),\n"
|
|
s += " \"warps\": [\n"
|
|
for (i, o) in outs.enumerated() {
|
|
s += " {\"base_nonce\": \(bases[i]), \"expected\": [\n"
|
|
for row in 0..<4 {
|
|
s += " " + (0..<8).map { jhex64(o[row * 8 + $0]) }.joined(separator: ", ") + (row == 3 ? "\n" : ",\n")
|
|
}
|
|
s += i == outs.count - 1 ? " ]}\n" : " ]},\n"
|
|
}
|
|
s += " ],\n"
|
|
s += " \"dataset_head\": [\(head.map(jhex).joined(separator: ", "))],\n"
|
|
s += " \"dataset_last_index\": \(mask),\n"
|
|
s += " \"dataset_last\": \(jhex(last))\n"
|
|
s += "}\n"
|
|
return s
|
|
}
|
|
|
|
// Runs the Metal kernel for each base nonce (one 32-thread threadgroup each) and compares with `expected`.
|
|
func metalCrossCheck(_ p: Program, datasetLog2: Int, day: (UInt32, UInt32), bases: [UInt32], expected: [[UInt64]]) -> (ok: Bool, detail: String) {
|
|
guard let device = MTLCreateSystemDefaultDevice(), let queue = device.makeCommandQueue() else { return (false, "no Metal device") }
|
|
let words = 1 << datasetLog2
|
|
guard let dataset = device.makeBuffer(length: words * 4, options: .storageModePrivate),
|
|
let outBuf = device.makeBuffer(length: 32 * 8, options: .storageModeShared) else { return (false, "buffer allocation failed") }
|
|
do {
|
|
let flib = try device.makeLibrary(source: fillMSL, options: MTLCompileOptions())
|
|
let fpipe = try device.makeComputePipelineState(function: flib.makeFunction(name: "igneum_fill")!)
|
|
let hlib = try device.makeLibrary(source: generateMSL(p, datasetLog2: datasetLog2), options: MTLCompileOptions())
|
|
let hpipe = try device.makeComputePipelineState(function: hlib.makeFunction(name: "igneum_hash")!)
|
|
let cb = queue.makeCommandBuffer()!
|
|
let enc = cb.makeComputeCommandEncoder()!
|
|
enc.setComputePipelineState(fpipe)
|
|
enc.setBuffer(dataset, offset: 0, index: 0)
|
|
var d = (day.0, day.1)
|
|
enc.setBytes(&d, length: 8, index: 1)
|
|
enc.dispatchThreadgroups(MTLSize(width: words / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
|
|
enc.endEncoding()
|
|
cb.commit(); cb.waitUntilCompleted()
|
|
if let e = cb.error { return (false, "fill error \(e)") }
|
|
var bad = [String]()
|
|
for (i, base) in bases.enumerated() {
|
|
let cb2 = queue.makeCommandBuffer()!
|
|
let e2 = cb2.makeComputeCommandEncoder()!
|
|
e2.setComputePipelineState(hpipe)
|
|
e2.setBuffer(dataset, offset: 0, index: 0)
|
|
e2.setBuffer(outBuf, offset: 0, index: 1)
|
|
var b = base
|
|
e2.setBytes(&b, length: 4, index: 2)
|
|
e2.dispatchThreadgroups(MTLSize(width: 1, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 32, height: 1, depth: 1))
|
|
e2.endEncoding()
|
|
cb2.commit(); cb2.waitUntilCompleted()
|
|
if let e = cb2.error { return (false, "hash error \(e)") }
|
|
let ptr = outBuf.contents().bindMemory(to: UInt64.self, capacity: 32)
|
|
let got = (0..<32).map { ptr[$0] }
|
|
if got != expected[i] { bad.append("base \(base)") }
|
|
}
|
|
return (bad.isEmpty, bad.isEmpty ? "Metal GPU cross-check PASS \(bases.count)/\(bases.count) warps" : "Metal GPU cross-check FAIL: \(bad.joined(separator: ", "))")
|
|
} catch {
|
|
return (false, "Metal compile error \(error)")
|
|
}
|
|
}
|
|
|
|
func exportPack(_ opts: Options) -> Never {
|
|
let dir = opts.exportPack!
|
|
let program = generateProgram(seedString: opts.seed)
|
|
let dayWords = seedWords("day/" + opts.day)
|
|
let day = (dayWords[0], dayWords[1])
|
|
let mask = UInt32((1 << opts.datasetLog2) - 1)
|
|
print("igneum-bench --export-pack \(dir)")
|
|
print("seed \"\(opts.seed)\", day \"\(opts.day)\", dataset 2^\(opts.datasetLog2) words, loads/hash \(program.loadsPerHash)")
|
|
print("op mix: " + program.histogram.map { "\($0.0)=\($0.1)" }.joined(separator: " "))
|
|
|
|
let outs = packVectorBases.map { cpuWarp(program, baseNonce: $0, day: day, mask: mask) }
|
|
let head = (0..<16).map { datasetElem(UInt32($0), day.0, day.1) }
|
|
let last = datasetElem(mask, day.0, day.1)
|
|
|
|
var source = "proto-metal CPU interpreter (cpuWarp) on Apple M5 Max"
|
|
let check = metalCrossCheck(program, datasetLog2: opts.datasetLog2, day: day, bases: packVectorBases, expected: outs)
|
|
print(check.detail)
|
|
source += "; " + check.detail
|
|
if !check.ok { print("FAIL: vectors do not match the Metal GPU, pack not written"); exit(1) }
|
|
|
|
let files: [(String, String)] = [
|
|
("program.json", generateProgramJSON(program, dayString: opts.day, day: day, datasetLog2: opts.datasetLog2)),
|
|
("vectors.json", generateVectorsJSON(program, dayString: opts.day, datasetLog2: opts.datasetLog2, bases: packVectorBases, outs: outs, head: head, last: last, mask: mask, source: source)),
|
|
("kernel.cu", generateCUDA(program)),
|
|
("program.h", generateProgramHeader(program, dayString: opts.day, day: day, datasetLog2: opts.datasetLog2)),
|
|
("vectors.h", generateVectorsHeader(program, bases: packVectorBases, outs: outs, head: head, last: last, mask: mask, source: source)),
|
|
("program.metal", generateMSL(program, datasetLog2: opts.datasetLog2)),
|
|
]
|
|
do {
|
|
try FileManager.default.createDirectory(atPath: dir, withIntermediateDirectories: true)
|
|
for (name, text) in files {
|
|
try text.write(toFile: "\(dir)/\(name)", atomically: true, encoding: .utf8)
|
|
print("wrote \(dir)/\(name) (\(text.utf8.count) bytes)")
|
|
}
|
|
} catch {
|
|
print("FAIL: write error \(error)"); exit(1)
|
|
}
|
|
for (i, b) in packVectorBases.enumerated() {
|
|
print("vector warp base \(b): lane0 \(String(format: "%016llx", outs[i][0])) lane31 \(String(format: "%016llx", outs[i][31]))")
|
|
}
|
|
print("OVERALL: PASS (pack written)")
|
|
exit(0)
|
|
}
|
|
|
|
// MARK: - Timing
|
|
|
|
@inline(__always) func nowNs() -> UInt64 { clock_gettime_nsec_np(CLOCK_UPTIME_RAW) }
|
|
func ms(_ a: UInt64, _ b: UInt64) -> Double { Double(b - a) / 1e6 }
|
|
func fmt(_ v: Double, _ digits: Int = 2) -> String { String(format: "%.\(digits)f", v) }
|
|
|
|
// MARK: - GPU context
|
|
|
|
final class GPU {
|
|
let device: MTLDevice
|
|
let queue: MTLCommandQueue
|
|
init() {
|
|
guard let d = MTLCreateSystemDefaultDevice(), let q = d.makeCommandQueue() else {
|
|
print("FAIL: no Metal device"); exit(1)
|
|
}
|
|
device = d; queue = q
|
|
}
|
|
}
|
|
|
|
struct EpochResult {
|
|
var seed: String
|
|
var libraryMs: Double
|
|
var pipelineMs: Double
|
|
var hashesPerSecWall: Double
|
|
var hashesPerSecGPU: Double
|
|
var gbpsWall: Double
|
|
var gbpsGPU: Double
|
|
var loadsPerHash: Int
|
|
var verify: [(warp: Int, pass: Bool, ms: Double, repMs: Double)]
|
|
var allPass: Bool { verify.allSatisfy { $0.pass } }
|
|
}
|
|
|
|
func runEpoch(gpu: GPU, opts: Options, seedString: String, dataset: MTLBuffer, day: (UInt32, UInt32)) -> EpochResult {
|
|
let program = generateProgram(seedString: seedString)
|
|
let msl = generateMSL(program, datasetLog2: opts.datasetLog2)
|
|
if let dir = opts.dumpDir {
|
|
try? FileManager.default.createDirectory(atPath: dir, withIntermediateDirectories: true)
|
|
let safe = seedString.replacingOccurrences(of: "/", with: "_")
|
|
try? msl.write(toFile: "\(dir)/program-\(safe).metal", atomically: true, encoding: .utf8)
|
|
}
|
|
|
|
print("\n=== epoch seed \"\(seedString)\" ===")
|
|
print("program: \(Program.count) instructions x \(Program.iterations) iterations, loads/hash = \(program.loadsPerHash)")
|
|
print("op mix: " + program.histogram.map { "\($0.0)=\($0.1)" }.joined(separator: " "))
|
|
|
|
// Runtime compile
|
|
let t0 = nowNs()
|
|
let library: MTLLibrary
|
|
do {
|
|
let copts = MTLCompileOptions()
|
|
library = try gpu.device.makeLibrary(source: msl, options: copts)
|
|
} catch {
|
|
print("FAIL: Metal compile error:\n\(error)")
|
|
exit(1)
|
|
}
|
|
let t1 = nowNs()
|
|
guard let fn = library.makeFunction(name: "igneum_hash") else { print("FAIL: no igneum_hash"); exit(1) }
|
|
let pipeline: MTLComputePipelineState
|
|
do { pipeline = try gpu.device.makeComputePipelineState(function: fn) } catch {
|
|
print("FAIL: pipeline error: \(error)"); exit(1)
|
|
}
|
|
let t2 = nowNs()
|
|
let libMs = ms(t0, t1), pipeMs = ms(t1, t2)
|
|
print("compile: library \(fmt(libMs)) ms, pipeline \(fmt(pipeMs)) ms, total \(fmt(libMs + pipeMs)) ms")
|
|
print("threadExecutionWidth = \(pipeline.threadExecutionWidth), maxTotalThreadsPerThreadgroup = \(pipeline.maxTotalThreadsPerThreadgroup)")
|
|
if pipeline.threadExecutionWidth != 32 {
|
|
print("WARNING: threadExecutionWidth is not 32; the one-warp-per-threadgroup assumption does not hold on this device")
|
|
}
|
|
|
|
// Buffers
|
|
let n = 1 << opts.batchLog2
|
|
let groups = n / 32
|
|
guard let outBuf = gpu.device.makeBuffer(length: n * 8, options: .storageModeShared) else { print("FAIL: out buffer"); exit(1) }
|
|
|
|
func encodeBatch(_ cb: MTLCommandBuffer, base: UInt32) {
|
|
let enc = cb.makeComputeCommandEncoder()!
|
|
enc.setComputePipelineState(pipeline)
|
|
enc.setBuffer(dataset, offset: 0, index: 0)
|
|
enc.setBuffer(outBuf, offset: 0, index: 1)
|
|
var b = base
|
|
enc.setBytes(&b, length: 4, index: 2)
|
|
enc.dispatchThreadgroups(MTLSize(width: groups, height: 1, depth: 1),
|
|
threadsPerThreadgroup: MTLSize(width: 32, height: 1, depth: 1))
|
|
enc.endEncoding()
|
|
}
|
|
|
|
// Batch 0: warm-up and verification source (baseNonce 0)
|
|
do {
|
|
let cb = gpu.queue.makeCommandBuffer()!
|
|
encodeBatch(cb, base: 0)
|
|
let w0 = nowNs()
|
|
cb.commit(); cb.waitUntilCompleted()
|
|
let w1 = nowNs()
|
|
if let e = cb.error { print("FAIL: batch 0 error \(e)"); exit(1) }
|
|
print("warm-up batch: \(n) hashes in \(fmt(ms(w0, w1))) ms wall, \(fmt((cb.gpuEndTime - cb.gpuStartTime) * 1000)) ms GPU")
|
|
}
|
|
// Pick warps to verify from batch 0
|
|
var warps = [0, groups / 2 + 1, groups - 1]
|
|
var vr = SplitMix64(s: UInt64(program.seed[4]) | (UInt64(program.seed[5]) << 32))
|
|
while warps.count < opts.verifyWarps { warps.append(vr.below(groups)) }
|
|
warps = Array(warps.prefix(max(opts.verifyWarps, 1)))
|
|
let outPtr = outBuf.contents().bindMemory(to: UInt64.self, capacity: n)
|
|
var gpuOutputs = [[UInt64]]()
|
|
for w in warps { gpuOutputs.append((0..<32).map { outPtr[w * 32 + $0] }) }
|
|
|
|
// Timed batches
|
|
var cbs = [MTLCommandBuffer]()
|
|
for b in 0..<opts.batches {
|
|
let cb = gpu.queue.makeCommandBuffer()!
|
|
encodeBatch(cb, base: UInt32(truncatingIfNeeded: (b + 1) * n))
|
|
cbs.append(cb)
|
|
}
|
|
let s0 = nowNs()
|
|
for cb in cbs { cb.commit() }
|
|
cbs.last!.waitUntilCompleted()
|
|
let s1 = nowNs()
|
|
for cb in cbs { if let e = cb.error { print("FAIL: batch error \(e)"); exit(1) } }
|
|
let gpuSeconds = cbs.reduce(0.0) { $0 + ($1.gpuEndTime - $1.gpuStartTime) }
|
|
let wallSeconds = Double(s1 - s0) / 1e9
|
|
let totalHashes = Double(n * opts.batches)
|
|
let hpsWall = totalHashes / wallSeconds
|
|
let hpsGPU = totalHashes / gpuSeconds
|
|
let bytesPerHash = Double(program.loadsPerHash * 4)
|
|
let gbpsWall = hpsWall * bytesPerHash / 1e9
|
|
let gbpsGPU = hpsGPU * bytesPerHash / 1e9
|
|
print("timed: \(opts.batches) batches x \(n) hashes = \(Int(totalHashes)) hashes")
|
|
print(" wall \(fmt(wallSeconds * 1000)) ms -> \(fmt(hpsWall / 1e6, 3)) Mhash/s, \(fmt(gbpsWall)) GB/s useful (loads x 4 B)")
|
|
print(" GPU \(fmt(gpuSeconds * 1000)) ms -> \(fmt(hpsGPU / 1e6, 3)) Mhash/s, \(fmt(gbpsGPU)) GB/s useful (loads x 4 B)")
|
|
|
|
// CPU verification
|
|
let mask = UInt32((1 << opts.datasetLog2) - 1)
|
|
var verify = [(warp: Int, pass: Bool, ms: Double, repMs: Double)]()
|
|
for (i, w) in warps.enumerated() {
|
|
let base = UInt32(w * 32)
|
|
let c0 = nowNs()
|
|
let cpu = cpuWarp(program, baseNonce: base, day: day, mask: mask)
|
|
let c1 = nowNs()
|
|
// repeated runs for a steadier figure
|
|
let reps = 20
|
|
let r0 = nowNs()
|
|
var sink: UInt64 = 0
|
|
for _ in 0..<reps { sink ^= cpuWarp(program, baseNonce: base, day: day, mask: mask)[0] }
|
|
let r1 = nowNs()
|
|
let pass = cpu == gpuOutputs[i] && sink != 1
|
|
let single = ms(c0, c1), rep = ms(r0, r1) / Double(reps)
|
|
verify.append((w, pass, single, rep))
|
|
var detail = ""
|
|
if !pass {
|
|
let bad = (0..<32).filter { cpu[$0] != gpuOutputs[i][$0] }
|
|
detail = " mismatched lanes: \(bad) first: cpu=\(String(format: "%016llx", cpu[bad.first ?? 0])) gpu=\(String(format: "%016llx", gpuOutputs[i][bad.first ?? 0]))"
|
|
}
|
|
print("verify warp \(w) (nonces \(base)..\(base + 31)): \(pass ? "PASS" : "FAIL") cpu \(fmt(single, 3)) ms single, \(fmt(rep, 3)) ms avg of \(reps)\(detail)")
|
|
}
|
|
|
|
return EpochResult(seed: seedString, libraryMs: libMs, pipelineMs: pipeMs,
|
|
hashesPerSecWall: hpsWall, hashesPerSecGPU: hpsGPU, gbpsWall: gbpsWall, gbpsGPU: gbpsGPU,
|
|
loadsPerHash: program.loadsPerHash, verify: verify)
|
|
}
|
|
|
|
// MARK: - Hardening tests: shared helpers
|
|
//
|
|
// Added 3 October 2026. Everything below reuses generateProgram, generateMSL, fillMSL and cpuWarp
|
|
// unchanged; the helpers only wrap compile, fill and dispatch so the tests can run many programs
|
|
// and many warps cheaply. Nothing in the bench path (runEpoch) or the pack exporter calls these.
|
|
|
|
struct CompiledHash {
|
|
let pipeline: MTLComputePipelineState
|
|
let libraryMs: Double
|
|
let pipelineMs: Double
|
|
var totalMs: Double { libraryMs + pipelineMs }
|
|
}
|
|
|
|
struct IgneumError: Error, CustomStringConvertible {
|
|
let description: String
|
|
init(_ s: String) { description = s }
|
|
}
|
|
|
|
func compileHash(_ gpu: GPU, msl: String) throws -> CompiledHash {
|
|
let t0 = nowNs()
|
|
let lib = try gpu.device.makeLibrary(source: msl, options: MTLCompileOptions())
|
|
let t1 = nowNs()
|
|
guard let fn = lib.makeFunction(name: "igneum_hash") else { throw IgneumError("no igneum_hash function in library") }
|
|
let pipe = try gpu.device.makeComputePipelineState(function: fn)
|
|
let t2 = nowNs()
|
|
return CompiledHash(pipeline: pipe, libraryMs: ms(t0, t1), pipelineMs: ms(t1, t2))
|
|
}
|
|
|
|
// Allocates a private 2^log2-word dataset and fills it on the GPU with the closed form for `day`.
|
|
func makeDataset(_ gpu: GPU, log2: Int, day: (UInt32, UInt32)) -> MTLBuffer {
|
|
let words = 1 << log2
|
|
guard let buf = gpu.device.makeBuffer(length: words * 4, options: .storageModePrivate) else {
|
|
print("FAIL: cannot allocate 2^\(log2) word dataset"); exit(1)
|
|
}
|
|
do {
|
|
let lib = try gpu.device.makeLibrary(source: fillMSL, options: MTLCompileOptions())
|
|
let pipe = try gpu.device.makeComputePipelineState(function: lib.makeFunction(name: "igneum_fill")!)
|
|
let cb = gpu.queue.makeCommandBuffer()!
|
|
let enc = cb.makeComputeCommandEncoder()!
|
|
enc.setComputePipelineState(pipe)
|
|
enc.setBuffer(buf, offset: 0, index: 0)
|
|
var d = (day.0, day.1)
|
|
enc.setBytes(&d, length: 8, index: 1)
|
|
enc.dispatchThreadgroups(MTLSize(width: words / 256, height: 1, depth: 1),
|
|
threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
|
|
enc.endEncoding()
|
|
cb.commit(); cb.waitUntilCompleted()
|
|
if let e = cb.error { print("FAIL: fill error \(e)"); exit(1) }
|
|
} catch {
|
|
print("FAIL: fill kernel \(error)"); exit(1)
|
|
}
|
|
return buf
|
|
}
|
|
|
|
// One 32-thread threadgroup per base nonce, all dispatched from one encoder. Warp i lands at byte offset
|
|
// i * 256 of the output buffer, which is pre-filled with a sentinel so an unwritten lane is visible.
|
|
func gpuWarps(_ gpu: GPU, _ k: CompiledHash, dataset: MTLBuffer, bases: [UInt32]) -> [[UInt64]]? {
|
|
guard !bases.isEmpty, let outBuf = gpu.device.makeBuffer(length: bases.count * 256, options: .storageModeShared) else { return nil }
|
|
memset(outBuf.contents(), 0xAA, outBuf.length)
|
|
let cb = gpu.queue.makeCommandBuffer()!
|
|
let enc = cb.makeComputeCommandEncoder()!
|
|
enc.setComputePipelineState(k.pipeline)
|
|
enc.setBuffer(dataset, offset: 0, index: 0)
|
|
for (i, base) in bases.enumerated() {
|
|
enc.setBuffer(outBuf, offset: i * 256, index: 1)
|
|
var b = base
|
|
enc.setBytes(&b, length: 4, index: 2)
|
|
enc.dispatchThreadgroups(MTLSize(width: 1, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 32, height: 1, depth: 1))
|
|
}
|
|
enc.endEncoding()
|
|
cb.commit(); cb.waitUntilCompleted()
|
|
if cb.error != nil { return nil }
|
|
let p = outBuf.contents().bindMemory(to: UInt64.self, capacity: bases.count * 32)
|
|
return (0..<bases.count).map { i in (0..<32).map { p[i * 32 + $0] } }
|
|
}
|
|
|
|
// `count` consecutive nonces from `base` (count a multiple of 32) into `out`. Returns GPU time in ms.
|
|
func gpuRange(_ gpu: GPU, _ k: CompiledHash, dataset: MTLBuffer, base: UInt32, count: Int, out: MTLBuffer) -> Double? {
|
|
let cb = gpu.queue.makeCommandBuffer()!
|
|
let enc = cb.makeComputeCommandEncoder()!
|
|
enc.setComputePipelineState(k.pipeline)
|
|
enc.setBuffer(dataset, offset: 0, index: 0)
|
|
enc.setBuffer(out, offset: 0, index: 1)
|
|
var b = base
|
|
enc.setBytes(&b, length: 4, index: 2)
|
|
enc.dispatchThreadgroups(MTLSize(width: count / 32, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 32, height: 1, depth: 1))
|
|
enc.endEncoding()
|
|
cb.commit(); cb.waitUntilCompleted()
|
|
if cb.error != nil { return nil }
|
|
return (cb.gpuEndTime - cb.gpuStartTime) * 1000
|
|
}
|
|
|
|
func h64(_ v: UInt64) -> String { String(format: "%016llx", v) }
|
|
func pad(_ s: String, _ n: Int) -> String { s.count >= n ? s : s + String(repeating: " ", count: n - s.count) }
|
|
|
|
func describeProgram(_ p: Program) -> String {
|
|
var s = " program seed \"\(p.seedString)\" words [\(p.seed.map(hex).joined(separator: ", "))], \(p.instrs.count) instructions\n"
|
|
for (k, i) in p.instrs.enumerated() {
|
|
s += " \(pad(String(k), 3)) \(pad(i.op.rawValue, 5)) dst=r\(i.dst) src=r\(i.a) src2=r\(i.b) imm=\(hex(i.imm)) imm2=\(hex(i.imm2)) rot=\(i.rot) bit=\(i.bit) mask=\(i.mask)\n"
|
|
}
|
|
return s
|
|
}
|
|
|
|
func fnv64(_ ptr: UnsafeRawPointer, _ count: Int) -> UInt64 {
|
|
var h: UInt64 = 0xcbf29ce484222325
|
|
let b = ptr.bindMemory(to: UInt8.self, capacity: count)
|
|
for i in 0..<count { h ^= UInt64(b[i]); h &*= 0x100000001b3 }
|
|
return h
|
|
}
|
|
|
|
func regexCount(_ pattern: String, in text: String) -> Int {
|
|
let re = try! NSRegularExpression(pattern: pattern)
|
|
return re.numberOfMatches(in: text, range: NSRange(text.startIndex..., in: text))
|
|
}
|
|
|
|
// Static check: every dataset access in the generated MSL is `dataset[rN & MASK]`, and the identifier
|
|
// `dataset` appears nowhere else except the kernel parameter.
|
|
func maskCheckMSL(_ msl: String) -> (ok: Bool, detail: String) {
|
|
let total = regexCount("dataset\\[", in: msl)
|
|
let masked = regexCount("dataset\\[r[0-7] & MASK\\]", in: msl)
|
|
let words = regexCount("\\bdataset\\b", in: msl)
|
|
let ok = total == masked && words == total + 1
|
|
return (ok, "MSL: \(total) dataset[ accesses, \(masked) of the form dataset[rN & MASK], identifier appears \(words) times (expected \(total + 1))")
|
|
}
|
|
|
|
// Same for the CUDA twin: hash accesses are `ds[rN & mask]`; the fill kernel's one write is guarded by `if (i < n)`.
|
|
func maskCheckCUDA(_ cu: String) -> (ok: Bool, detail: String) {
|
|
let total = regexCount("\\bds\\[", in: cu)
|
|
let masked = regexCount("\\bds\\[r[0-7] & mask\\]", in: cu)
|
|
let fill = regexCount("if \\(i < n\\) ds\\[i\\] = ds_elem", in: cu)
|
|
let ok = total == masked + fill && fill == 1
|
|
return (ok, "CUDA: \(total) ds[ accesses, \(masked) of the form ds[rN & mask], \(fill) guarded fill write")
|
|
}
|
|
|
|
// MARK: - --fuzz
|
|
|
|
func runFuzz(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
|
|
let n = max(opts.fuzz ?? 200, 1)
|
|
let master = opts.fuzzSeed
|
|
print("\n=== fuzz: \(n) random programs, master seed \"\(master)\", 4 random warps each ===")
|
|
let sizes = [24, 26, 28]
|
|
let d0 = nowNs()
|
|
var datasets = [Int: MTLBuffer]()
|
|
for s in sizes { datasets[s] = makeDataset(gpu, log2: s, day: day) }
|
|
print("datasets " + sizes.map { "2^\($0) (\((1 << $0) * 4 / (1 << 20)) MiB)" }.joined(separator: ", ") + " filled in \(fmt(ms(d0, nowNs()), 1)) ms")
|
|
|
|
let mw = seedWords("fuzz/" + master)
|
|
var rng = SplitMix64(s: UInt64(mw[0]) | (UInt64(mw[1]) << 32))
|
|
var pass = 0, fail = 0, compileFail = 0, staticFail = 0, contractFail = 0
|
|
var perSize = [Int: (pass: Int, fail: Int)]()
|
|
var compileMs = [Double]()
|
|
var cpuNs: UInt64 = 0, gpuNs: UInt64 = 0
|
|
var warps = 0
|
|
var opCount = [String: Int]()
|
|
var loadsMin = Int.max, loadsMax = 0
|
|
let t0 = nowNs()
|
|
for i in 0..<n {
|
|
let seedString = "\(master)/\(i)/\(h64(rng.next()))"
|
|
let log2 = sizes[rng.below(sizes.count)]
|
|
let bases = (0..<4).map { _ in UInt32(truncatingIfNeeded: rng.next()) }
|
|
let program = generateProgram(seedString: seedString)
|
|
for ins in program.instrs {
|
|
opCount[ins.op.rawValue, default: 0] += 1
|
|
// Generator contract, relied on by the MSL emitter: rotl amount 1..31, shuffle mask a power of two <= 16,
|
|
// source register never the destination.
|
|
if ins.rot < 1 || ins.rot > 31 || ![1, 2, 4, 8, 16].contains(ins.mask) || ins.a == ins.dst || ins.dst > 7 || ins.a > 7 || ins.b > 7 {
|
|
contractFail += 1
|
|
print("CONTRACT FAIL seed \"\(seedString)\": \(ins)")
|
|
}
|
|
}
|
|
loadsMin = min(loadsMin, program.loadsPerHash); loadsMax = max(loadsMax, program.loadsPerHash)
|
|
let msl = generateMSL(program, datasetLog2: log2)
|
|
let sc = maskCheckMSL(msl)
|
|
if !sc.ok { staticFail += 1; print("STATIC MASK FAIL seed \"\(seedString)\": \(sc.detail)") }
|
|
let k: CompiledHash
|
|
do { k = try compileHash(gpu, msl: msl) } catch {
|
|
compileFail += 1
|
|
print("COMPILE FAIL seed \"\(seedString)\" dataset 2^\(log2):\n\(error)\n\(describeProgram(program))")
|
|
continue
|
|
}
|
|
compileMs.append(k.totalMs)
|
|
let g0 = nowNs()
|
|
guard let gpuOut = gpuWarps(gpu, k, dataset: datasets[log2]!, bases: bases) else {
|
|
fail += 1; print("GPU RUN FAIL seed \"\(seedString)\" dataset 2^\(log2)"); continue
|
|
}
|
|
let g1 = nowNs()
|
|
let mask = UInt32((1 << log2) - 1)
|
|
var ok = true
|
|
for (w, base) in bases.enumerated() {
|
|
let cpu = cpuWarp(program, baseNonce: base, day: day, mask: mask)
|
|
warps += 1
|
|
if cpu != gpuOut[w] {
|
|
ok = false
|
|
let bad = (0..<32).filter { cpu[$0] != gpuOut[w][$0] }
|
|
print("MISMATCH seed \"\(seedString)\" dataset 2^\(log2) warp \(w) base nonce \(base) (\(hex(base))) lanes \(bad)")
|
|
for l in bad { print(" lane \(l) nonce \(base &+ UInt32(l)): gpu \(h64(gpuOut[w][l])) cpu \(h64(cpu[l]))") }
|
|
print(describeProgram(program))
|
|
}
|
|
}
|
|
let g2 = nowNs()
|
|
gpuNs += g1 - g0; cpuNs += g2 - g1
|
|
if ok { pass += 1 } else { fail += 1 }
|
|
var ps = perSize[log2] ?? (0, 0)
|
|
if ok { ps.pass += 1 } else { ps.fail += 1 }
|
|
perSize[log2] = ps
|
|
if (i + 1) % 100 == 0 || i + 1 == n {
|
|
print(" \(i + 1)/\(n): pass \(pass) fail \(fail) compile-fail \(compileFail), \(fmt(Double(nowNs() - t0) / 1e9, 1)) s elapsed")
|
|
}
|
|
}
|
|
let total = Double(nowNs() - t0) / 1e9
|
|
let cAvg = compileMs.isEmpty ? 0 : compileMs.reduce(0, +) / Double(compileMs.count)
|
|
print("\n| Dataset | Programs | Pass | Fail |")
|
|
print("|---|---|---|---|")
|
|
for s in sizes {
|
|
let ps = perSize[s] ?? (0, 0)
|
|
print("| 2^\(s) words (\((1 << s) * 4 / (1 << 20)) MiB) | \(ps.pass + ps.fail) | \(ps.pass) | \(ps.fail) |")
|
|
}
|
|
print("| all | \(pass + fail) | \(pass) | \(fail) |")
|
|
print("programs \(n): pass \(pass), mismatch \(fail), compile failures \(compileFail), static mask failures \(staticFail), generator contract failures \(contractFail)")
|
|
print("warps compared \(warps) (\(warps * 32) hashes), loads/hash range \(loadsMin)..\(loadsMax)")
|
|
print("op totals over all programs: " + opCount.sorted { $0.value != $1.value ? $0.value > $1.value : $0.key < $1.key }.map { "\($0.key)=\($0.value)" }.joined(separator: " "))
|
|
print("compile ms (library+pipeline): min \(fmt(compileMs.min() ?? 0, 1)) avg \(fmt(cAvg, 1)) max \(fmt(compileMs.max() ?? 0, 1)); GPU dispatch total \(fmt(Double(gpuNs) / 1e6, 1)) ms; CPU interpreter total \(fmt(Double(cpuNs) / 1e6, 1)) ms; wall \(fmt(total, 1)) s")
|
|
let ok = fail == 0 && compileFail == 0 && staticFail == 0 && contractFail == 0 && pass == n
|
|
print("FUZZ: \(ok ? "PASS" : "FAIL")")
|
|
return ok
|
|
}
|
|
|
|
// MARK: - --edge
|
|
|
|
struct EdgeCase {
|
|
let name: String
|
|
let instrs: [Instr]
|
|
// (instruction index, what must hold, check on lane-0 registers as they are just before that instruction)
|
|
let pre: [(Int, String, ([UInt32]) -> Bool)]
|
|
let informational: Bool // reported but not counted: exercises something the generator never emits
|
|
}
|
|
|
|
// Instruction builder for hand-made programs. imm2 = imm so the `add` is a constant regardless of the selector bit.
|
|
func I(_ op: Op, _ dst: Int, _ a: Int, b: Int = 0, imm: UInt32 = 0, rot: UInt32 = 1, bit: Int = 0, mask: Int = 1) -> Instr {
|
|
Instr(op: op, dst: dst, a: a, b: b, imm: imm, imm2: imm, rot: rot, bit: bit, mask: mask)
|
|
}
|
|
func zero(_ r: Int) -> Instr { I(.sub, r, r) } // r = r - r = 0 (src == dst, never generated, legal MSL)
|
|
func set(_ r: Int, _ v: UInt32) -> [Instr] { [zero(r), I(.add, r, 7, imm: v)] } // needs r7 == 0
|
|
|
|
func edgeCases(mask: UInt32) -> [EdgeCase] {
|
|
let M = mask
|
|
var c = [EdgeCase]()
|
|
c.append(EdgeCase(name: "rotl immediate by 1 and by 31",
|
|
instrs: [I(.rotl, 1, 0, rot: 1), I(.rotl, 2, 0, rot: 31), I(.xor, 3, 1), I(.xor, 4, 2), I(.rotl, 5, 0, rot: 1), I(.rotl, 6, 0, rot: 31)],
|
|
pre: [], informational: false))
|
|
c.append(EdgeCase(name: "rotr by register == 0",
|
|
instrs: [zero(7), I(.rotr, 3, 7), I(.xor, 4, 3)],
|
|
pre: [(1, "r7 == 0", { $0[7] == 0 })], informational: false))
|
|
c.append(EdgeCase(name: "rotr by register == 32 (32 mod 32 = 0)",
|
|
instrs: [zero(7)] + (set(1, 32) + [I(.rotr, 3, 1), I(.xor, 4, 3)]),
|
|
pre: [(3, "r1 == 32", { $0[1] == 32 })], informational: false))
|
|
c.append(EdgeCase(name: "rotr by register == 0xFFFFFFE0 (-32, 0 mod 32)",
|
|
instrs: [zero(7)] + (set(1, 0xFFFFFFE0) + [I(.rotr, 3, 1), I(.xor, 4, 3)]),
|
|
pre: [(3, "r1 == 0xFFFFFFE0", { $0[1] == 0xFFFFFFE0 })], informational: false))
|
|
var rr: [Instr] = [zero(7)]
|
|
rr += set(1, 31); rr += [I(.rotr, 3, 1), I(.add, 1, 7, imm: 32), I(.rotr, 4, 1)]
|
|
rr += set(2, 1); rr += [I(.rotr, 5, 2), I(.xor, 6, 5)]
|
|
c.append(EdgeCase(name: "rotr by register == 31 and == 63 and == 1",
|
|
instrs: rr,
|
|
pre: [(3, "r1 == 31", { $0[1] == 31 }), (5, "r1 == 63", { $0[1] == 63 }), (8, "r2 == 1", { $0[2] == 1 })], informational: false))
|
|
var mh: [Instr] = [zero(7)]
|
|
mh += set(1, 0xFFFFFFFF); mh += set(2, 0xFFFFFFFF)
|
|
mh += [I(.mulhi, 1, 2), I(.xor, 3, 1)]
|
|
mh += set(4, 0x80000000); mh += set(5, 2)
|
|
mh += [I(.mulhi, 4, 5), I(.xor, 3, 4), I(.mulhi, 6, 7), I(.xor, 0, 6)]
|
|
c.append(EdgeCase(name: "mulhi 0xFFFFFFFF x 0xFFFFFFFF, 0x80000000 x 2, x 0",
|
|
instrs: mh,
|
|
pre: [(5, "r1 == r2 == 0xFFFFFFFF", { $0[1] == 0xFFFFFFFF && $0[2] == 0xFFFFFFFF }),
|
|
(6, "mulhi result r1 == 0xFFFFFFFE", { $0[1] == 0xFFFFFFFE }),
|
|
(11, "r4 == 0x80000000, r5 == 2", { $0[4] == 0x80000000 && $0[5] == 2 }),
|
|
(12, "mulhi result r4 == 1", { $0[4] == 1 }),
|
|
(13, "r7 == 0", { $0[7] == 0 }),
|
|
(14, "mulhi by 0 gives r6 == 0", { $0[6] == 0 })], informational: false))
|
|
c.append(EdgeCase(name: "shfl_xor every mask 1..16 in sequence (generator uses only 1,2,4,8,16)",
|
|
instrs: (1...16).map { I(.shfl, $0 % 8, ($0 + 1) % 8, mask: $0) },
|
|
pre: [], informational: false))
|
|
var l0: [Instr] = [zero(7), I(.load, 3, 7)]
|
|
l0 += set(1, M &+ 1); l0 += [I(.load, 4, 1), I(.xor, 5, 4)]
|
|
c.append(EdgeCase(name: "load at index 0 (register 0, and register MASK+1 which masks to 0)",
|
|
instrs: l0,
|
|
pre: [(1, "r7 & MASK == 0", { $0[7] & M == 0 }),
|
|
(4, "r1 == MASK+1, so unmasked index is out of range and masked index is 0", { $0[1] == M &+ 1 && ($0[1] & M) == 0 })],
|
|
informational: false))
|
|
var lm: [Instr] = [zero(7)]
|
|
lm += set(1, M); lm += [I(.load, 3, 1)]
|
|
lm += set(2, 0xFFFFFFFF); lm += [I(.load, 4, 2), I(.xor, 5, 4)]
|
|
c.append(EdgeCase(name: "load at index MASK (register MASK, and register 0xFFFFFFFF which masks to MASK)",
|
|
instrs: lm,
|
|
pre: [(3, "r1 == MASK", { $0[1] == M }),
|
|
(6, "r2 == 0xFFFFFFFF, masked index == MASK", { $0[2] == 0xFFFFFFFF && ($0[2] & M) == M })],
|
|
informational: false))
|
|
c.append(EdgeCase(name: "add wraparound 0xFFFFFFFF + 1",
|
|
instrs: [zero(7)] + (set(1, 0xFFFFFFFF) + [I(.add, 1, 7, imm: 1), I(.xor, 2, 1)]),
|
|
pre: [(3, "r1 == 0xFFFFFFFF", { $0[1] == 0xFFFFFFFF }), (4, "r1 == 0 after add", { $0[1] == 0 })], informational: false))
|
|
c.append(EdgeCase(name: "sub wraparound 0 - 1",
|
|
instrs: [zero(7), zero(1)] + (set(2, 1) + [I(.sub, 1, 2), I(.xor, 3, 1)]),
|
|
pre: [(4, "r1 == 0, r2 == 1", { $0[1] == 0 && $0[2] == 1 }), (5, "r1 == 0xFFFFFFFF after sub", { $0[1] == 0xFFFFFFFF })], informational: false))
|
|
var mm: [Instr] = [zero(7)]
|
|
mm += set(1, 0xFFFFFFFF); mm += set(2, 0xFFFFFFFF)
|
|
mm += [I(.mul, 1, 2), I(.xor, 3, 1)]
|
|
mm += set(4, 0xFFFFFFFF); mm += set(5, 0xFFFFFFFF); mm += set(6, 5)
|
|
mm += [I(.mad, 6, 4, b: 5), I(.xor, 0, 6)]
|
|
c.append(EdgeCase(name: "mul and mad wraparound 0xFFFFFFFF x 0xFFFFFFFF",
|
|
instrs: mm,
|
|
pre: [(5, "r1 == r2 == 0xFFFFFFFF", { $0[1] == 0xFFFFFFFF && $0[2] == 0xFFFFFFFF }),
|
|
(6, "mul low result r1 == 1", { $0[1] == 1 }),
|
|
(13, "r4 == r5 == 0xFFFFFFFF, r6 == 5", { $0[4] == 0xFFFFFFFF && $0[5] == 0xFFFFFFFF && $0[6] == 5 }),
|
|
(14, "mad result r6 == 6", { $0[6] == 6 })], informational: false))
|
|
let z = generateProgram(seedString: "edge/zero-loads")
|
|
c.append(EdgeCase(name: "generated program with every load replaced by xor (zero loads)",
|
|
instrs: z.instrs.map { ins in var m = ins; if m.op == .load { m.op = .xor }; return m },
|
|
pre: [], informational: false))
|
|
c.append(EdgeCase(name: "64 loads and nothing else",
|
|
instrs: (0..<64).map { I(.load, $0 % 8, ($0 + 3) % 8) },
|
|
pre: [], informational: false))
|
|
c.append(EdgeCase(name: "rotl immediate by 0 (outside the generator's 1..31 contract; MSL shifts by 32)",
|
|
instrs: [I(.rotl, 1, 0, rot: 0), I(.xor, 2, 1)],
|
|
pre: [], informational: true))
|
|
return c
|
|
}
|
|
|
|
func runEdge(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
|
|
let log2 = opts.datasetLog2
|
|
let mask = UInt32((1 << log2) - 1)
|
|
print("\n=== edge cases, dataset 2^\(log2) words, MASK \(hex(mask)) ===")
|
|
let dataset = makeDataset(gpu, log2: log2, day: day)
|
|
let bases: [UInt32] = [0, 1 << 20, 0x7FFFFFF0, 0xFFFFFFE0]
|
|
print("warps: base nonces " + bases.map { hex($0) }.joined(separator: ", ") + " (the last two straddle 2^31 and wrap past 2^32)")
|
|
var allOk = true
|
|
var rows = [String]()
|
|
for ec in edgeCases(mask: mask) {
|
|
let program = Program(seedString: "edge/\(ec.name)", seed: seedWords("edge/\(ec.name)"), instrs: ec.instrs)
|
|
let msl = generateMSL(program, datasetLog2: log2)
|
|
var status = "", detail = ""
|
|
var ok = true
|
|
// Preconditions, checked on lane 0 of every warp in every iteration.
|
|
var preOk = true
|
|
var preNotes = [String]()
|
|
if !ec.pre.isEmpty {
|
|
for base in bases {
|
|
var hits = [Int: Int]()
|
|
var misses = [Int: Int]()
|
|
_ = cpuWarpTraced(program, baseNonce: base, day: day, mask: mask) { _, k, regs in
|
|
for (idx, _, check) in ec.pre where idx == k {
|
|
if check(regs) { hits[idx, default: 0] += 1 } else { misses[idx, default: 0] += 1 }
|
|
}
|
|
}
|
|
for (idx, what, _) in ec.pre {
|
|
if (misses[idx] ?? 0) > 0 || (hits[idx] ?? 0) != Program.iterations {
|
|
preOk = false
|
|
preNotes.append("base \(hex(base)) instr \(idx) '\(what)' held \(hits[idx] ?? 0)/\(Program.iterations) iterations")
|
|
}
|
|
}
|
|
}
|
|
if preOk { preNotes = ec.pre.map { "instr \($0.0): \($0.1)" } }
|
|
}
|
|
do {
|
|
let k = try compileHash(gpu, msl: msl)
|
|
guard let gpuOut = gpuWarps(gpu, k, dataset: dataset, bases: bases) else { throw IgneumError("GPU run failed") }
|
|
var badLanes = 0
|
|
var first = ""
|
|
for (w, base) in bases.enumerated() {
|
|
let cpu = cpuWarp(program, baseNonce: base, day: day, mask: mask)
|
|
for l in 0..<32 where cpu[l] != gpuOut[w][l] {
|
|
badLanes += 1
|
|
if first.isEmpty { first = "first: base \(hex(base)) lane \(l) gpu \(h64(gpuOut[w][l])) cpu \(h64(cpu[l]))" }
|
|
}
|
|
}
|
|
ok = badLanes == 0 && preOk
|
|
status = badLanes == 0 ? "GPU == CPU 128/128 lanes" : "MISMATCH \(badLanes)/128 lanes, \(first)"
|
|
detail = "compile \(fmt(k.totalMs, 1)) ms"
|
|
if badLanes > 0 { print(describeProgram(program)) }
|
|
} catch {
|
|
ok = false
|
|
status = "COMPILE FAIL: \(error)"
|
|
}
|
|
let pre = ec.pre.isEmpty ? "none needed" : (preOk ? "held (all 8 iterations, lane 0, 4 warps)" : "NOT HELD")
|
|
let verdict = ec.informational ? (ok ? "info: agrees" : "info: differs") : (ok ? "PASS" : "FAIL")
|
|
if !ec.informational && !ok { allOk = false }
|
|
print("\(verdict): \(ec.name)")
|
|
print(" \(ec.instrs.count) instructions, loads/hash \(program.loadsPerHash), \(status), \(detail)")
|
|
for n in preNotes { print(" precondition \(n)") }
|
|
rows.append("| \(ec.name) | \(ec.instrs.count) | \(program.loadsPerHash) | \(pre) | \(status) | \(verdict) |")
|
|
}
|
|
print("\n| Case | Instrs | Loads/hash | Preconditions | GPU vs CPU | Result |")
|
|
print("|---|---|---|---|---|---|")
|
|
for r in rows { print(r) }
|
|
print("EDGE: \(allOk ? "PASS" : "FAIL")")
|
|
return allOk
|
|
}
|
|
|
|
// MARK: - --stats
|
|
|
|
func popcount64(_ v: UInt64) -> Int { v.nonzeroBitCount }
|
|
|
|
func runStats(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
|
|
let log2 = opts.datasetLog2
|
|
let mask = UInt32((1 << log2) - 1)
|
|
let n = 1 << 20
|
|
print("\n=== output statistics, 2^20 consecutive nonces per seed, dataset 2^\(log2) words ===")
|
|
print("This is a sanity check for obvious structural bias. It is not a proof of cryptographic strength.")
|
|
let dataset = makeDataset(gpu, log2: log2, day: day)
|
|
guard let outBuf = gpu.device.makeBuffer(length: n * 8, options: .storageModeShared) else { print("FAIL: out buffer"); return false }
|
|
let seeds = [opts.seed, "\(opts.seed)/stats1", "\(opts.seed)/stats2"]
|
|
var allOk = true
|
|
var rows = [String]()
|
|
for seedString in seeds {
|
|
let program = generateProgram(seedString: seedString)
|
|
let k: CompiledHash
|
|
do { k = try compileHash(gpu, msl: generateMSL(program, datasetLog2: log2)) } catch { print("FAIL: compile \(error)"); return false }
|
|
memset(outBuf.contents(), 0, n * 8)
|
|
guard let gms = gpuRange(gpu, k, dataset: dataset, base: 0, count: n, out: outBuf) else { print("FAIL: GPU run"); return false }
|
|
let p = outBuf.contents().bindMemory(to: UInt64.self, capacity: n)
|
|
let outs = (0..<n).map { p[$0] }
|
|
// Spot check 2 warps against the CPU so the statistics are known to describe the verified function.
|
|
var spot = true
|
|
for w in [0, (n / 32) - 1] {
|
|
let cpu = cpuWarp(program, baseNonce: UInt32(w * 32), day: day, mask: mask)
|
|
if cpu != Array(outs[(w * 32)..<(w * 32 + 32)]) { spot = false }
|
|
}
|
|
|
|
// (a) bit frequency per output bit position
|
|
var ones = [Int](repeating: 0, count: 64)
|
|
for v in outs { var x = v; var b = 0; while x != 0 { if x & 1 == 1 { ones[b] += 1 }; x >>= 1; b += 1 } }
|
|
let expected = Double(n) / 2, sigma = (Double(n) * 0.25).squareRoot()
|
|
var maxDev = 0.0, maxBit = 0
|
|
for b in 0..<64 { let d = abs(Double(ones[b]) - expected); if d > maxDev { maxDev = d; maxBit = b } }
|
|
let maxZ = maxDev / sigma
|
|
let minFreq = Double(ones.min()!) / Double(n), maxFreq = Double(ones.max()!) / Double(n)
|
|
|
|
// (c) chi-square over 65536 buckets for each 16-bit window of the output
|
|
var chiRows = [String]()
|
|
var chiWorstZ = 0.0
|
|
for shift in [0, 16, 32, 48] {
|
|
var buckets = [Int](repeating: 0, count: 65536)
|
|
for v in outs { buckets[Int((v >> UInt64(shift)) & 0xFFFF)] += 1 }
|
|
let e = Double(n) / 65536
|
|
var chi = 0.0
|
|
for c in buckets { let d = Double(c) - e; chi += d * d / e }
|
|
let df = 65535.0
|
|
let z = (chi - df) / (2 * df).squareRoot()
|
|
chiWorstZ = max(chiWorstZ, abs(z))
|
|
chiRows.append("bits \(shift)..\(shift + 15): chi2 \(fmt(chi, 0)) (df 65535, z \(fmt(z, 2)))")
|
|
}
|
|
|
|
// (d) duplicates
|
|
let sorted = outs.sorted()
|
|
var dups = 0
|
|
for i in 1..<n where sorted[i] == sorted[i - 1] { dups += 1 }
|
|
|
|
// (b) avalanche: 1000 random nonces, flip bit (i mod 32), count changed output bits. Run on the GPU.
|
|
let sw = seedWords("avalanche/" + seedString)
|
|
var rng = SplitMix64(s: UInt64(sw[0]) | (UInt64(sw[1]) << 32))
|
|
let trials = 1000
|
|
var bases = [UInt32]()
|
|
var flipped = [Int]()
|
|
for t in 0..<trials {
|
|
let nonce = UInt32(truncatingIfNeeded: rng.next())
|
|
let bit = t % 32
|
|
bases.append(nonce); bases.append(nonce ^ (1 << UInt32(bit)))
|
|
flipped.append(bit)
|
|
}
|
|
guard let av = gpuWarps(gpu, k, dataset: dataset, bases: bases) else { print("FAIL: avalanche GPU run"); return false }
|
|
var diffs = [Int]()
|
|
var perBitSum = [Int](repeating: 0, count: 32), perBitN = [Int](repeating: 0, count: 32)
|
|
var minDiff = 64, maxDiff = 0
|
|
for t in 0..<trials {
|
|
let d = popcount64(av[2 * t][0] ^ av[2 * t + 1][0])
|
|
diffs.append(d)
|
|
perBitSum[flipped[t]] += d; perBitN[flipped[t]] += 1
|
|
minDiff = min(minDiff, d); maxDiff = max(maxDiff, d)
|
|
}
|
|
let mean = Double(diffs.reduce(0, +)) / Double(trials)
|
|
let variance = diffs.reduce(0.0) { $0 + (Double($1) - mean) * (Double($1) - mean) } / Double(trials - 1)
|
|
let std = variance.squareRoot()
|
|
var perBitMin = 64.0, perBitMax = 0.0
|
|
for b in 0..<32 where perBitN[b] > 0 { let m = Double(perBitSum[b]) / Double(perBitN[b]); perBitMin = min(perBitMin, m); perBitMax = max(perBitMax, m) }
|
|
// Expected for an ideal function: mean 32, std 4 (binomial 64 x 0.5). Standard error of the mean over 1000 trials is 0.13.
|
|
let avOk = abs(mean - 32) < 0.6 && std > 3.3 && std < 4.7
|
|
let freqOk = maxZ < 4.5
|
|
let chiOk = chiWorstZ < 4.5
|
|
let dupOk = dups == 0
|
|
let ok = avOk && freqOk && chiOk && dupOk && spot
|
|
if !ok { allOk = false }
|
|
|
|
print("\nseed \"\(seedString)\": loads/hash \(program.loadsPerHash), GPU \(fmt(gms, 1)) ms for 2^20 hashes, CPU spot check 2 warps \(spot ? "PASS" : "FAIL")")
|
|
print(" (a) bit frequency: min \(fmt(minFreq, 4)) max \(fmt(maxFreq, 4)); largest deviation \(fmt(maxDev, 0)) counts at bit \(maxBit) = \(fmt(maxZ, 2)) sigma (sigma \(fmt(sigma, 0)), 64 bits, expect max under about 3.5)")
|
|
print(" (b) avalanche over \(trials) single-bit nonce flips: mean \(fmt(mean, 2)) std \(fmt(std, 2)) min \(minDiff) max \(maxDiff) of 64 bits (expect mean 32, std 4); per-input-bit mean range \(fmt(perBitMin, 1))..\(fmt(perBitMax, 1))")
|
|
for r in chiRows { print(" (c) \(r)") }
|
|
print(" (d) duplicate 64-bit outputs among 2^20: \(dups) (expected about 3e-8)")
|
|
print(" verdict: \(ok ? "no obvious bias" : "SUSPECT")")
|
|
rows.append("| \(seedString) | \(program.loadsPerHash) | \(fmt(minFreq, 4))..\(fmt(maxFreq, 4)) | \(fmt(maxZ, 2)) | \(fmt(mean, 2)) | \(fmt(std, 2)) | \(fmt(chiWorstZ, 2)) | \(dups) | \(ok ? "uniform-looking" : "SUSPECT") |")
|
|
}
|
|
print("\n| Seed | Loads/hash | Bit freq min..max | Max bit z | Avalanche mean | Avalanche std | Worst chi2 z (4 windows) | Dups | Verdict |")
|
|
print("|---|---|---|---|---|---|---|---|---|")
|
|
for r in rows { print(r) }
|
|
print("STATS: \(allOk ? "PASS (no obvious structural bias; not a security proof)" : "FAIL (something looks biased)")")
|
|
return allOk
|
|
}
|
|
|
|
// MARK: - --determinism
|
|
|
|
func runDeterminism(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
|
|
let log2 = opts.datasetLog2
|
|
let mask = UInt32((1 << log2) - 1)
|
|
let n = 1 << 20
|
|
print("\n=== determinism, seed \"\(opts.seed)\", 2^20 nonces from base 0, dataset 2^\(log2) words ===")
|
|
let dataset = makeDataset(gpu, log2: log2, day: day)
|
|
guard let outBuf = gpu.device.makeBuffer(length: n * 8, options: .storageModeShared) else { print("FAIL: out buffer"); return false }
|
|
var ok = true
|
|
|
|
// Generator and emitter determinism: two independent generations give identical MSL text.
|
|
let p1 = generateProgram(seedString: opts.seed), p2 = generateProgram(seedString: opts.seed)
|
|
let msl1 = generateMSL(p1, datasetLog2: log2), msl2 = generateMSL(p2, datasetLog2: log2)
|
|
let sameSource = msl1 == msl2
|
|
print("generator: two generations of the program give identical MSL source: \(sameSource ? "yes" : "NO") (\(msl1.utf8.count) bytes)")
|
|
if !sameSource { ok = false }
|
|
|
|
// Two separate compiles of the same source.
|
|
let k1: CompiledHash, k2: CompiledHash
|
|
do { k1 = try compileHash(gpu, msl: msl1); k2 = try compileHash(gpu, msl: msl2) } catch { print("FAIL: compile \(error)"); return false }
|
|
print("compiled twice: \(fmt(k1.totalMs, 1)) ms and \(fmt(k2.totalMs, 1)) ms")
|
|
|
|
func runOnce(_ k: CompiledHash) -> (fp: UInt64, sentinels: Int, ms: Double)? {
|
|
memset(outBuf.contents(), 0xAA, n * 8)
|
|
guard let gms = gpuRange(gpu, k, dataset: dataset, base: 0, count: n, out: outBuf) else { return nil }
|
|
let p = outBuf.contents().bindMemory(to: UInt64.self, capacity: n)
|
|
var s = 0
|
|
for i in 0..<n where p[i] == 0xAAAAAAAAAAAAAAAA { s += 1 }
|
|
return (fnv64(outBuf.contents(), n * 8), s, gms)
|
|
}
|
|
var reference = [UInt64]()
|
|
var fps = [UInt64]()
|
|
for run in 0..<5 {
|
|
guard let r = runOnce(k1) else { print("FAIL: GPU run \(run)"); return false }
|
|
let p = outBuf.contents().bindMemory(to: UInt64.self, capacity: n)
|
|
if run == 0 { reference = (0..<n).map { p[$0] } }
|
|
var differ = 0
|
|
for i in 0..<n where p[i] != reference[i] { differ += 1 }
|
|
fps.append(r.fp)
|
|
let same = differ == 0 && r.sentinels == 0
|
|
if !same { ok = false }
|
|
print("run \(run + 1)/5 (compile 1): fingerprint \(h64(r.fp)), \(differ) of \(n) outputs differ from run 1, \(r.sentinels) unwritten lanes, GPU \(fmt(r.ms, 1)) ms: \(same ? "identical" : "DIFFERENT")")
|
|
}
|
|
guard let r2 = runOnce(k2) else { print("FAIL: GPU run on compile 2"); return false }
|
|
let p = outBuf.contents().bindMemory(to: UInt64.self, capacity: n)
|
|
var differ2 = 0
|
|
for i in 0..<n where p[i] != reference[i] { differ2 += 1 }
|
|
if differ2 != 0 || r2.sentinels != 0 { ok = false }
|
|
print("run on compile 2: fingerprint \(h64(r2.fp)), \(differ2) outputs differ from compile 1 run 1, \(r2.sentinels) unwritten lanes: \(differ2 == 0 ? "identical" : "DIFFERENT")")
|
|
|
|
// CPU reference on the first and last warp and 6 others, so the fingerprint is tied to the verified function.
|
|
var cpuBad = 0
|
|
var vr = SplitMix64(s: 0x1234_5678_9abc_def0)
|
|
var warps = [0, n / 32 - 1]
|
|
while warps.count < 8 { warps.append(vr.below(n / 32)) }
|
|
for w in warps {
|
|
let cpu = cpuWarp(p1, baseNonce: UInt32(w * 32), day: day, mask: mask)
|
|
if cpu != Array(reference[(w * 32)..<(w * 32 + 32)]) { cpuBad += 1 }
|
|
}
|
|
if cpuBad != 0 { ok = false }
|
|
print("CPU interpreter on \(warps.count) warps of the reference run: \(cpuBad == 0 ? "all match" : "\(cpuBad) MISMATCH")")
|
|
|
|
// Dataset fill determinism and GPU-vs-CPU agreement of the dataset itself: fill a second buffer, blit both
|
|
// to shared memory, fingerprint, and compare sampled words (including 0 and MASK) with datasetElem.
|
|
let words = 1 << log2
|
|
let dataset2 = makeDataset(gpu, log2: log2, day: day)
|
|
var fillFps = [UInt64]()
|
|
var sampleBad = 0
|
|
if let shared = gpu.device.makeBuffer(length: words * 4, options: .storageModeShared) {
|
|
for (idx, ds) in [dataset, dataset2].enumerated() {
|
|
let cb = gpu.queue.makeCommandBuffer()!
|
|
let blit = cb.makeBlitCommandEncoder()!
|
|
blit.copy(from: ds, sourceOffset: 0, to: shared, destinationOffset: 0, size: words * 4)
|
|
blit.endEncoding()
|
|
cb.commit(); cb.waitUntilCompleted()
|
|
fillFps.append(fnv64(shared.contents(), words * 4))
|
|
if idx == 0 {
|
|
let dp = shared.contents().bindMemory(to: UInt32.self, capacity: words)
|
|
var sr = SplitMix64(s: 0xfeed_beef)
|
|
var idxs: [UInt32] = [0, 1, mask - 1, mask]
|
|
while idxs.count < 4096 { idxs.append(UInt32(sr.below(words))) }
|
|
for i in idxs where dp[Int(i)] != datasetElem(i, day.0, day.1) { sampleBad += 1 }
|
|
}
|
|
}
|
|
let sameFill = fillFps[0] == fillFps[1]
|
|
if !sameFill || sampleBad != 0 { ok = false }
|
|
print("dataset fill: two fills fingerprint \(h64(fillFps[0])) and \(h64(fillFps[1])): \(sameFill ? "identical" : "DIFFERENT"); 4096 sampled words (incl. 0, 1, MASK-1, MASK) vs CPU datasetElem: \(sampleBad == 0 ? "all match" : "\(sampleBad) MISMATCH")")
|
|
} else {
|
|
print("dataset fill check skipped: could not allocate a shared copy")
|
|
}
|
|
print("DETERMINISM: \(ok ? "PASS" : "FAIL")")
|
|
return ok
|
|
}
|
|
|
|
// MARK: - --memcheck
|
|
|
|
func runMemcheck(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
|
|
print("\n=== memcheck, seed \"\(opts.seed)\" ===")
|
|
var ok = true
|
|
let program = generateProgram(seedString: opts.seed)
|
|
// Static: every dataset index in the generated sources is masked. Checked at three dataset sizes because
|
|
// the MASK literal changes with size.
|
|
for log2 in [20, 24, 28] {
|
|
let msl = generateMSL(program, datasetLog2: log2)
|
|
let r = maskCheckMSL(msl)
|
|
if !r.ok { ok = false }
|
|
print("static 2^\(log2): \(r.ok ? "PASS" : "FAIL") \(r.detail)")
|
|
}
|
|
let cu = maskCheckCUDA(generateCUDA(program))
|
|
if !cu.ok { ok = false }
|
|
print("static CUDA twin: \(cu.ok ? "PASS" : "FAIL") \(cu.detail)")
|
|
print("program has \(program.instrs.filter { $0.op == .load }.count) load instructions (\(program.loadsPerHash) loads/hash)")
|
|
|
|
// Dynamic: a 4 MiB dataset with nonces at the top of the 32-bit range (they wrap to 0 inside the batch),
|
|
// one full batch of 2^20 nonces plus 4 warps verified against the CPU. Metal does not bounds-check device
|
|
// buffers, so "no crash" is weak evidence by itself; the static check above is the real guarantee.
|
|
let log2 = 20
|
|
let mask = UInt32((1 << log2) - 1)
|
|
let dataset = makeDataset(gpu, log2: log2, day: day)
|
|
let k: CompiledHash
|
|
do { k = try compileHash(gpu, msl: generateMSL(program, datasetLog2: log2)) } catch { print("FAIL: compile \(error)"); return false }
|
|
let n = 1 << 20
|
|
guard let outBuf = gpu.device.makeBuffer(length: n * 8, options: .storageModeShared) else { print("FAIL: out buffer"); return false }
|
|
for base: UInt32 in [0xFFF00000, 0xFFFFFFE0, 0x80000000, 0] {
|
|
if let gms = gpuRange(gpu, k, dataset: dataset, base: base, count: n, out: outBuf) {
|
|
print("dynamic 4 MiB: 2^20 nonces from base \(hex(base)) (last nonce \(hex(base &+ UInt32(n - 1)))): completed, GPU \(fmt(gms, 1)) ms")
|
|
} else { ok = false; print("dynamic 4 MiB: base \(hex(base)): GPU ERROR") }
|
|
}
|
|
let bases: [UInt32] = [0xFFFFFFE0, 0xFFFFFFFF, 0x80000000, 0xFFF00000]
|
|
guard let gpuOut = gpuWarps(gpu, k, dataset: dataset, bases: bases) else { print("FAIL: GPU warps"); return false }
|
|
var overMask = 0, loads = 0
|
|
for (w, base) in bases.enumerated() {
|
|
let cpu = cpuWarpTraced(program, baseNonce: base, day: day, mask: mask) { _, kk, regs in
|
|
let ins = program.instrs[kk]
|
|
if ins.op == .load { loads += 1; if regs[ins.a] > mask { overMask += 1 } }
|
|
}
|
|
let match = cpu == gpuOut[w]
|
|
if !match { ok = false }
|
|
print("verify warp base \(hex(base)): GPU vs CPU \(match ? "PASS" : "FAIL")")
|
|
}
|
|
print("in those 4 warps (lane 0, all iterations) \(overMask) of \(loads) load indices were above MASK before masking, so the mask was exercised")
|
|
print("MEMCHECK: \(ok ? "PASS" : "FAIL")")
|
|
return ok
|
|
}
|
|
|
|
// MARK: - Test dispatcher
|
|
|
|
func runTests(_ opts: Options) -> Never {
|
|
let gpu = GPU()
|
|
print("igneum-bench hardening tests")
|
|
print("GPU: \(gpu.device.name) (maxBufferLength \(gpu.device.maxBufferLength / (1 << 20)) MiB, unified memory \(gpu.device.hasUnifiedMemory)), day \"\(opts.day)\"")
|
|
let dayWords = seedWords("day/" + opts.day)
|
|
let day = (dayWords[0], dayWords[1])
|
|
var results = [(String, Bool)]()
|
|
let t0 = nowNs()
|
|
if opts.fuzz != nil { results.append(("fuzz", runFuzz(opts, gpu: gpu, day: day))) }
|
|
if opts.edge { results.append(("edge", runEdge(opts, gpu: gpu, day: day))) }
|
|
if opts.stats { results.append(("stats", runStats(opts, gpu: gpu, day: day))) }
|
|
if opts.determinism { results.append(("determinism", runDeterminism(opts, gpu: gpu, day: day))) }
|
|
if opts.memcheck { results.append(("memcheck", runMemcheck(opts, gpu: gpu, day: day))) }
|
|
print("\n=== tests summary (\(fmt(Double(nowNs() - t0) / 1e9, 1)) s) ===")
|
|
for (name, ok) in results { print("\(pad(name, 12)) \(ok ? "PASS" : "FAIL")") }
|
|
let all = results.allSatisfy { $0.1 }
|
|
print("OVERALL: \(all ? "PASS" : "FAIL")")
|
|
exit(all ? 0 : 1)
|
|
}
|
|
|
|
// MARK: - Main
|
|
|
|
let opts = parseArgs()
|
|
if opts.exportPack != nil { exportPack(opts) }
|
|
if opts.anyTest { runTests(opts) }
|
|
let gpu = GPU()
|
|
print("igneum-bench")
|
|
print("GPU: \(gpu.device.name) (maxBufferLength \(gpu.device.maxBufferLength / (1 << 20)) MiB, unified memory \(gpu.device.hasUnifiedMemory))")
|
|
print("dataset: 2^\(opts.datasetLog2) uint32 = \(fmt(Double(1 << opts.datasetLog2) * 4 / Double(1 << 20), 0)) MiB, day \"\(opts.day)\"")
|
|
|
|
let dayWords = seedWords("day/" + opts.day)
|
|
let day = (dayWords[0], dayWords[1])
|
|
let datasetWords = 1 << opts.datasetLog2
|
|
guard let dataset = gpu.device.makeBuffer(length: datasetWords * 4, options: .storageModePrivate) else {
|
|
print("FAIL: cannot allocate dataset buffer"); exit(1)
|
|
}
|
|
|
|
// Fill the dataset on the GPU, timed.
|
|
var fillMsWall = 0.0, fillMsGPU = 0.0
|
|
do {
|
|
let f0 = nowNs()
|
|
let lib = try gpu.device.makeLibrary(source: fillMSL, options: MTLCompileOptions())
|
|
let pipe = try gpu.device.makeComputePipelineState(function: lib.makeFunction(name: "igneum_fill")!)
|
|
let f1 = nowNs()
|
|
let cb = gpu.queue.makeCommandBuffer()!
|
|
let enc = cb.makeComputeCommandEncoder()!
|
|
enc.setComputePipelineState(pipe)
|
|
enc.setBuffer(dataset, offset: 0, index: 0)
|
|
var d = (day.0, day.1)
|
|
enc.setBytes(&d, length: 8, index: 1)
|
|
let tg = 256
|
|
enc.dispatchThreadgroups(MTLSize(width: datasetWords / tg, height: 1, depth: 1),
|
|
threadsPerThreadgroup: MTLSize(width: tg, height: 1, depth: 1))
|
|
enc.endEncoding()
|
|
let f2 = nowNs()
|
|
cb.commit(); cb.waitUntilCompleted()
|
|
let f3 = nowNs()
|
|
if let e = cb.error { print("FAIL: fill error \(e)"); exit(1) }
|
|
fillMsWall = ms(f2, f3)
|
|
fillMsGPU = (cb.gpuEndTime - cb.gpuStartTime) * 1000
|
|
let gib = Double(datasetWords * 4) / Double(1 << 30)
|
|
print("dataset fill: compile \(fmt(ms(f0, f1))) ms; fill \(fmt(fillMsWall)) ms wall, \(fmt(fillMsGPU)) ms GPU -> \(fmt(gib / (fillMsGPU / 1000))) GB/s write (GPU time)")
|
|
} catch {
|
|
print("FAIL: fill kernel: \(error)"); exit(1)
|
|
}
|
|
|
|
var results = [EpochResult]()
|
|
for epoch in 0..<max(opts.hours, 1) {
|
|
let seedString = epoch == 0 ? opts.seed : "\(opts.seed)/epoch\(epoch)"
|
|
results.append(runEpoch(gpu: gpu, opts: opts, seedString: seedString, dataset: dataset, day: day))
|
|
}
|
|
|
|
// Summary table
|
|
print("\n=== summary (\(gpu.device.name), dataset 2^\(opts.datasetLog2) words, batch 2^\(opts.batchLog2) x \(opts.batches)) ===")
|
|
print("| seed | compile ms (lib+pipe) | Mhash/s (wall) | GB/s useful (wall) | loads/hash | CPU verify ms/warp (avg) | verify |")
|
|
print("|---|---|---|---|---|---|---|")
|
|
for r in results {
|
|
let avg = r.verify.map { $0.repMs }.reduce(0, +) / Double(max(r.verify.count, 1))
|
|
print("| \(r.seed) | \(fmt(r.libraryMs + r.pipelineMs, 1)) | \(fmt(r.hashesPerSecWall / 1e6, 3)) | \(fmt(r.gbpsWall)) | \(r.loadsPerHash) | \(fmt(avg, 3)) | \(r.allPass ? "PASS" : "FAIL") (\(r.verify.count) warps) |")
|
|
}
|
|
let overall = results.allSatisfy { $0.allPass }
|
|
print("dataset fill: \(fmt(fillMsGPU)) ms GPU time for \(datasetWords * 4 / (1 << 20)) MiB")
|
|
print("OVERALL: \(overall ? "PASS" : "FAIL")")
|
|
exit(overall ? 0 : 1)
|