igneum/proto-metal/main.swift
igneum-labs 28635b165f Metal worker compiles a class v3 program from the pack a prepare line names; igneum-pow chain_dataset_day seam; pow_genesis_dataset_log2 in the override files
main.swift: servePackProgram reads program.h with the packfile.h checks (generator 2 or 3, the class line against
the generator, the seed bytes, IGNEUM_SEEDW_INIT against attempt_words, class and era against the line) and
compiles program_bound.metal; the program store keys on (seed, class, era); a v3 job with no resident v3 pack
program answers need + error; v2 lines unchanged (Swift generation, the variant race); a pack program never races.
verify.rs: Epoch::chain_dataset_day(day, class, days_since_genesis, genesis_dataset_log2) and days_since_genesis,
the entry the node builds every day cache through (the ca2-mixer growth rule fills the body).

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
2026-10-05 20:51:24 +00:00

3328 lines
182 KiB
Swift

// igneum-bench: first prototype of Igneum's random-program GPU proof-of-work.
// One file. Build: swiftc -O -o igneum-bench main.swift -framework Metal
// Metal shaders are compiled at runtime from generated source (no Xcode needed).
import Foundation
import Metal
// MARK: - Options
struct Options {
var seed = "igneum-genesis"
var day = "2026-10-03"
var hours = 2 // number of epochs (seeds) run in sequence; default 2 so verification covers 2 seeds
var batchLog2 = 22 // nonces per batch
var batches = 4 // timed batches
var datasetLog2 = 28 // 2^28 uint32 = 1 GiB
var verifyWarps = 3
var dumpDir: String? = nil
var exportPack: String? = nil // write a CUDA program pack for --seed into this directory and exit
var serve = false // --serve: GPU worker for igneum-miner (jobs on stdin, results on stdout), 3 October 2026
var noPrepare = false // --no-prepare: serve without the prepare command (ready line says "prepare 0"), to test the miner's fallback
// Hardening tests (added 3 October 2026). Any of these runs instead of the bench.
var fuzz: Int? = nil // --fuzz N: N random programs, GPU vs CPU on 4 random warps each
var fuzzSeed = "igneum-fuzz-2026-10-03"
var edge = false // --edge: hand-built edge-case programs
var stats = false // --stats: output distribution sanity checks on 2^20 nonces
var determinism = false // --determinism: 5 identical GPU runs + double compile
var memcheck = false // --memcheck: static mask check + 4 MiB run with wrapping nonces
// Shortcut measurement (bench variant, not a test): compute dataset elements inline instead of loading them.
// Closed-form mode: every load computes ds_elem. Memory-hard mode: every load derives the item from the cache.
var inlineDataset = false
// Dataset construction (added 3 October 2026). Default is the memory-hard cache construction (MEMHARD.md);
// --closed-form selects the original six-operation closed form so the two can be compared.
var closedForm = false
// Generator levers (MEMHARD.md section 7). Defaults reproduce the original generator exactly.
var loadWeight = 25 // --load-weight W: percent weight of the load op (default 25)
var wideFrac = 0 // --wide-frac P: percent of load instructions emitted as warp-coalesced wide loads
// Variant racing (4 October 2026, evening; docs/design/miner-tuning.md): in --serve every prepared program is
// compiled in several variants, each checked bit for bit against the base kernel and timed for about two
// seconds with the job loop paused; the fastest serves the hour. --race-test runs the race alone and prints it.
var race = "on" // --race on|off|a,b,c
var raceBenchMs = 2000 // --race-bench-ms
var raceBudgetS = 120 // --race-budget-s (the prepare lead is 600 DAA; base is kept when the budget runs out)
var raceRounds = 0 // --race-rounds (0 = 1 in --serve, 3 in --race-test)
var pinnedVariant: String? = nil // --variant: this one, no race
var tuningPath: String? = nil // --tuning <file> (default IGNEUM_TUNING_FILE)
var raceTest = false // --race-test: the race for --seed on --day, rounds, table, exit
var anyTest: Bool { fuzz != nil || edge || stats || determinism || memcheck }
}
func parseArgs() -> Options {
var o = Options()
var args = Array(CommandLine.arguments.dropFirst())
func take() -> String { args.isEmpty ? "" : args.removeFirst() }
while !args.isEmpty {
let a = take()
switch a {
case "--seed": o.seed = take()
case "--day": o.day = take()
case "--hours": o.hours = Int(take()) ?? o.hours
case "--batch-log2": o.batchLog2 = Int(take()) ?? o.batchLog2
case "--batches": o.batches = Int(take()) ?? o.batches
case "--dataset-log2": o.datasetLog2 = Int(take()) ?? o.datasetLog2
case "--verify-warps": o.verifyWarps = Int(take()) ?? o.verifyWarps
case "--dump": o.dumpDir = take()
case "--export-pack": o.exportPack = take()
case "--serve": o.serve = true
case "--no-prepare": o.noPrepare = true
case "--fuzz": o.fuzz = Int(take()) ?? 200
case "--fuzz-seed": o.fuzzSeed = take()
case "--edge": o.edge = true
case "--stats": o.stats = true
case "--determinism": o.determinism = true
case "--memcheck": o.memcheck = true
case "--inline-dataset": o.inlineDataset = true
case "--closed-form": o.closedForm = true
case "--load-weight": o.loadWeight = Int(take()) ?? o.loadWeight
case "--wide-frac": o.wideFrac = Int(take()) ?? o.wideFrac
case "--race": o.race = take()
case "--race-bench-ms": o.raceBenchMs = Int(take()) ?? o.raceBenchMs
case "--race-budget-s": o.raceBudgetS = Int(take()) ?? o.raceBudgetS
case "--race-rounds": o.raceRounds = Int(take()) ?? o.raceRounds
case "--variant": o.pinnedVariant = take()
case "--tuning": o.tuningPath = take()
case "--race-test": o.raceTest = true
case "-h", "--help":
print("""
igneum-bench [--seed <string>] [--hours N] [--batch-log2 22] [--batches 4]
[--dataset-log2 28] [--verify-warps 3] [--dump <dir>] [--day <string>]
[--closed-form] original closed-form dataset (default: memory-hard cache construction, see MEMHARD.md)
[--load-weight W] generator lever (a): percent weight of the load op (default 25)
[--wide-frac P] generator lever (b): percent of loads emitted as warp-coalesced 128-byte loads (default 0)
[--export-pack <dir>] write the CUDA + OpenCL program pack for --seed, then exit
[--serve] GPU worker for igneum-miner --worker: reads "job ..." lines on stdin (see runServe)
hardening tests (run instead of the bench; several may be combined; exit 0 only if all pass):
[--fuzz N [--fuzz-seed <string>]] N random programs, GPU vs CPU, 4 random warps each,
dataset size drawn from 64 MiB, 256 MiB, 1 GiB
[--edge] hand-built edge-case programs, GPU vs CPU
[--stats] output distribution sanity checks on 2^20 nonces, 3 seeds
[--determinism] 5 identical GPU runs of 2^20 nonces, double compile, dataset fill check
[--memcheck] static dataset-index mask check, 4 MiB run with wrapping nonces
shortcut measurement:
[--inline-dataset] bench variant: every load computes ds_elem(index) inline, no memory read
variant racing (4 October 2026; in --serve every prepared program is raced, the fastest variant serves the hour):
[--race on|off|a,b,c] [--race-bench-ms 2000] [--race-budget-s 120] [--race-rounds N]
[--variant <name>] use this variant, no race [--tuning <file>] per-card tuning (IGNEUM_TUNING_FILE)
[--race-test] the race alone for --seed on --day (3 rounds): a table per variant, exit 0/1
""")
exit(0)
default:
print("unknown argument \(a)"); exit(2)
}
}
return o
}
// MARK: - Integer helpers (CPU side, must match MSL bit for bit)
@inline(__always) func rotl32(_ x: UInt32, _ n: UInt32) -> UInt32 {
let n = n & 31
return n == 0 ? x : (x << n) | (x >> (32 - n))
}
@inline(__always) func rotr32(_ x: UInt32, _ n: UInt32) -> UInt32 {
let n = n & 31
return n == 0 ? x : (x >> n) | (x << (32 - n))
}
@inline(__always) func mulhi32(_ a: UInt32, _ b: UInt32) -> UInt32 {
UInt32(truncatingIfNeeded: (UInt64(a) &* UInt64(b)) >> 32)
}
@inline(__always) func splitmix32(_ v: UInt32) -> UInt32 {
var x = v
x ^= x >> 16; x &*= 0x7feb352d
x ^= x >> 15; x &*= 0x846ca68b
x ^= x >> 16
return x
}
// Dataset element, closed form of (daySeed, index). Same formula is emitted into the MSL.
@inline(__always) func datasetElem(_ i: UInt32, _ d0: UInt32, _ d1: UInt32) -> UInt32 {
var x = i ^ d0
x &*= 0x9E3779B1; x ^= x >> 15
x &+= d1
x &*= 0x85EBCA77; x ^= x >> 13
x &*= 0xC2B2AE3D; x ^= x >> 16
return x
}
// 32-byte seed (8 x uint32) from bytes: FNV-1a 64 with four salts, each finalised (seed_words_from_bytes in igneum-pow).
func seedWordsBytes(_ bytes: [UInt8]) -> [UInt32] {
var words = [UInt32]()
for salt in 0..<4 {
var h: UInt64 = 0xcbf29ce484222325 ^ (UInt64(salt) &* 0x9E3779B97F4A7C15)
for b in bytes { h ^= UInt64(b); h &*= 0x100000001b3 }
h ^= h >> 33; h &*= 0xff51afd7ed558ccd; h ^= h >> 33
words.append(UInt32(truncatingIfNeeded: h))
words.append(UInt32(truncatingIfNeeded: h >> 32))
}
return words
}
// The same from a string (its UTF-8 bytes).
func seedWords(_ s: String) -> [UInt32] { seedWordsBytes(Array(s.utf8)) }
struct SplitMix64 {
var s: UInt64
mutating func next() -> UInt64 {
s &+= 0x9E3779B97F4A7C15
var z = s
z = (z ^ (z >> 30)) &* 0xBF58476D1CE4E5B9
z = (z ^ (z >> 27)) &* 0x94D049BB133111EB
return z ^ (z >> 31)
}
mutating func below(_ n: Int) -> Int { Int(next() % UInt64(n)) }
}
// MARK: - Memory-hard dataset (added 3 October 2026, spec in MEMHARD.md)
//
// Cache: 2^26 words (256 MiB) = 2^22 lines of 16 words, in 2^16 segments of 64 lines. Each segment is a
// sequential chain of ChaCha12 blocks with feed-forward: in_j = prev_line ^ (sigma || K || seg || j || tag),
// line_j = core(in_j) + in_j, prev_0 = 0. Recomputing line j costs j + 1 block evaluations.
// Item t (16 words = 64 bytes): s = (K[0..7], t * MUL[i] + RC[i]), then 8 rounds of
// s = M_r(s); a = s[0] & (2^22 - 1); s ^= cache line a
// and a final M_8. M_r is the seed-parameterised mixer: per word (s ^ (RC + (r+1) * 0x9E3779B9)) * MUL,
// then one ChaCha-shaped column round and diagonal round with seed-drawn rotations.
// Dataset word w = item(w >> 4)[w & 15]. The hash kernel is unchanged: it still does dataset[r & MASK].
let cacheLog2Words = 26
let cacheSegmentLog2Lines = 6
let cacheWords = 1 << cacheLog2Words
let cacheLines = cacheWords >> 4
let cacheLinesPerSegment = 1 << cacheSegmentLog2Lines
let cacheSegments = cacheLines >> cacheSegmentLog2Lines
let cacheLineMask = UInt32(cacheLines - 1)
let itemRounds = 8
let chachaRounds = 12
let chachaSigma: [UInt32] = [0x61707865, 0x3320646e, 0x79622d32, 0x6b206574]
let cacheTag: [UInt32] = [0x49676e65, 0x756d4d48] // "Igne", "umMH"
// Rotation without the n == 0 check: every caller passes 1..31.
@inline(__always) func rotlc(_ x: UInt32, _ n: UInt32) -> UInt32 { (x << n) | (x >> (32 - n)) }
@inline(__always) func qr(_ s: UnsafeMutablePointer<UInt32>, _ a: Int, _ b: Int, _ c: Int, _ d: Int,
_ r1: UInt32, _ r2: UInt32, _ r3: UInt32, _ r4: UInt32) {
s[a] = s[a] &+ s[b]; s[d] ^= s[a]; s[d] = rotlc(s[d], r1)
s[c] = s[c] &+ s[d]; s[b] ^= s[c]; s[b] = rotlc(s[b], r2)
s[a] = s[a] &+ s[b]; s[d] ^= s[a]; s[d] = rotlc(s[d], r3)
s[c] = s[c] &+ s[d]; s[b] ^= s[c]; s[b] = rotlc(s[b], r4)
}
// y = ChaCha12 core(x) + x. Standard quarter-round rotations 16, 12, 8, 7; column then diagonal.
@inline(__always) func chachaBlock(_ x: UnsafePointer<UInt32>, _ y: UnsafeMutablePointer<UInt32>) {
for i in 0..<16 { y[i] = x[i] }
for _ in 0..<(chachaRounds / 2) {
qr(y, 0, 4, 8, 12, 16, 12, 8, 7); qr(y, 1, 5, 9, 13, 16, 12, 8, 7)
qr(y, 2, 6, 10, 14, 16, 12, 8, 7); qr(y, 3, 7, 11, 15, 16, 12, 8, 7)
qr(y, 0, 5, 10, 15, 16, 12, 8, 7); qr(y, 1, 6, 11, 12, 16, 12, 8, 7)
qr(y, 2, 7, 8, 13, 16, 12, 8, 7); qr(y, 3, 4, 9, 14, 16, 12, 8, 7)
}
for i in 0..<16 { y[i] = y[i] &+ x[i] }
}
// Mixer parameters drawn from the day key. Draw order: ROT[0..7] (1..31), MUL[0..15] (odd), RC[0..15].
final class MixParams {
let key: UnsafeMutablePointer<UInt32> // 8
let rot: UnsafeMutablePointer<UInt32> // 8
let mul: UnsafeMutablePointer<UInt32> // 16
let rc: UnsafeMutablePointer<UInt32> // 16
let keyWords: [UInt32]
var rotWords: [UInt32] { (0..<8).map { rot[$0] } }
var mulWords: [UInt32] { (0..<16).map { mul[$0] } }
var rcWords: [UInt32] { (0..<16).map { rc[$0] } }
init(key k: [UInt32]) {
precondition(k.count == 8)
keyWords = k
key = UnsafeMutablePointer<UInt32>.allocate(capacity: 8)
rot = UnsafeMutablePointer<UInt32>.allocate(capacity: 8)
mul = UnsafeMutablePointer<UInt32>.allocate(capacity: 16)
rc = UnsafeMutablePointer<UInt32>.allocate(capacity: 16)
for i in 0..<8 { key[i] = k[i] }
var rng = SplitMix64(s: UInt64(k[0]) | (UInt64(k[1]) << 32))
for i in 0..<8 { rot[i] = UInt32(1 + rng.below(31)) }
for i in 0..<16 { mul[i] = UInt32(truncatingIfNeeded: rng.next()) | 1 }
for i in 0..<16 { rc[i] = UInt32(truncatingIfNeeded: rng.next()) }
}
}
// M_r on 16 words in place. rk = (r + 1) * 0x9E3779B9 mod 2^32.
@inline(__always) func mixer(_ s: UnsafeMutablePointer<UInt32>, _ rk: UInt32, _ mp: MixParams) {
let mul = mp.mul, rc = mp.rc, R = mp.rot
for i in 0..<16 { s[i] = (s[i] ^ (rc[i] &+ rk)) &* mul[i] }
qr(s, 0, 4, 8, 12, R[0], R[1], R[2], R[3]); qr(s, 1, 5, 9, 13, R[0], R[1], R[2], R[3])
qr(s, 2, 6, 10, 14, R[0], R[1], R[2], R[3]); qr(s, 3, 7, 11, 15, R[0], R[1], R[2], R[3])
qr(s, 0, 5, 10, 15, R[4], R[5], R[6], R[7]); qr(s, 1, 6, 11, 12, R[4], R[5], R[6], R[7])
qr(s, 2, 7, 8, 13, R[4], R[5], R[6], R[7]); qr(s, 3, 4, 9, 14, R[4], R[5], R[6], R[7])
}
@inline(__always) func roundKey(_ r: Int) -> UInt32 { UInt32(r + 1) &* 0x9E3779B9 }
// One segment of the cache: 64 chained lines written at cache[seg * 1024 ...].
func cpuFillSegment(_ cache: UnsafeMutablePointer<UInt32>, seg: Int, key: UnsafePointer<UInt32>) {
var inp = [UInt32](repeating: 0, count: 16)
inp.withUnsafeMutableBufferPointer { ib in
let x = ib.baseAddress!
var prev: UnsafePointer<UInt32>? = nil
for j in 0..<cacheLinesPerSegment {
let line = cache + ((seg << cacheSegmentLog2Lines) + j) * 16
for i in 0..<4 { x[i] = chachaSigma[i] }
for i in 0..<8 { x[4 + i] = key[i] }
x[12] = UInt32(seg); x[13] = UInt32(j); x[14] = cacheTag[0]; x[15] = cacheTag[1]
if let p = prev { for i in 0..<16 { x[i] ^= p[i] } }
chachaBlock(x, line)
prev = UnsafePointer(line)
}
}
}
// The whole 256 MiB cache on one CPU core. Returns wall milliseconds.
func cpuFillCache(_ cache: UnsafeMutablePointer<UInt32>, key: [UInt32]) -> Double {
let t0 = nowNs()
key.withUnsafeBufferPointer { kb in
for seg in 0..<cacheSegments { cpuFillSegment(cache, seg: seg, key: kb.baseAddress!) }
}
return Double(nowNs() - t0) / 1e6
}
// Derive `n` items (indices ts[0..n)) into out[k * 16 ...], all n chains interleaved round by round so the
// cache-line misses of independent items overlap in the memory system. This is how the verifier reaches
// memory-level parallelism without threads: the 32 lanes of a warp are independent.
func deriveItems(_ ts: UnsafePointer<UInt32>, _ n: Int, _ mp: MixParams, cache: UnsafePointer<UInt32>, out: UnsafeMutablePointer<UInt32>) {
let key = mp.key, mul = mp.mul, rc = mp.rc
for k in 0..<n {
let s = out + k * 16
let t = ts[k]
for i in 0..<8 { s[i] = key[i] }
for i in 0..<8 { s[8 + i] = t &* mul[i] &+ rc[i] }
}
for r in 0..<itemRounds {
let rk = roundKey(r)
for k in 0..<n { mixer(out + k * 16, rk, mp) }
for k in 0..<n {
let s = out + k * 16
let line = cache + Int(s[0] & cacheLineMask) * 16
for i in 0..<16 { s[i] ^= line[i] }
}
}
let rk = roundKey(itemRounds)
for k in 0..<n { mixer(out + k * 16, rk, mp) }
}
func deriveItem(_ t: UInt32, _ mp: MixParams, cache: UnsafePointer<UInt32>) -> [UInt32] {
var out = [UInt32](repeating: 0, count: 16)
var tt = t
out.withUnsafeMutableBufferPointer { ob in deriveItems(&tt, 1, mp, cache: cache, out: ob.baseAddress!) }
return out
}
// The CPU verifier's view of the memory-hard dataset: the 256 MiB cache and nothing else.
final class MemhardCPU {
let mp: MixParams
let cache: UnsafeMutablePointer<UInt32>
let fillMs: Double
private let items = UnsafeMutablePointer<UInt32>.allocate(capacity: 64 * 16)
private let uniq = UnsafeMutablePointer<UInt32>.allocate(capacity: 64)
private let slot = UnsafeMutablePointer<Int>.allocate(capacity: 64)
var derivations = 0 // items derived so far (statistics)
init(key: [UInt32]) {
mp = MixParams(key: key)
cache = UnsafeMutablePointer<UInt32>.allocate(capacity: cacheWords)
fillMs = cpuFillCache(cache, key: key)
}
func word(_ w: UInt32) -> UInt32 { deriveItem(w >> 4, mp, cache: UnsafePointer(cache))[Int(w & 15)] }
// out[k] = dataset[idx[k]] for k < n (n <= 64). Equal items are derived once.
func fetch(_ idx: UnsafePointer<UInt32>, _ n: Int, _ out: UnsafeMutablePointer<UInt32>) {
var u = 0
for k in 0..<n {
let t = idx[k] >> 4
var found = -1
for j in 0..<u where uniq[j] == t { found = j; break }
if found < 0 { uniq[u] = t; found = u; u += 1 }
slot[k] = found
}
deriveItems(UnsafePointer(uniq), u, mp, cache: UnsafePointer(cache), out: items)
derivations += u
for k in 0..<n { out[k] = items[slot[k] * 16 + Int(idx[k] & 15)] }
}
}
// What the CPU interpreter reads dataset words from. Closed form (day words d0, d1) or the memory-hard cache.
final class DatasetSource {
let mask: UInt32
let day: (UInt32, UInt32)
let memhard: MemhardCPU?
init(mask: UInt32, day: (UInt32, UInt32), memhard: MemhardCPU?) { self.mask = mask; self.day = day; self.memhard = memhard }
var modeName: String { memhard == nil ? "closed-form" : "memory-hard" }
func word(_ w: UInt32) -> UInt32 { memhard?.word(w & mask) ?? datasetElem(w & mask, day.0, day.1) }
func fetch(_ idx: UnsafePointer<UInt32>, _ n: Int, _ out: UnsafeMutablePointer<UInt32>) {
if let m = memhard { m.fetch(idx, n, out) }
else { for k in 0..<n { out[k] = datasetElem(idx[k], day.0, day.1) } }
}
}
// MARK: - Program
// wload (added 3 October 2026, lever b): warp-coalesced load. All 32 lanes read consecutive words of one
// 128-byte block whose address comes from lane 0's source register. Never emitted unless --wide-frac > 0.
enum Op: String { case add, sub, mul, mulhi, xor, or, rotl, rotr, mad, shfl, load, wload }
struct Instr {
var op: Op
var dst: Int
var a: Int // source register, never equal to dst
var b: Int // second source (mad only)
var imm: UInt32 // add immediate A
var imm2: UInt32 // add immediate B
var rot: UInt32 // rotl amount 1..31
var bit: Int // selector bit of r0 for add
var mask: Int // shuffle xor mask: 1,2,4,8,16
}
struct Program {
let seedString: String
let seed: [UInt32]
let instrs: [Instr]
var generator = 1 // 2 for every current program (generateProgramV2); 1 for the retired lever generator
var attempt: UInt32 = 0 // attempt index under the acceptance rule (0 = the bare seed)
static let iterations = 8
static let count = 64
var loadsPerHash: Int { instrs.filter { $0.op == .load || $0.op == .wload }.count * Program.iterations }
var wideLoadsPerHash: Int { instrs.filter { $0.op == .wload }.count * Program.iterations }
var hasWide: Bool { instrs.contains { $0.op == .wload } }
// Distinct dataset items a 32-lane warp touches per hash: 32 per plain load, 2 per wide load (128 B = 2 items).
var itemsPerWarp: Int { (loadsPerHash - wideLoadsPerHash) * 32 + wideLoadsPerHash * 2 }
var histogram: [(String, Int)] {
var d = [String: Int]()
for i in instrs { d[i.op.rawValue, default: 0] += 1 }
return d.sorted { $0.1 != $1.1 ? $0.1 > $1.1 : $0.0 < $1.0 } // count desc, then name, so output is deterministic
}
}
// Weights sum to 100. Loads are 25 percent so the kernel leans on memory.
let opWeights: [(Op, Int)] = [(.load, 25), (.add, 12), (.xor, 10), (.mul, 8), (.mad, 8), (.shfl, 8),
(.rotl, 7), (.sub, 6), (.mulhi, 6), (.rotr, 6), (.or, 4)]
// Generator levers (3 October 2026). The defaults reproduce the original generator instruction for instruction:
// with loadWeight 25 the weight table above is used unchanged, and with wideFrac 0 no load becomes a wload.
// Neither lever consumes extra random draws, so a program differs from the default one only where the lever acts.
struct GeneratorConfig {
var loadWeight = 25
var wideFrac = 0
// Scaled weights: load gets loadWeight, the other ten ops share the rest in their original proportions,
// rounded by largest remainder so the table still sums to 100.
var weights: [(Op, Int)] {
if loadWeight == 25 { return opWeights }
let others = opWeights.dropFirst()
let total = others.reduce(0) { $0 + $1.1 } // 75
let budget = 100 - loadWeight
var scaled = others.map { (op: $0.0, floor: ($0.1 * budget) / total, rem: ($0.1 * budget) % total) }
var sum = scaled.reduce(0) { $0 + $1.floor }
let order = scaled.indices.sorted { scaled[$0].rem != scaled[$1].rem ? scaled[$0].rem > scaled[$1].rem : $0 < $1 }
var k = 0
while sum < budget { scaled[order[k]].floor += 1; sum += 1; k += 1 }
return [(.load, loadWeight)] + scaled.map { ($0.op, $0.floor) }
}
}
var generatorConfig = GeneratorConfig()
// The program of a seed string: generator version 2 with the acceptance rule (below), the same program the Rust crate
// derives. The retired version 1 generator is used only when a lever (--load-weight, --wide-frac) is set, for the
// MEMHARD.md section 2.4 measurements; those programs are not the lottery hash.
func generateProgram(seedString: String) -> Program {
if generatorConfig.loadWeight != 25 || generatorConfig.wideFrac != 0 {
return generateProgramV1(seedString: seedString, words: seedWords(seedString))
}
return generateProgramV2(seedString: seedString, bytes: Array(seedString.utf8))
}
// Version 1 (retired 4 October 2026): op rolled per instruction against the 11-family table, load count free.
func generateProgramV1(seedString: String, words sw: [UInt32]) -> Program {
var rng = SplitMix64(s: (UInt64(sw[0]) | (UInt64(sw[1]) << 32)) ^ ((UInt64(sw[2]) | (UInt64(sw[3]) << 32)) &* 0x9E3779B97F4A7C15))
var instrs = [Instr]()
let weights = generatorConfig.weights
for _ in 0..<Program.count {
var roll = rng.below(100)
var op = Op.add
for (o, w) in weights { if roll < w { op = o; break }; roll -= w }
let dst = rng.below(8)
var a = rng.below(7); if a >= dst { a += 1 }
let b = rng.below(8)
let imm = UInt32(truncatingIfNeeded: rng.next())
let imm2 = UInt32(truncatingIfNeeded: rng.next())
let rot = UInt32(1 + rng.below(31))
let bit = rng.below(32)
let mask = 1 << rng.below(5)
// Lever (b): the already-drawn selector bit decides whether a load is wide, so the draw stream is unchanged.
if op == .load && bit * 100 < generatorConfig.wideFrac * 32 { op = .wload }
instrs.append(Instr(op: op, dst: dst, a: a, b: b, imm: imm, imm2: imm2, rot: rot, bit: bit, mask: mask))
}
return Program(seedString: seedString, seed: sw, instrs: instrs)
}
// MARK: - Generator version 2 and the acceptance rule (4 October 2026)
//
// Draw for draw the Rust generator (igneum-pow/src/generator.rs, candidate_from_words) and acceptance rule
// (igneum-pow/src/accept.rs), spec 01 sections 1.4.3 and 1.4.6. Exactly 16 load slots drawn first from instructions
// 1..63; a load's source is drawn from the registers other than dst written by an earlier instruction and not read by
// a load since; a candidate that fails the rule is replaced by attempt k + 1, seedWordsBytes(seed || k_le32).
let generatorVersion = 2
let loadSlots = 16
let maxAttempts: UInt32 = 32
let nonloadWeights: [(Op, Int)] = [(.add, 12), (.xor, 10), (.mul, 8), (.mad, 8), (.shfl, 8),
(.rotl, 7), (.sub, 6), (.mulhi, 6), (.rotr, 6), (.or, 4)]
let acceptUnits = 64
let acceptDatasetLog2 = 28
let acceptMaxSaturated: UInt32 = 164
let acceptBiasTolerance: UInt32 = 136
let acceptMinDistinctSum: UInt64 = 245_760
func fnv1a64Bytes(_ bytes: [UInt8]) -> UInt64 {
var h: UInt64 = 0xcbf29ce484222325
for b in bytes { h ^= UInt64(b); h &*= 0x100000001b3 }
return h
}
func le32(_ v: UInt32) -> [UInt8] { [UInt8(v & 0xff), UInt8((v >> 8) & 0xff), UInt8((v >> 16) & 0xff), UInt8((v >> 24) & 0xff)] }
// The seed words of attempt k: seedWordsBytes(seed) for k = 0, seedWordsBytes(seed || k_le32) otherwise.
func attemptWords(_ seedBytes: [UInt8], _ attempt: UInt32) -> [UInt32] {
attempt == 0 ? seedWordsBytes(seedBytes) : seedWordsBytes(seedBytes + le32(attempt))
}
// FNV-1a 64 over "igneum-program/" || generator_le32 || seed words LE || attempt_le32 (Program::program_id in Rust).
func programId(_ p: Program) -> UInt64 {
var b = Array("igneum-program/".utf8) + le32(UInt32(p.generator))
for w in p.seed { b += le32(w) }
b += le32(p.attempt)
return fnv1a64Bytes(b)
}
// One version 2 candidate from its seed words, before the acceptance rule.
func candidateProgram(seedString: String, words sw: [UInt32], attempt: UInt32) -> Program {
var rng = SplitMix64(s: (UInt64(sw[0]) | (UInt64(sw[1]) << 32)) ^ ((UInt64(sw[2]) | (UInt64(sw[3]) << 32)) &* 0x9E3779B97F4A7C15))
var slots = Array(1..<Program.count)
for i in 0..<loadSlots {
let j = i + rng.below(Program.count - 1 - i)
slots.swapAt(i, j)
}
var isLoad = [Bool](repeating: false, count: Program.count)
for i in 0..<loadSlots { isLoad[slots[i]] = true }
var fresh = [Bool](repeating: false, count: 8)
var instrs = [Instr]()
for k in 0..<Program.count {
var roll = rng.below(75)
var op = Op.add
for (o, w) in nonloadWeights { if roll < w { op = o; break }; roll -= w }
if isLoad[k] { op = .load }
let dst = rng.below(8)
var a: Int
if op == .load {
let eligible = (0..<8).filter { $0 != dst && fresh[$0] }
if eligible.isEmpty {
a = rng.below(7); if a >= dst { a += 1 }
} else {
a = eligible[rng.below(eligible.count)]
}
} else {
a = rng.below(7); if a >= dst { a += 1 }
}
let b = rng.below(8)
let imm = UInt32(truncatingIfNeeded: rng.next())
let imm2 = UInt32(truncatingIfNeeded: rng.next())
let rot = UInt32(1 + rng.below(31))
let bit = rng.below(32)
let mask = 1 << rng.below(5)
if op == .load { fresh[a] = false }
fresh[dst] = true
instrs.append(Instr(op: op, dst: dst, a: a, b: b, imm: imm, imm2: imm2, rot: rot, bit: bit, mask: mask))
}
return Program(seedString: seedString, seed: sw, instrs: instrs, generator: generatorVersion, attempt: attempt)
}
@inline(__always) func opInjects(_ op: Op) -> Bool {
switch op { case .add, .sub, .xor, .mad, .shfl, .load, .wload: return true; default: return false }
}
// Parts (a) and (b) of the rule. Returns nil when the program passes, else the reason.
func acceptStatic(_ p: Program) -> String? {
var pending = [Bool](repeating: false, count: 8)
for _ in 0..<2 {
for (k, ins) in p.instrs.enumerated() {
let isLoad = ins.op == .load || ins.op == .wload
if isLoad && pending[ins.a] { return "(a) load at instruction \(k) reads r\(ins.a), unwritten since the previous load from it" }
pending[ins.dst] = false
if isLoad { pending[ins.a] = true }
}
}
var injected = [Bool](repeating: false, count: 8)
for ins in p.instrs where opInjects(ins.op) { injected[ins.dst] = true }
for r in 0..<8 where !injected[r] { return "(b) r\(r) has no add, sub, xor, mad, shfl or load write" }
return nil
}
// The 64 base nonces of the dynamic test: SplitMix64 seeded with FNV-1a 64("igneum-accept/" || seed words LE).
func acceptBaseNonces(_ seed: [UInt32]) -> [UInt32] {
var b = Array("igneum-accept/".utf8)
for w in seed { b += le32(w) }
var rng = SplitMix64(s: fnv1a64Bytes(b))
return (0..<acceptUnits).map { _ in UInt32(truncatingIfNeeded: rng.next()) & ~31 }
}
// Part (c): 64 units on the closed-form dataset keyed by the seed words, init words = seed words. Returns nil
// when the program passes, else the reason. Mirrors run_unit and check_dynamic in igneum-pow/src/accept.rs.
func acceptDynamic(_ p: Program) -> String? {
let lanes = 32
let loads = p.loadsPerHash
let mask: UInt32 = (1 << UInt32(acceptDatasetLog2)) - 1
let (d0, d1) = (p.seed[0], p.seed[1])
var andAcc = [UInt32](repeating: 0xffffffff, count: 8)
var orAcc = [UInt32](repeating: 0, count: 8)
var saturated: UInt32 = 0
var bitOnes = [UInt32](repeating: 0, count: 64)
var distinctSum: UInt64 = 0
var r = [UInt32](repeating: 0, count: 8 * lanes) // r[reg * 32 + lane]
var laneAddrs = [UInt32](repeating: 0, count: lanes * loads)
var sel = [UInt32](repeating: 0, count: lanes)
var idx = [UInt32](repeating: 0, count: lanes)
var tmp = [UInt32](repeating: 0, count: lanes)
for (unit, base) in acceptBaseNonces(p.seed).enumerated() {
for lane in 0..<lanes {
let nonce = base &+ UInt32(lane)
for i in 0..<8 {
var x = nonce ^ p.seed[i]
x &+= 0x9e3779b9 &* UInt32(i + 1)
x = splitmix32(x)
r[i * lanes + lane] = x ^ p.seed[(i + 1) & 7]
}
}
var nload = 0
for it in 0..<Program.iterations {
for lane in 0..<lanes { sel[lane] = r[lane] }
for (k, ins) in p.instrs.enumerated() {
let d = ins.dst * lanes, a = ins.a * lanes
switch ins.op {
case .add:
for lane in 0..<lanes {
let s = (sel[lane] >> UInt32(ins.bit)) & 1
r[d + lane] = r[d + lane] &+ r[a + lane] &+ (s != 0 ? ins.imm2 : ins.imm)
}
case .sub: for lane in 0..<lanes { r[d + lane] = r[d + lane] &- r[a + lane] }
case .mul: for lane in 0..<lanes { r[d + lane] = r[d + lane] &* r[a + lane] }
case .mulhi: for lane in 0..<lanes { r[d + lane] = mulhi32(r[d + lane], r[a + lane]) }
case .xor: for lane in 0..<lanes { r[d + lane] ^= r[a + lane] }
case .or: for lane in 0..<lanes { r[d + lane] |= r[a + lane] }
case .rotl: for lane in 0..<lanes { r[d + lane] = rotl32(r[d + lane], ins.rot) }
case .rotr: for lane in 0..<lanes { r[d + lane] = rotr32(r[d + lane], r[a + lane]) }
case .mad:
let b = ins.b * lanes
for lane in 0..<lanes { r[d + lane] = (r[a + lane] &* r[b + lane]) &+ r[d + lane] }
case .shfl:
for lane in 0..<lanes { tmp[lane] = r[a + lane] }
for lane in 0..<lanes { r[d + lane] ^= tmp[lane ^ ins.mask] }
case .load:
for lane in 0..<lanes { idx[lane] = r[a + lane] & mask }
var same = true
for lane in 1..<lanes where idx[lane] != idx[0] { same = false; break }
if same { return "(c) load at iteration \(it) instruction \(k) reads one address in all lanes of unit \(unit)" }
for lane in 0..<lanes {
r[d + lane] ^= datasetElem(idx[lane], d0, d1)
laneAddrs[lane * loads + nload] = idx[lane]
}
nload += 1
case .wload:
let b = (r[a] & mask) & ~31
for lane in 0..<lanes {
idx[lane] = b + UInt32(lane)
r[d + lane] ^= datasetElem(idx[lane], d0, d1)
laneAddrs[lane * loads + nload] = idx[lane]
}
nload += 1
}
}
}
for i in 0..<8 {
for lane in 0..<lanes {
let v = r[i * lanes + lane]
andAcc[i] &= v; orAcc[i] |= v
if v == 0 || v == 0xffffffff { saturated += 1 }
}
}
for lane in 0..<lanes {
let lo = r[lane] ^ rotl32(r[lanes + lane], 7) ^ rotl32(r[2 * lanes + lane], 14) ^ rotl32(r[3 * lanes + lane], 21)
let hi = r[4 * lanes + lane] ^ rotl32(r[5 * lanes + lane], 9) ^ rotl32(r[6 * lanes + lane], 18) ^ rotl32(r[7 * lanes + lane], 27)
let h = (UInt64(hi) << 32) | UInt64(lo)
for j in 0..<64 { bitOnes[j] += UInt32((h >> UInt64(j)) & 1) }
var sl = Array(laneAddrs[lane * loads..<(lane + 1) * loads])
sl.sort()
var distinct: UInt64 = 0
for k in 0..<loads where k == 0 || sl[k] != sl[k - 1] { distinct += 1 }
distinctSum += distinct
}
}
for i in 0..<8 {
let bits = (andAcc[i] | ~orAcc[i]).nonzeroBitCount
if bits != 0 { return "(c) r\(i) has \(bits) nonce-independent bits" }
}
if saturated >= acceptMaxSaturated { return "(c) \(saturated) of 16384 final register values saturated (limit 163)" }
let half = UInt32(acceptUnits * lanes / 2)
for j in 0..<64 {
let d = bitOnes[j] > half ? bitOnes[j] - half : half - bitOnes[j]
if d > acceptBiasTolerance { return "(c) output bit \(j) set in \(bitOnes[j]) of 2048 hashes" }
}
if distinctSum <= acceptMinDistinctSum { return "(c) distinct addresses \(distinctSum) over 2048 hashes (needs above 245760)" }
return nil
}
func acceptProgram(_ p: Program) -> String? { acceptStatic(p) ?? acceptDynamic(p) }
// The program of a seed under version 2: the first accepted candidate over attempts 0, 1, 2, ...
func generateProgramV2(seedString: String, bytes: [UInt8]) -> Program {
for attempt in 0..<maxAttempts {
let p = candidateProgram(seedString: seedString, words: attemptWords(bytes, attempt), attempt: attempt)
if acceptProgram(p) == nil { return p }
}
fatalError("seed \(seedString): \(maxAttempts) consecutive candidates rejected (consensus fault)")
}
// MARK: - MSL generation
func hex(_ v: UInt32) -> String { String(format: "0x%08xu", v) }
// The memory-hard core as source text, in Metal, CUDA C++ or OpenCL C. Mixer parameters are
// literals so the GPU kernels, the CUDA pack, the OpenCL pack and the host reference share one text. Names are prefixed mh_.
// In the CUDA dialect every function is IGNEUM_HD (host and device) so host.cu can derive items too.
// In the OpenCL dialect the cache pointers carry the __global address space (OpenCL C 1.2 has no generic space).
enum CoreDialect { case metal, cuda, opencl }
func emitMemhardCore(_ mp: MixParams, cuda: Bool) -> String { emitMemhardCore(mp, dialect: cuda ? .cuda : .metal) }
func emitMemhardCore(_ mp: MixParams, dialect: CoreDialect) -> String {
let U: String, fn: String, cptr: String, wptr: String, lptr: String, lcptr: String
switch dialect {
case .metal: (U, fn, cptr, wptr, lptr, lcptr) = ("uint", "inline", "device const uint*", "device uint*", "thread uint*", "const thread uint*")
case .cuda: (U, fn, cptr, wptr, lptr, lcptr) = ("uint32_t", "IGNEUM_HD", "const uint32_t*", "uint32_t*", "uint32_t*", "const uint32_t*")
case .opencl: (U, fn, cptr, wptr, lptr, lcptr) = ("uint", "static inline", "__global const uint*", "__global uint*", "uint*", "const uint*")
}
let K = mp.keyWords, R = mp.rotWords, M = mp.mulWords, C = mp.rcWords
var s = """
// Memory-hard dataset core (MEMHARD.md). Cache: 2^\(cacheLog2Words) words in 2^\(Int(log2(Double(cacheSegments)))) segments of \(cacheLinesPerSegment) chained ChaCha\(chachaRounds) lines.
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
#define MH_CACHE_LINE_MASK \(hex(cacheLineMask))
#define MH_SEGMENT_LINES \(cacheLinesPerSegment)u
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
\(fn) \(U) mh_rotl(\(U) x, \(U) n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
// y = ChaCha\(chachaRounds) core(x) + x
\(fn) void mh_chacha_block(\(lcptr) x, \(lptr) y) {
for (\(U) i = 0u; i < 16u; ++i) y[i] = x[i];
for (\(U) r = 0u; r < \(chachaRounds / 2)u; ++r) {
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
}
for (\(U) i = 0u; i < 16u; ++i) y[i] += x[i];
}
// One cache segment: \(cacheLinesPerSegment) chained lines written at cache[seg * \(cacheLinesPerSegment * 16)]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
\(fn) void mh_cache_segment(\(wptr) cache, \(U) seg) {
\(U) prev[16]; \(U) x[16]; \(U) y[16];
for (\(U) i = 0u; i < 16u; ++i) prev[i] = 0u;
for (\(U) j = 0u; j < MH_SEGMENT_LINES; ++j) {
x[0] = \(hex(chachaSigma[0])) ^ prev[0]; x[1] = \(hex(chachaSigma[1])) ^ prev[1]; x[2] = \(hex(chachaSigma[2])) ^ prev[2]; x[3] = \(hex(chachaSigma[3])) ^ prev[3];
"""
for i in 0..<8 { s += " x[\(4 + i)] = \(hex(K[i])) ^ prev[\(4 + i)];\n" }
s += """
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = \(hex(cacheTag[0])) ^ prev[14]; x[15] = \(hex(cacheTag[1])) ^ prev[15];
mh_chacha_block(x, y);
\(wptr) line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
for (\(U) i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
}
}
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
\(fn) void mh_mixer(\(lptr) s, \(U) rk) {
"""
for i in 0..<16 { s += " s[\(i)] = (s[\(i)] ^ (\(hex(C[i])) + rk)) * \(hex(M[i]));\n" }
let c = (0..<4).map { "\(R[$0])u" }.joined(separator: ", "), d = (4..<8).map { "\(R[$0])u" }.joined(separator: ", ")
s += """
MH_QR(s[0], s[4], s[8], s[12], \(c)) MH_QR(s[1], s[5], s[9], s[13], \(c))
MH_QR(s[2], s[6], s[10], s[14], \(c)) MH_QR(s[3], s[7], s[11], s[15], \(c))
MH_QR(s[0], s[5], s[10], s[15], \(d)) MH_QR(s[1], s[6], s[11], s[12], \(d))
MH_QR(s[2], s[7], s[8], s[13], \(d)) MH_QR(s[3], s[4], s[9], s[14], \(d))
}
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); \(itemRounds) rounds of mixer + cache line s[0] & mask; final mixer.
\(fn) void mh_item(\(cptr) cache, \(U) t, \(lptr) s) {
"""
for i in 0..<8 { s += " s[\(i)] = \(hex(K[i]));\n" }
for i in 0..<8 { s += " s[\(8 + i)] = t * \(hex(M[i])) + \(hex(C[i]));\n" }
s += """
for (\(U) r = 0u; r < \(itemRounds)u; ++r) {
mh_mixer(s, 0x9E3779B9u * (r + 1u));
\(cptr) line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
for (\(U) i = 0u; i < 16u; ++i) s[i] ^= line[i];
}
mh_mixer(s, 0x9E3779B9u * \(itemRounds + 1)u);
}
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
\(fn) \(U) mh_word(\(cptr) cache, \(U) w) { \(U) s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
"""
return s
}
// Metal library with the cache fill and dataset build kernels for one day key.
func memhardMSL(_ mp: MixParams) -> String {
return """
#include <metal_stdlib>
using namespace metal;
\(emitMemhardCore(mp, cuda: false))
// One thread per segment (2^\(Int(log2(Double(cacheSegments)))) threads).
kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) {
mh_cache_segment(cache, gid);
}
// One thread per 64-byte item (dataset words / 16 threads).
kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],
uint gid [[thread_position_in_grid]]) {
uint s[16];
mh_item(cache, gid, s);
device uint* d = dataset + gid * 16u;
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
}
"""
}
// How the hash kernel gets dataset words. .stored reads the buffer (the honest kernel). The two inline
// variants are the shortcut measurements: every load recomputes the word instead of reading the dataset.
enum LoadSource {
case stored
case inlineClosed(UInt32, UInt32) // closed form: ds_elem(index, d0, d1), no memory read at all
case inlineMemhard(MixParams) // memory-hard: mh_word(cache, index), 8 dependent cache reads per word
}
// A kernel variant for the race (4 October 2026): the same instruction text, a different shape for the compiler.
// Names are stable: the tuning file and the fleet records use them. "base" is the kernel as it has always shipped.
struct MetalVariant {
let name: String
var unroll = 0 // 0: the iteration loop as emitted; N: "#pragma unroll N" before it (8 = fully unrolled)
var maxThreads = 0 // N > 0: [[max_total_threads_per_threadgroup(N)]] (fewer threads per group, more registers per thread)
var groupWidth = 32 // threads per threadgroup at dispatch (a multiple of 32; SIMD groups stay 32 wide)
var sizeOpt = false // MTLCompileOptions.optimizationLevel = .size
}
func metalVariants() -> [MetalVariant] {
[MetalVariant(name: "base"),
MetalVariant(name: "g64", groupWidth: 64), MetalVariant(name: "g128", groupWidth: 128), MetalVariant(name: "g256", groupWidth: 256),
MetalVariant(name: "u2", unroll: 2), MetalVariant(name: "u8", unroll: 8),
MetalVariant(name: "mt256", maxThreads: 256), MetalVariant(name: "mt512", maxThreads: 512), MetalVariant(name: "mt1024", maxThreads: 1024),
MetalVariant(name: "osize", sizeOpt: true),
MetalVariant(name: "u2-g128", unroll: 2, groupWidth: 128), MetalVariant(name: "u8-g128", unroll: 8, groupWidth: 128),
MetalVariant(name: "mt256-g128", maxThreads: 256, groupWidth: 128), MetalVariant(name: "mt512-g256", maxThreads: 512, groupWidth: 256)]
}
func generateMSL(_ p: Program, datasetLog2: Int, source: LoadSource = .stored, bound: Bool = false, variant: MetalVariant? = nil) -> String {
let mask = UInt32((1 << datasetLog2) - 1)
var s = """
#include <metal_stdlib>
using namespace metal;
#define MASK \(hex(mask))
constant uint SEEDW[8] = { \(p.seed.map(hex).joined(separator: ", ")) };
inline uint splitmix32(uint x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
inline uint ds_elem(uint i, uint d0, uint d1) {
uint x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
"""
if p.hasWide { s += "#define WMASK (MASK & ~31u)\n\n" }
var buffer0 = "device const uint* dataset [[buffer(0)]]"
if case .inlineMemhard(let mp) = source {
s += emitMemhardCore(mp, cuda: false) + "\n"
buffer0 = "device const uint* cache [[buffer(0)]]"
}
// Header-bound variant (3 October 2026, igneum-pow/src/bind.rs): same body, the init words come from buffer 3.
let kernelName = bound ? "igneum_hash_bound" : "igneum_hash"
let initArg = bound ? " constant uint* initw [[buffer(3)]],\n" : ""
let iw = bound ? "initw" : "SEEDW"
if let v = variant, v.maxThreads > 0 { s += "[[max_total_threads_per_threadgroup(\(v.maxThreads))]]\n" }
s += """
kernel void \(kernelName)(\(buffer0),
device ulong* out [[buffer(1)]],
constant uint& baseNonce [[buffer(2)]],
\(initArg) uint gid [[thread_position_in_grid]]) {
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
"""
if p.hasWide { s += " uint lane = gid & 31u;\n" }
for i in 0..<8 {
s += " { uint x = nonce ^ \(iw)[\(i)]; x += 0x9e3779b9u * \(i + 1)u; x = splitmix32(x); r\(i) = x ^ \(iw)[\((i + 1) & 7)]; }\n"
}
if let v = variant, v.unroll > 0 { s += "\n#pragma unroll \(v.unroll)" }
s += "\n for (uint it = 0u; it < \(Program.iterations)u; ++it) {\n uint sel = r0;\n"
// The word index expression for a load: plain = a & MASK; wide = lane 0's a, aligned to 32 words, plus lane.
func wordIndex(_ a: String, wide: Bool) -> String { wide ? "(simd_broadcast(\(a), 0) & WMASK) + lane" : "\(a) & MASK" }
func fetch(_ idx: String) -> String {
switch source {
case .stored: return "dataset[\(idx)]"
case .inlineClosed(let d0, let d1): return "ds_elem(\(idx), \(hex(d0)), \(hex(d1)))"
case .inlineMemhard: return "mh_word(cache, \(idx))"
}
}
for (k, ins) in p.instrs.enumerated() {
let d = "r\(ins.dst)", a = "r\(ins.a)", b = "r\(ins.b)"
var line: String
switch ins.op {
case .add: line = "\(d) = \(d) + \(a) + select(\(hex(ins.imm)), \(hex(ins.imm2)), ((sel >> \(ins.bit)u) & 1u) != 0u);"
case .sub: line = "\(d) = \(d) - \(a);"
case .mul: line = "\(d) = \(d) * \(a);"
case .mulhi: line = "\(d) = mulhi(\(d), \(a));"
case .xor: line = "\(d) = \(d) ^ \(a);"
case .or: line = "\(d) = \(d) | \(a);"
case .rotl: line = "\(d) = rotl_imm(\(d), \(ins.rot)u);"
case .rotr: line = "\(d) = rotr_var(\(d), \(a));"
case .mad: line = "\(d) = \(a) * \(b) + \(d);"
case .shfl: line = "\(d) = \(d) ^ simd_shuffle_xor(\(a), (ushort)\(ins.mask));"
case .load: line = "\(d) = \(d) ^ \(fetch(wordIndex(a, wide: false)));"
case .wload: line = "\(d) = \(d) ^ \(fetch(wordIndex(a, wide: true)));"
}
s += " \(line) // \(k)\n"
}
s += """
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}
"""
return s
}
let fillMSL = """
#include <metal_stdlib>
using namespace metal;
inline uint ds_elem(uint i, uint d0, uint d1) {
uint x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
kernel void igneum_fill(device uint* dataset [[buffer(0)]],
constant uint2& day [[buffer(1)]],
uint gid [[thread_position_in_grid]]) {
dataset[gid] = ds_elem(gid, day.x, day.y);
}
"""
// MARK: - CPU reference interpreter for one 32-lane warp
func cpuWarp(_ p: Program, baseNonce: UInt32, ds: DatasetSource) -> [UInt64] {
cpuWarpTraced(p, baseNonce: baseNonce, ds: ds, trace: nil)
}
// Same interpreter with an optional hook. When `trace` is set it is called before every instruction with
// (iteration, instruction index, the 8 registers of lane 0). The edge-case tests use it to prove that the
// operand values they were built to produce really occurred. The bench passes nil.
// Loads are batched across the 32 lanes: the indices are gathered, DatasetSource.fetch answers all 32, then
// the xors are applied. For the closed form this is the old per-lane formula; for the memory-hard dataset it
// lets the 32 independent item derivations overlap their cache misses (see deriveItems).
func cpuWarpTraced(_ p: Program, baseNonce: UInt32, ds: DatasetSource,
trace: ((Int, Int, [UInt32]) -> Void)?) -> [UInt64] {
let lanes = 32
let mask = ds.mask
var r = [UInt32](repeating: 0, count: lanes * 8) // r[lane*8 + reg]
for lane in 0..<lanes {
let nonce = baseNonce &+ UInt32(lane)
for i in 0..<8 {
var x = nonce ^ p.seed[i]
x &+= 0x9e3779b9 &* UInt32(i + 1)
x = splitmix32(x)
r[lane * 8 + i] = x ^ p.seed[(i + 1) & 7]
}
}
var tmp = [UInt32](repeating: 0, count: lanes)
let idx = UnsafeMutablePointer<UInt32>.allocate(capacity: lanes)
let val = UnsafeMutablePointer<UInt32>.allocate(capacity: lanes)
defer { idx.deallocate(); val.deallocate() }
for it in 0..<Program.iterations {
for lane in 0..<lanes { tmp[lane] = r[lane * 8] } // sel = r0 at the top of the iteration
let sel = tmp
for (k, ins) in p.instrs.enumerated() {
if let t = trace { t(it, k, Array(r[0..<8])) }
switch ins.op {
case .shfl:
for lane in 0..<lanes { tmp[lane] = r[lane * 8 + ins.a] }
for lane in 0..<lanes { r[lane * 8 + ins.dst] ^= tmp[lane ^ ins.mask] }
case .load:
for lane in 0..<lanes { idx[lane] = r[lane * 8 + ins.a] & mask }
ds.fetch(UnsafePointer(idx), lanes, val)
for lane in 0..<lanes { r[lane * 8 + ins.dst] ^= val[lane] }
case .wload:
// Lane 0's register, masked, aligned down to 32 words; lane l reads word base + l.
let base = (r[ins.a] & mask) & ~31
for lane in 0..<lanes { idx[lane] = base + UInt32(lane) }
ds.fetch(UnsafePointer(idx), lanes, val)
for lane in 0..<lanes { r[lane * 8 + ins.dst] ^= val[lane] }
default:
for lane in 0..<lanes {
let base = lane * 8
let d = r[base + ins.dst], a = r[base + ins.a]
var v: UInt32
switch ins.op {
case .add:
let s = (sel[lane] >> UInt32(ins.bit)) & 1
v = d &+ a &+ (s != 0 ? ins.imm2 : ins.imm)
case .sub: v = d &- a
case .mul: v = d &* a
case .mulhi: v = mulhi32(d, a)
case .xor: v = d ^ a
case .or: v = d | a
case .rotl: v = rotl32(d, ins.rot)
case .rotr: v = rotr32(d, a)
case .mad: v = (a &* r[base + ins.b]) &+ d
case .load, .wload, .shfl: v = d // unreachable, handled above
}
r[base + ins.dst] = v
}
}
}
}
var out = [UInt64](repeating: 0, count: lanes)
for lane in 0..<lanes {
let b = lane * 8
let lo = r[b] ^ rotl32(r[b + 1], 7) ^ rotl32(r[b + 2], 14) ^ rotl32(r[b + 3], 21)
let hi = r[b + 4] ^ rotl32(r[b + 5], 9) ^ rotl32(r[b + 6], 18) ^ rotl32(r[b + 7], 27)
out[lane] = (UInt64(hi) << 32) | UInt64(lo)
}
return out
}
// MARK: - Program pack export (CUDA twin of the Metal kernel)
//
// Writes, for one seed: program.json, vectors.json, kernel.cu, kernel.cl, program.h, vectors.h, program.metal.
// The CUDA kernel is emitted from the same Instr list as the MSL above, line for line.
// Differences by design: the dataset mask is a kernel argument (so the host can sweep dataset
// sizes with one ahead-of-time compile), seeds are inlined as literals, and the host launch
// wrappers live in kernel.cu so host.cu never declares a __global__ across translation units.
let packVectorBases: [UInt32] = [0, 4096, 1000000]
func hex64(_ v: UInt64) -> String { String(format: "0x%016llxull", v) }
func jhex(_ v: UInt32) -> String { String(format: "\"0x%08x\"", v) }
func jhex64(_ v: UInt64) -> String { String(format: "\"0x%016llx\"", v) }
func jstr(_ s: String) -> String {
var o = "\""
for c in s.unicodeScalars {
switch c {
case "\"": o += "\\\""
case "\\": o += "\\\\"
case "\n": o += "\\n"
default: o.unicodeScalars.append(c)
}
}
return o + "\""
}
// memhard: nil for a closed-form pack (the original fill kernel), MixParams for the memory-hard pack
// (cache fill and dataset build kernels; the shared core comes from memhard.h in the same pack).
func generateCUDA(_ p: Program, memhard: MixParams?) -> String {
var s = """
// Generated by proto-metal/igneum-bench --export-pack for seed "\(p.seedString)". Do not edit by hand.
// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal).
// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC.
#include <cuda_runtime.h>
#include <cstdint>
#include "program.h"
\(memhard != nil ? "#include \"memhard.h\"\n" : "")
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31.
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x.
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) {
uint32_t x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
"""
if memhard == nil {
s += """
// dataset[i] = ds_elem(i, d0, d1) for i < n. Same closed form as the Metal igneum_fill kernel.
__global__ void igneum_fill(uint32_t* ds, uint32_t n, uint32_t d0, uint32_t d1) {
uint32_t i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < n) ds[i] = ds_elem(i, d0, d1);
}
"""
} else {
s += """
// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item.
// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host.
__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) {
uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;
if (seg < nSegments) mh_cache_segment(cache, seg);
}
__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
uint32_t t = blockIdx.x * blockDim.x + threadIdx.x;
if (t < nItems) {
uint32_t s[16];
mh_item(cache, t, s);
uint32_t* d = ds + (size_t)t * 16u;
for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i];
}
}
"""
}
s += """
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask) {
uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;
uint32_t nonce = baseNonce + gid;
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
"""
if p.hasWide { s += " uint32_t lane = threadIdx.x & 31u;\n uint32_t wmask = mask & ~31u;\n" }
for i in 0..<8 {
let addc = 0x9e3779b9 &* UInt32(i + 1)
s += " { uint32_t x = nonce ^ \(hex(p.seed[i])); x += \(hex(addc)); x = splitmix32(x); r\(i) = x ^ \(hex(p.seed[(i + 1) & 7])); } // SEEDW[\(i)], 0x9e3779b9u * \(i + 1)u, SEEDW[\((i + 1) & 7)]\n"
}
s += "\n for (uint32_t it = 0u; it < \(Program.iterations)u; ++it) {\n uint32_t sel = r0;\n"
for (k, ins) in p.instrs.enumerated() {
let d = "r\(ins.dst)", a = "r\(ins.a)", b = "r\(ins.b)"
var line: String
switch ins.op {
// Metal select(A, B, c) returns c ? B : A, so the true branch is imm2 here as well.
case .add: line = "\(d) = \(d) + \(a) + ((((sel >> \(ins.bit)u) & 1u) != 0u) ? \(hex(ins.imm2)) : \(hex(ins.imm)));"
case .sub: line = "\(d) = \(d) - \(a);"
case .mul: line = "\(d) = \(d) * \(a);"
case .mulhi: line = "\(d) = __umulhi(\(d), \(a));"
case .xor: line = "\(d) = \(d) ^ \(a);"
case .or: line = "\(d) = \(d) | \(a);"
case .rotl: line = "\(d) = rotl_imm(\(d), \(ins.rot)u);"
case .rotr: line = "\(d) = rotr_var(\(d), \(a));"
case .mad: line = "\(d) = \(a) * \(b) + \(d);"
case .shfl: line = "\(d) = \(d) ^ __shfl_xor_sync(0xffffffffu, \(a), \(ins.mask));"
case .load: line = "\(d) = \(d) ^ ds[\(a) & mask];"
case .wload: line = "\(d) = \(d) ^ ds[(__shfl_sync(0xffffffffu, \(a), 0) & wmask) + lane];"
}
s += " \(line) // \(k) \(ins.op.rawValue)\n"
}
s += """
}
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
}
// Host-side launch wrappers. Declared in program.h, called from host.cu.
"""
if memhard == nil {
s += """
cudaError_t igneum_launch_fill(uint32_t* ds, uint32_t nWords, uint32_t d0, uint32_t d1) {
if (nWords == 0u) return cudaErrorInvalidValue;
uint32_t block = 256u;
uint32_t grid = (nWords + block - 1u) / block;
igneum_fill<<<grid, block>>>(ds, nWords, d0, d1);
return cudaGetLastError();
}
"""
} else {
s += """
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) {
if (nSegments == 0u) return cudaErrorInvalidValue;
uint32_t block = 256u;
uint32_t grid = (nSegments + block - 1u) / block;
igneum_cache_fill<<<grid, block>>>(cache, nSegments);
return cudaGetLastError();
}
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
if (nItems == 0u) return cudaErrorInvalidValue;
uint32_t block = 256u;
uint32_t grid = (nItems + block - 1u) / block;
igneum_build<<<grid, block>>>(ds, cache, nItems);
return cudaGetLastError();
}
"""
}
s += """
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
uint32_t nonces, uint32_t blockWarps) {
if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;
uint32_t block = 32u * blockWarps;
if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;
igneum_hash<<<nonces / block, block>>>(ds, out, baseNonce, mask);
return cudaGetLastError();
}
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
cudaFuncAttributes attr;
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash);
if (e != cudaSuccess) return e;
*numRegs = attr.numRegs;
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0);
}
"""
return s
}
// kernel.cl: the same program as OpenCL C 1.2, built from source at runtime by proto-opencl/host.c.
// The 32-lane exchange is selected at compile time (IGNEUM_EXCHANGE): sub-group shuffles where the device has them
// and its sub-group size is exactly 32, or a local-memory exchange with a barrier on every other device
// (proto-opencl/WAVEFRONT.md). The same text is compiled as C++ by proto-opencl/emu with a 32- or 64-wide sub-group.
func generateOpenCL(_ p: Program, memhard: MixParams?) -> String {
var s = """
// Generated by proto-metal/igneum-bench --export-pack for seed "\(p.seedString)". Do not edit by hand.
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
#ifndef IGNEUM_GROUP
#define IGNEUM_GROUP 32
#endif
#ifndef IGNEUM_EXCHANGE
#define IGNEUM_EXCHANGE 0
#endif
#ifdef __OPENCL_VERSION__
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
#if IGNEUM_EXCHANGE == 1
#ifdef cl_khr_subgroups
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
#endif
#ifdef cl_khr_subgroup_shuffle
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
#endif
#elif IGNEUM_EXCHANGE == 2
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
#endif
#else
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
#include "emu_opencl.h"
#endif
#if IGNEUM_EXCHANGE == 1
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
#elif IGNEUM_EXCHANGE == 2
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
#else
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
#endif
static inline uint splitmix32(uint x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
static inline uint ds_elem(uint i, uint d0, uint d1) {
uint x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
"""
if let mp = memhard {
s += emitMemhardCore(mp, dialect: .opencl) + "\n"
s += """
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
// The same constants as memhard.h in this pack (one emitter, three dialects).
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
uint seg = (uint)get_global_id(0);
if (seg < nSegments) mh_cache_segment(cache, seg);
}
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
uint t = (uint)get_global_id(0);
if (t < nItems) {
uint s[16];
mh_item(cache, t, s);
__global uint* d = ds + ((ulong)t * 16u);
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
}
}
"""
} else {
s += """
// dataset[i] = ds_elem(i, d0, d1) for i < n. Same closed form as the Metal igneum_fill kernel.
__kernel void igneum_fill(__global uint* ds, uint n, uint d0, uint d1) {
uint i = (uint)get_global_id(0);
if (i < n) ds[i] = ds_elem(i, d0, d1);
}
"""
}
s += """
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask) {
uint gid = (uint)get_global_id(0);
uint lid = (uint)get_local_id(0);
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
#if IGNEUM_EXCHANGE == 0
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
uint xk = 0u;
#else
(void)lid;
#endif
"""
if p.hasWide { s += " uint lane = lid & 31u;\n uint wmask = mask & ~31u;\n" }
for i in 0..<8 {
let addc = 0x9e3779b9 &* UInt32(i + 1)
s += " { uint x = nonce ^ \(hex(p.seed[i])); x += \(hex(addc)); x = splitmix32(x); r\(i) = x ^ \(hex(p.seed[(i + 1) & 7])); } // SEEDW[\(i)], 0x9e3779b9u * \(i + 1)u, SEEDW[\((i + 1) & 7)]\n"
}
s += "\n for (uint it = 0u; it < \(Program.iterations)u; ++it) {\n uint sel = r0;\n"
for (k, ins) in p.instrs.enumerated() {
let d = "r\(ins.dst)", a = "r\(ins.a)", b = "r\(ins.b)"
var line: String
switch ins.op {
// Metal select(A, B, c) returns c ? B : A, so the true branch is imm2 here as well.
case .add: line = "\(d) = \(d) + \(a) + ((((sel >> \(ins.bit)u) & 1u) != 0u) ? \(hex(ins.imm2)) : \(hex(ins.imm)));"
case .sub: line = "\(d) = \(d) - \(a);"
case .mul: line = "\(d) = \(d) * \(a);"
case .mulhi: line = "\(d) = mul_hi(\(d), \(a));"
case .xor: line = "\(d) = \(d) ^ \(a);"
case .or: line = "\(d) = \(d) | \(a);"
case .rotl: line = "\(d) = rotl_imm(\(d), \(ins.rot)u);"
case .rotr: line = "\(d) = rotr_var(\(d), \(a));"
case .mad: line = "\(d) = \(a) * \(b) + \(d);"
case .shfl: line = "{ uint t_; IGNEUM_SHFL_XOR(t_, \(a), \(ins.mask)u); \(d) = \(d) ^ t_; }"
case .load: line = "\(d) = \(d) ^ ds[\(a) & mask];"
case .wload: line = "{ uint t_; IGNEUM_BCAST0(t_, \(a)); \(d) = \(d) ^ ds[(t_ & wmask) + lane]; }"
}
s += " \(line) // \(k) \(ins.op.rawValue)\n"
}
s += """
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}
#if IGNEUM_EXCHANGE != 0
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
}
#endif
"""
return s
}
func generateProgramHeader(_ p: Program, dayString: String, day: (UInt32, UInt32), datasetLog2: Int, memhard: MixParams?) -> String {
let mask = UInt32((1 << datasetLog2) - 1)
let mix = p.histogram.map { "\($0.0)=\($0.1)" }.joined(separator: " ")
var s = """
// Generated by proto-metal/igneum-bench --export-pack for seed "\(p.seedString)". Do not edit by hand.
// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.
// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros.
#pragma once
#ifdef __cplusplus
#include <cstdint>
#else
#include <stdint.h>
#endif
#ifndef IGNEUM_NO_CUDA
#include <cuda_runtime.h>
#endif
#define IGNEUM_SEED_STRING \(jstr(p.seedString))
#define IGNEUM_DAY_STRING \(jstr(dayString))
#define IGNEUM_DAY0 \(hex(day.0))
#define IGNEUM_DAY1 \(hex(day.1))
#define IGNEUM_DATASET_LOG2 \(datasetLog2)
#define IGNEUM_MASK \(hex(mask))
#define IGNEUM_LANES 32
#define IGNEUM_ITERATIONS \(Program.iterations)
#define IGNEUM_INSTR_COUNT \(Program.count)
#define IGNEUM_LOADS_PER_HASH \(p.loadsPerHash)
#define IGNEUM_WIDE_LOADS_PER_HASH \(p.wideLoadsPerHash)
#define IGNEUM_OP_MIX \(jstr(mix))
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
#define IGNEUM_DATASET_MODE \(memhard == nil ? 0 : 1)
#define IGNEUM_SEEDW_INIT { \(p.seed.map(hex).joined(separator: ", ")) }
"""
if let mp = memhard {
s += """
#define IGNEUM_KEY_INIT { \(mp.keyWords.map(hex).joined(separator: ", ")) }
#define IGNEUM_CACHE_LOG2_WORDS \(cacheLog2Words)
#define IGNEUM_CACHE_SEGMENT_LOG2_LINES \(cacheSegmentLog2Lines)
#define IGNEUM_CACHE_SEGMENTS \(cacheSegments)u
#define IGNEUM_ITEM_ROUNDS \(itemRounds)
#define IGNEUM_MIX_ROT_INIT { \(mp.rotWords.map { "\($0)u" }.joined(separator: ", ")) }
#define IGNEUM_MIX_MUL_INIT { \(mp.mulWords.map(hex).joined(separator: ", ")) }
#define IGNEUM_MIX_RC_INIT { \(mp.rcWords.map(hex).joined(separator: ", ")) }
#ifndef IGNEUM_NO_CUDA
// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);
"""
} else {
s += """
#ifndef IGNEUM_NO_CUDA
// Defined in kernel.cu. Both launch on the default stream and return cudaGetLastError().
cudaError_t igneum_launch_fill(uint32_t* ds, uint32_t nWords, uint32_t d0, uint32_t d1);
"""
}
s += """
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
uint32_t nonces, uint32_t blockWarps);
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
#endif
"""
return s
}
// memhard.h for a memory-hard pack: the core in CUDA C++, compiled for host and device.
func generateMemhardHeader(_ p: Program, _ mp: MixParams) -> String {
return """
// Generated by proto-metal/igneum-bench --export-pack for seed "\(p.seedString)". Do not edit by hand.
// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against.
// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference).
// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C.
#pragma once
#ifdef __cplusplus
#include <cstdint>
#else
#include <stdint.h>
#endif
#if defined(__CUDACC__)
#define IGNEUM_HD __host__ __device__ __forceinline__
#elif defined(_MSC_VER) && !defined(__cplusplus)
#define IGNEUM_HD static __inline
#else
#define IGNEUM_HD static inline
#endif
\(emitMemhardCore(mp, dialect: .cuda))
"""
}
struct PackVectors {
var head: [UInt32] // dataset[0..15]
var last: UInt32 // dataset[MASK]
var sampleIdx: [UInt32] // 64 dataset indices
var sampleVal: [UInt32]
var cacheHead: [UInt32] // cache[0..15] (memhard only)
var cacheLast: [UInt32] // last cache line (memhard only)
var cacheFNV: UInt64 // FNV-1a 64 over the whole cache (memhard only)
}
func generateVectorsHeader(_ p: Program, bases: [UInt32], outs: [[UInt64]], v: PackVectors, mask: UInt32, source: String, memhard: Bool) -> String {
var s = """
// Generated by proto-metal/igneum-bench --export-pack for seed "\(p.seedString)". Do not edit by hand.
// Expected outputs: \(source)
#pragma once
#ifdef __cplusplus
#include <cstdint>
#else
#include <stdint.h>
#endif
#define IGNEUM_VEC_WARPS \(bases.count)
static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { \(bases.map { "\($0)u" }.joined(separator: ", ")) };
static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {
"""
for (i, o) in outs.enumerated() {
s += " { // base nonce \(bases[i])\n"
for row in 0..<4 {
s += " " + (0..<8).map { hex64(o[row * 8 + $0]) }.joined(separator: ", ") + (row == 3 ? "\n" : ",\n")
}
s += i == outs.count - 1 ? " }\n" : " },\n"
}
s += """
};
// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (\(mask)).
static const uint32_t IGNEUM_DS_HEAD[16] = {
\((0..<8).map { hex(v.head[$0]) }.joined(separator: ", ")),
\((8..<16).map { hex(v.head[$0]) }.joined(separator: ", "))
};
static const uint32_t IGNEUM_DS_LAST_INDEX = \(mask)u;
static const uint32_t IGNEUM_DS_LAST = \(hex(v.last));
// 64 sampled dataset words (index, value) computed on the Mac.
#define IGNEUM_DS_SAMPLES \(v.sampleIdx.count)
static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = {
\(v.sampleIdx.map { "\($0)u" }.joined(separator: ", "))
};
static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = {
\(v.sampleVal.map(hex).joined(separator: ", "))
};
"""
if memhard {
s += """
// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^\(cacheLog2Words) words.
static const uint32_t IGNEUM_CACHE_HEAD[16] = {
\((0..<8).map { hex(v.cacheHead[$0]) }.joined(separator: ", ")),
\((8..<16).map { hex(v.cacheHead[$0]) }.joined(separator: ", "))
};
static const uint32_t IGNEUM_CACHE_LAST[16] = {
\((0..<8).map { hex(v.cacheLast[$0]) }.joined(separator: ", ")),
\((8..<16).map { hex(v.cacheLast[$0]) }.joined(separator: ", "))
};
static const uint64_t IGNEUM_CACHE_FNV64 = \(hex64(v.cacheFNV));
"""
}
return s
}
func generateProgramJSON(_ p: Program, dayString: String, day: (UInt32, UInt32), datasetLog2: Int, memhard: MixParams?) -> String {
let mask = UInt32((1 << datasetLog2) - 1)
var s = "{\n"
s += " \"format\": \"igneum-program-pack-2\",\n"
s += " \"dataset_mode\": \(jstr(memhard == nil ? "closed-form" : "memory-hard")),\n"
s += " \"seed\": \(jstr(p.seedString)),\n"
s += " \"seed_words\": [\(p.seed.map(jhex).joined(separator: ", "))],\n"
s += " \"seed_derivation\": \"FNV-1a 64 over UTF-8 of seed, basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32\",\n"
s += " \"lanes\": 32,\n"
s += " \"registers\": 8,\n"
s += " \"iterations\": \(Program.iterations),\n"
s += " \"instruction_count\": \(Program.count),\n"
s += " \"loads_per_hash\": \(p.loadsPerHash),\n"
s += " \"op_mix\": {\(p.histogram.map { "\(jstr($0.0)): \($0.1)" }.joined(separator: ", "))},\n"
s += " \"register_init\": \"for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]\",\n"
s += " \"splitmix32\": \"x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16\",\n"
s += " \"iteration\": \"sel = r0 sampled once at the top of each iteration, then all instructions in order\",\n"
s += " \"output\": \"lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo\",\n"
s += " \"op_semantics\": {\n"
s += " \"add\": \"dst = dst + src + (bit `bit` of sel ? imm2 : imm)\",\n"
s += " \"sub\": \"dst = dst - src\",\n"
s += " \"mul\": \"dst = dst * src (low 32)\",\n"
s += " \"mulhi\": \"dst = high 32 bits of dst * src\",\n"
s += " \"xor\": \"dst = dst ^ src\",\n"
s += " \"or\": \"dst = dst | src\",\n"
s += " \"rotl\": \"dst = rotl(dst, rot), rot in 1..31\",\n"
s += " \"rotr\": \"dst = rotr(dst, src & 31)\",\n"
s += " \"mad\": \"dst = src * src2 + dst\",\n"
s += " \"shfl\": \"dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp\",\n"
s += " \"load\": \"dst = dst ^ dataset[src & dataset.mask]\",\n"
s += " \"wload\": \"base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)\"\n"
s += " },\n"
s += " \"dataset\": {\n"
s += " \"log2_words\": \(datasetLog2),\n"
s += " \"bytes\": \(UInt64(1) << UInt64(datasetLog2 + 2)),\n"
s += " \"mask\": \(jhex(mask)),\n"
s += " \"day\": \(jstr(dayString)),\n"
s += " \"day_words_from\": \(jstr("day/" + dayString)),\n"
s += " \"d0\": \(jhex(day.0)),\n"
s += " \"d1\": \(jhex(day.1)),\n"
if let mp = memhard {
s += " \"mode\": \"memory-hard\",\n"
s += " \"spec\": \"proto-metal/MEMHARD.md\",\n"
s += " \"key\": [\(mp.keyWords.map(jhex).joined(separator: ", "))],\n"
s += " \"key_derivation\": \"the 8 words of seedWords(\\\"day/\\\" + day); d0, d1 are key[0], key[1]\",\n"
s += " \"cache\": {\"log2_words\": \(cacheLog2Words), \"bytes\": \(UInt64(cacheWords) * 4), \"line_words\": 16, \"segment_lines\": \(cacheLinesPerSegment), \"segments\": \(cacheSegments), \"block\": \"ChaCha\(chachaRounds) core + feed-forward, rotations 16 12 8 7\", \"sigma\": [\(chachaSigma.map(jhex).joined(separator: ", "))], \"tag\": [\(cacheTag.map(jhex).joined(separator: ", "))], \"chain\": \"in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0\"},\n"
s += " \"mixer\": {\"draw\": \"SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())\", \"rot\": [\(mp.rotWords.map { "\($0)" }.joined(separator: ", "))], \"mul\": [\(mp.mulWords.map(jhex).joined(separator: ", "))], \"rc\": [\(mp.rcWords.map(jhex).joined(separator: ", "))], \"round\": \"for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]\", \"quarter_round\": \"a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)\"},\n"
s += " \"item\": \"s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..\(itemRounds - 1): s = M_r(s); line = s[0] & \(jhex(cacheLineMask)); s[i] ^= cache[line * 16 + i]; then s = M_\(itemRounds)(s); item(t) = s\",\n"
s += " \"word\": \"dataset[w] = item(w >> 4)[w & 15]\"\n"
} else {
s += " \"mode\": \"closed-form\",\n"
s += " \"formula\": \"x = i ^ d0; x *= 0x9E3779B1; x ^= x>>15; x += d1; x *= 0x85EBCA77; x ^= x>>13; x *= 0xC2B2AE3D; x ^= x>>16 (all mod 2^32)\"\n"
}
s += " },\n"
s += " \"instructions\": [\n"
for (k, ins) in p.instrs.enumerated() {
s += " {\"i\": \(k), \"op\": \(jstr(ins.op.rawValue)), \"dst\": \(ins.dst), \"src\": \(ins.a), \"src2\": \(ins.b), \"imm\": \(jhex(ins.imm)), \"imm2\": \(jhex(ins.imm2)), \"rot\": \(ins.rot), \"bit\": \(ins.bit), \"mask\": \(ins.mask)}"
s += k == p.instrs.count - 1 ? "\n" : ",\n"
}
s += " ]\n}\n"
return s
}
func generateVectorsJSON(_ p: Program, dayString: String, datasetLog2: Int, bases: [UInt32], outs: [[UInt64]], v: PackVectors, mask: UInt32, source: String, memhard: Bool) -> String {
var s = "{\n"
s += " \"seed\": \(jstr(p.seedString)),\n"
s += " \"day\": \(jstr(dayString)),\n"
s += " \"dataset_mode\": \(jstr(memhard ? "memory-hard" : "closed-form")),\n"
s += " \"dataset_log2_words\": \(datasetLog2),\n"
s += " \"mask\": \(jhex(mask)),\n"
s += " \"lanes\": 32,\n"
s += " \"source\": \(jstr(source)),\n"
s += " \"warps\": [\n"
for (i, o) in outs.enumerated() {
s += " {\"base_nonce\": \(bases[i]), \"expected\": [\n"
for row in 0..<4 {
s += " " + (0..<8).map { jhex64(o[row * 8 + $0]) }.joined(separator: ", ") + (row == 3 ? "\n" : ",\n")
}
s += i == outs.count - 1 ? " ]}\n" : " ]},\n"
}
s += " ],\n"
s += " \"dataset_head\": [\(v.head.map(jhex).joined(separator: ", "))],\n"
s += " \"dataset_last_index\": \(mask),\n"
s += " \"dataset_last\": \(jhex(v.last)),\n"
s += " \"dataset_samples\": [\(zip(v.sampleIdx, v.sampleVal).map { "{\"index\": \($0.0), \"value\": \(jhex($0.1))}" }.joined(separator: ", "))]"
if memhard {
s += ",\n \"cache_head\": [\(v.cacheHead.map(jhex).joined(separator: ", "))],\n"
s += " \"cache_last_line\": [\(v.cacheLast.map(jhex).joined(separator: ", "))],\n"
s += " \"cache_fnv1a64\": \(jhex64(v.cacheFNV))\n"
} else { s += "\n" }
s += "}\n"
return s
}
// Runs the Metal kernel for each base nonce (one 32-thread threadgroup each) and compares with `expected`.
// The dataset comes from the context (closed form or memory-hard), so the GPU build path is covered too.
func metalCrossCheck(_ ctx: DatasetContext, _ p: Program, datasetLog2: Int, bases: [UInt32], expected: [[UInt64]]) -> (ok: Bool, detail: String) {
let dataset = ctx.makeDataset(log2: datasetLog2)
do {
let k = try compileHash(ctx.gpu, msl: generateMSL(p, datasetLog2: datasetLog2))
guard let got = gpuWarps(ctx.gpu, k, dataset: dataset, bases: bases) else { return (false, "GPU run failed") }
var bad = [String]()
for (i, base) in bases.enumerated() where got[i] != expected[i] { bad.append("base \(base)") }
return (bad.isEmpty, bad.isEmpty ? "Metal GPU cross-check PASS \(bases.count)/\(bases.count) warps" : "Metal GPU cross-check FAIL: \(bad.joined(separator: ", "))")
} catch {
return (false, "Metal compile error \(error)")
}
}
func exportPack(_ opts: Options) -> Never {
let dir = opts.exportPack!
let program = generateProgram(seedString: opts.seed)
let gpu = GPU()
let ctx = DatasetContext(gpu: gpu, closedForm: opts.closedForm, dayString: opts.day)
let day = ctx.day
let mask = UInt32((1 << opts.datasetLog2) - 1)
print("igneum-bench --export-pack \(dir)")
print("seed \"\(opts.seed)\", day \"\(opts.day)\", dataset 2^\(opts.datasetLog2) words (\(ctx.modeName)), loads/hash \(program.loadsPerHash), wide loads/hash \(program.wideLoadsPerHash)")
print("op mix: " + program.histogram.map { "\($0.0)=\($0.1)" }.joined(separator: " "))
if !ctx.closed { print("cache: GPU fill \(fmt(ctx.cacheFillGPUms, 2)) ms GPU time; CPU fill \(fmt(ctx.cpuSide()!.fillMs, 1)) ms one core") }
let ds = ctx.source(log2: opts.datasetLog2)
let outs = packVectorBases.map { cpuWarp(program, baseNonce: $0, ds: ds) }
var v = PackVectors(head: (0..<16).map { ds.word(UInt32($0)) }, last: ds.word(mask), sampleIdx: [], sampleVal: [],
cacheHead: [], cacheLast: [], cacheFNV: 0)
var sr = SplitMix64(s: 0x6d68_7361_6d70_6c65) // "mhsample"
for _ in 0..<64 { let i = UInt32(truncatingIfNeeded: sr.next()) & mask; v.sampleIdx.append(i); v.sampleVal.append(ds.word(i)) }
if let cpu = ctx.cpuSide() {
v.cacheHead = (0..<16).map { cpu.cache[$0] }
v.cacheLast = (0..<16).map { cpu.cache[cacheWords - 16 + $0] }
v.cacheFNV = fnv64(UnsafeRawPointer(cpu.cache), cacheWords * 4)
let cc = ctx.cacheCheck()
print(cc.detail)
if !cc.ok { print("FAIL: GPU cache differs from the CPU cache, pack not written"); exit(1) }
}
var source = "proto-metal CPU interpreter (cpuWarp, \(ctx.modeName) dataset) on Apple M5 Max"
let check = metalCrossCheck(ctx, program, datasetLog2: opts.datasetLog2, bases: packVectorBases, expected: outs)
print(check.detail)
source += "; " + check.detail
if !check.ok { print("FAIL: vectors do not match the Metal GPU, pack not written"); exit(1) }
// The sampled dataset words against the GPU-built dataset as well.
let sampleCheck = ctx.sampleCheck(log2: opts.datasetLog2, indices: v.sampleIdx + [0, mask])
print(sampleCheck.detail)
if !sampleCheck.ok { print("FAIL: GPU dataset words differ from the CPU derivation, pack not written"); exit(1) }
var files: [(String, String)] = [
("program.json", generateProgramJSON(program, dayString: opts.day, day: day, datasetLog2: opts.datasetLog2, memhard: ctx.mp)),
("vectors.json", generateVectorsJSON(program, dayString: opts.day, datasetLog2: opts.datasetLog2, bases: packVectorBases, outs: outs, v: v, mask: mask, source: source, memhard: !ctx.closed)),
("kernel.cu", generateCUDA(program, memhard: ctx.mp)),
("kernel.cl", generateOpenCL(program, memhard: ctx.mp)),
("program.h", generateProgramHeader(program, dayString: opts.day, day: day, datasetLog2: opts.datasetLog2, memhard: ctx.mp)),
("vectors.h", generateVectorsHeader(program, bases: packVectorBases, outs: outs, v: v, mask: mask, source: source, memhard: !ctx.closed)),
("program.metal", generateMSL(program, datasetLog2: opts.datasetLog2)),
]
if let mp = ctx.mp {
files.append(("memhard.h", generateMemhardHeader(program, mp)))
files.append(("memhard.metal", memhardMSL(mp)))
}
do {
try FileManager.default.createDirectory(atPath: dir, withIntermediateDirectories: true)
for (name, text) in files {
try text.write(toFile: "\(dir)/\(name)", atomically: true, encoding: .utf8)
print("wrote \(dir)/\(name) (\(text.utf8.count) bytes)")
}
} catch {
print("FAIL: write error \(error)"); exit(1)
}
for (i, b) in packVectorBases.enumerated() {
print("vector warp base \(b): lane0 \(String(format: "%016llx", outs[i][0])) lane31 \(String(format: "%016llx", outs[i][31]))")
}
print("OVERALL: PASS (pack written)")
exit(0)
}
// MARK: - Timing
@inline(__always) func nowNs() -> UInt64 { clock_gettime_nsec_np(CLOCK_UPTIME_RAW) }
func ms(_ a: UInt64, _ b: UInt64) -> Double { Double(b - a) / 1e6 }
func fmt(_ v: Double, _ digits: Int = 2) -> String { String(format: "%.\(digits)f", v) }
// MARK: - GPU context
final class GPU {
let device: MTLDevice
let queue: MTLCommandQueue
init() {
guard let d = MTLCreateSystemDefaultDevice(), let q = d.makeCommandQueue() else {
print("FAIL: no Metal device"); exit(1)
}
device = d; queue = q
}
}
// Everything about the dataset for one day: which construction, the GPU kernels that build it, the GPU cache
// (memory-hard mode), and the CPU side the verifier uses. Tests and the bench share one of these.
final class DatasetContext {
let gpu: GPU
let closed: Bool
let dayString: String
let key: [UInt32] // the 8 words of seedWords("day/" + day)
var day: (UInt32, UInt32) { (key[0], key[1]) }
let mp: MixParams? // memory-hard mixer parameters (nil in closed-form mode)
var modeName: String { closed ? "closed-form" : "memory-hard" }
private var closedFillPipe: MTLComputePipelineState?
private var cacheFillPipe: MTLComputePipelineState?
private var buildPipe: MTLComputePipelineState?
var compileMs = 0.0
var gpuCache: MTLBuffer? // 2^26 words, private
var cacheFillGPUms = 0.0, cacheFillWallMs = 0.0
var lastBuildGPUms = 0.0, lastBuildWallMs = 0.0
private var cpu: MemhardCPU?
convenience init(gpu: GPU, closedForm: Bool, dayString: String) {
self.init(gpu: gpu, closedForm: closedForm, dayString: dayString, key: seedWords("day/" + dayString))
}
// From already-derived key words (serve mode: seed_words_from_bytes of the day seed bytes the miner sends).
init(gpu: GPU, closedForm: Bool, dayString: String, key: [UInt32]) {
self.gpu = gpu; closed = closedForm; self.dayString = dayString
self.key = key
mp = closedForm ? nil : MixParams(key: key)
let t0 = nowNs()
do {
if closedForm {
let lib = try gpu.device.makeLibrary(source: fillMSL, options: MTLCompileOptions())
closedFillPipe = try gpu.device.makeComputePipelineState(function: lib.makeFunction(name: "igneum_fill")!)
} else {
let lib = try gpu.device.makeLibrary(source: memhardMSL(mp!), options: MTLCompileOptions())
cacheFillPipe = try gpu.device.makeComputePipelineState(function: lib.makeFunction(name: "igneum_cache_fill")!)
buildPipe = try gpu.device.makeComputePipelineState(function: lib.makeFunction(name: "igneum_build")!)
}
} catch { print("FAIL: dataset kernel compile: \(error)"); exit(1) }
compileMs = ms(t0, nowNs())
if !closedForm {
guard let c = gpu.device.makeBuffer(length: cacheWords * 4, options: .storageModePrivate) else { print("FAIL: cannot allocate the cache"); exit(1) }
gpuCache = c
let cb = gpu.queue.makeCommandBuffer()!
let enc = cb.makeComputeCommandEncoder()!
enc.setComputePipelineState(cacheFillPipe!)
enc.setBuffer(c, offset: 0, index: 0)
enc.dispatchThreadgroups(MTLSize(width: cacheSegments / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
enc.endEncoding()
let w0 = nowNs()
cb.commit(); cb.waitUntilCompleted()
cacheFillWallMs = ms(w0, nowNs())
if let e = cb.error { print("FAIL: cache fill error \(e)"); exit(1) }
cacheFillGPUms = (cb.gpuEndTime - cb.gpuStartTime) * 1000
}
}
// Fills the GPU cache again (for repeat timings). Returns GPU ms.
func refillCache() -> Double {
guard let c = gpuCache else { return 0 }
let cb = gpu.queue.makeCommandBuffer()!
let enc = cb.makeComputeCommandEncoder()!
enc.setComputePipelineState(cacheFillPipe!)
enc.setBuffer(c, offset: 0, index: 0)
enc.dispatchThreadgroups(MTLSize(width: cacheSegments / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
enc.endEncoding()
cb.commit(); cb.waitUntilCompleted()
return (cb.gpuEndTime - cb.gpuStartTime) * 1000
}
// Allocates a private 2^log2-word dataset and builds it on the GPU. Timing lands in lastBuild*.
func makeDataset(log2: Int) -> MTLBuffer {
let words = 1 << log2
guard let buf = gpu.device.makeBuffer(length: words * 4, options: .storageModePrivate) else {
print("FAIL: cannot allocate 2^\(log2) word dataset"); exit(1)
}
build(into: buf, words: words)
return buf
}
func build(into buf: MTLBuffer, words: Int) {
let cb = gpu.queue.makeCommandBuffer()!
let enc = cb.makeComputeCommandEncoder()!
if closed {
enc.setComputePipelineState(closedFillPipe!)
enc.setBuffer(buf, offset: 0, index: 0)
var d = day
enc.setBytes(&d, length: 8, index: 1)
enc.dispatchThreadgroups(MTLSize(width: words / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
} else {
enc.setComputePipelineState(buildPipe!)
enc.setBuffer(gpuCache!, offset: 0, index: 0)
enc.setBuffer(buf, offset: 0, index: 1)
let items = words / 16
enc.dispatchThreadgroups(MTLSize(width: max(items / 256, 1), height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: min(items, 256), height: 1, depth: 1))
}
enc.endEncoding()
let w0 = nowNs()
cb.commit(); cb.waitUntilCompleted()
lastBuildWallMs = ms(w0, nowNs())
if let e = cb.error { print("FAIL: dataset build error \(e)"); exit(1) }
lastBuildGPUms = (cb.gpuEndTime - cb.gpuStartTime) * 1000
}
// The CPU verifier's side: the 256 MiB cache computed on one core, once per process. Prints the fill time.
func cpuSide() -> MemhardCPU? {
if closed { return nil }
if let c = cpu { return c }
let c = MemhardCPU(key: key)
print("cache: CPU fill \(fmt(c.fillMs, 1)) ms on one core (2^\(cacheLog2Words) words, \(cacheSegments) chains of \(cacheLinesPerSegment) ChaCha\(chachaRounds) blocks)")
cpu = c
return c
}
func source(log2: Int) -> DatasetSource {
DatasetSource(mask: UInt32((1 << log2) - 1), day: day, memhard: cpuSide())
}
// The buffer the inline (shortcut) kernel binds at index 0: the cache in memory-hard mode, the dataset otherwise.
func inlineBuffer(dataset: MTLBuffer) -> MTLBuffer { closed ? dataset : gpuCache! }
var inlineSource: LoadSource { closed ? .inlineClosed(day.0, day.1) : .inlineMemhard(mp!) }
// Blit the GPU cache to shared memory and compare every word with the CPU cache.
func cacheCheck() -> (ok: Bool, detail: String) {
guard let gc = gpuCache, let c = cpuSide() else { return (true, "cache check: not applicable (closed form)") }
guard let shared = gpu.device.makeBuffer(length: cacheWords * 4, options: .storageModeShared) else { return (false, "cache check: no shared buffer") }
let cb = gpu.queue.makeCommandBuffer()!
let blit = cb.makeBlitCommandEncoder()!
blit.copy(from: gc, sourceOffset: 0, to: shared, destinationOffset: 0, size: cacheWords * 4)
blit.endEncoding()
cb.commit(); cb.waitUntilCompleted()
let same = memcmp(shared.contents(), c.cache, cacheWords * 4) == 0
let fp = fnv64(shared.contents(), cacheWords * 4)
return (same, "cache check: GPU cache \(same ? "==" : "!=") CPU cache, all \(cacheWords) words compared, FNV-1a 64 \(h64(fp))")
}
// Reads the given words of a GPU-built dataset back and compares with the CPU derivation.
func sampleCheck(log2: Int, indices: [UInt32]) -> (ok: Bool, detail: String) {
let words = 1 << log2
let dataset = makeDataset(log2: log2)
guard let shared = gpu.device.makeBuffer(length: words * 4, options: .storageModeShared) else { return (false, "sample check: no shared buffer") }
let cb = gpu.queue.makeCommandBuffer()!
let blit = cb.makeBlitCommandEncoder()!
blit.copy(from: dataset, sourceOffset: 0, to: shared, destinationOffset: 0, size: words * 4)
blit.endEncoding()
cb.commit(); cb.waitUntilCompleted()
let p = shared.contents().bindMemory(to: UInt32.self, capacity: words)
let ds = source(log2: log2)
var bad = 0
for i in indices where p[Int(i)] != ds.word(i) { bad += 1 }
return (bad == 0, "dataset sample check (\(modeName), 2^\(log2) words): \(indices.count) GPU words vs CPU derivation, \(bad) mismatches")
}
}
struct EpochResult {
var seed: String
var libraryMs: Double
var pipelineMs: Double
var hashesPerSecWall: Double
var hashesPerSecGPU: Double
var gbpsWall: Double
var gbpsGPU: Double
var loadsPerHash: Int
var itemsPerWarp: Int
var verify: [(warp: Int, pass: Bool, ms: Double, repMs: Double)]
var allPass: Bool { verify.allSatisfy { $0.pass } }
}
func runEpoch(gpu: GPU, opts: Options, seedString: String, dataset: MTLBuffer, ctx: DatasetContext) -> EpochResult {
let program = generateProgram(seedString: seedString)
let msl = generateMSL(program, datasetLog2: opts.datasetLog2, source: opts.inlineDataset ? ctx.inlineSource : .stored)
if opts.inlineDataset {
print(ctx.closed ? "\nNOTE: --inline-dataset: loads compute ds_elem inline, the dataset buffer is never read"
: "\nNOTE: --inline-dataset: loads derive the item from the 256 MiB cache (8 dependent 64-byte reads + 9 mixers), the dataset buffer is never read")
}
if let dir = opts.dumpDir {
try? FileManager.default.createDirectory(atPath: dir, withIntermediateDirectories: true)
let safe = seedString.replacingOccurrences(of: "/", with: "_")
try? msl.write(toFile: "\(dir)/program-\(safe).metal", atomically: true, encoding: .utf8)
}
print("\n=== epoch seed \"\(seedString)\" ===")
print("program: \(Program.count) instructions x \(Program.iterations) iterations, loads/hash = \(program.loadsPerHash) (wide \(program.wideLoadsPerHash)), distinct items per warp = \(program.itemsPerWarp)")
print("op mix: " + program.histogram.map { "\($0.0)=\($0.1)" }.joined(separator: " "))
// Runtime compile
let t0 = nowNs()
let library: MTLLibrary
do {
let copts = MTLCompileOptions()
library = try gpu.device.makeLibrary(source: msl, options: copts)
} catch {
print("FAIL: Metal compile error:\n\(error)")
exit(1)
}
let t1 = nowNs()
guard let fn = library.makeFunction(name: "igneum_hash") else { print("FAIL: no igneum_hash"); exit(1) }
let pipeline: MTLComputePipelineState
do { pipeline = try gpu.device.makeComputePipelineState(function: fn) } catch {
print("FAIL: pipeline error: \(error)"); exit(1)
}
let t2 = nowNs()
let libMs = ms(t0, t1), pipeMs = ms(t1, t2)
print("compile: library \(fmt(libMs)) ms, pipeline \(fmt(pipeMs)) ms, total \(fmt(libMs + pipeMs)) ms")
print("threadExecutionWidth = \(pipeline.threadExecutionWidth), maxTotalThreadsPerThreadgroup = \(pipeline.maxTotalThreadsPerThreadgroup)")
if pipeline.threadExecutionWidth != 32 {
print("WARNING: threadExecutionWidth is not 32; the one-warp-per-threadgroup assumption does not hold on this device")
}
// Buffers
let n = 1 << opts.batchLog2
let groups = n / 32
guard let outBuf = gpu.device.makeBuffer(length: n * 8, options: .storageModeShared) else { print("FAIL: out buffer"); exit(1) }
let buffer0 = opts.inlineDataset ? ctx.inlineBuffer(dataset: dataset) : dataset
func encodeBatch(_ cb: MTLCommandBuffer, base: UInt32) {
let enc = cb.makeComputeCommandEncoder()!
enc.setComputePipelineState(pipeline)
enc.setBuffer(buffer0, offset: 0, index: 0)
enc.setBuffer(outBuf, offset: 0, index: 1)
var b = base
enc.setBytes(&b, length: 4, index: 2)
enc.dispatchThreadgroups(MTLSize(width: groups, height: 1, depth: 1),
threadsPerThreadgroup: MTLSize(width: 32, height: 1, depth: 1))
enc.endEncoding()
}
// Batch 0: warm-up and verification source (baseNonce 0)
do {
let cb = gpu.queue.makeCommandBuffer()!
encodeBatch(cb, base: 0)
let w0 = nowNs()
cb.commit(); cb.waitUntilCompleted()
let w1 = nowNs()
if let e = cb.error { print("FAIL: batch 0 error \(e)"); exit(1) }
print("warm-up batch: \(n) hashes in \(fmt(ms(w0, w1))) ms wall, \(fmt((cb.gpuEndTime - cb.gpuStartTime) * 1000)) ms GPU")
}
// Pick warps to verify from batch 0
var warps = [0, groups / 2 + 1, groups - 1]
var vr = SplitMix64(s: UInt64(program.seed[4]) | (UInt64(program.seed[5]) << 32))
while warps.count < opts.verifyWarps { warps.append(vr.below(groups)) }
warps = Array(warps.prefix(max(opts.verifyWarps, 1)))
let outPtr = outBuf.contents().bindMemory(to: UInt64.self, capacity: n)
var gpuOutputs = [[UInt64]]()
for w in warps { gpuOutputs.append((0..<32).map { outPtr[w * 32 + $0] }) }
// Timed batches
var cbs = [MTLCommandBuffer]()
for b in 0..<opts.batches {
let cb = gpu.queue.makeCommandBuffer()!
encodeBatch(cb, base: UInt32(truncatingIfNeeded: (b + 1) * n))
cbs.append(cb)
}
let s0 = nowNs()
for cb in cbs { cb.commit() }
cbs.last!.waitUntilCompleted()
let s1 = nowNs()
for cb in cbs { if let e = cb.error { print("FAIL: batch error \(e)"); exit(1) } }
let gpuSeconds = cbs.reduce(0.0) { $0 + ($1.gpuEndTime - $1.gpuStartTime) }
let wallSeconds = Double(s1 - s0) / 1e9
let totalHashes = Double(n * opts.batches)
let hpsWall = totalHashes / wallSeconds
let hpsGPU = totalHashes / gpuSeconds
let bytesPerHash = Double(program.loadsPerHash * 4)
let gbpsWall = hpsWall * bytesPerHash / 1e9
let gbpsGPU = hpsGPU * bytesPerHash / 1e9
print("timed: \(opts.batches) batches x \(n) hashes = \(Int(totalHashes)) hashes")
print(" wall \(fmt(wallSeconds * 1000)) ms -> \(fmt(hpsWall / 1e6, 3)) Mhash/s, \(fmt(gbpsWall)) GB/s useful (loads x 4 B)")
print(" GPU \(fmt(gpuSeconds * 1000)) ms -> \(fmt(hpsGPU / 1e6, 3)) Mhash/s, \(fmt(gbpsGPU)) GB/s useful (loads x 4 B)")
// CPU verification. The verifier holds the 256 MiB cache (memory-hard mode) or nothing (closed form), never
// the dataset; every dataset word a load needs is derived on demand inside cpuWarp.
let ds = ctx.source(log2: opts.datasetLog2)
var verify = [(warp: Int, pass: Bool, ms: Double, repMs: Double)]()
for (i, w) in warps.enumerated() {
let base = UInt32(w * 32)
let d0 = ds.memhard?.derivations ?? 0
let c0 = nowNs()
let cpu = cpuWarp(program, baseNonce: base, ds: ds)
let c1 = nowNs()
let derived = (ds.memhard?.derivations ?? 0) - d0
// repeated runs for a steadier figure
let reps = 20
let r0 = nowNs()
var sink: UInt64 = 0
for _ in 0..<reps { sink ^= cpuWarp(program, baseNonce: base, ds: ds)[0] }
let r1 = nowNs()
let pass = cpu == gpuOutputs[i] && sink != 1
let single = ms(c0, c1), rep = ms(r0, r1) / Double(reps)
verify.append((w, pass, single, rep))
var detail = ""
if !pass {
let bad = (0..<32).filter { cpu[$0] != gpuOutputs[i][$0] }
detail = " mismatched lanes: \(bad) first: cpu=\(String(format: "%016llx", cpu[bad.first ?? 0])) gpu=\(String(format: "%016llx", gpuOutputs[i][bad.first ?? 0]))"
}
let items = ds.memhard == nil ? "" : ", \(derived) items derived"
print("verify warp \(w) (nonces \(base)..\(base + 31)): \(pass ? "PASS" : "FAIL") cpu \(fmt(single, 3)) ms single, \(fmt(rep, 3)) ms avg of \(reps)\(items)\(detail)")
}
return EpochResult(seed: seedString, libraryMs: libMs, pipelineMs: pipeMs,
hashesPerSecWall: hpsWall, hashesPerSecGPU: hpsGPU, gbpsWall: gbpsWall, gbpsGPU: gbpsGPU,
loadsPerHash: program.loadsPerHash, itemsPerWarp: program.itemsPerWarp, verify: verify)
}
// MARK: - Hardening tests: shared helpers
//
// Added 3 October 2026. Everything below reuses generateProgram, generateMSL, fillMSL and cpuWarp
// unchanged; the helpers only wrap compile, fill and dispatch so the tests can run many programs
// and many warps cheaply. Nothing in the bench path (runEpoch) or the pack exporter calls these.
struct CompiledHash {
let pipeline: MTLComputePipelineState
let libraryMs: Double
let pipelineMs: Double
var groupWidth = 32 // threads per threadgroup at dispatch (the variant's)
var variant = "base"
var totalMs: Double { libraryMs + pipelineMs }
}
struct IgneumError: Error, CustomStringConvertible {
let description: String
init(_ s: String) { description = s }
}
func compileHash(_ gpu: GPU, msl: String) throws -> CompiledHash {
let t0 = nowNs()
let lib = try gpu.device.makeLibrary(source: msl, options: MTLCompileOptions())
let t1 = nowNs()
guard let fn = lib.makeFunction(name: "igneum_hash") else { throw IgneumError("no igneum_hash function in library") }
let pipe = try gpu.device.makeComputePipelineState(function: fn)
let t2 = nowNs()
return CompiledHash(pipeline: pipe, libraryMs: ms(t0, t1), pipelineMs: ms(t1, t2))
}
// Dataset allocation and build moved into DatasetContext.makeDataset (3 October 2026), which handles both constructions.
// One 32-thread threadgroup per base nonce, all dispatched from one encoder. Warp i lands at byte offset
// i * 256 of the output buffer, which is pre-filled with a sentinel so an unwritten lane is visible.
func gpuWarps(_ gpu: GPU, _ k: CompiledHash, dataset: MTLBuffer, bases: [UInt32]) -> [[UInt64]]? {
guard !bases.isEmpty, let outBuf = gpu.device.makeBuffer(length: bases.count * 256, options: .storageModeShared) else { return nil }
memset(outBuf.contents(), 0xAA, outBuf.length)
let cb = gpu.queue.makeCommandBuffer()!
let enc = cb.makeComputeCommandEncoder()!
enc.setComputePipelineState(k.pipeline)
enc.setBuffer(dataset, offset: 0, index: 0)
for (i, base) in bases.enumerated() {
enc.setBuffer(outBuf, offset: i * 256, index: 1)
var b = base
enc.setBytes(&b, length: 4, index: 2)
enc.dispatchThreadgroups(MTLSize(width: 1, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 32, height: 1, depth: 1))
}
enc.endEncoding()
cb.commit(); cb.waitUntilCompleted()
if cb.error != nil { return nil }
let p = outBuf.contents().bindMemory(to: UInt64.self, capacity: bases.count * 32)
return (0..<bases.count).map { i in (0..<32).map { p[i * 32 + $0] } }
}
// `count` consecutive nonces from `base` (count a multiple of 32) into `out`. Returns GPU time in ms.
func gpuRange(_ gpu: GPU, _ k: CompiledHash, dataset: MTLBuffer, base: UInt32, count: Int, out: MTLBuffer) -> Double? {
let cb = gpu.queue.makeCommandBuffer()!
let enc = cb.makeComputeCommandEncoder()!
enc.setComputePipelineState(k.pipeline)
enc.setBuffer(dataset, offset: 0, index: 0)
enc.setBuffer(out, offset: 0, index: 1)
var b = base
enc.setBytes(&b, length: 4, index: 2)
enc.dispatchThreadgroups(MTLSize(width: count / 32, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 32, height: 1, depth: 1))
enc.endEncoding()
cb.commit(); cb.waitUntilCompleted()
if cb.error != nil { return nil }
return (cb.gpuEndTime - cb.gpuStartTime) * 1000
}
func h64(_ v: UInt64) -> String { String(format: "%016llx", v) }
func pad(_ s: String, _ n: Int) -> String { s.count >= n ? s : s + String(repeating: " ", count: n - s.count) }
func describeProgram(_ p: Program) -> String {
var s = " program seed \"\(p.seedString)\" words [\(p.seed.map(hex).joined(separator: ", "))], \(p.instrs.count) instructions\n"
for (k, i) in p.instrs.enumerated() {
s += " \(pad(String(k), 3)) \(pad(i.op.rawValue, 5)) dst=r\(i.dst) src=r\(i.a) src2=r\(i.b) imm=\(hex(i.imm)) imm2=\(hex(i.imm2)) rot=\(i.rot) bit=\(i.bit) mask=\(i.mask)\n"
}
return s
}
func fnv64(_ ptr: UnsafeRawPointer, _ count: Int) -> UInt64 {
var h: UInt64 = 0xcbf29ce484222325
let b = ptr.bindMemory(to: UInt8.self, capacity: count)
for i in 0..<count { h ^= UInt64(b[i]); h &*= 0x100000001b3 }
return h
}
func regexCount(_ pattern: String, in text: String) -> Int {
let re = try! NSRegularExpression(pattern: pattern)
return re.numberOfMatches(in: text, range: NSRange(text.startIndex..., in: text))
}
// Static check: every dataset access in the generated MSL is `dataset[rN & MASK]`, and the identifier
// `dataset` appears nowhere else except the kernel parameter.
// A wide load (lever b) is `dataset[(simd_broadcast(rN, 0) & WMASK) + lane]` with WMASK = MASK & ~31 and lane < 32,
// so its index is at most MASK as well.
func maskCheckMSL(_ msl: String) -> (ok: Bool, detail: String) {
let total = regexCount("dataset\\[", in: msl)
let masked = regexCount("dataset\\[r[0-7] & MASK\\]", in: msl)
let wide = regexCount("dataset\\[\\(simd_broadcast\\(r[0-7], 0\\) & WMASK\\) \\+ lane\\]", in: msl)
let words = regexCount("\\bdataset\\b", in: msl)
let ok = total == masked + wide && words == total + 1
return (ok, "MSL: \(total) dataset[ accesses, \(masked) of the form dataset[rN & MASK], \(wide) wide loads dataset[(simd_broadcast(rN, 0) & WMASK) + lane], identifier appears \(words) times (expected \(total + 1))")
}
// Same for the CUDA twin: hash accesses are `ds[rN & mask]`; the fill kernel's one write is guarded by `if (i < n)`.
// The closed-form pack has one fill write guarded by `if (i < n)`; the memory-hard pack writes the dataset through a
// `d` pointer in igneum_build and has no `ds[` write at all.
func maskCheckCUDA(_ cu: String, memhard: Bool) -> (ok: Bool, detail: String) {
let total = regexCount("\\bds\\[", in: cu)
let masked = regexCount("\\bds\\[r[0-7] & mask\\]", in: cu)
let wide = regexCount("\\bds\\[\\(__shfl_sync\\(0xffffffffu, r[0-7], 0\\) & wmask\\) \\+ lane\\]", in: cu)
let fill = regexCount("if \\(i < n\\) ds\\[i\\] = ds_elem", in: cu)
let ok = total == masked + wide + fill && fill == (memhard ? 0 : 1)
return (ok, "CUDA: \(total) ds[ accesses, \(masked) of the form ds[rN & mask], \(wide) wide, \(fill) guarded fill write (expected \(memhard ? 0 : 1))")
}
// MARK: - --fuzz
func runFuzz(_ opts: Options, ctx: DatasetContext) -> Bool {
let gpu = ctx.gpu
let n = max(opts.fuzz ?? 200, 1)
let master = opts.fuzzSeed
print("\n=== fuzz: \(n) random programs, master seed \"\(master)\", 4 random warps each ===")
let sizes = [24, 26, 28]
let d0 = nowNs()
var datasets = [Int: MTLBuffer]()
for s in sizes { datasets[s] = ctx.makeDataset(log2: s) }
print("datasets " + sizes.map { "2^\($0) (\((1 << $0) * 4 / (1 << 20)) MiB)" }.joined(separator: ", ") + " filled in \(fmt(ms(d0, nowNs()), 1)) ms")
let mw = seedWords("fuzz/" + master)
var rng = SplitMix64(s: UInt64(mw[0]) | (UInt64(mw[1]) << 32))
var pass = 0, fail = 0, compileFail = 0, staticFail = 0, contractFail = 0
var perSize = [Int: (pass: Int, fail: Int)]()
var compileMs = [Double]()
var cpuNs: UInt64 = 0, gpuNs: UInt64 = 0
var warps = 0
var opCount = [String: Int]()
var loadsMin = Int.max, loadsMax = 0
let t0 = nowNs()
for i in 0..<n {
let seedString = "\(master)/\(i)/\(h64(rng.next()))"
let log2 = sizes[rng.below(sizes.count)]
let bases = (0..<4).map { _ in UInt32(truncatingIfNeeded: rng.next()) }
let program = generateProgram(seedString: seedString)
for ins in program.instrs {
opCount[ins.op.rawValue, default: 0] += 1
// Generator contract, relied on by the MSL emitter: rotl amount 1..31, shuffle mask a power of two <= 16,
// source register never the destination.
if ins.rot < 1 || ins.rot > 31 || ![1, 2, 4, 8, 16].contains(ins.mask) || ins.a == ins.dst || ins.dst > 7 || ins.a > 7 || ins.b > 7 {
contractFail += 1
print("CONTRACT FAIL seed \"\(seedString)\": \(ins)")
}
}
loadsMin = min(loadsMin, program.loadsPerHash); loadsMax = max(loadsMax, program.loadsPerHash)
let msl = generateMSL(program, datasetLog2: log2)
let sc = maskCheckMSL(msl)
if !sc.ok { staticFail += 1; print("STATIC MASK FAIL seed \"\(seedString)\": \(sc.detail)") }
let k: CompiledHash
do { k = try compileHash(gpu, msl: msl) } catch {
compileFail += 1
print("COMPILE FAIL seed \"\(seedString)\" dataset 2^\(log2):\n\(error)\n\(describeProgram(program))")
continue
}
compileMs.append(k.totalMs)
let g0 = nowNs()
guard let gpuOut = gpuWarps(gpu, k, dataset: datasets[log2]!, bases: bases) else {
fail += 1; print("GPU RUN FAIL seed \"\(seedString)\" dataset 2^\(log2)"); continue
}
let g1 = nowNs()
let ds = ctx.source(log2: log2)
var ok = true
for (w, base) in bases.enumerated() {
let cpu = cpuWarp(program, baseNonce: base, ds: ds)
warps += 1
if cpu != gpuOut[w] {
ok = false
let bad = (0..<32).filter { cpu[$0] != gpuOut[w][$0] }
print("MISMATCH seed \"\(seedString)\" dataset 2^\(log2) warp \(w) base nonce \(base) (\(hex(base))) lanes \(bad)")
for l in bad { print(" lane \(l) nonce \(base &+ UInt32(l)): gpu \(h64(gpuOut[w][l])) cpu \(h64(cpu[l]))") }
print(describeProgram(program))
}
}
let g2 = nowNs()
gpuNs += g1 - g0; cpuNs += g2 - g1
if ok { pass += 1 } else { fail += 1 }
var ps = perSize[log2] ?? (0, 0)
if ok { ps.pass += 1 } else { ps.fail += 1 }
perSize[log2] = ps
if (i + 1) % 100 == 0 || i + 1 == n {
print(" \(i + 1)/\(n): pass \(pass) fail \(fail) compile-fail \(compileFail), \(fmt(Double(nowNs() - t0) / 1e9, 1)) s elapsed")
}
}
let total = Double(nowNs() - t0) / 1e9
let cAvg = compileMs.isEmpty ? 0 : compileMs.reduce(0, +) / Double(compileMs.count)
print("\n| Dataset | Programs | Pass | Fail |")
print("|---|---|---|---|")
for s in sizes {
let ps = perSize[s] ?? (0, 0)
print("| 2^\(s) words (\((1 << s) * 4 / (1 << 20)) MiB) | \(ps.pass + ps.fail) | \(ps.pass) | \(ps.fail) |")
}
print("| all | \(pass + fail) | \(pass) | \(fail) |")
print("programs \(n): pass \(pass), mismatch \(fail), compile failures \(compileFail), static mask failures \(staticFail), generator contract failures \(contractFail)")
print("warps compared \(warps) (\(warps * 32) hashes), loads/hash range \(loadsMin)..\(loadsMax)")
print("op totals over all programs: " + opCount.sorted { $0.value != $1.value ? $0.value > $1.value : $0.key < $1.key }.map { "\($0.key)=\($0.value)" }.joined(separator: " "))
print("compile ms (library+pipeline): min \(fmt(compileMs.min() ?? 0, 1)) avg \(fmt(cAvg, 1)) max \(fmt(compileMs.max() ?? 0, 1)); GPU dispatch total \(fmt(Double(gpuNs) / 1e6, 1)) ms; CPU interpreter total \(fmt(Double(cpuNs) / 1e6, 1)) ms; wall \(fmt(total, 1)) s")
let ok = fail == 0 && compileFail == 0 && staticFail == 0 && contractFail == 0 && pass == n
print("FUZZ: \(ok ? "PASS" : "FAIL")")
return ok
}
// MARK: - --edge
struct EdgeCase {
let name: String
let instrs: [Instr]
// (instruction index, what must hold, check on lane-0 registers as they are just before that instruction)
let pre: [(Int, String, ([UInt32]) -> Bool)]
let informational: Bool // reported but not counted: exercises something the generator never emits
}
// Instruction builder for hand-made programs. imm2 = imm so the `add` is a constant regardless of the selector bit.
func I(_ op: Op, _ dst: Int, _ a: Int, b: Int = 0, imm: UInt32 = 0, rot: UInt32 = 1, bit: Int = 0, mask: Int = 1) -> Instr {
Instr(op: op, dst: dst, a: a, b: b, imm: imm, imm2: imm, rot: rot, bit: bit, mask: mask)
}
func zero(_ r: Int) -> Instr { I(.sub, r, r) } // r = r - r = 0 (src == dst, never generated, legal MSL)
func set(_ r: Int, _ v: UInt32) -> [Instr] { [zero(r), I(.add, r, 7, imm: v)] } // needs r7 == 0
func edgeCases(mask: UInt32) -> [EdgeCase] {
let M = mask
var c = [EdgeCase]()
c.append(EdgeCase(name: "rotl immediate by 1 and by 31",
instrs: [I(.rotl, 1, 0, rot: 1), I(.rotl, 2, 0, rot: 31), I(.xor, 3, 1), I(.xor, 4, 2), I(.rotl, 5, 0, rot: 1), I(.rotl, 6, 0, rot: 31)],
pre: [], informational: false))
c.append(EdgeCase(name: "rotr by register == 0",
instrs: [zero(7), I(.rotr, 3, 7), I(.xor, 4, 3)],
pre: [(1, "r7 == 0", { $0[7] == 0 })], informational: false))
c.append(EdgeCase(name: "rotr by register == 32 (32 mod 32 = 0)",
instrs: [zero(7)] + (set(1, 32) + [I(.rotr, 3, 1), I(.xor, 4, 3)]),
pre: [(3, "r1 == 32", { $0[1] == 32 })], informational: false))
c.append(EdgeCase(name: "rotr by register == 0xFFFFFFE0 (-32, 0 mod 32)",
instrs: [zero(7)] + (set(1, 0xFFFFFFE0) + [I(.rotr, 3, 1), I(.xor, 4, 3)]),
pre: [(3, "r1 == 0xFFFFFFE0", { $0[1] == 0xFFFFFFE0 })], informational: false))
var rr: [Instr] = [zero(7)]
rr += set(1, 31); rr += [I(.rotr, 3, 1), I(.add, 1, 7, imm: 32), I(.rotr, 4, 1)]
rr += set(2, 1); rr += [I(.rotr, 5, 2), I(.xor, 6, 5)]
c.append(EdgeCase(name: "rotr by register == 31 and == 63 and == 1",
instrs: rr,
pre: [(3, "r1 == 31", { $0[1] == 31 }), (5, "r1 == 63", { $0[1] == 63 }), (8, "r2 == 1", { $0[2] == 1 })], informational: false))
var mh: [Instr] = [zero(7)]
mh += set(1, 0xFFFFFFFF); mh += set(2, 0xFFFFFFFF)
mh += [I(.mulhi, 1, 2), I(.xor, 3, 1)]
mh += set(4, 0x80000000); mh += set(5, 2)
mh += [I(.mulhi, 4, 5), I(.xor, 3, 4), I(.mulhi, 6, 7), I(.xor, 0, 6)]
c.append(EdgeCase(name: "mulhi 0xFFFFFFFF x 0xFFFFFFFF, 0x80000000 x 2, x 0",
instrs: mh,
pre: [(5, "r1 == r2 == 0xFFFFFFFF", { $0[1] == 0xFFFFFFFF && $0[2] == 0xFFFFFFFF }),
(6, "mulhi result r1 == 0xFFFFFFFE", { $0[1] == 0xFFFFFFFE }),
(11, "r4 == 0x80000000, r5 == 2", { $0[4] == 0x80000000 && $0[5] == 2 }),
(12, "mulhi result r4 == 1", { $0[4] == 1 }),
(13, "r7 == 0", { $0[7] == 0 }),
(14, "mulhi by 0 gives r6 == 0", { $0[6] == 0 })], informational: false))
c.append(EdgeCase(name: "shfl_xor every mask 1..16 in sequence (generator uses only 1,2,4,8,16)",
instrs: (1...16).map { I(.shfl, $0 % 8, ($0 + 1) % 8, mask: $0) },
pre: [], informational: false))
var l0: [Instr] = [zero(7), I(.load, 3, 7)]
l0 += set(1, M &+ 1); l0 += [I(.load, 4, 1), I(.xor, 5, 4)]
c.append(EdgeCase(name: "load at index 0 (register 0, and register MASK+1 which masks to 0)",
instrs: l0,
pre: [(1, "r7 & MASK == 0", { $0[7] & M == 0 }),
(4, "r1 == MASK+1, so unmasked index is out of range and masked index is 0", { $0[1] == M &+ 1 && ($0[1] & M) == 0 })],
informational: false))
var lm: [Instr] = [zero(7)]
lm += set(1, M); lm += [I(.load, 3, 1)]
lm += set(2, 0xFFFFFFFF); lm += [I(.load, 4, 2), I(.xor, 5, 4)]
c.append(EdgeCase(name: "load at index MASK (register MASK, and register 0xFFFFFFFF which masks to MASK)",
instrs: lm,
pre: [(3, "r1 == MASK", { $0[1] == M }),
(6, "r2 == 0xFFFFFFFF, masked index == MASK", { $0[2] == 0xFFFFFFFF && ($0[2] & M) == M })],
informational: false))
c.append(EdgeCase(name: "add wraparound 0xFFFFFFFF + 1",
instrs: [zero(7)] + (set(1, 0xFFFFFFFF) + [I(.add, 1, 7, imm: 1), I(.xor, 2, 1)]),
pre: [(3, "r1 == 0xFFFFFFFF", { $0[1] == 0xFFFFFFFF }), (4, "r1 == 0 after add", { $0[1] == 0 })], informational: false))
c.append(EdgeCase(name: "sub wraparound 0 - 1",
instrs: [zero(7), zero(1)] + (set(2, 1) + [I(.sub, 1, 2), I(.xor, 3, 1)]),
pre: [(4, "r1 == 0, r2 == 1", { $0[1] == 0 && $0[2] == 1 }), (5, "r1 == 0xFFFFFFFF after sub", { $0[1] == 0xFFFFFFFF })], informational: false))
var mm: [Instr] = [zero(7)]
mm += set(1, 0xFFFFFFFF); mm += set(2, 0xFFFFFFFF)
mm += [I(.mul, 1, 2), I(.xor, 3, 1)]
mm += set(4, 0xFFFFFFFF); mm += set(5, 0xFFFFFFFF); mm += set(6, 5)
mm += [I(.mad, 6, 4, b: 5), I(.xor, 0, 6)]
c.append(EdgeCase(name: "mul and mad wraparound 0xFFFFFFFF x 0xFFFFFFFF",
instrs: mm,
pre: [(5, "r1 == r2 == 0xFFFFFFFF", { $0[1] == 0xFFFFFFFF && $0[2] == 0xFFFFFFFF }),
(6, "mul low result r1 == 1", { $0[1] == 1 }),
(13, "r4 == r5 == 0xFFFFFFFF, r6 == 5", { $0[4] == 0xFFFFFFFF && $0[5] == 0xFFFFFFFF && $0[6] == 5 }),
(14, "mad result r6 == 6", { $0[6] == 6 })], informational: false))
let z = generateProgram(seedString: "edge/zero-loads")
c.append(EdgeCase(name: "generated program with every load replaced by xor (zero loads)",
instrs: z.instrs.map { ins in var m = ins; if m.op == .load { m.op = .xor }; return m },
pre: [], informational: false))
c.append(EdgeCase(name: "64 loads and nothing else",
instrs: (0..<64).map { I(.load, $0 % 8, ($0 + 3) % 8) },
pre: [], informational: false))
c.append(EdgeCase(name: "rotl immediate by 0 (outside the generator's 1..31 contract; MSL shifts by 32)",
instrs: [I(.rotl, 1, 0, rot: 0), I(.xor, 2, 1)],
pre: [], informational: true))
return c
}
func runEdge(_ opts: Options, ctx: DatasetContext) -> Bool {
let gpu = ctx.gpu
let log2 = opts.datasetLog2
let mask = UInt32((1 << log2) - 1)
let ds = ctx.source(log2: log2)
print("\n=== edge cases, dataset 2^\(log2) words, MASK \(hex(mask)) ===")
let dataset = ctx.makeDataset(log2: log2)
let bases: [UInt32] = [0, 1 << 20, 0x7FFFFFF0, 0xFFFFFFE0]
print("warps: base nonces " + bases.map { hex($0) }.joined(separator: ", ") + " (the last two straddle 2^31 and wrap past 2^32)")
var allOk = true
var rows = [String]()
for ec in edgeCases(mask: mask) {
let program = Program(seedString: "edge/\(ec.name)", seed: seedWords("edge/\(ec.name)"), instrs: ec.instrs)
let msl = generateMSL(program, datasetLog2: log2)
var status = "", detail = ""
var ok = true
// Preconditions, checked on lane 0 of every warp in every iteration.
var preOk = true
var preNotes = [String]()
if !ec.pre.isEmpty {
for base in bases {
var hits = [Int: Int]()
var misses = [Int: Int]()
_ = cpuWarpTraced(program, baseNonce: base, ds: ds) { _, k, regs in
for (idx, _, check) in ec.pre where idx == k {
if check(regs) { hits[idx, default: 0] += 1 } else { misses[idx, default: 0] += 1 }
}
}
for (idx, what, _) in ec.pre {
if (misses[idx] ?? 0) > 0 || (hits[idx] ?? 0) != Program.iterations {
preOk = false
preNotes.append("base \(hex(base)) instr \(idx) '\(what)' held \(hits[idx] ?? 0)/\(Program.iterations) iterations")
}
}
}
if preOk { preNotes = ec.pre.map { "instr \($0.0): \($0.1)" } }
}
do {
let k = try compileHash(gpu, msl: msl)
guard let gpuOut = gpuWarps(gpu, k, dataset: dataset, bases: bases) else { throw IgneumError("GPU run failed") }
var badLanes = 0
var first = ""
for (w, base) in bases.enumerated() {
let cpu = cpuWarp(program, baseNonce: base, ds: ds)
for l in 0..<32 where cpu[l] != gpuOut[w][l] {
badLanes += 1
if first.isEmpty { first = "first: base \(hex(base)) lane \(l) gpu \(h64(gpuOut[w][l])) cpu \(h64(cpu[l]))" }
}
}
ok = badLanes == 0 && preOk
status = badLanes == 0 ? "GPU == CPU 128/128 lanes" : "MISMATCH \(badLanes)/128 lanes, \(first)"
detail = "compile \(fmt(k.totalMs, 1)) ms"
if badLanes > 0 { print(describeProgram(program)) }
} catch {
ok = false
status = "COMPILE FAIL: \(error)"
}
let pre = ec.pre.isEmpty ? "none needed" : (preOk ? "held (all 8 iterations, lane 0, 4 warps)" : "NOT HELD")
let verdict = ec.informational ? (ok ? "info: agrees" : "info: differs") : (ok ? "PASS" : "FAIL")
if !ec.informational && !ok { allOk = false }
print("\(verdict): \(ec.name)")
print(" \(ec.instrs.count) instructions, loads/hash \(program.loadsPerHash), \(status), \(detail)")
for n in preNotes { print(" precondition \(n)") }
rows.append("| \(ec.name) | \(ec.instrs.count) | \(program.loadsPerHash) | \(pre) | \(status) | \(verdict) |")
}
print("\n| Case | Instrs | Loads/hash | Preconditions | GPU vs CPU | Result |")
print("|---|---|---|---|---|---|")
for r in rows { print(r) }
print("EDGE: \(allOk ? "PASS" : "FAIL")")
return allOk
}
// MARK: - --stats
func popcount64(_ v: UInt64) -> Int { v.nonzeroBitCount }
func runStats(_ opts: Options, ctx: DatasetContext) -> Bool {
let gpu = ctx.gpu
let log2 = opts.datasetLog2
let ds = ctx.source(log2: log2)
let n = 1 << 20
print("\n=== output statistics, 2^20 consecutive nonces per seed, dataset 2^\(log2) words ===")
print("This is a sanity check for obvious structural bias. It is not a proof of cryptographic strength.")
let dataset = ctx.makeDataset(log2: log2)
guard let outBuf = gpu.device.makeBuffer(length: n * 8, options: .storageModeShared) else { print("FAIL: out buffer"); return false }
let seeds = [opts.seed, "\(opts.seed)/stats1", "\(opts.seed)/stats2"]
var allOk = true
var rows = [String]()
for seedString in seeds {
let program = generateProgram(seedString: seedString)
let k: CompiledHash
do { k = try compileHash(gpu, msl: generateMSL(program, datasetLog2: log2)) } catch { print("FAIL: compile \(error)"); return false }
memset(outBuf.contents(), 0, n * 8)
guard let gms = gpuRange(gpu, k, dataset: dataset, base: 0, count: n, out: outBuf) else { print("FAIL: GPU run"); return false }
let p = outBuf.contents().bindMemory(to: UInt64.self, capacity: n)
let outs = (0..<n).map { p[$0] }
// Spot check 2 warps against the CPU so the statistics are known to describe the verified function.
var spot = true
for w in [0, (n / 32) - 1] {
let cpu = cpuWarp(program, baseNonce: UInt32(w * 32), ds: ds)
if cpu != Array(outs[(w * 32)..<(w * 32 + 32)]) { spot = false }
}
// (a) bit frequency per output bit position
var ones = [Int](repeating: 0, count: 64)
for v in outs { var x = v; var b = 0; while x != 0 { if x & 1 == 1 { ones[b] += 1 }; x >>= 1; b += 1 } }
let expected = Double(n) / 2, sigma = (Double(n) * 0.25).squareRoot()
var maxDev = 0.0, maxBit = 0
for b in 0..<64 { let d = abs(Double(ones[b]) - expected); if d > maxDev { maxDev = d; maxBit = b } }
let maxZ = maxDev / sigma
let minFreq = Double(ones.min()!) / Double(n), maxFreq = Double(ones.max()!) / Double(n)
// (c) chi-square over 65536 buckets for each 16-bit window of the output
var chiRows = [String]()
var chiWorstZ = 0.0
for shift in [0, 16, 32, 48] {
var buckets = [Int](repeating: 0, count: 65536)
for v in outs { buckets[Int((v >> UInt64(shift)) & 0xFFFF)] += 1 }
let e = Double(n) / 65536
var chi = 0.0
for c in buckets { let d = Double(c) - e; chi += d * d / e }
let df = 65535.0
let z = (chi - df) / (2 * df).squareRoot()
chiWorstZ = max(chiWorstZ, abs(z))
chiRows.append("bits \(shift)..\(shift + 15): chi2 \(fmt(chi, 0)) (df 65535, z \(fmt(z, 2)))")
}
// (d) duplicates
let sorted = outs.sorted()
var dups = 0
for i in 1..<n where sorted[i] == sorted[i - 1] { dups += 1 }
// (b) avalanche: random nonces, flip bit (t mod 32), count changed output bits. Run on the GPU.
// The brief asks for 1,000 trials; a 16,000-trial pass (500 per input bit) is added because the
// standard error of the mean at 1,000 trials is 0.13 bits, too coarse to see a small bias.
func avalanche(_ trials: Int, salt: String) -> (mean: Double, std: Double, minD: Int, maxD: Int, perBitMin: Double, perBitMax: Double, perOutMin: Double, perOutMax: Double)? {
let sw = seedWords("avalanche/\(salt)/" + seedString)
var rng = SplitMix64(s: UInt64(sw[0]) | (UInt64(sw[1]) << 32))
var bases = [UInt32]()
var flipped = [Int]()
for t in 0..<trials {
let nonce = UInt32(truncatingIfNeeded: rng.next())
let bit = t % 32
bases.append(nonce); bases.append(nonce ^ (1 << UInt32(bit)))
flipped.append(bit)
}
guard let av = gpuWarps(gpu, k, dataset: dataset, bases: bases) else { return nil }
var diffs = [Int]()
var perBitSum = [Int](repeating: 0, count: 32), perBitN = [Int](repeating: 0, count: 32)
var perOut = [Int](repeating: 0, count: 64)
var minDiff = 64, maxDiff = 0
for t in 0..<trials {
let x = av[2 * t][0] ^ av[2 * t + 1][0]
let d = popcount64(x)
diffs.append(d)
perBitSum[flipped[t]] += d; perBitN[flipped[t]] += 1
for b in 0..<64 where (x >> UInt64(b)) & 1 == 1 { perOut[b] += 1 }
minDiff = min(minDiff, d); maxDiff = max(maxDiff, d)
}
let mean = Double(diffs.reduce(0, +)) / Double(trials)
let variance = diffs.reduce(0.0) { $0 + (Double($1) - mean) * (Double($1) - mean) } / Double(trials - 1)
var pbMin = 64.0, pbMax = 0.0
for b in 0..<32 where perBitN[b] > 0 { let m = Double(perBitSum[b]) / Double(perBitN[b]); pbMin = min(pbMin, m); pbMax = max(pbMax, m) }
let poMin = Double(perOut.min()!) / Double(trials), poMax = Double(perOut.max()!) / Double(trials)
return (mean, variance.squareRoot(), minDiff, maxDiff, pbMin, pbMax, poMin, poMax)
}
guard let a1 = avalanche(1000, salt: "small"), let a2 = avalanche(16000, salt: "large") else { print("FAIL: avalanche GPU run"); return false }
// Expected for an ideal function: mean 32, std 4 (binomial 64 x 0.5). Standard error of the mean:
// 0.13 bits at 1,000 trials, 0.03 bits at 16,000. Thresholds are about 4.5 standard errors.
let avOk = abs(a1.mean - 32) < 0.6 && a1.std > 3.3 && a1.std < 4.7 && abs(a2.mean - 32) < 0.15 && a2.std > 3.6 && a2.std < 4.4
let freqOk = maxZ < 4.5
let chiOk = chiWorstZ < 4.5
let dupOk = dups == 0
let ok = avOk && freqOk && chiOk && dupOk && spot
if !ok { allOk = false }
print("\nseed \"\(seedString)\": loads/hash \(program.loadsPerHash), GPU \(fmt(gms, 1)) ms for 2^20 hashes, CPU spot check 2 warps \(spot ? "PASS" : "FAIL")")
print(" (a) bit frequency: min \(fmt(minFreq, 4)) max \(fmt(maxFreq, 4)); largest deviation \(fmt(maxDev, 0)) counts at bit \(maxBit) = \(fmt(maxZ, 2)) sigma (sigma \(fmt(sigma, 0)), 64 bits, expect max under about 3.5)")
print(" (b) avalanche, 1000 single-bit nonce flips: mean \(fmt(a1.mean, 2)) std \(fmt(a1.std, 2)) min \(a1.minD) max \(a1.maxD) of 64 bits (expect mean 32, std 4); per-input-bit mean range \(fmt(a1.perBitMin, 1))..\(fmt(a1.perBitMax, 1))")
print(" (b) avalanche, 16000 flips (500 per input bit): mean \(fmt(a2.mean, 3)) std \(fmt(a2.std, 2)) min \(a2.minD) max \(a2.maxD); per-input-bit mean range \(fmt(a2.perBitMin, 2))..\(fmt(a2.perBitMax, 2)); per-output-bit flip probability range \(fmt(a2.perOutMin, 3))..\(fmt(a2.perOutMax, 3)) (expect 0.5, sigma 0.004)")
for r in chiRows { print(" (c) \(r)") }
print(" (d) duplicate 64-bit outputs among 2^20: \(dups) (expected about 3e-8)")
print(" verdict: \(ok ? "no obvious bias" : "SUSPECT")")
rows.append("| \(seedString) | \(program.loadsPerHash) | \(fmt(minFreq, 4))..\(fmt(maxFreq, 4)) | \(fmt(maxZ, 2)) | \(fmt(a1.mean, 2)) / \(fmt(a1.std, 2)) | \(fmt(a2.mean, 3)) / \(fmt(a2.std, 2)) | \(fmt(a2.perOutMin, 3))..\(fmt(a2.perOutMax, 3)) | \(fmt(chiWorstZ, 2)) | \(dups) | \(ok ? "uniform-looking" : "SUSPECT") |")
}
print("\n| Seed | Loads/hash | Bit freq min..max | Max bit z | Avalanche 1k mean / std | Avalanche 16k mean / std | Per-output-bit flip prob | Worst chi2 z (4 windows) | Dups | Verdict |")
print("|---|---|---|---|---|---|---|---|---|---|")
for r in rows { print(r) }
print("STATS: \(allOk ? "PASS (no obvious structural bias; not a security proof)" : "FAIL (something looks biased)")")
return allOk
}
// MARK: - --determinism
func runDeterminism(_ opts: Options, ctx: DatasetContext) -> Bool {
let gpu = ctx.gpu
let log2 = opts.datasetLog2
let mask = UInt32((1 << log2) - 1)
let ds = ctx.source(log2: log2)
let n = 1 << 20
print("\n=== determinism, seed \"\(opts.seed)\", 2^20 nonces from base 0, dataset 2^\(log2) words ===")
let dataset = ctx.makeDataset(log2: log2)
guard let outBuf = gpu.device.makeBuffer(length: n * 8, options: .storageModeShared) else { print("FAIL: out buffer"); return false }
var ok = true
// Generator and emitter determinism: two independent generations give identical MSL text.
let p1 = generateProgram(seedString: opts.seed), p2 = generateProgram(seedString: opts.seed)
let msl1 = generateMSL(p1, datasetLog2: log2), msl2 = generateMSL(p2, datasetLog2: log2)
let sameSource = msl1 == msl2
print("generator: two generations of the program give identical MSL source: \(sameSource ? "yes" : "NO") (\(msl1.utf8.count) bytes)")
if !sameSource { ok = false }
// Three compiles: the same source twice (Metal's shader cache may serve the second), and once more with a
// comment tag appended so the cache misses and the compiler really runs again on an identical kernel.
let tag = "\n// recompile tag \(h64(nowNs()))\n"
let k1: CompiledHash, k2: CompiledHash, k3: CompiledHash
do {
k1 = try compileHash(gpu, msl: msl1); k2 = try compileHash(gpu, msl: msl2); k3 = try compileHash(gpu, msl: msl2 + tag)
} catch { print("FAIL: compile \(error)"); return false }
print("compiled three times: identical source \(fmt(k1.totalMs, 1)) ms and \(fmt(k2.totalMs, 1)) ms (a sub-millisecond second figure means the system shader cache answered), tagged source \(fmt(k3.totalMs, 1)) ms (forced recompile)")
func runOnce(_ k: CompiledHash) -> (fp: UInt64, sentinels: Int, ms: Double)? {
memset(outBuf.contents(), 0xAA, n * 8)
guard let gms = gpuRange(gpu, k, dataset: dataset, base: 0, count: n, out: outBuf) else { return nil }
let p = outBuf.contents().bindMemory(to: UInt64.self, capacity: n)
var s = 0
for i in 0..<n where p[i] == 0xAAAAAAAAAAAAAAAA { s += 1 }
return (fnv64(outBuf.contents(), n * 8), s, gms)
}
var reference = [UInt64]()
var fps = [UInt64]()
for run in 0..<5 {
guard let r = runOnce(k1) else { print("FAIL: GPU run \(run)"); return false }
let p = outBuf.contents().bindMemory(to: UInt64.self, capacity: n)
if run == 0 { reference = (0..<n).map { p[$0] } }
var differ = 0
for i in 0..<n where p[i] != reference[i] { differ += 1 }
fps.append(r.fp)
let same = differ == 0 && r.sentinels == 0
if !same { ok = false }
print("run \(run + 1)/5 (compile 1): fingerprint \(h64(r.fp)), \(differ) of \(n) outputs differ from run 1, \(r.sentinels) unwritten lanes, GPU \(fmt(r.ms, 1)) ms: \(same ? "identical" : "DIFFERENT")")
}
guard let r2 = runOnce(k2) else { print("FAIL: GPU run on compile 2"); return false }
let p = outBuf.contents().bindMemory(to: UInt64.self, capacity: n)
var differ2 = 0
for i in 0..<n where p[i] != reference[i] { differ2 += 1 }
if differ2 != 0 || r2.sentinels != 0 { ok = false }
print("run on compile 2 (identical source): fingerprint \(h64(r2.fp)), \(differ2) outputs differ from compile 1 run 1, \(r2.sentinels) unwritten lanes: \(differ2 == 0 ? "identical" : "DIFFERENT")")
guard let r3 = runOnce(k3) else { print("FAIL: GPU run on compile 3"); return false }
var differ3 = 0
for i in 0..<n where p[i] != reference[i] { differ3 += 1 }
if differ3 != 0 || r3.sentinels != 0 { ok = false }
print("run on compile 3 (forced recompile): fingerprint \(h64(r3.fp)), \(differ3) outputs differ from compile 1 run 1, \(r3.sentinels) unwritten lanes: \(differ3 == 0 ? "identical" : "DIFFERENT")")
// CPU reference on the first and last warp and 6 others, so the fingerprint is tied to the verified function.
var cpuBad = 0
var vr = SplitMix64(s: 0x1234_5678_9abc_def0)
var warps = [0, n / 32 - 1]
while warps.count < 8 { warps.append(vr.below(n / 32)) }
for w in warps {
let cpu = cpuWarp(p1, baseNonce: UInt32(w * 32), ds: ds)
if cpu != Array(reference[(w * 32)..<(w * 32 + 32)]) { cpuBad += 1 }
}
if cpuBad != 0 { ok = false }
print("CPU interpreter on \(warps.count) warps of the reference run: \(cpuBad == 0 ? "all match" : "\(cpuBad) MISMATCH")")
// Dataset fill determinism and GPU-vs-CPU agreement of the dataset itself: fill a second buffer, blit both
// to shared memory, fingerprint, and compare sampled words (including 0 and MASK) with datasetElem.
let words = 1 << log2
let dataset2 = ctx.makeDataset(log2: log2)
var fillFps = [UInt64]()
var sampleBad = 0
if let shared = gpu.device.makeBuffer(length: words * 4, options: .storageModeShared) {
for (idx, dsBuf) in [dataset, dataset2].enumerated() {
let cb = gpu.queue.makeCommandBuffer()!
let blit = cb.makeBlitCommandEncoder()!
blit.copy(from: dsBuf, sourceOffset: 0, to: shared, destinationOffset: 0, size: words * 4)
blit.endEncoding()
cb.commit(); cb.waitUntilCompleted()
fillFps.append(fnv64(shared.contents(), words * 4))
if idx == 0 {
let dp = shared.contents().bindMemory(to: UInt32.self, capacity: words)
var sr = SplitMix64(s: 0xfeed_beef)
var idxs: [UInt32] = [0, 1, mask - 1, mask]
while idxs.count < 4096 { idxs.append(UInt32(sr.below(words))) }
for i in idxs where dp[Int(i)] != ds.word(i) { sampleBad += 1 }
}
}
let sameFill = fillFps[0] == fillFps[1]
if !sameFill || sampleBad != 0 { ok = false }
print("dataset fill: two fills fingerprint \(h64(fillFps[0])) and \(h64(fillFps[1])): \(sameFill ? "identical" : "DIFFERENT"); 4096 sampled words (incl. 0, 1, MASK-1, MASK) vs CPU dataset word (\(ds.modeName)): \(sampleBad == 0 ? "all match" : "\(sampleBad) MISMATCH")")
} else {
print("dataset fill check skipped: could not allocate a shared copy")
}
print("DETERMINISM: \(ok ? "PASS" : "FAIL")")
return ok
}
// MARK: - --memcheck
func runMemcheck(_ opts: Options, ctx: DatasetContext) -> Bool {
let gpu = ctx.gpu
print("\n=== memcheck, seed \"\(opts.seed)\" ===")
var ok = true
let program = generateProgram(seedString: opts.seed)
// Static: every dataset index in the generated sources is masked. Checked at three dataset sizes because
// the MASK literal changes with size.
for log2 in [20, 24, 28] {
let msl = generateMSL(program, datasetLog2: log2)
let r = maskCheckMSL(msl)
if !r.ok { ok = false }
print("static 2^\(log2): \(r.ok ? "PASS" : "FAIL") \(r.detail)")
}
let cu = maskCheckCUDA(generateCUDA(program, memhard: ctx.mp), memhard: !ctx.closed)
if !cu.ok { ok = false }
print("static CUDA twin: \(cu.ok ? "PASS" : "FAIL") \(cu.detail)")
print("program has \(program.instrs.filter { $0.op == .load }.count) load instructions (\(program.loadsPerHash) loads/hash)")
// Dynamic: a 4 MiB dataset with nonces at the top of the 32-bit range (they wrap to 0 inside the batch),
// one full batch of 2^20 nonces plus 4 warps verified against the CPU. Metal does not bounds-check device
// buffers, so "no crash" is weak evidence by itself; the static check above is the real guarantee.
let log2 = 20
let mask = UInt32((1 << log2) - 1)
let ds = ctx.source(log2: log2)
let dataset = ctx.makeDataset(log2: log2)
let k: CompiledHash
do { k = try compileHash(gpu, msl: generateMSL(program, datasetLog2: log2)) } catch { print("FAIL: compile \(error)"); return false }
let n = 1 << 20
guard let outBuf = gpu.device.makeBuffer(length: n * 8, options: .storageModeShared) else { print("FAIL: out buffer"); return false }
for base: UInt32 in [0xFFF00000, 0xFFFFFFE0, 0x80000000, 0] {
if let gms = gpuRange(gpu, k, dataset: dataset, base: base, count: n, out: outBuf) {
print("dynamic 4 MiB: 2^20 nonces from base \(hex(base)) (last nonce \(hex(base &+ UInt32(n - 1)))): completed, GPU \(fmt(gms, 1)) ms")
} else { ok = false; print("dynamic 4 MiB: base \(hex(base)): GPU ERROR") }
}
let bases: [UInt32] = [0xFFFFFFE0, 0xFFFFFFFF, 0x80000000, 0xFFF00000]
guard let gpuOut = gpuWarps(gpu, k, dataset: dataset, bases: bases) else { print("FAIL: GPU warps"); return false }
var overMask = 0, loads = 0
for (w, base) in bases.enumerated() {
let cpu = cpuWarpTraced(program, baseNonce: base, ds: ds) { _, kk, regs in
let ins = program.instrs[kk]
if ins.op == .load { loads += 1; if regs[ins.a] > mask { overMask += 1 } }
}
let match = cpu == gpuOut[w]
if !match { ok = false }
print("verify warp base \(hex(base)): GPU vs CPU \(match ? "PASS" : "FAIL")")
}
print("in those 4 warps (lane 0, all iterations) \(overMask) of \(loads) load indices were above MASK before masking, so the mask was exercised")
print("MEMCHECK: \(ok ? "PASS" : "FAIL")")
return ok
}
// MARK: - Test dispatcher
func runTests(_ opts: Options) -> Never {
let gpu = GPU()
print("igneum-bench hardening tests")
print("GPU: \(gpu.device.name) (maxBufferLength \(gpu.device.maxBufferLength / (1 << 20)) MiB, unified memory \(gpu.device.hasUnifiedMemory)), day \"\(opts.day)\"")
let ctx = DatasetContext(gpu: gpu, closedForm: opts.closedForm, dayString: opts.day)
print("dataset construction: \(ctx.modeName)" + (ctx.closed ? "" : "; cache GPU fill \(fmt(ctx.cacheFillGPUms, 2)) ms GPU time"))
if generatorConfig.loadWeight != 25 || generatorConfig.wideFrac != 0 {
print("generator levers: load weight \(generatorConfig.loadWeight), wide fraction \(generatorConfig.wideFrac) percent (NOT the default generator)")
}
var results = [(String, Bool)]()
let t0 = nowNs()
if !ctx.closed {
let cc = ctx.cacheCheck()
print(cc.detail)
results.append(("cache", cc.ok))
}
if opts.fuzz != nil { results.append(("fuzz", runFuzz(opts, ctx: ctx))) }
if opts.edge { results.append(("edge", runEdge(opts, ctx: ctx))) }
if opts.stats { results.append(("stats", runStats(opts, ctx: ctx))) }
if opts.determinism { results.append(("determinism", runDeterminism(opts, ctx: ctx))) }
if opts.memcheck { results.append(("memcheck", runMemcheck(opts, ctx: ctx))) }
print("\n=== tests summary (\(fmt(Double(nowNs() - t0) / 1e9, 1)) s) ===")
for (name, ok) in results { print("\(pad(name, 12)) \(ok ? "PASS" : "FAIL")") }
let all = results.allSatisfy { $0.1 }
print("OVERALL: \(all ? "PASS" : "FAIL")")
exit(all ? 0 : 1)
}
// MARK: - Serve mode (GPU worker for igneum-miner --worker), 3 October 2026
//
// Protocol, one line each. Only "found", "done", "error", "prepared", "prepare-failed" and "ready" are parsed by the
// miner; every other line is logged.
// stdin: job <job_id> <header_prehash_hex 64> <target_hex 16> <nonce_start u64> <nonce_count u64> <epoch_seed_hex 64> <day_seed_hex>
// prepare <epoch_seed_hex 64> <day_seed_hex> [<pack_dir>] compile that program and build that day's dataset in the
// background while jobs on the current seeds keep running
// quit
// stdout: ready metal <device> dataset-log2 <n> batch <n> prepare 1
// found <job_id> <nonce u64> <hash_hex 16> every nonce whose 64-bit hash is <= target (hash <= target64)
// done <job_id> <hashes> <ms> end of the job (wall ms, dispatch plus scan)
// error <job_id> <text>
// prepared <epoch_seed_hex> <day_seed_hex> <ms> program <ms> dataset <ms> ... the pair is resident
// prepare-failed <epoch_seed_hex> <day_seed_hex> <text>
// The program for an epoch seed is generateProgramV2(bytes: epoch_seed) (version 2 with the acceptance rule, the same
// derivation as igneum-pow), compiled once and kept; the
// cache and 1 GiB dataset for a day seed come from seedWordsBytes(day_seed_bytes), built once and kept. At most two
// programs and two datasets are resident: the current job's pair and one more (the prepared pair, or the previous
// pair until the first job on the new one is done). A job whose pair is resident switches instantly; a job whose pair
// is not (no prepare, or a pair nobody predicted) compiles inline as before. The pack_dir of a prepare is ignored
// here (Metal compiles from the seed); ahead-of-time workers build their kernel from it.
// The init words of a dispatch are seedWordsBytes("igneum-block/" || prehash || nonce_hi_le32) and go to buffer 3 of
// igneum_hash_bound; the lane nonce is baseNonce + gid as in the bench kernel.
let emitLock = NSLock()
func emit(_ line: String) { emitLock.lock(); print(line); fflush(stdout); emitLock.unlock() }
func unhex(_ s: String) -> [UInt8]? {
let chars = Array(s.utf8)
if chars.count % 2 != 0 { return nil }
var out = [UInt8](); out.reserveCapacity(chars.count / 2)
var i = 0
while i < chars.count {
guard let hi = UInt8(String(UnicodeScalar(chars[i])), radix: 16), let lo = UInt8(String(UnicodeScalar(chars[i + 1])), radix: 16) else { return nil }
out.append(hi << 4 | lo); i += 2
}
return out
}
func blockInitWords(prehash: [UInt8], nonceHi: UInt32) -> [UInt32] {
var b = Array("igneum-block/".utf8)
b += prehash
b += [UInt8(nonceHi & 0xff), UInt8((nonceHi >> 8) & 0xff), UInt8((nonceHi >> 16) & 0xff), UInt8((nonceHi >> 24) & 0xff)]
return seedWordsBytes(b)
}
final class ServeProgram {
let seedHex: String
/// The Swift-generated program (class v2); nil for a program compiled from a pack (class v3, Counter ASIC 2.0),
/// which has no Swift instruction list and so never races variants
let program: Program?
/// Counter ASIC 2.0: the class ("v2" or "v3") and, for a pack program, its era seed hex and loads per hash
let programClass: String
let eraHex: String
let loadsPerHash: Int
private var slot: CompiledHash
private let lock = NSLock()
/// true until a race ran for this program (an inline compile races after the first job on it)
var raceDue = true
var raceLine = ""
init(seedHex: String, program: Program, compiled: CompiledHash) {
self.seedHex = seedHex; self.program = program; self.slot = compiled
self.programClass = "v2"; self.eraHex = ""; self.loadsPerHash = program.loadsPerHash
}
init(seedHex: String, packClass: String, eraHex: String, loadsPerHash: Int, compiled: CompiledHash) {
self.seedHex = seedHex; self.program = nil; self.slot = compiled
self.programClass = packClass; self.eraHex = eraHex; self.loadsPerHash = loadsPerHash; self.raceDue = false
}
/// Whether this program is the one a line naming `cls` (and `era`) wants; empty names accept any (class v2 lines).
func matches(cls: String, era: String) -> Bool {
if cls != "" && cls != programClass { return false }
if era != "" && programClass == "v3" && era.lowercased() != eraHex.lowercased() { return false }
return true
}
var compiled: CompiledHash { lock.lock(); defer { lock.unlock() }; return slot }
func install(_ c: CompiledHash) { lock.lock(); slot = c; lock.unlock() }
}
// The job loop and a race take turns on the card: a variant is timed with no job running (exclusive numbers), and
// mining resumes between variants. Held per chunk by the job loop, per variant window by the race.
let gpuLock = NSLock()
// The tuning file: {"cards": {"<device name with underscores>": {"variant": "u2", "race": false, "candidates": [...]}}}
struct MetalTuning {
var found = false
var variant = ""
var race = true
var candidates = [String]()
}
func readMetalTuning(path: String?, device: String) -> MetalTuning {
var t = MetalTuning()
guard let path = path, let data = FileManager.default.contents(atPath: path),
let root = try? JSONSerialization.jsonObject(with: data) as? [String: Any],
let cards = root["cards"] as? [String: Any], let e = cards[device] as? [String: Any] else { return t }
t.found = true
t.variant = e["variant"] as? String ?? ""
t.race = e["race"] as? Bool ?? true
t.candidates = e["candidates"] as? [String] ?? []
return t
}
// Dispatches `count` lane nonces (a multiple of 32) from `base` with the kernel's group width; a tail under the
// width goes in 32-wide groups (same pipeline, a threadgroup size is a dispatch parameter in Metal).
func encodeHash(_ enc: MTLComputeCommandEncoder, _ k: CompiledHash, dataset: MTLBuffer, out: MTLBuffer, base: UInt32, initw: inout [UInt32], count: Int) {
enc.setComputePipelineState(k.pipeline)
enc.setBuffer(dataset, offset: 0, index: 0)
enc.setBuffer(out, offset: 0, index: 1)
var b = base
enc.setBytes(&b, length: 4, index: 2)
enc.setBytes(&initw, length: 32, index: 3)
let w = k.groupWidth
let main = count - count % w
if main > 0 { enc.dispatchThreadgroups(MTLSize(width: main / w, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: w, height: 1, depth: 1)) }
if main < count {
var b2 = base &+ UInt32(main)
enc.setBytes(&b2, length: 4, index: 2)
enc.setBuffer(out, offset: main * 8, index: 1)
enc.dispatchThreadgroups(MTLSize(width: (count - main) / 32, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 32, height: 1, depth: 1))
}
}
struct RaceEntry {
let v: MetalVariant
var compiled: CompiledHash? = nil
var mhs = 0.0 // best round
var note = "" // why it is out, or a detail
var ok: Bool { compiled != nil && note.isEmpty }
}
// Races the variants of `program` on `dataset` and returns the race line; the winner is installed in `program`.
// Every variant must equal the base kernel bit for bit over 2^16 nonces (the base is the kernel the miner's CPU
// re-check has always covered); a window is `benchMs` of launches of `batch` nonces, the first launch warming up.
// `rounds` interleaved rounds, best per variant. Stops compiling and timing after `budgetS` (base is always kept).
func raceProgram(_ gpu: GPU, _ program: ServeProgram, dataset: MTLBuffer, datasetLog2: Int, opts: Options, rounds: Int, table: Bool = false) -> String {
let t0 = nowNs()
let device = gpu.device.name.replacingOccurrences(of: " ", with: "_")
let tuning = readMetalTuning(path: opts.tuningPath ?? ProcessInfo.processInfo.environment["IGNEUM_TUNING_FILE"], device: device)
let all = metalVariants()
let pinned = opts.pinnedVariant ?? (tuning.found && !tuning.race ? tuning.variant : "")
var order = [MetalVariant]()
func push(_ n: String) { if let v = all.first(where: { $0.name == n }), !order.contains(where: { $0.name == n }) { order.append(v) } }
push("base")
if !pinned.isEmpty { push(pinned) }
else {
tuning.candidates.forEach(push)
if opts.race == "on" { all.forEach { push($0.name) } }
else if opts.race != "off" { opts.race.split(separator: ",").forEach { push(String($0)) } }
}
let pinnedOnly = !pinned.isEmpty && order.count == 2
let batch = 1 << opts.batchLog2
let checkN = 1 << 16
let deadline = t0 + UInt64(opts.raceBudgetS) * 1_000_000_000
guard let outBuf = gpu.device.makeBuffer(length: batch * 8, options: .storageModeShared),
let refBuf = gpu.device.makeBuffer(length: checkN * 8, options: .storageModeShared) else { return "race \(program.seedHex.prefix(16)) failed: no buffers" }
var entries = order.map { RaceEntry(v: $0) }
entries[0].compiled = program.compiled
// Compile (the base is already compiled)
for i in 1..<entries.count {
if nowNs() > deadline { entries[i].note = "not compiled: the race budget ran out"; continue }
guard let swiftProgram = program.program else { return "race \(program.seedHex.prefix(16)) skipped: a pack program (class \(program.programClass)) has no variants to race" }
do { entries[i].compiled = try compileBound(gpu, msl: generateMSL(swiftProgram, datasetLog2: datasetLog2, source: .stored, bound: true, variant: entries[i].v), variant: entries[i].v) }
catch { entries[i].note = "compile: \(error)".prefix(200).description }
}
let compileMs = ms(t0, nowNs())
var initw = blockInitWords(prehash: [UInt8](repeating: 0x5a, count: 32), nonceHi: 7)
func run(_ k: CompiledHash, base: UInt32, count: Int, into: MTLBuffer) -> Bool {
let cb = gpu.queue.makeCommandBuffer()!
let enc = cb.makeComputeCommandEncoder()!
encodeHash(enc, k, dataset: dataset, out: into, base: base, initw: &initw, count: count)
enc.endEncoding()
cb.commit(); cb.waitUntilCompleted()
return cb.error == nil
}
// Reference output of the base kernel (exclusive window)
gpuLock.lock()
let refOk = run(entries[0].compiled!, base: 0x1000_0000, count: checkN, into: refBuf)
gpuLock.unlock()
if !refOk { return "race \(program.seedHex.prefix(16)) failed: the base kernel did not run" }
let ref = refBuf.contents().bindMemory(to: UInt64.self, capacity: checkN)
let out = outBuf.contents().bindMemory(to: UInt64.self, capacity: batch)
for round in 0..<max(1, rounds) {
for i in 0..<entries.count {
guard entries[i].ok, let k = entries[i].compiled else { continue }
if pinnedOnly && i == 0 { continue }
if round == 0 && i > 0 && nowNs() > deadline { entries[i].note = "not timed: the race budget ran out"; continue }
gpuLock.lock()
// NSLock is not fair: after the window the job loop gets the card (measured 4 October 2026: without the
// pause a queued job waited the whole race, 36 s)
defer { gpuLock.unlock(); Thread.sleep(forTimeInterval: 0.15) }
if round == 0 && i > 0 {
if !run(k, base: 0x1000_0000, count: checkN, into: outBuf) { entries[i].note = "did not run"; continue }
var bad = -1
for j in 0..<checkN where out[j] != ref[j] { bad = j; break }
if bad >= 0 { entries[i].note = String(format: "lane %d: %016llx, base %016llx (discarded)", bad, out[bad], ref[bad]); continue }
}
if pinnedOnly { continue }
var launches = 0, hashes = 0, tStart: UInt64 = 0
while true {
if !run(k, base: 0x2000_0000 &+ UInt32(launches * batch), count: batch, into: outBuf) { entries[i].note = "bench: did not run"; break }
let now = nowNs()
if launches == 0 { tStart = now } else { hashes += batch }
launches += 1
if launches >= 3 && ms(tStart, now) >= Double(opts.raceBenchMs) { entries[i].mhs = max(entries[i].mhs, Double(hashes) / ms(tStart, now) / 1000.0); break }
}
}
}
var win = 0
if pinnedOnly && entries.count == 2 && entries[1].ok { win = 1 }
else { for i in 1..<entries.count where entries[i].ok && entries[i].mhs > entries[win].mhs * (win == 0 ? 1.005 : 1.0) { win = i } }
let baseMhs = entries[0].mhs, winMhs = entries[win].mhs
if win != 0, let k = entries[win].compiled { program.install(k) }
let total = ms(t0, nowNs())
let os = ProcessInfo.processInfo.operatingSystemVersion
var line = "race \(program.seedHex.prefix(16)) device \(device) driver macos-\(os.majorVersion).\(os.minorVersion).\(os.patchVersion) arch metal loads \(program.loadsPerHash) wide \(program.program?.wideLoadsPerHash ?? 0) variants \(entries.count)"
for e in entries { line += e.ok && (e.mhs > 0 || pinnedOnly) ? " \(e.v.name)=\(fmt(e.mhs, 3))/\(e.compiled!.pipeline.maxTotalThreadsPerThreadgroup)t/\(e.v.groupWidth)w" : " \(e.v.name)=-" }
let gain = baseMhs > 0 ? (winMhs / baseMhs - 1) * 100 : 0
line += " winner \(entries[win].v.name) \(fmt(winMhs, 3)) base \(fmt(baseMhs, 3)) gain \(gain >= 0 ? "+" : "")\(fmt(gain, 2))% compile \(fmt(compileMs, 0)) bench \(fmt(total - compileMs, 0)) total \(fmt(total, 0)) ms"
if pinnedOnly { line += " pinned by tuning" } else if tuning.found { line += " tuned order" }
for e in entries where !e.note.isEmpty { line += " | \(e.v.name): \(e.note)" }
program.raceDue = false
program.raceLine = line
if table {
print("| variant | threads/group | max threads | MH/s | vs base | note |")
print("|---|---|---|---|---|---|")
for e in entries { print("| \(e.v.name) | \(e.v.groupWidth) | \(e.compiled.map { String($0.pipeline.maxTotalThreadsPerThreadgroup) } ?? "-") | \(e.ok ? fmt(e.mhs, 3) : "-") | \(e.ok && baseMhs > 0 ? (e.mhs / baseMhs - 1 >= 0 ? "+" : "") + fmt((e.mhs / baseMhs - 1) * 100, 2) + "%" : "-") | \(e.note) |") }
}
return line
}
final class ServeDataset {
let dayHex: String
let ctx: DatasetContext
let buffer: MTLBuffer
init(dayHex: String, ctx: DatasetContext, buffer: MTLBuffer) { self.dayHex = dayHex; self.ctx = ctx; self.buffer = buffer }
}
// The resident programs and datasets, shared by the job loop (main thread) and the prepare queue (background).
final class ServeStore {
private let lock = NSLock()
private var programs = [ServeProgram]()
private var datasets = [ServeDataset]()
/// The pair of the last job (epoch seed hex, day seed hex); never evicted
var current: (String, String)? = nil
func program(_ seedHex: String, cls: String = "", era: String = "") -> ServeProgram? { lock.lock(); defer { lock.unlock() }; return programs.first { $0.seedHex == seedHex && $0.matches(cls: cls, era: era) } }
func dataset(_ dayHex: String) -> ServeDataset? { lock.lock(); defer { lock.unlock() }; return datasets.first { $0.dayHex == dayHex } }
func counts() -> (Int, Int) { lock.lock(); defer { lock.unlock() }; return (programs.count, datasets.count) }
/// Adds a program; with more than two resident, the oldest one that is not the current job's goes.
func add(_ p: ServeProgram) {
lock.lock(); defer { lock.unlock() }
if programs.contains(where: { $0.seedHex == p.seedHex && $0.programClass == p.programClass && $0.eraHex == p.eraHex }) { return }
programs.append(p)
while programs.count > 2, let i = programs.firstIndex(where: { $0.seedHex != current?.0 && $0.seedHex != p.seedHex }) { programs.remove(at: i) }
}
func add(_ d: ServeDataset) {
lock.lock(); defer { lock.unlock() }
if datasets.contains(where: { $0.dayHex == d.dayHex }) { return }
datasets.append(d)
while datasets.count > 2, let i = datasets.firstIndex(where: { $0.dayHex != current?.1 && $0.dayHex != d.dayHex }) { datasets.remove(at: i) }
}
/// After the first job on a new pair: drop everything but that pair (the old program and dataset are released).
func prune(to pair: (String, String)) -> (Int, Int) {
lock.lock(); defer { lock.unlock() }
let before = (programs.count, datasets.count)
programs.removeAll { $0.seedHex != pair.0 }
datasets.removeAll { $0.dayHex != pair.1 }
return (before.0 - programs.count, before.1 - datasets.count)
}
}
func compileBound(_ gpu: GPU, msl: String, variant: MetalVariant? = nil) throws -> CompiledHash {
let t0 = nowNs()
let copts = MTLCompileOptions()
if let v = variant, v.sizeOpt { if #available(macOS 13.0, *) { copts.optimizationLevel = .size } }
let lib = try gpu.device.makeLibrary(source: msl, options: copts)
let t1 = nowNs()
guard let fn = lib.makeFunction(name: "igneum_hash_bound") else { throw IgneumError("no igneum_hash_bound function in library") }
let pipe = try gpu.device.makeComputePipelineState(function: fn)
let t2 = nowNs()
var c = CompiledHash(pipeline: pipe, libraryMs: ms(t0, t1), pipelineMs: ms(t1, t2))
if let v = variant { c.groupWidth = v.groupWidth; c.variant = v.name }
if c.groupWidth > pipe.maxTotalThreadsPerThreadgroup { throw IgneumError("variant \(c.variant): group width \(c.groupWidth) is over the pipeline's maxTotalThreadsPerThreadgroup \(pipe.maxTotalThreadsPerThreadgroup)") }
return c
}
// Builds the program for an epoch seed (hex) unless resident. Returns (program, compile ms) or throws.
func serveProgram(_ gpu: GPU, _ store: ServeStore, seedHex: String, seed: [UInt8], datasetLog2: Int) throws -> (ServeProgram, Double) {
if let p = store.program(seedHex) { return (p, 0) }
let t0 = nowNs()
let p = generateProgramV2(seedString: "epoch/" + seedHex, bytes: seed)
let msl = generateMSL(p, datasetLog2: datasetLog2, source: .stored, bound: true)
let c = try compileBound(gpu, msl: msl)
let sp = ServeProgram(seedHex: seedHex, program: p, compiled: c)
store.add(sp)
return (sp, ms(t0, nowNs()))
}
// Counter ASIC 2.0 (5 October 2026): a class v3 program comes from the pack igneum-miner export-pack / --prepare-packs
// wrote (program_bound.metal, the Rust emitter's text; program.h for the identity), never from the Swift generator.
// The pack is checked the way the one-click workers check it (proto-cuda/nvrtc/packfile.h): IGNEUM_GENERATOR 2 or 3,
// the class line against the generator, the seed bytes against the line's epoch seed, IGNEUM_SEEDW_INIT against the
// words of IGNEUM_PROGRAM_ATTEMPT of that seed, and the class and era against what the line names.
func servePackProgram(_ gpu: GPU, _ store: ServeStore, seedHex: String, seed: [UInt8], dir: String, wantClass: String, wantEra: String) throws -> (ServeProgram, Double) {
if let p = store.program(seedHex, cls: wantClass, era: wantEra) { return (p, 0) }
let t0 = nowNs()
guard let programH = try? String(contentsOfFile: dir + "/program.h", encoding: .utf8) else { throw NSError(domain: "pack", code: 1, userInfo: [NSLocalizedDescriptionKey: "cannot read \(dir)/program.h"]) }
func defineU32(_ name: String) -> UInt32? {
guard let re = try? NSRegularExpression(pattern: "#define \(name) ([0-9a-fA-Fx]+)"), let m = re.firstMatch(in: programH, range: NSRange(programH.startIndex..., in: programH)) else { return nil }
let v = String(programH[Range(m.range(at: 1), in: programH)!]).replacingOccurrences(of: "u", with: "")
return v.hasPrefix("0x") ? UInt32(v.dropFirst(2), radix: 16) : UInt32(v)
}
func defineStr(_ name: String) -> String? {
guard let re = try? NSRegularExpression(pattern: "#define \(name) \"([^\"]*)\""), let m = re.firstMatch(in: programH, range: NSRange(programH.startIndex..., in: programH)) else { return nil }
return String(programH[Range(m.range(at: 1), in: programH)!])
}
func defineWords(_ name: String) -> [UInt32]? {
guard let re = try? NSRegularExpression(pattern: "#define \(name) \\{([^}]*)\\}"), let m = re.firstMatch(in: programH, range: NSRange(programH.startIndex..., in: programH)) else { return nil }
let body = String(programH[Range(m.range(at: 1), in: programH)!])
let words = body.split(separator: ",").compactMap { t -> UInt32? in let v = t.trimmingCharacters(in: .whitespaces).replacingOccurrences(of: "u", with: ""); return v.hasPrefix("0x") ? UInt32(v.dropFirst(2), radix: 16) : UInt32(v) }
return words.count == 8 ? words : nil
}
func refuse(_ why: String) -> NSError { NSError(domain: "pack", code: 2, userInfo: [NSLocalizedDescriptionKey: "pack \(dir): \(why)"]) }
let generator = defineU32("IGNEUM_GENERATOR") ?? 1
guard generator == 2 || generator == 3 else { throw refuse("program pack generator \(generator) is not a generator version this worker runs (2 or 3)") }
let packClass = generator == 3 ? "v3" : "v2"
if let named = defineStr("IGNEUM_PROGRAM_CLASS"), named != packClass { throw refuse("program pack IGNEUM_PROGRAM_CLASS \"\(named)\" does not match IGNEUM_GENERATOR \(generator)") }
let eraHex = defineStr("IGNEUM_ERA_SEED_HEX") ?? ""
if wantClass != "" && wantClass != packClass { throw refuse("program class mismatch: this pack is class \(packClass), the line names class \(wantClass) (export the pack again)") }
if wantEra != "" && packClass == "v3" && wantEra.lowercased() != eraHex.lowercased() { throw refuse("era seed mismatch: this pack was drawn under era \(eraHex.isEmpty ? "(none)" : String(eraHex.prefix(16))), the line names era \(wantEra.prefix(16)) (export the pack again)") }
if let packSeed = defineStr("IGNEUM_SEED_BYTES_HEX"), packSeed.lowercased() != seedHex.lowercased() { throw refuse("the pack is for epoch \(packSeed.prefix(16)), not \(seedHex.prefix(16))") }
let attempt = defineU32("IGNEUM_PROGRAM_ATTEMPT") ?? 0
guard let seedw = defineWords("IGNEUM_SEEDW_INIT") else { throw refuse("program.h has no IGNEUM_SEEDW_INIT with 8 words") }
let want = attemptWords(seed, attempt)
if want != seedw { throw refuse("program pack and its seeds disagree: IGNEUM_SEEDW_INIT is not attempt \(attempt) of the epoch seed \(seedHex.prefix(16))") }
let loads = Int(defineU32("IGNEUM_LOADS_PER_HASH") ?? 128)
guard let msl = try? String(contentsOfFile: dir + "/program_bound.metal", encoding: .utf8) else { throw refuse("cannot read program_bound.metal") }
let c = try compileBound(gpu, msl: msl)
let sp = ServeProgram(seedHex: seedHex, packClass: packClass, eraHex: eraHex, loadsPerHash: loads, compiled: c)
store.add(sp)
return (sp, ms(t0, nowNs()))
}
// Builds the cache and dataset for a day seed (hex) unless resident. Returns (dataset, build ms).
func serveDataset(_ gpu: GPU, _ store: ServeStore, dayHex: String, day: [UInt8], datasetLog2: Int) -> (ServeDataset, Double) {
if let d = store.dataset(dayHex) { return (d, 0) }
let t0 = nowNs()
let key = seedWordsBytes(day)
let ctx = DatasetContext(gpu: gpu, closedForm: false, dayString: "day/" + dayHex, key: key)
let buf = ctx.makeDataset(log2: datasetLog2)
let sd = ServeDataset(dayHex: dayHex, ctx: ctx, buffer: buf)
store.add(sd)
return (sd, ms(t0, nowNs()))
}
func runServe(_ opts: Options) -> Never {
let gpu = GPU()
let datasetLog2 = opts.datasetLog2
let batch = 1 << opts.batchLog2 // nonces per dispatch
let store = ServeStore()
let prepareQueue = DispatchQueue(label: "igneum.prepare") // one prepare at a time, off the job loop
guard let outBuf = gpu.device.makeBuffer(length: batch * 8, options: .storageModeShared) else { emit("error 0 cannot allocate the output buffer"); exit(1) }
emit("ready metal \(gpu.device.name.replacingOccurrences(of: " ", with: "_")) dataset-log2 \(datasetLog2) batch \(batch) prepare \(opts.noPrepare ? 0 : 1) race \(opts.race)")
var lastPair: (String, String)? = nil
while let line = readLine(strippingNewline: true) {
var f = line.split(separator: " ").map(String.init)
if f.isEmpty { continue }
if f[0] == "quit" { break }
// Counter ASIC 2.0 (5 October 2026): a job or prepare line of a class v3 epoch ends with `class=v3 era=<hex>`
// (a class v2 line is the line of before). A class v2 program is generated here from the seed (the Swift
// version 2 generator); a class v3 program is compiled from the pack the prepare line names (servePackProgram),
// so a v3 job on seeds with no prepared v3 pack asks for one (`need`) instead of mining the wrong program.
var wantClass = "", wantEra = ""
while let last = f.last, last.hasPrefix("class=") || last.hasPrefix("era=") {
if last.hasPrefix("class=") { wantClass = String(last.dropFirst(6)) } else { wantEra = String(last.dropFirst(4)) }
f.removeLast()
}
if wantClass != "" && wantClass != "v2" && wantClass != "v3" {
let id = f.count > 1 ? f[1] : "0"
if f[0] == "prepare" { emit("prepare-failed \(id) \(f.count > 2 ? f[2] : "0") program class \(wantClass) is not one this worker runs (v2 or v3)") }
else { emit("error \(id) program class \(wantClass) is not one this worker runs (v2 or v3)") }
continue
}
if f[0] == "prepare" {
if opts.noPrepare { emit("info ignored (started with --no-prepare): \(line)"); continue }
if f.count < 3 { emit("prepare-failed 0 0 malformed prepare line (need epoch_seed_hex and day_seed_hex)"); continue }
guard let epochSeed = unhex(f[1]), epochSeed.count == 32, let daySeed = unhex(f[2]) else {
emit("prepare-failed \(f[1]) \(f[2]) bad field (epoch_seed 64 hex, day_seed hex)"); continue
}
let (epochHex, dayHex) = (f[1], f[2])
let have = (store.program(epochHex, cls: wantClass, era: wantEra) != nil, store.dataset(dayHex) != nil)
if have.0 && have.1 { emit("prepared \(epochHex) \(dayHex) 0 program 0 dataset 0 (already resident)"); continue }
let packDir = f.count > 3 ? f[3...].joined(separator: " ") : ""
if wantClass == "v3" && packDir.isEmpty { emit("prepare-failed \(epochHex) \(dayHex) a class v3 program is compiled from a pack: send prepare with a pack directory (igneum-miner --prepare-packs <dir>)"); continue }
prepareQueue.async {
let t0 = nowNs()
do {
let (sp, progMs) = wantClass == "v3"
? try servePackProgram(gpu, store, seedHex: epochHex, seed: epochSeed, dir: packDir, wantClass: wantClass, wantEra: wantEra)
: try serveProgram(gpu, store, seedHex: epochHex, seed: epochSeed, datasetLog2: datasetLog2)
let (sd, dsMs) = serveDataset(gpu, store, dayHex: dayHex, day: daySeed, datasetLog2: datasetLog2)
// The race: the prepared program against its own dataset, exclusive windows between jobs
var raceMs = 0.0
if sp.raceDue && opts.race != "off" {
let r0 = nowNs()
emit(raceProgram(gpu, sp, dataset: sd.buffer, datasetLog2: datasetLog2, opts: opts, rounds: opts.raceRounds))
raceMs = ms(r0, nowNs())
}
let (np, nd) = store.counts()
emit("prepared \(epochHex) \(dayHex) \(fmt(ms(t0, nowNs()), 1)) program \(fmt(progMs, 1)) dataset \(fmt(dsMs, 1)) race \(fmt(raceMs, 1)) variant \(sp.compiled.variant) class \(sp.programClass) loads/hash \(sp.loadsPerHash) cache-fill \(fmt(sd.ctx.cacheFillGPUms, 1)) resident \(np) programs \(nd) datasets")
} catch { emit("prepare-failed \(epochHex) \(dayHex) \(wantClass == "v3" ? "" : "Metal compile failed: ")\(error.localizedDescription)") }
}
continue
}
if f[0] != "job" { emit("info ignored: \(line)"); continue }
if f.count < 8 { emit("error \(f.count > 1 ? f[1] : "0") malformed job line (need 7 fields after job)"); continue }
let jobId = f[1]
guard let prehash = unhex(f[2]), prehash.count == 32, let target = UInt64(f[3], radix: 16),
let nonceStart = UInt64(f[4]), let nonceCount = UInt64(f[5]),
let epochSeed = unhex(f[6]), epochSeed.count == 32, let daySeed = unhex(f[7]) else {
emit("error \(jobId) bad field (prehash 64 hex, target 16 hex, nonce_start u64, nonce_count u64, epoch_seed 64 hex, day_seed hex)"); continue
}
if nonceCount == 0 || nonceCount % 32 != 0 || (nonceStart & 31) != 0 { emit("error \(jobId) nonce_start must be 32-aligned and nonce_count a non-zero multiple of 32"); continue }
let t0 = nowNs()
let pair = (f[6], f[7])
let switched = lastPair == nil || lastPair! != pair
if switched { store.current = pair }
// Program and dataset for the pair: resident (prepared, or the current pair) or compiled inline now. A prepare
// of the same pair may be in flight on the queue; waiting for the queue makes this a join instead of a double build.
let program: ServeProgram
let dataset: ServeDataset
do {
if store.program(pair.0, cls: wantClass, era: wantEra) == nil || store.dataset(pair.1) == nil { prepareQueue.sync {} }
let sp: ServeProgram
if wantClass == "v3" {
guard let resident = store.program(pair.0, cls: wantClass, era: wantEra) else {
emit("need \(pair.0) \(pair.1)")
emit("error \(jobId) pack (metal): program class mismatch: no class v3 program for epoch \(pair.0.prefix(16)) is resident (this worker compiles v3 from a pack); send prepare with a pack directory")
continue
}
sp = resident
} else {
let (p, progMs) = try serveProgram(gpu, store, seedHex: pair.0, seed: epochSeed, datasetLog2: datasetLog2)
if progMs > 0 { emit("info program epoch \(pair.0.prefix(16)) loads/hash \(p.loadsPerHash) compiled inline in \(fmt(progMs, 1)) ms (not prepared)") }
sp = p
}
let (sd, dsMs) = serveDataset(gpu, store, dayHex: pair.1, day: daySeed, datasetLog2: datasetLog2)
if dsMs > 0 { emit("info dataset day \(pair.1) built inline in \(fmt(dsMs, 1)) ms (cache fill \(fmt(sd.ctx.cacheFillGPUms, 1)) ms, build \(fmt(sd.ctx.lastBuildGPUms, 1)) ms GPU; not prepared)") }
program = sp; dataset = sd
} catch { emit("error \(jobId) Metal compile failed: \(error)"); continue }
if switched, let prev = lastPair {
emit("info switched from epoch \(prev.0.prefix(16)) day \(prev.1) to epoch \(pair.0.prefix(16)) day \(pair.1) in \(fmt(ms(t0, nowNs()), 2)) ms (resident: \(store.counts().0) programs, \(store.counts().1) datasets)")
}
// Mine: chunks of at most `batch` lane nonces that share one high word
var remaining = nonceCount
var hi = UInt32(truncatingIfNeeded: nonceStart >> 32)
var lo = UInt32(truncatingIfNeeded: nonceStart)
var hashes: UInt64 = 0
var found = 0
var failed = false
let outPtr = outBuf.contents().bindMemory(to: UInt64.self, capacity: batch)
let kernel = program.compiled // the pair's kernel for this job (a race may swap it for the next)
while remaining > 0 {
let room = UInt64(UInt32.max - lo) + 1 // lane nonces left before the high word steps
let chunk = Int(min(min(remaining, UInt64(batch)), room))
var initw = blockInitWords(prehash: prehash, nonceHi: hi)
gpuLock.lock() // a race's exclusive windows fall between chunks
let cb = gpu.queue.makeCommandBuffer()!
let enc = cb.makeComputeCommandEncoder()!
encodeHash(enc, kernel, dataset: dataset.buffer, out: outBuf, base: lo, initw: &initw, count: chunk)
enc.endEncoding()
cb.commit(); cb.waitUntilCompleted()
gpuLock.unlock()
if let e = cb.error { emit("error \(jobId) dispatch failed: \(e)"); failed = true; break }
for i in 0..<chunk where outPtr[i] <= target {
let nonce = (UInt64(hi) << 32) | UInt64(lo &+ UInt32(i))
emit("found \(jobId) \(nonce) \(String(format: "%016llx", outPtr[i]))")
found += 1
}
hashes += UInt64(chunk)
remaining -= UInt64(chunk)
let (newLo, wrapped) = lo.addingReportingOverflow(UInt32(truncatingIfNeeded: chunk))
lo = newLo
if wrapped || (chunk == Int(room)) { hi &+= 1; lo = 0 }
}
if failed { continue }
emit("done \(jobId) \(hashes) \(fmt(ms(t0, nowNs()), 2))")
if switched, lastPair != nil {
// The first job on the new pair is done: the old pair goes (at most two of each were resident until now)
let (dp, dd) = store.prune(to: pair)
if dp + dd > 0 { emit("info dropped \(dp) program(s) and \(dd) dataset(s) of the previous pair") }
}
if program.raceDue && opts.race != "off" {
// A pair compiled inline (nobody prepared it) races now, in the background, exclusive windows between chunks
program.raceDue = false
prepareQueue.async { emit(raceProgram(gpu, program, dataset: dataset.buffer, datasetLog2: datasetLog2, opts: opts, rounds: opts.raceRounds)) }
}
lastPair = pair
}
exit(0)
}
// MARK: - Main
// The race alone (the Mac measurement, docs/bench-log.md "miner performance: variant racing"): the memory-hard
// dataset for --day, the version-2 program for --seed, every variant timed for --race-rounds rounds, a table.
func runRaceTest(_ opts: Options) -> Never {
let gpu = GPU()
print("igneum-bench --race-test on \(gpu.device.name): seed \"\(opts.seed)\", day \"\(opts.day)\", dataset 2^\(opts.datasetLog2) words, batch 2^\(opts.batchLog2), \(opts.raceRounds) rounds, \(opts.raceBenchMs) ms per window")
let ctx = DatasetContext(gpu: gpu, closedForm: false, dayString: opts.day)
let t0 = nowNs()
let dataset = ctx.makeDataset(log2: opts.datasetLog2)
print("dataset built in \(fmt(ms(t0, nowNs()), 0)) ms (cache fill \(fmt(ctx.cacheFillGPUms, 0)) ms GPU, build \(fmt(ctx.lastBuildGPUms, 0)) ms GPU)")
let p = generateProgramV2(seedString: opts.seed, bytes: Array(opts.seed.utf8))
print("program: \(describeProgram(p))")
let base: CompiledHash
do { base = try compileBound(gpu, msl: generateMSL(p, datasetLog2: opts.datasetLog2, source: .stored, bound: true)) } catch { print("FAIL: \(error)"); exit(1) }
let sp = ServeProgram(seedHex: String(repeating: "0", count: 64), program: p, compiled: base)
let line = raceProgram(gpu, sp, dataset: dataset, datasetLog2: opts.datasetLog2, opts: opts, rounds: opts.raceRounds, table: true)
print(line)
exit(line.contains("winner ") ? 0 : 1)
}
var opts = parseArgs()
if opts.raceRounds == 0 { opts.raceRounds = opts.raceTest ? 3 : 1 }
generatorConfig = GeneratorConfig(loadWeight: opts.loadWeight, wideFrac: opts.wideFrac)
if opts.exportPack != nil { exportPack(opts) }
if opts.raceTest { runRaceTest(opts) }
if opts.serve { runServe(opts) }
if opts.anyTest { runTests(opts) }
let gpu = GPU()
print("igneum-bench")
print("GPU: \(gpu.device.name) (maxBufferLength \(gpu.device.maxBufferLength / (1 << 20)) MiB, unified memory \(gpu.device.hasUnifiedMemory))")
print("dataset: 2^\(opts.datasetLog2) uint32 = \(fmt(Double(1 << opts.datasetLog2) * 4 / Double(1 << 20), 0)) MiB, day \"\(opts.day)\", construction \(opts.closedForm ? "closed-form (--closed-form)" : "memory-hard (default; MEMHARD.md)")")
if generatorConfig.loadWeight != 25 || generatorConfig.wideFrac != 0 {
print("generator levers: load weight \(generatorConfig.loadWeight) percent, wide-load fraction \(generatorConfig.wideFrac) percent (NOT the default generator)")
print("generator weights: " + generatorConfig.weights.map { "\($0.0.rawValue)=\($0.1)" }.joined(separator: " "))
}
// Dataset construction, timed. Memory-hard: compile the cache fill + build kernels, fill the 256 MiB cache on the
// GPU, build the dataset from it. Closed form: the original fill kernel. The CPU cache (verifier side) is computed
// afterwards on one core and compared word for word with the GPU cache.
let ctx = DatasetContext(gpu: gpu, closedForm: opts.closedForm, dayString: opts.day)
let datasetWords = 1 << opts.datasetLog2
guard let dataset = gpu.device.makeBuffer(length: datasetWords * 4, options: .storageModePrivate) else {
print("FAIL: cannot allocate dataset buffer"); exit(1)
}
print("dataset kernels: compile \(fmt(ctx.compileMs)) ms")
var cacheRefillMs = 0.0
if !ctx.closed {
cacheRefillMs = ctx.refillCache()
print("cache fill (GPU): \(fmt(ctx.cacheFillGPUms)) ms GPU time first (\(fmt(ctx.cacheFillWallMs)) ms wall), \(fmt(cacheRefillMs)) ms GPU time second; \(cacheSegments) chains x \(cacheLinesPerSegment) ChaCha\(chachaRounds) blocks, 256 MiB written")
}
ctx.build(into: dataset, words: datasetWords)
let build1GPU = ctx.lastBuildGPUms, build1Wall = ctx.lastBuildWallMs
ctx.build(into: dataset, words: datasetWords)
let build2GPU = ctx.lastBuildGPUms
let gib = Double(datasetWords * 4) / Double(1 << 30)
if ctx.closed {
print("dataset fill (closed form): \(fmt(build1GPU)) ms GPU time first (\(fmt(build1Wall)) ms wall), \(fmt(build2GPU)) ms second -> \(fmt(gib / (build2GPU / 1000))) GB/s write (second, GPU time)")
} else {
let items = Double(datasetWords / 16)
print("dataset build (memory-hard): \(fmt(build1GPU)) ms GPU time first (\(fmt(build1Wall)) ms wall), \(fmt(build2GPU)) ms second; \(Int(items)) items, \(fmt(items / (build2GPU / 1000) / 1e6, 1)) M items/s, \(fmt(items * 8 / (build2GPU / 1000) / 1e9, 2)) G cache-line reads/s (second)")
let cc = ctx.cacheCheck()
print(cc.detail)
if !cc.ok { print("FAIL: GPU and CPU cache differ"); exit(1) }
let sc = ctx.sampleCheck(log2: min(opts.datasetLog2, 24), indices: [0, 1, 15, 16, UInt32((1 << min(opts.datasetLog2, 24)) - 1)] + (0..<1019).map { _ in UInt32.random(in: 0..<UInt32(1 << min(opts.datasetLog2, 24))) })
print(sc.detail)
if !sc.ok { print("FAIL: GPU dataset words differ from the CPU derivation"); exit(1) }
}
var results = [EpochResult]()
for epoch in 0..<max(opts.hours, 1) {
let seedString = epoch == 0 ? opts.seed : "\(opts.seed)/epoch\(epoch)"
results.append(runEpoch(gpu: gpu, opts: opts, seedString: seedString, dataset: dataset, ctx: ctx))
}
// Summary table
print("\n=== summary (\(gpu.device.name), dataset 2^\(opts.datasetLog2) words \(ctx.modeName)\(opts.inlineDataset ? " INLINE shortcut kernel" : ""), batch 2^\(opts.batchLog2) x \(opts.batches)) ===")
print("| seed | compile ms (lib+pipe) | Mhash/s (wall) | Mhash/s (GPU) | GB/s useful (wall) | loads/hash | items/warp | CPU verify ms/warp (avg of 20) | verify |")
print("|---|---|---|---|---|---|---|---|---|")
for r in results {
let avg = r.verify.map { $0.repMs }.reduce(0, +) / Double(max(r.verify.count, 1))
print("| \(r.seed) | \(fmt(r.libraryMs + r.pipelineMs, 1)) | \(fmt(r.hashesPerSecWall / 1e6, 3)) | \(fmt(r.hashesPerSecGPU / 1e6, 3)) | \(fmt(r.gbpsWall)) | \(r.loadsPerHash) | \(r.itemsPerWarp) | \(fmt(avg, 3)) | \(r.allPass ? "PASS" : "FAIL") (\(r.verify.count) warps) |")
}
let overall = results.allSatisfy { $0.allPass }
if ctx.closed {
print("dataset fill: \(fmt(build2GPU)) ms GPU time for \(datasetWords * 4 / (1 << 20)) MiB (closed form)")
} else {
print("cache fill: \(fmt(cacheRefillMs)) ms GPU, \(fmt(ctx.cpuSide()!.fillMs, 1)) ms one CPU core; dataset build: \(fmt(build2GPU)) ms GPU for \(datasetWords * 4 / (1 << 20)) MiB (memory-hard)")
}
print("OVERALL: \(overall ? "PASS" : "FAIL")")
exit(overall ? 0 : 1)