Exporter writes kernel.cl next to kernel.cu (same instruction list; memory-hard core emitted in a third, OpenCL C dialect with the same literals as memhard.h). Pack headers are now C99-safe so a plain C host can include them. proto-opencl/host.c: C99 + OpenCL 1.2 API, device list, runtime build, cache fill and FNV check, dataset build and self-test, 3 vector warps standalone and in batch, bench and sweep as host.cu, whole-batch fingerprint. The 32-lane exchange is sub_group_shuffle_xor only when the queried sub-group size for a 32-item work-group is exactly 32; otherwise a local-memory exchange with one barrier per exchange, so wave64 hardware cannot change the hash (WAVEFRONT.md). build.sh (macOS, Linux), build.bat (MSVC), README with the exact AMD-rig commands. Proven without AMD silicon: Apple OpenCL 1.2 on the M5 Max 96/96 on all three packs (45.0 Mhash/s at 1 GiB, Apple number, not AMD); pocl 7.2 CPU device 96/96 on both exchange paths including the real sub_group_shuffle_xor text; CPU emulator 7 configurations incl. 64-wide sub-groups, identical fingerprint f99fb375b3abeaf5 everywhere. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
2504 lines
131 KiB
Swift
2504 lines
131 KiB
Swift
// igneum-bench: first prototype of Igneum's random-program GPU proof-of-work.
|
|
// One file. Build: swiftc -O -o igneum-bench main.swift -framework Metal
|
|
// Metal shaders are compiled at runtime from generated source (no Xcode needed).
|
|
|
|
import Foundation
|
|
import Metal
|
|
|
|
// MARK: - Options
|
|
|
|
struct Options {
|
|
var seed = "igneum-genesis"
|
|
var day = "2026-10-03"
|
|
var hours = 2 // number of epochs (seeds) run in sequence; default 2 so verification covers 2 seeds
|
|
var batchLog2 = 22 // nonces per batch
|
|
var batches = 4 // timed batches
|
|
var datasetLog2 = 28 // 2^28 uint32 = 1 GiB
|
|
var verifyWarps = 3
|
|
var dumpDir: String? = nil
|
|
var exportPack: String? = nil // write a CUDA program pack for --seed into this directory and exit
|
|
// Hardening tests (added 3 October 2026). Any of these runs instead of the bench.
|
|
var fuzz: Int? = nil // --fuzz N: N random programs, GPU vs CPU on 4 random warps each
|
|
var fuzzSeed = "igneum-fuzz-2026-10-03"
|
|
var edge = false // --edge: hand-built edge-case programs
|
|
var stats = false // --stats: output distribution sanity checks on 2^20 nonces
|
|
var determinism = false // --determinism: 5 identical GPU runs + double compile
|
|
var memcheck = false // --memcheck: static mask check + 4 MiB run with wrapping nonces
|
|
// Shortcut measurement (bench variant, not a test): compute dataset elements inline instead of loading them.
|
|
// Closed-form mode: every load computes ds_elem. Memory-hard mode: every load derives the item from the cache.
|
|
var inlineDataset = false
|
|
// Dataset construction (added 3 October 2026). Default is the memory-hard cache construction (MEMHARD.md);
|
|
// --closed-form selects the original six-operation closed form so the two can be compared.
|
|
var closedForm = false
|
|
// Generator levers (MEMHARD.md section 7). Defaults reproduce the original generator exactly.
|
|
var loadWeight = 25 // --load-weight W: percent weight of the load op (default 25)
|
|
var wideFrac = 0 // --wide-frac P: percent of load instructions emitted as warp-coalesced wide loads
|
|
var anyTest: Bool { fuzz != nil || edge || stats || determinism || memcheck }
|
|
}
|
|
|
|
func parseArgs() -> Options {
|
|
var o = Options()
|
|
var args = Array(CommandLine.arguments.dropFirst())
|
|
func take() -> String { args.isEmpty ? "" : args.removeFirst() }
|
|
while !args.isEmpty {
|
|
let a = take()
|
|
switch a {
|
|
case "--seed": o.seed = take()
|
|
case "--day": o.day = take()
|
|
case "--hours": o.hours = Int(take()) ?? o.hours
|
|
case "--batch-log2": o.batchLog2 = Int(take()) ?? o.batchLog2
|
|
case "--batches": o.batches = Int(take()) ?? o.batches
|
|
case "--dataset-log2": o.datasetLog2 = Int(take()) ?? o.datasetLog2
|
|
case "--verify-warps": o.verifyWarps = Int(take()) ?? o.verifyWarps
|
|
case "--dump": o.dumpDir = take()
|
|
case "--export-pack": o.exportPack = take()
|
|
case "--fuzz": o.fuzz = Int(take()) ?? 200
|
|
case "--fuzz-seed": o.fuzzSeed = take()
|
|
case "--edge": o.edge = true
|
|
case "--stats": o.stats = true
|
|
case "--determinism": o.determinism = true
|
|
case "--memcheck": o.memcheck = true
|
|
case "--inline-dataset": o.inlineDataset = true
|
|
case "--closed-form": o.closedForm = true
|
|
case "--load-weight": o.loadWeight = Int(take()) ?? o.loadWeight
|
|
case "--wide-frac": o.wideFrac = Int(take()) ?? o.wideFrac
|
|
case "-h", "--help":
|
|
print("""
|
|
igneum-bench [--seed <string>] [--hours N] [--batch-log2 22] [--batches 4]
|
|
[--dataset-log2 28] [--verify-warps 3] [--dump <dir>] [--day <string>]
|
|
[--closed-form] original closed-form dataset (default: memory-hard cache construction, see MEMHARD.md)
|
|
[--load-weight W] generator lever (a): percent weight of the load op (default 25)
|
|
[--wide-frac P] generator lever (b): percent of loads emitted as warp-coalesced 128-byte loads (default 0)
|
|
[--export-pack <dir>] write the CUDA + OpenCL program pack for --seed, then exit
|
|
hardening tests (run instead of the bench; several may be combined; exit 0 only if all pass):
|
|
[--fuzz N [--fuzz-seed <string>]] N random programs, GPU vs CPU, 4 random warps each,
|
|
dataset size drawn from 64 MiB, 256 MiB, 1 GiB
|
|
[--edge] hand-built edge-case programs, GPU vs CPU
|
|
[--stats] output distribution sanity checks on 2^20 nonces, 3 seeds
|
|
[--determinism] 5 identical GPU runs of 2^20 nonces, double compile, dataset fill check
|
|
[--memcheck] static dataset-index mask check, 4 MiB run with wrapping nonces
|
|
shortcut measurement:
|
|
[--inline-dataset] bench variant: every load computes ds_elem(index) inline, no memory read
|
|
""")
|
|
exit(0)
|
|
default:
|
|
print("unknown argument \(a)"); exit(2)
|
|
}
|
|
}
|
|
return o
|
|
}
|
|
|
|
// MARK: - Integer helpers (CPU side, must match MSL bit for bit)
|
|
|
|
@inline(__always) func rotl32(_ x: UInt32, _ n: UInt32) -> UInt32 {
|
|
let n = n & 31
|
|
return n == 0 ? x : (x << n) | (x >> (32 - n))
|
|
}
|
|
@inline(__always) func rotr32(_ x: UInt32, _ n: UInt32) -> UInt32 {
|
|
let n = n & 31
|
|
return n == 0 ? x : (x >> n) | (x << (32 - n))
|
|
}
|
|
@inline(__always) func mulhi32(_ a: UInt32, _ b: UInt32) -> UInt32 {
|
|
UInt32(truncatingIfNeeded: (UInt64(a) &* UInt64(b)) >> 32)
|
|
}
|
|
@inline(__always) func splitmix32(_ v: UInt32) -> UInt32 {
|
|
var x = v
|
|
x ^= x >> 16; x &*= 0x7feb352d
|
|
x ^= x >> 15; x &*= 0x846ca68b
|
|
x ^= x >> 16
|
|
return x
|
|
}
|
|
// Dataset element, closed form of (daySeed, index). Same formula is emitted into the MSL.
|
|
@inline(__always) func datasetElem(_ i: UInt32, _ d0: UInt32, _ d1: UInt32) -> UInt32 {
|
|
var x = i ^ d0
|
|
x &*= 0x9E3779B1; x ^= x >> 15
|
|
x &+= d1
|
|
x &*= 0x85EBCA77; x ^= x >> 13
|
|
x &*= 0xC2B2AE3D; x ^= x >> 16
|
|
return x
|
|
}
|
|
|
|
// 32-byte seed (8 x uint32) from a string: FNV-1a 64 with four salts, each finalised.
|
|
func seedWords(_ s: String) -> [UInt32] {
|
|
var words = [UInt32]()
|
|
for salt in 0..<4 {
|
|
var h: UInt64 = 0xcbf29ce484222325 ^ (UInt64(salt) &* 0x9E3779B97F4A7C15)
|
|
for b in s.utf8 { h ^= UInt64(b); h &*= 0x100000001b3 }
|
|
h ^= h >> 33; h &*= 0xff51afd7ed558ccd; h ^= h >> 33
|
|
words.append(UInt32(truncatingIfNeeded: h))
|
|
words.append(UInt32(truncatingIfNeeded: h >> 32))
|
|
}
|
|
return words
|
|
}
|
|
|
|
struct SplitMix64 {
|
|
var s: UInt64
|
|
mutating func next() -> UInt64 {
|
|
s &+= 0x9E3779B97F4A7C15
|
|
var z = s
|
|
z = (z ^ (z >> 30)) &* 0xBF58476D1CE4E5B9
|
|
z = (z ^ (z >> 27)) &* 0x94D049BB133111EB
|
|
return z ^ (z >> 31)
|
|
}
|
|
mutating func below(_ n: Int) -> Int { Int(next() % UInt64(n)) }
|
|
}
|
|
|
|
// MARK: - Memory-hard dataset (added 3 October 2026, spec in MEMHARD.md)
|
|
//
|
|
// Cache: 2^26 words (256 MiB) = 2^22 lines of 16 words, in 2^16 segments of 64 lines. Each segment is a
|
|
// sequential chain of ChaCha12 blocks with feed-forward: in_j = prev_line ^ (sigma || K || seg || j || tag),
|
|
// line_j = core(in_j) + in_j, prev_0 = 0. Recomputing line j costs j + 1 block evaluations.
|
|
// Item t (16 words = 64 bytes): s = (K[0..7], t * MUL[i] + RC[i]), then 8 rounds of
|
|
// s = M_r(s); a = s[0] & (2^22 - 1); s ^= cache line a
|
|
// and a final M_8. M_r is the seed-parameterised mixer: per word (s ^ (RC + (r+1) * 0x9E3779B9)) * MUL,
|
|
// then one ChaCha-shaped column round and diagonal round with seed-drawn rotations.
|
|
// Dataset word w = item(w >> 4)[w & 15]. The hash kernel is unchanged: it still does dataset[r & MASK].
|
|
|
|
let cacheLog2Words = 26
|
|
let cacheSegmentLog2Lines = 6
|
|
let cacheWords = 1 << cacheLog2Words
|
|
let cacheLines = cacheWords >> 4
|
|
let cacheLinesPerSegment = 1 << cacheSegmentLog2Lines
|
|
let cacheSegments = cacheLines >> cacheSegmentLog2Lines
|
|
let cacheLineMask = UInt32(cacheLines - 1)
|
|
let itemRounds = 8
|
|
let chachaRounds = 12
|
|
let chachaSigma: [UInt32] = [0x61707865, 0x3320646e, 0x79622d32, 0x6b206574]
|
|
let cacheTag: [UInt32] = [0x49676e65, 0x756d4d48] // "Igne", "umMH"
|
|
|
|
// Rotation without the n == 0 check: every caller passes 1..31.
|
|
@inline(__always) func rotlc(_ x: UInt32, _ n: UInt32) -> UInt32 { (x << n) | (x >> (32 - n)) }
|
|
|
|
@inline(__always) func qr(_ s: UnsafeMutablePointer<UInt32>, _ a: Int, _ b: Int, _ c: Int, _ d: Int,
|
|
_ r1: UInt32, _ r2: UInt32, _ r3: UInt32, _ r4: UInt32) {
|
|
s[a] = s[a] &+ s[b]; s[d] ^= s[a]; s[d] = rotlc(s[d], r1)
|
|
s[c] = s[c] &+ s[d]; s[b] ^= s[c]; s[b] = rotlc(s[b], r2)
|
|
s[a] = s[a] &+ s[b]; s[d] ^= s[a]; s[d] = rotlc(s[d], r3)
|
|
s[c] = s[c] &+ s[d]; s[b] ^= s[c]; s[b] = rotlc(s[b], r4)
|
|
}
|
|
|
|
// y = ChaCha12 core(x) + x. Standard quarter-round rotations 16, 12, 8, 7; column then diagonal.
|
|
@inline(__always) func chachaBlock(_ x: UnsafePointer<UInt32>, _ y: UnsafeMutablePointer<UInt32>) {
|
|
for i in 0..<16 { y[i] = x[i] }
|
|
for _ in 0..<(chachaRounds / 2) {
|
|
qr(y, 0, 4, 8, 12, 16, 12, 8, 7); qr(y, 1, 5, 9, 13, 16, 12, 8, 7)
|
|
qr(y, 2, 6, 10, 14, 16, 12, 8, 7); qr(y, 3, 7, 11, 15, 16, 12, 8, 7)
|
|
qr(y, 0, 5, 10, 15, 16, 12, 8, 7); qr(y, 1, 6, 11, 12, 16, 12, 8, 7)
|
|
qr(y, 2, 7, 8, 13, 16, 12, 8, 7); qr(y, 3, 4, 9, 14, 16, 12, 8, 7)
|
|
}
|
|
for i in 0..<16 { y[i] = y[i] &+ x[i] }
|
|
}
|
|
|
|
// Mixer parameters drawn from the day key. Draw order: ROT[0..7] (1..31), MUL[0..15] (odd), RC[0..15].
|
|
final class MixParams {
|
|
let key: UnsafeMutablePointer<UInt32> // 8
|
|
let rot: UnsafeMutablePointer<UInt32> // 8
|
|
let mul: UnsafeMutablePointer<UInt32> // 16
|
|
let rc: UnsafeMutablePointer<UInt32> // 16
|
|
let keyWords: [UInt32]
|
|
var rotWords: [UInt32] { (0..<8).map { rot[$0] } }
|
|
var mulWords: [UInt32] { (0..<16).map { mul[$0] } }
|
|
var rcWords: [UInt32] { (0..<16).map { rc[$0] } }
|
|
init(key k: [UInt32]) {
|
|
precondition(k.count == 8)
|
|
keyWords = k
|
|
key = UnsafeMutablePointer<UInt32>.allocate(capacity: 8)
|
|
rot = UnsafeMutablePointer<UInt32>.allocate(capacity: 8)
|
|
mul = UnsafeMutablePointer<UInt32>.allocate(capacity: 16)
|
|
rc = UnsafeMutablePointer<UInt32>.allocate(capacity: 16)
|
|
for i in 0..<8 { key[i] = k[i] }
|
|
var rng = SplitMix64(s: UInt64(k[0]) | (UInt64(k[1]) << 32))
|
|
for i in 0..<8 { rot[i] = UInt32(1 + rng.below(31)) }
|
|
for i in 0..<16 { mul[i] = UInt32(truncatingIfNeeded: rng.next()) | 1 }
|
|
for i in 0..<16 { rc[i] = UInt32(truncatingIfNeeded: rng.next()) }
|
|
}
|
|
}
|
|
|
|
// M_r on 16 words in place. rk = (r + 1) * 0x9E3779B9 mod 2^32.
|
|
@inline(__always) func mixer(_ s: UnsafeMutablePointer<UInt32>, _ rk: UInt32, _ mp: MixParams) {
|
|
let mul = mp.mul, rc = mp.rc, R = mp.rot
|
|
for i in 0..<16 { s[i] = (s[i] ^ (rc[i] &+ rk)) &* mul[i] }
|
|
qr(s, 0, 4, 8, 12, R[0], R[1], R[2], R[3]); qr(s, 1, 5, 9, 13, R[0], R[1], R[2], R[3])
|
|
qr(s, 2, 6, 10, 14, R[0], R[1], R[2], R[3]); qr(s, 3, 7, 11, 15, R[0], R[1], R[2], R[3])
|
|
qr(s, 0, 5, 10, 15, R[4], R[5], R[6], R[7]); qr(s, 1, 6, 11, 12, R[4], R[5], R[6], R[7])
|
|
qr(s, 2, 7, 8, 13, R[4], R[5], R[6], R[7]); qr(s, 3, 4, 9, 14, R[4], R[5], R[6], R[7])
|
|
}
|
|
|
|
@inline(__always) func roundKey(_ r: Int) -> UInt32 { UInt32(r + 1) &* 0x9E3779B9 }
|
|
|
|
// One segment of the cache: 64 chained lines written at cache[seg * 1024 ...].
|
|
func cpuFillSegment(_ cache: UnsafeMutablePointer<UInt32>, seg: Int, key: UnsafePointer<UInt32>) {
|
|
var inp = [UInt32](repeating: 0, count: 16)
|
|
inp.withUnsafeMutableBufferPointer { ib in
|
|
let x = ib.baseAddress!
|
|
var prev: UnsafePointer<UInt32>? = nil
|
|
for j in 0..<cacheLinesPerSegment {
|
|
let line = cache + ((seg << cacheSegmentLog2Lines) + j) * 16
|
|
for i in 0..<4 { x[i] = chachaSigma[i] }
|
|
for i in 0..<8 { x[4 + i] = key[i] }
|
|
x[12] = UInt32(seg); x[13] = UInt32(j); x[14] = cacheTag[0]; x[15] = cacheTag[1]
|
|
if let p = prev { for i in 0..<16 { x[i] ^= p[i] } }
|
|
chachaBlock(x, line)
|
|
prev = UnsafePointer(line)
|
|
}
|
|
}
|
|
}
|
|
|
|
// The whole 256 MiB cache on one CPU core. Returns wall milliseconds.
|
|
func cpuFillCache(_ cache: UnsafeMutablePointer<UInt32>, key: [UInt32]) -> Double {
|
|
let t0 = nowNs()
|
|
key.withUnsafeBufferPointer { kb in
|
|
for seg in 0..<cacheSegments { cpuFillSegment(cache, seg: seg, key: kb.baseAddress!) }
|
|
}
|
|
return Double(nowNs() - t0) / 1e6
|
|
}
|
|
|
|
// Derive `n` items (indices ts[0..n)) into out[k * 16 ...], all n chains interleaved round by round so the
|
|
// cache-line misses of independent items overlap in the memory system. This is how the verifier reaches
|
|
// memory-level parallelism without threads: the 32 lanes of a warp are independent.
|
|
func deriveItems(_ ts: UnsafePointer<UInt32>, _ n: Int, _ mp: MixParams, cache: UnsafePointer<UInt32>, out: UnsafeMutablePointer<UInt32>) {
|
|
let key = mp.key, mul = mp.mul, rc = mp.rc
|
|
for k in 0..<n {
|
|
let s = out + k * 16
|
|
let t = ts[k]
|
|
for i in 0..<8 { s[i] = key[i] }
|
|
for i in 0..<8 { s[8 + i] = t &* mul[i] &+ rc[i] }
|
|
}
|
|
for r in 0..<itemRounds {
|
|
let rk = roundKey(r)
|
|
for k in 0..<n { mixer(out + k * 16, rk, mp) }
|
|
for k in 0..<n {
|
|
let s = out + k * 16
|
|
let line = cache + Int(s[0] & cacheLineMask) * 16
|
|
for i in 0..<16 { s[i] ^= line[i] }
|
|
}
|
|
}
|
|
let rk = roundKey(itemRounds)
|
|
for k in 0..<n { mixer(out + k * 16, rk, mp) }
|
|
}
|
|
|
|
func deriveItem(_ t: UInt32, _ mp: MixParams, cache: UnsafePointer<UInt32>) -> [UInt32] {
|
|
var out = [UInt32](repeating: 0, count: 16)
|
|
var tt = t
|
|
out.withUnsafeMutableBufferPointer { ob in deriveItems(&tt, 1, mp, cache: cache, out: ob.baseAddress!) }
|
|
return out
|
|
}
|
|
|
|
// The CPU verifier's view of the memory-hard dataset: the 256 MiB cache and nothing else.
|
|
final class MemhardCPU {
|
|
let mp: MixParams
|
|
let cache: UnsafeMutablePointer<UInt32>
|
|
let fillMs: Double
|
|
private let items = UnsafeMutablePointer<UInt32>.allocate(capacity: 64 * 16)
|
|
private let uniq = UnsafeMutablePointer<UInt32>.allocate(capacity: 64)
|
|
private let slot = UnsafeMutablePointer<Int>.allocate(capacity: 64)
|
|
var derivations = 0 // items derived so far (statistics)
|
|
init(key: [UInt32]) {
|
|
mp = MixParams(key: key)
|
|
cache = UnsafeMutablePointer<UInt32>.allocate(capacity: cacheWords)
|
|
fillMs = cpuFillCache(cache, key: key)
|
|
}
|
|
func word(_ w: UInt32) -> UInt32 { deriveItem(w >> 4, mp, cache: UnsafePointer(cache))[Int(w & 15)] }
|
|
// out[k] = dataset[idx[k]] for k < n (n <= 64). Equal items are derived once.
|
|
func fetch(_ idx: UnsafePointer<UInt32>, _ n: Int, _ out: UnsafeMutablePointer<UInt32>) {
|
|
var u = 0
|
|
for k in 0..<n {
|
|
let t = idx[k] >> 4
|
|
var found = -1
|
|
for j in 0..<u where uniq[j] == t { found = j; break }
|
|
if found < 0 { uniq[u] = t; found = u; u += 1 }
|
|
slot[k] = found
|
|
}
|
|
deriveItems(UnsafePointer(uniq), u, mp, cache: UnsafePointer(cache), out: items)
|
|
derivations += u
|
|
for k in 0..<n { out[k] = items[slot[k] * 16 + Int(idx[k] & 15)] }
|
|
}
|
|
}
|
|
|
|
// What the CPU interpreter reads dataset words from. Closed form (day words d0, d1) or the memory-hard cache.
|
|
final class DatasetSource {
|
|
let mask: UInt32
|
|
let day: (UInt32, UInt32)
|
|
let memhard: MemhardCPU?
|
|
init(mask: UInt32, day: (UInt32, UInt32), memhard: MemhardCPU?) { self.mask = mask; self.day = day; self.memhard = memhard }
|
|
var modeName: String { memhard == nil ? "closed-form" : "memory-hard" }
|
|
func word(_ w: UInt32) -> UInt32 { memhard?.word(w & mask) ?? datasetElem(w & mask, day.0, day.1) }
|
|
func fetch(_ idx: UnsafePointer<UInt32>, _ n: Int, _ out: UnsafeMutablePointer<UInt32>) {
|
|
if let m = memhard { m.fetch(idx, n, out) }
|
|
else { for k in 0..<n { out[k] = datasetElem(idx[k], day.0, day.1) } }
|
|
}
|
|
}
|
|
|
|
// MARK: - Program
|
|
|
|
// wload (added 3 October 2026, lever b): warp-coalesced load. All 32 lanes read consecutive words of one
|
|
// 128-byte block whose address comes from lane 0's source register. Never emitted unless --wide-frac > 0.
|
|
enum Op: String { case add, sub, mul, mulhi, xor, or, rotl, rotr, mad, shfl, load, wload }
|
|
|
|
struct Instr {
|
|
var op: Op
|
|
var dst: Int
|
|
var a: Int // source register, never equal to dst
|
|
var b: Int // second source (mad only)
|
|
var imm: UInt32 // add immediate A
|
|
var imm2: UInt32 // add immediate B
|
|
var rot: UInt32 // rotl amount 1..31
|
|
var bit: Int // selector bit of r0 for add
|
|
var mask: Int // shuffle xor mask: 1,2,4,8,16
|
|
}
|
|
|
|
struct Program {
|
|
let seedString: String
|
|
let seed: [UInt32]
|
|
let instrs: [Instr]
|
|
static let iterations = 8
|
|
static let count = 64
|
|
var loadsPerHash: Int { instrs.filter { $0.op == .load || $0.op == .wload }.count * Program.iterations }
|
|
var wideLoadsPerHash: Int { instrs.filter { $0.op == .wload }.count * Program.iterations }
|
|
var hasWide: Bool { instrs.contains { $0.op == .wload } }
|
|
// Distinct dataset items a 32-lane warp touches per hash: 32 per plain load, 2 per wide load (128 B = 2 items).
|
|
var itemsPerWarp: Int { (loadsPerHash - wideLoadsPerHash) * 32 + wideLoadsPerHash * 2 }
|
|
var histogram: [(String, Int)] {
|
|
var d = [String: Int]()
|
|
for i in instrs { d[i.op.rawValue, default: 0] += 1 }
|
|
return d.sorted { $0.1 != $1.1 ? $0.1 > $1.1 : $0.0 < $1.0 } // count desc, then name, so output is deterministic
|
|
}
|
|
}
|
|
|
|
// Weights sum to 100. Loads are 25 percent so the kernel leans on memory.
|
|
let opWeights: [(Op, Int)] = [(.load, 25), (.add, 12), (.xor, 10), (.mul, 8), (.mad, 8), (.shfl, 8),
|
|
(.rotl, 7), (.sub, 6), (.mulhi, 6), (.rotr, 6), (.or, 4)]
|
|
|
|
// Generator levers (3 October 2026). The defaults reproduce the original generator instruction for instruction:
|
|
// with loadWeight 25 the weight table above is used unchanged, and with wideFrac 0 no load becomes a wload.
|
|
// Neither lever consumes extra random draws, so a program differs from the default one only where the lever acts.
|
|
struct GeneratorConfig {
|
|
var loadWeight = 25
|
|
var wideFrac = 0
|
|
// Scaled weights: load gets loadWeight, the other ten ops share the rest in their original proportions,
|
|
// rounded by largest remainder so the table still sums to 100.
|
|
var weights: [(Op, Int)] {
|
|
if loadWeight == 25 { return opWeights }
|
|
let others = opWeights.dropFirst()
|
|
let total = others.reduce(0) { $0 + $1.1 } // 75
|
|
let budget = 100 - loadWeight
|
|
var scaled = others.map { (op: $0.0, floor: ($0.1 * budget) / total, rem: ($0.1 * budget) % total) }
|
|
var sum = scaled.reduce(0) { $0 + $1.floor }
|
|
let order = scaled.indices.sorted { scaled[$0].rem != scaled[$1].rem ? scaled[$0].rem > scaled[$1].rem : $0 < $1 }
|
|
var k = 0
|
|
while sum < budget { scaled[order[k]].floor += 1; sum += 1; k += 1 }
|
|
return [(.load, loadWeight)] + scaled.map { ($0.op, $0.floor) }
|
|
}
|
|
}
|
|
var generatorConfig = GeneratorConfig()
|
|
|
|
func generateProgram(seedString: String) -> Program {
|
|
let sw = seedWords(seedString)
|
|
var rng = SplitMix64(s: (UInt64(sw[0]) | (UInt64(sw[1]) << 32)) ^ ((UInt64(sw[2]) | (UInt64(sw[3]) << 32)) &* 0x9E3779B97F4A7C15))
|
|
var instrs = [Instr]()
|
|
let weights = generatorConfig.weights
|
|
for _ in 0..<Program.count {
|
|
var roll = rng.below(100)
|
|
var op = Op.add
|
|
for (o, w) in weights { if roll < w { op = o; break }; roll -= w }
|
|
let dst = rng.below(8)
|
|
var a = rng.below(7); if a >= dst { a += 1 }
|
|
let b = rng.below(8)
|
|
let imm = UInt32(truncatingIfNeeded: rng.next())
|
|
let imm2 = UInt32(truncatingIfNeeded: rng.next())
|
|
let rot = UInt32(1 + rng.below(31))
|
|
let bit = rng.below(32)
|
|
let mask = 1 << rng.below(5)
|
|
// Lever (b): the already-drawn selector bit decides whether a load is wide, so the draw stream is unchanged.
|
|
if op == .load && bit * 100 < generatorConfig.wideFrac * 32 { op = .wload }
|
|
instrs.append(Instr(op: op, dst: dst, a: a, b: b, imm: imm, imm2: imm2, rot: rot, bit: bit, mask: mask))
|
|
}
|
|
return Program(seedString: seedString, seed: sw, instrs: instrs)
|
|
}
|
|
|
|
// MARK: - MSL generation
|
|
|
|
func hex(_ v: UInt32) -> String { String(format: "0x%08xu", v) }
|
|
|
|
// The memory-hard core as source text, in Metal, CUDA C++ or OpenCL C. Mixer parameters are
|
|
// literals so the GPU kernels, the CUDA pack, the OpenCL pack and the host reference share one text. Names are prefixed mh_.
|
|
// In the CUDA dialect every function is IGNEUM_HD (host and device) so host.cu can derive items too.
|
|
// In the OpenCL dialect the cache pointers carry the __global address space (OpenCL C 1.2 has no generic space).
|
|
enum CoreDialect { case metal, cuda, opencl }
|
|
|
|
func emitMemhardCore(_ mp: MixParams, cuda: Bool) -> String { emitMemhardCore(mp, dialect: cuda ? .cuda : .metal) }
|
|
|
|
func emitMemhardCore(_ mp: MixParams, dialect: CoreDialect) -> String {
|
|
let U: String, fn: String, cptr: String, wptr: String, lptr: String, lcptr: String
|
|
switch dialect {
|
|
case .metal: (U, fn, cptr, wptr, lptr, lcptr) = ("uint", "inline", "device const uint*", "device uint*", "thread uint*", "const thread uint*")
|
|
case .cuda: (U, fn, cptr, wptr, lptr, lcptr) = ("uint32_t", "IGNEUM_HD", "const uint32_t*", "uint32_t*", "uint32_t*", "const uint32_t*")
|
|
case .opencl: (U, fn, cptr, wptr, lptr, lcptr) = ("uint", "static inline", "__global const uint*", "__global uint*", "uint*", "const uint*")
|
|
}
|
|
let K = mp.keyWords, R = mp.rotWords, M = mp.mulWords, C = mp.rcWords
|
|
var s = """
|
|
// Memory-hard dataset core (MEMHARD.md). Cache: 2^\(cacheLog2Words) words in 2^\(Int(log2(Double(cacheSegments)))) segments of \(cacheLinesPerSegment) chained ChaCha\(chachaRounds) lines.
|
|
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
|
|
#define MH_CACHE_LINE_MASK \(hex(cacheLineMask))
|
|
#define MH_SEGMENT_LINES \(cacheLinesPerSegment)u
|
|
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
|
\(fn) \(U) mh_rotl(\(U) x, \(U) n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
|
|
|
// y = ChaCha\(chachaRounds) core(x) + x
|
|
\(fn) void mh_chacha_block(\(lcptr) x, \(lptr) y) {
|
|
for (\(U) i = 0u; i < 16u; ++i) y[i] = x[i];
|
|
for (\(U) r = 0u; r < \(chachaRounds / 2)u; ++r) {
|
|
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
|
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
|
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
|
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
|
}
|
|
for (\(U) i = 0u; i < 16u; ++i) y[i] += x[i];
|
|
}
|
|
|
|
// One cache segment: \(cacheLinesPerSegment) chained lines written at cache[seg * \(cacheLinesPerSegment * 16)]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
|
\(fn) void mh_cache_segment(\(wptr) cache, \(U) seg) {
|
|
\(U) prev[16]; \(U) x[16]; \(U) y[16];
|
|
for (\(U) i = 0u; i < 16u; ++i) prev[i] = 0u;
|
|
for (\(U) j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
|
x[0] = \(hex(chachaSigma[0])) ^ prev[0]; x[1] = \(hex(chachaSigma[1])) ^ prev[1]; x[2] = \(hex(chachaSigma[2])) ^ prev[2]; x[3] = \(hex(chachaSigma[3])) ^ prev[3];
|
|
|
|
"""
|
|
for i in 0..<8 { s += " x[\(4 + i)] = \(hex(K[i])) ^ prev[\(4 + i)];\n" }
|
|
s += """
|
|
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = \(hex(cacheTag[0])) ^ prev[14]; x[15] = \(hex(cacheTag[1])) ^ prev[15];
|
|
mh_chacha_block(x, y);
|
|
\(wptr) line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
|
for (\(U) i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
|
}
|
|
}
|
|
|
|
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
|
\(fn) void mh_mixer(\(lptr) s, \(U) rk) {
|
|
|
|
"""
|
|
for i in 0..<16 { s += " s[\(i)] = (s[\(i)] ^ (\(hex(C[i])) + rk)) * \(hex(M[i]));\n" }
|
|
let c = (0..<4).map { "\(R[$0])u" }.joined(separator: ", "), d = (4..<8).map { "\(R[$0])u" }.joined(separator: ", ")
|
|
s += """
|
|
MH_QR(s[0], s[4], s[8], s[12], \(c)) MH_QR(s[1], s[5], s[9], s[13], \(c))
|
|
MH_QR(s[2], s[6], s[10], s[14], \(c)) MH_QR(s[3], s[7], s[11], s[15], \(c))
|
|
MH_QR(s[0], s[5], s[10], s[15], \(d)) MH_QR(s[1], s[6], s[11], s[12], \(d))
|
|
MH_QR(s[2], s[7], s[8], s[13], \(d)) MH_QR(s[3], s[4], s[9], s[14], \(d))
|
|
}
|
|
|
|
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); \(itemRounds) rounds of mixer + cache line s[0] & mask; final mixer.
|
|
\(fn) void mh_item(\(cptr) cache, \(U) t, \(lptr) s) {
|
|
|
|
"""
|
|
for i in 0..<8 { s += " s[\(i)] = \(hex(K[i]));\n" }
|
|
for i in 0..<8 { s += " s[\(8 + i)] = t * \(hex(M[i])) + \(hex(C[i]));\n" }
|
|
s += """
|
|
for (\(U) r = 0u; r < \(itemRounds)u; ++r) {
|
|
mh_mixer(s, 0x9E3779B9u * (r + 1u));
|
|
\(cptr) line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
|
for (\(U) i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
|
}
|
|
mh_mixer(s, 0x9E3779B9u * \(itemRounds + 1)u);
|
|
}
|
|
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
|
\(fn) \(U) mh_word(\(cptr) cache, \(U) w) { \(U) s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
|
|
|
"""
|
|
return s
|
|
}
|
|
|
|
// Metal library with the cache fill and dataset build kernels for one day key.
|
|
func memhardMSL(_ mp: MixParams) -> String {
|
|
return """
|
|
#include <metal_stdlib>
|
|
using namespace metal;
|
|
\(emitMemhardCore(mp, cuda: false))
|
|
// One thread per segment (2^\(Int(log2(Double(cacheSegments)))) threads).
|
|
kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) {
|
|
mh_cache_segment(cache, gid);
|
|
}
|
|
// One thread per 64-byte item (dataset words / 16 threads).
|
|
kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],
|
|
uint gid [[thread_position_in_grid]]) {
|
|
uint s[16];
|
|
mh_item(cache, gid, s);
|
|
device uint* d = dataset + gid * 16u;
|
|
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
|
|
}
|
|
|
|
"""
|
|
}
|
|
|
|
// How the hash kernel gets dataset words. .stored reads the buffer (the honest kernel). The two inline
|
|
// variants are the shortcut measurements: every load recomputes the word instead of reading the dataset.
|
|
enum LoadSource {
|
|
case stored
|
|
case inlineClosed(UInt32, UInt32) // closed form: ds_elem(index, d0, d1), no memory read at all
|
|
case inlineMemhard(MixParams) // memory-hard: mh_word(cache, index), 8 dependent cache reads per word
|
|
}
|
|
|
|
func generateMSL(_ p: Program, datasetLog2: Int, source: LoadSource = .stored) -> String {
|
|
let mask = UInt32((1 << datasetLog2) - 1)
|
|
var s = """
|
|
#include <metal_stdlib>
|
|
using namespace metal;
|
|
|
|
#define MASK \(hex(mask))
|
|
constant uint SEEDW[8] = { \(p.seed.map(hex).joined(separator: ", ")) };
|
|
|
|
inline uint splitmix32(uint x) {
|
|
x ^= x >> 16; x *= 0x7feb352du;
|
|
x ^= x >> 15; x *= 0x846ca68bu;
|
|
x ^= x >> 16;
|
|
return x;
|
|
}
|
|
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
|
|
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
|
inline uint ds_elem(uint i, uint d0, uint d1) {
|
|
uint x = i ^ d0;
|
|
x *= 0x9E3779B1u; x ^= x >> 15;
|
|
x += d1;
|
|
x *= 0x85EBCA77u; x ^= x >> 13;
|
|
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
|
return x;
|
|
}
|
|
|
|
|
|
"""
|
|
if p.hasWide { s += "#define WMASK (MASK & ~31u)\n\n" }
|
|
var buffer0 = "device const uint* dataset [[buffer(0)]]"
|
|
if case .inlineMemhard(let mp) = source {
|
|
s += emitMemhardCore(mp, cuda: false) + "\n"
|
|
buffer0 = "device const uint* cache [[buffer(0)]]"
|
|
}
|
|
s += """
|
|
kernel void igneum_hash(\(buffer0),
|
|
device ulong* out [[buffer(1)]],
|
|
constant uint& baseNonce [[buffer(2)]],
|
|
uint gid [[thread_position_in_grid]]) {
|
|
uint nonce = baseNonce + gid;
|
|
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
|
|
|
"""
|
|
if p.hasWide { s += " uint lane = gid & 31u;\n" }
|
|
for i in 0..<8 {
|
|
s += " { uint x = nonce ^ SEEDW[\(i)]; x += 0x9e3779b9u * \(i + 1)u; x = splitmix32(x); r\(i) = x ^ SEEDW[\((i + 1) & 7)]; }\n"
|
|
}
|
|
s += "\n for (uint it = 0u; it < \(Program.iterations)u; ++it) {\n uint sel = r0;\n"
|
|
// The word index expression for a load: plain = a & MASK; wide = lane 0's a, aligned to 32 words, plus lane.
|
|
func wordIndex(_ a: String, wide: Bool) -> String { wide ? "(simd_broadcast(\(a), 0) & WMASK) + lane" : "\(a) & MASK" }
|
|
func fetch(_ idx: String) -> String {
|
|
switch source {
|
|
case .stored: return "dataset[\(idx)]"
|
|
case .inlineClosed(let d0, let d1): return "ds_elem(\(idx), \(hex(d0)), \(hex(d1)))"
|
|
case .inlineMemhard: return "mh_word(cache, \(idx))"
|
|
}
|
|
}
|
|
for (k, ins) in p.instrs.enumerated() {
|
|
let d = "r\(ins.dst)", a = "r\(ins.a)", b = "r\(ins.b)"
|
|
var line: String
|
|
switch ins.op {
|
|
case .add: line = "\(d) = \(d) + \(a) + select(\(hex(ins.imm)), \(hex(ins.imm2)), ((sel >> \(ins.bit)u) & 1u) != 0u);"
|
|
case .sub: line = "\(d) = \(d) - \(a);"
|
|
case .mul: line = "\(d) = \(d) * \(a);"
|
|
case .mulhi: line = "\(d) = mulhi(\(d), \(a));"
|
|
case .xor: line = "\(d) = \(d) ^ \(a);"
|
|
case .or: line = "\(d) = \(d) | \(a);"
|
|
case .rotl: line = "\(d) = rotl_imm(\(d), \(ins.rot)u);"
|
|
case .rotr: line = "\(d) = rotr_var(\(d), \(a));"
|
|
case .mad: line = "\(d) = \(a) * \(b) + \(d);"
|
|
case .shfl: line = "\(d) = \(d) ^ simd_shuffle_xor(\(a), (ushort)\(ins.mask));"
|
|
case .load: line = "\(d) = \(d) ^ \(fetch(wordIndex(a, wide: false)));"
|
|
case .wload: line = "\(d) = \(d) ^ \(fetch(wordIndex(a, wide: true)));"
|
|
}
|
|
s += " \(line) // \(k)\n"
|
|
}
|
|
s += """
|
|
}
|
|
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
|
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
|
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
|
}
|
|
|
|
"""
|
|
return s
|
|
}
|
|
|
|
let fillMSL = """
|
|
#include <metal_stdlib>
|
|
using namespace metal;
|
|
inline uint ds_elem(uint i, uint d0, uint d1) {
|
|
uint x = i ^ d0;
|
|
x *= 0x9E3779B1u; x ^= x >> 15;
|
|
x += d1;
|
|
x *= 0x85EBCA77u; x ^= x >> 13;
|
|
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
|
return x;
|
|
}
|
|
kernel void igneum_fill(device uint* dataset [[buffer(0)]],
|
|
constant uint2& day [[buffer(1)]],
|
|
uint gid [[thread_position_in_grid]]) {
|
|
dataset[gid] = ds_elem(gid, day.x, day.y);
|
|
}
|
|
"""
|
|
|
|
// MARK: - CPU reference interpreter for one 32-lane warp
|
|
|
|
func cpuWarp(_ p: Program, baseNonce: UInt32, ds: DatasetSource) -> [UInt64] {
|
|
cpuWarpTraced(p, baseNonce: baseNonce, ds: ds, trace: nil)
|
|
}
|
|
|
|
// Same interpreter with an optional hook. When `trace` is set it is called before every instruction with
|
|
// (iteration, instruction index, the 8 registers of lane 0). The edge-case tests use it to prove that the
|
|
// operand values they were built to produce really occurred. The bench passes nil.
|
|
// Loads are batched across the 32 lanes: the indices are gathered, DatasetSource.fetch answers all 32, then
|
|
// the xors are applied. For the closed form this is the old per-lane formula; for the memory-hard dataset it
|
|
// lets the 32 independent item derivations overlap their cache misses (see deriveItems).
|
|
func cpuWarpTraced(_ p: Program, baseNonce: UInt32, ds: DatasetSource,
|
|
trace: ((Int, Int, [UInt32]) -> Void)?) -> [UInt64] {
|
|
let lanes = 32
|
|
let mask = ds.mask
|
|
var r = [UInt32](repeating: 0, count: lanes * 8) // r[lane*8 + reg]
|
|
for lane in 0..<lanes {
|
|
let nonce = baseNonce &+ UInt32(lane)
|
|
for i in 0..<8 {
|
|
var x = nonce ^ p.seed[i]
|
|
x &+= 0x9e3779b9 &* UInt32(i + 1)
|
|
x = splitmix32(x)
|
|
r[lane * 8 + i] = x ^ p.seed[(i + 1) & 7]
|
|
}
|
|
}
|
|
var tmp = [UInt32](repeating: 0, count: lanes)
|
|
let idx = UnsafeMutablePointer<UInt32>.allocate(capacity: lanes)
|
|
let val = UnsafeMutablePointer<UInt32>.allocate(capacity: lanes)
|
|
defer { idx.deallocate(); val.deallocate() }
|
|
for it in 0..<Program.iterations {
|
|
for lane in 0..<lanes { tmp[lane] = r[lane * 8] } // sel = r0 at the top of the iteration
|
|
let sel = tmp
|
|
for (k, ins) in p.instrs.enumerated() {
|
|
if let t = trace { t(it, k, Array(r[0..<8])) }
|
|
switch ins.op {
|
|
case .shfl:
|
|
for lane in 0..<lanes { tmp[lane] = r[lane * 8 + ins.a] }
|
|
for lane in 0..<lanes { r[lane * 8 + ins.dst] ^= tmp[lane ^ ins.mask] }
|
|
case .load:
|
|
for lane in 0..<lanes { idx[lane] = r[lane * 8 + ins.a] & mask }
|
|
ds.fetch(UnsafePointer(idx), lanes, val)
|
|
for lane in 0..<lanes { r[lane * 8 + ins.dst] ^= val[lane] }
|
|
case .wload:
|
|
// Lane 0's register, masked, aligned down to 32 words; lane l reads word base + l.
|
|
let base = (r[ins.a] & mask) & ~31
|
|
for lane in 0..<lanes { idx[lane] = base + UInt32(lane) }
|
|
ds.fetch(UnsafePointer(idx), lanes, val)
|
|
for lane in 0..<lanes { r[lane * 8 + ins.dst] ^= val[lane] }
|
|
default:
|
|
for lane in 0..<lanes {
|
|
let base = lane * 8
|
|
let d = r[base + ins.dst], a = r[base + ins.a]
|
|
var v: UInt32
|
|
switch ins.op {
|
|
case .add:
|
|
let s = (sel[lane] >> UInt32(ins.bit)) & 1
|
|
v = d &+ a &+ (s != 0 ? ins.imm2 : ins.imm)
|
|
case .sub: v = d &- a
|
|
case .mul: v = d &* a
|
|
case .mulhi: v = mulhi32(d, a)
|
|
case .xor: v = d ^ a
|
|
case .or: v = d | a
|
|
case .rotl: v = rotl32(d, ins.rot)
|
|
case .rotr: v = rotr32(d, a)
|
|
case .mad: v = (a &* r[base + ins.b]) &+ d
|
|
case .load, .wload, .shfl: v = d // unreachable, handled above
|
|
}
|
|
r[base + ins.dst] = v
|
|
}
|
|
}
|
|
}
|
|
}
|
|
var out = [UInt64](repeating: 0, count: lanes)
|
|
for lane in 0..<lanes {
|
|
let b = lane * 8
|
|
let lo = r[b] ^ rotl32(r[b + 1], 7) ^ rotl32(r[b + 2], 14) ^ rotl32(r[b + 3], 21)
|
|
let hi = r[b + 4] ^ rotl32(r[b + 5], 9) ^ rotl32(r[b + 6], 18) ^ rotl32(r[b + 7], 27)
|
|
out[lane] = (UInt64(hi) << 32) | UInt64(lo)
|
|
}
|
|
return out
|
|
}
|
|
|
|
// MARK: - Program pack export (CUDA twin of the Metal kernel)
|
|
//
|
|
// Writes, for one seed: program.json, vectors.json, kernel.cu, kernel.cl, program.h, vectors.h, program.metal.
|
|
// The CUDA kernel is emitted from the same Instr list as the MSL above, line for line.
|
|
// Differences by design: the dataset mask is a kernel argument (so the host can sweep dataset
|
|
// sizes with one ahead-of-time compile), seeds are inlined as literals, and the host launch
|
|
// wrappers live in kernel.cu so host.cu never declares a __global__ across translation units.
|
|
|
|
let packVectorBases: [UInt32] = [0, 4096, 1000000]
|
|
|
|
func hex64(_ v: UInt64) -> String { String(format: "0x%016llxull", v) }
|
|
func jhex(_ v: UInt32) -> String { String(format: "\"0x%08x\"", v) }
|
|
func jhex64(_ v: UInt64) -> String { String(format: "\"0x%016llx\"", v) }
|
|
func jstr(_ s: String) -> String {
|
|
var o = "\""
|
|
for c in s.unicodeScalars {
|
|
switch c {
|
|
case "\"": o += "\\\""
|
|
case "\\": o += "\\\\"
|
|
case "\n": o += "\\n"
|
|
default: o.unicodeScalars.append(c)
|
|
}
|
|
}
|
|
return o + "\""
|
|
}
|
|
|
|
// memhard: nil for a closed-form pack (the original fill kernel), MixParams for the memory-hard pack
|
|
// (cache fill and dataset build kernels; the shared core comes from memhard.h in the same pack).
|
|
func generateCUDA(_ p: Program, memhard: MixParams?) -> String {
|
|
var s = """
|
|
// Generated by proto-metal/igneum-bench --export-pack for seed "\(p.seedString)". Do not edit by hand.
|
|
// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal).
|
|
// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC.
|
|
#include <cuda_runtime.h>
|
|
#include <cstdint>
|
|
#include "program.h"
|
|
\(memhard != nil ? "#include \"memhard.h\"\n" : "")
|
|
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
|
|
x ^= x >> 16; x *= 0x7feb352du;
|
|
x ^= x >> 15; x *= 0x846ca68bu;
|
|
x ^= x >> 16;
|
|
return x;
|
|
}
|
|
// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31.
|
|
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
|
|
// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x.
|
|
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
|
__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) {
|
|
uint32_t x = i ^ d0;
|
|
x *= 0x9E3779B1u; x ^= x >> 15;
|
|
x += d1;
|
|
x *= 0x85EBCA77u; x ^= x >> 13;
|
|
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
|
return x;
|
|
}
|
|
|
|
|
|
"""
|
|
if memhard == nil {
|
|
s += """
|
|
// dataset[i] = ds_elem(i, d0, d1) for i < n. Same closed form as the Metal igneum_fill kernel.
|
|
__global__ void igneum_fill(uint32_t* ds, uint32_t n, uint32_t d0, uint32_t d1) {
|
|
uint32_t i = blockIdx.x * blockDim.x + threadIdx.x;
|
|
if (i < n) ds[i] = ds_elem(i, d0, d1);
|
|
}
|
|
|
|
|
|
"""
|
|
} else {
|
|
s += """
|
|
// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item.
|
|
// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host.
|
|
__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) {
|
|
uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;
|
|
if (seg < nSegments) mh_cache_segment(cache, seg);
|
|
}
|
|
__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
|
|
uint32_t t = blockIdx.x * blockDim.x + threadIdx.x;
|
|
if (t < nItems) {
|
|
uint32_t s[16];
|
|
mh_item(cache, t, s);
|
|
uint32_t* d = ds + (size_t)t * 16u;
|
|
for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i];
|
|
}
|
|
}
|
|
|
|
|
|
"""
|
|
}
|
|
s += """
|
|
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
|
|
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
|
|
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
|
|
__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask) {
|
|
uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;
|
|
uint32_t nonce = baseNonce + gid;
|
|
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
|
|
|
|
"""
|
|
if p.hasWide { s += " uint32_t lane = threadIdx.x & 31u;\n uint32_t wmask = mask & ~31u;\n" }
|
|
for i in 0..<8 {
|
|
let addc = 0x9e3779b9 &* UInt32(i + 1)
|
|
s += " { uint32_t x = nonce ^ \(hex(p.seed[i])); x += \(hex(addc)); x = splitmix32(x); r\(i) = x ^ \(hex(p.seed[(i + 1) & 7])); } // SEEDW[\(i)], 0x9e3779b9u * \(i + 1)u, SEEDW[\((i + 1) & 7)]\n"
|
|
}
|
|
s += "\n for (uint32_t it = 0u; it < \(Program.iterations)u; ++it) {\n uint32_t sel = r0;\n"
|
|
for (k, ins) in p.instrs.enumerated() {
|
|
let d = "r\(ins.dst)", a = "r\(ins.a)", b = "r\(ins.b)"
|
|
var line: String
|
|
switch ins.op {
|
|
// Metal select(A, B, c) returns c ? B : A, so the true branch is imm2 here as well.
|
|
case .add: line = "\(d) = \(d) + \(a) + ((((sel >> \(ins.bit)u) & 1u) != 0u) ? \(hex(ins.imm2)) : \(hex(ins.imm)));"
|
|
case .sub: line = "\(d) = \(d) - \(a);"
|
|
case .mul: line = "\(d) = \(d) * \(a);"
|
|
case .mulhi: line = "\(d) = __umulhi(\(d), \(a));"
|
|
case .xor: line = "\(d) = \(d) ^ \(a);"
|
|
case .or: line = "\(d) = \(d) | \(a);"
|
|
case .rotl: line = "\(d) = rotl_imm(\(d), \(ins.rot)u);"
|
|
case .rotr: line = "\(d) = rotr_var(\(d), \(a));"
|
|
case .mad: line = "\(d) = \(a) * \(b) + \(d);"
|
|
case .shfl: line = "\(d) = \(d) ^ __shfl_xor_sync(0xffffffffu, \(a), \(ins.mask));"
|
|
case .load: line = "\(d) = \(d) ^ ds[\(a) & mask];"
|
|
case .wload: line = "\(d) = \(d) ^ ds[(__shfl_sync(0xffffffffu, \(a), 0) & wmask) + lane];"
|
|
}
|
|
s += " \(line) // \(k) \(ins.op.rawValue)\n"
|
|
}
|
|
s += """
|
|
}
|
|
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
|
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
|
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
|
|
}
|
|
|
|
// Host-side launch wrappers. Declared in program.h, called from host.cu.
|
|
|
|
"""
|
|
if memhard == nil {
|
|
s += """
|
|
cudaError_t igneum_launch_fill(uint32_t* ds, uint32_t nWords, uint32_t d0, uint32_t d1) {
|
|
if (nWords == 0u) return cudaErrorInvalidValue;
|
|
uint32_t block = 256u;
|
|
uint32_t grid = (nWords + block - 1u) / block;
|
|
igneum_fill<<<grid, block>>>(ds, nWords, d0, d1);
|
|
return cudaGetLastError();
|
|
}
|
|
|
|
|
|
"""
|
|
} else {
|
|
s += """
|
|
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) {
|
|
if (nSegments == 0u) return cudaErrorInvalidValue;
|
|
uint32_t block = 256u;
|
|
uint32_t grid = (nSegments + block - 1u) / block;
|
|
igneum_cache_fill<<<grid, block>>>(cache, nSegments);
|
|
return cudaGetLastError();
|
|
}
|
|
|
|
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
|
|
if (nItems == 0u) return cudaErrorInvalidValue;
|
|
uint32_t block = 256u;
|
|
uint32_t grid = (nItems + block - 1u) / block;
|
|
igneum_build<<<grid, block>>>(ds, cache, nItems);
|
|
return cudaGetLastError();
|
|
}
|
|
|
|
|
|
"""
|
|
}
|
|
s += """
|
|
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
|
uint32_t nonces, uint32_t blockWarps) {
|
|
if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;
|
|
uint32_t block = 32u * blockWarps;
|
|
if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;
|
|
igneum_hash<<<nonces / block, block>>>(ds, out, baseNonce, mask);
|
|
return cudaGetLastError();
|
|
}
|
|
|
|
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
|
|
cudaFuncAttributes attr;
|
|
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash);
|
|
if (e != cudaSuccess) return e;
|
|
*numRegs = attr.numRegs;
|
|
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0);
|
|
}
|
|
|
|
"""
|
|
return s
|
|
}
|
|
|
|
// kernel.cl: the same program as OpenCL C 1.2, built from source at runtime by proto-opencl/host.c.
|
|
// The 32-lane exchange is selected at compile time (IGNEUM_EXCHANGE): sub-group shuffles where the device has them
|
|
// and its sub-group size is exactly 32, or a local-memory exchange with a barrier on every other device
|
|
// (proto-opencl/WAVEFRONT.md). The same text is compiled as C++ by proto-opencl/emu with a 32- or 64-wide sub-group.
|
|
func generateOpenCL(_ p: Program, memhard: MixParams?) -> String {
|
|
var s = """
|
|
// Generated by proto-metal/igneum-bench --export-pack for seed "\(p.seedString)". Do not edit by hand.
|
|
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
|
|
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
|
|
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
|
|
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
|
|
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
|
|
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
|
|
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
|
|
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
|
|
#ifndef IGNEUM_GROUP
|
|
#define IGNEUM_GROUP 32
|
|
#endif
|
|
#ifndef IGNEUM_EXCHANGE
|
|
#define IGNEUM_EXCHANGE 0
|
|
#endif
|
|
#ifdef __OPENCL_VERSION__
|
|
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
|
|
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
|
|
#if IGNEUM_EXCHANGE == 1
|
|
#ifdef cl_khr_subgroups
|
|
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
|
|
#endif
|
|
#ifdef cl_khr_subgroup_shuffle
|
|
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
|
|
#endif
|
|
#elif IGNEUM_EXCHANGE == 2
|
|
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
|
|
#endif
|
|
#else
|
|
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
|
|
#include "emu_opencl.h"
|
|
#endif
|
|
|
|
#if IGNEUM_EXCHANGE == 1
|
|
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
|
|
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
|
#elif IGNEUM_EXCHANGE == 2
|
|
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
|
|
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
|
#else
|
|
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
|
|
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
|
|
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
|
|
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
|
|
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
|
|
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
|
|
#endif
|
|
|
|
static inline uint splitmix32(uint x) {
|
|
x ^= x >> 16; x *= 0x7feb352du;
|
|
x ^= x >> 15; x *= 0x846ca68bu;
|
|
x ^= x >> 16;
|
|
return x;
|
|
}
|
|
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
|
|
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
|
|
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
|
|
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
|
|
static inline uint ds_elem(uint i, uint d0, uint d1) {
|
|
uint x = i ^ d0;
|
|
x *= 0x9E3779B1u; x ^= x >> 15;
|
|
x += d1;
|
|
x *= 0x85EBCA77u; x ^= x >> 13;
|
|
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
|
return x;
|
|
}
|
|
|
|
|
|
"""
|
|
if let mp = memhard {
|
|
s += emitMemhardCore(mp, dialect: .opencl) + "\n"
|
|
s += """
|
|
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
|
|
// The same constants as memhard.h in this pack (one emitter, three dialects).
|
|
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
|
|
uint seg = (uint)get_global_id(0);
|
|
if (seg < nSegments) mh_cache_segment(cache, seg);
|
|
}
|
|
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
|
|
uint t = (uint)get_global_id(0);
|
|
if (t < nItems) {
|
|
uint s[16];
|
|
mh_item(cache, t, s);
|
|
__global uint* d = ds + ((ulong)t * 16u);
|
|
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
|
|
}
|
|
}
|
|
|
|
|
|
"""
|
|
} else {
|
|
s += """
|
|
// dataset[i] = ds_elem(i, d0, d1) for i < n. Same closed form as the Metal igneum_fill kernel.
|
|
__kernel void igneum_fill(__global uint* ds, uint n, uint d0, uint d1) {
|
|
uint i = (uint)get_global_id(0);
|
|
if (i < n) ds[i] = ds_elem(i, d0, d1);
|
|
}
|
|
|
|
|
|
"""
|
|
}
|
|
s += """
|
|
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
|
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
|
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
|
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask) {
|
|
uint gid = (uint)get_global_id(0);
|
|
uint lid = (uint)get_local_id(0);
|
|
uint nonce = baseNonce + gid;
|
|
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
|
#if IGNEUM_EXCHANGE == 0
|
|
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
|
uint xk = 0u;
|
|
#else
|
|
(void)lid;
|
|
#endif
|
|
|
|
"""
|
|
if p.hasWide { s += " uint lane = lid & 31u;\n uint wmask = mask & ~31u;\n" }
|
|
for i in 0..<8 {
|
|
let addc = 0x9e3779b9 &* UInt32(i + 1)
|
|
s += " { uint x = nonce ^ \(hex(p.seed[i])); x += \(hex(addc)); x = splitmix32(x); r\(i) = x ^ \(hex(p.seed[(i + 1) & 7])); } // SEEDW[\(i)], 0x9e3779b9u * \(i + 1)u, SEEDW[\((i + 1) & 7)]\n"
|
|
}
|
|
s += "\n for (uint it = 0u; it < \(Program.iterations)u; ++it) {\n uint sel = r0;\n"
|
|
for (k, ins) in p.instrs.enumerated() {
|
|
let d = "r\(ins.dst)", a = "r\(ins.a)", b = "r\(ins.b)"
|
|
var line: String
|
|
switch ins.op {
|
|
// Metal select(A, B, c) returns c ? B : A, so the true branch is imm2 here as well.
|
|
case .add: line = "\(d) = \(d) + \(a) + ((((sel >> \(ins.bit)u) & 1u) != 0u) ? \(hex(ins.imm2)) : \(hex(ins.imm)));"
|
|
case .sub: line = "\(d) = \(d) - \(a);"
|
|
case .mul: line = "\(d) = \(d) * \(a);"
|
|
case .mulhi: line = "\(d) = mul_hi(\(d), \(a));"
|
|
case .xor: line = "\(d) = \(d) ^ \(a);"
|
|
case .or: line = "\(d) = \(d) | \(a);"
|
|
case .rotl: line = "\(d) = rotl_imm(\(d), \(ins.rot)u);"
|
|
case .rotr: line = "\(d) = rotr_var(\(d), \(a));"
|
|
case .mad: line = "\(d) = \(a) * \(b) + \(d);"
|
|
case .shfl: line = "{ uint t_; IGNEUM_SHFL_XOR(t_, \(a), \(ins.mask)u); \(d) = \(d) ^ t_; }"
|
|
case .load: line = "\(d) = \(d) ^ ds[\(a) & mask];"
|
|
case .wload: line = "{ uint t_; IGNEUM_BCAST0(t_, \(a)); \(d) = \(d) ^ ds[(t_ & wmask) + lane]; }"
|
|
}
|
|
s += " \(line) // \(k) \(ins.op.rawValue)\n"
|
|
}
|
|
s += """
|
|
}
|
|
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
|
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
|
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
|
}
|
|
|
|
#if IGNEUM_EXCHANGE != 0
|
|
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
|
|
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
|
|
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
|
|
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
|
|
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
|
|
}
|
|
#endif
|
|
|
|
"""
|
|
return s
|
|
}
|
|
|
|
func generateProgramHeader(_ p: Program, dayString: String, day: (UInt32, UInt32), datasetLog2: Int, memhard: MixParams?) -> String {
|
|
let mask = UInt32((1 << datasetLog2) - 1)
|
|
let mix = p.histogram.map { "\($0.0)=\($0.1)" }.joined(separator: " ")
|
|
var s = """
|
|
// Generated by proto-metal/igneum-bench --export-pack for seed "\(p.seedString)". Do not edit by hand.
|
|
// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.
|
|
// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros.
|
|
#pragma once
|
|
#ifdef __cplusplus
|
|
#include <cstdint>
|
|
#else
|
|
#include <stdint.h>
|
|
#endif
|
|
#ifndef IGNEUM_NO_CUDA
|
|
#include <cuda_runtime.h>
|
|
#endif
|
|
|
|
#define IGNEUM_SEED_STRING \(jstr(p.seedString))
|
|
#define IGNEUM_DAY_STRING \(jstr(dayString))
|
|
#define IGNEUM_DAY0 \(hex(day.0))
|
|
#define IGNEUM_DAY1 \(hex(day.1))
|
|
#define IGNEUM_DATASET_LOG2 \(datasetLog2)
|
|
#define IGNEUM_MASK \(hex(mask))
|
|
#define IGNEUM_LANES 32
|
|
#define IGNEUM_ITERATIONS \(Program.iterations)
|
|
#define IGNEUM_INSTR_COUNT \(Program.count)
|
|
#define IGNEUM_LOADS_PER_HASH \(p.loadsPerHash)
|
|
#define IGNEUM_WIDE_LOADS_PER_HASH \(p.wideLoadsPerHash)
|
|
#define IGNEUM_OP_MIX \(jstr(mix))
|
|
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
|
|
#define IGNEUM_DATASET_MODE \(memhard == nil ? 0 : 1)
|
|
|
|
#define IGNEUM_SEEDW_INIT { \(p.seed.map(hex).joined(separator: ", ")) }
|
|
|
|
"""
|
|
if let mp = memhard {
|
|
s += """
|
|
#define IGNEUM_KEY_INIT { \(mp.keyWords.map(hex).joined(separator: ", ")) }
|
|
#define IGNEUM_CACHE_LOG2_WORDS \(cacheLog2Words)
|
|
#define IGNEUM_CACHE_SEGMENT_LOG2_LINES \(cacheSegmentLog2Lines)
|
|
#define IGNEUM_CACHE_SEGMENTS \(cacheSegments)u
|
|
#define IGNEUM_ITEM_ROUNDS \(itemRounds)
|
|
#define IGNEUM_MIX_ROT_INIT { \(mp.rotWords.map { "\($0)u" }.joined(separator: ", ")) }
|
|
#define IGNEUM_MIX_MUL_INIT { \(mp.mulWords.map(hex).joined(separator: ", ")) }
|
|
#define IGNEUM_MIX_RC_INIT { \(mp.rcWords.map(hex).joined(separator: ", ")) }
|
|
|
|
#ifndef IGNEUM_NO_CUDA
|
|
// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().
|
|
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);
|
|
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);
|
|
|
|
"""
|
|
} else {
|
|
s += """
|
|
#ifndef IGNEUM_NO_CUDA
|
|
// Defined in kernel.cu. Both launch on the default stream and return cudaGetLastError().
|
|
cudaError_t igneum_launch_fill(uint32_t* ds, uint32_t nWords, uint32_t d0, uint32_t d1);
|
|
|
|
"""
|
|
}
|
|
s += """
|
|
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
|
uint32_t nonces, uint32_t blockWarps);
|
|
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
|
|
#endif
|
|
|
|
"""
|
|
return s
|
|
}
|
|
|
|
// memhard.h for a memory-hard pack: the core in CUDA C++, compiled for host and device.
|
|
func generateMemhardHeader(_ p: Program, _ mp: MixParams) -> String {
|
|
return """
|
|
// Generated by proto-metal/igneum-bench --export-pack for seed "\(p.seedString)". Do not edit by hand.
|
|
// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against.
|
|
// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference).
|
|
// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C.
|
|
#pragma once
|
|
#ifdef __cplusplus
|
|
#include <cstdint>
|
|
#else
|
|
#include <stdint.h>
|
|
#endif
|
|
#if defined(__CUDACC__)
|
|
#define IGNEUM_HD __host__ __device__ __forceinline__
|
|
#elif defined(_MSC_VER) && !defined(__cplusplus)
|
|
#define IGNEUM_HD static __inline
|
|
#else
|
|
#define IGNEUM_HD static inline
|
|
#endif
|
|
\(emitMemhardCore(mp, dialect: .cuda))
|
|
"""
|
|
}
|
|
|
|
struct PackVectors {
|
|
var head: [UInt32] // dataset[0..15]
|
|
var last: UInt32 // dataset[MASK]
|
|
var sampleIdx: [UInt32] // 64 dataset indices
|
|
var sampleVal: [UInt32]
|
|
var cacheHead: [UInt32] // cache[0..15] (memhard only)
|
|
var cacheLast: [UInt32] // last cache line (memhard only)
|
|
var cacheFNV: UInt64 // FNV-1a 64 over the whole cache (memhard only)
|
|
}
|
|
|
|
func generateVectorsHeader(_ p: Program, bases: [UInt32], outs: [[UInt64]], v: PackVectors, mask: UInt32, source: String, memhard: Bool) -> String {
|
|
var s = """
|
|
// Generated by proto-metal/igneum-bench --export-pack for seed "\(p.seedString)". Do not edit by hand.
|
|
// Expected outputs: \(source)
|
|
#pragma once
|
|
#ifdef __cplusplus
|
|
#include <cstdint>
|
|
#else
|
|
#include <stdint.h>
|
|
#endif
|
|
|
|
#define IGNEUM_VEC_WARPS \(bases.count)
|
|
static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { \(bases.map { "\($0)u" }.joined(separator: ", ")) };
|
|
static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {
|
|
|
|
"""
|
|
for (i, o) in outs.enumerated() {
|
|
s += " { // base nonce \(bases[i])\n"
|
|
for row in 0..<4 {
|
|
s += " " + (0..<8).map { hex64(o[row * 8 + $0]) }.joined(separator: ", ") + (row == 3 ? "\n" : ",\n")
|
|
}
|
|
s += i == outs.count - 1 ? " }\n" : " },\n"
|
|
}
|
|
s += """
|
|
};
|
|
|
|
// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (\(mask)).
|
|
static const uint32_t IGNEUM_DS_HEAD[16] = {
|
|
\((0..<8).map { hex(v.head[$0]) }.joined(separator: ", ")),
|
|
\((8..<16).map { hex(v.head[$0]) }.joined(separator: ", "))
|
|
};
|
|
static const uint32_t IGNEUM_DS_LAST_INDEX = \(mask)u;
|
|
static const uint32_t IGNEUM_DS_LAST = \(hex(v.last));
|
|
// 64 sampled dataset words (index, value) computed on the Mac.
|
|
#define IGNEUM_DS_SAMPLES \(v.sampleIdx.count)
|
|
static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = {
|
|
\(v.sampleIdx.map { "\($0)u" }.joined(separator: ", "))
|
|
};
|
|
static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = {
|
|
\(v.sampleVal.map(hex).joined(separator: ", "))
|
|
};
|
|
|
|
"""
|
|
if memhard {
|
|
s += """
|
|
// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^\(cacheLog2Words) words.
|
|
static const uint32_t IGNEUM_CACHE_HEAD[16] = {
|
|
\((0..<8).map { hex(v.cacheHead[$0]) }.joined(separator: ", ")),
|
|
\((8..<16).map { hex(v.cacheHead[$0]) }.joined(separator: ", "))
|
|
};
|
|
static const uint32_t IGNEUM_CACHE_LAST[16] = {
|
|
\((0..<8).map { hex(v.cacheLast[$0]) }.joined(separator: ", ")),
|
|
\((8..<16).map { hex(v.cacheLast[$0]) }.joined(separator: ", "))
|
|
};
|
|
static const uint64_t IGNEUM_CACHE_FNV64 = \(hex64(v.cacheFNV));
|
|
|
|
"""
|
|
}
|
|
return s
|
|
}
|
|
|
|
func generateProgramJSON(_ p: Program, dayString: String, day: (UInt32, UInt32), datasetLog2: Int, memhard: MixParams?) -> String {
|
|
let mask = UInt32((1 << datasetLog2) - 1)
|
|
var s = "{\n"
|
|
s += " \"format\": \"igneum-program-pack-2\",\n"
|
|
s += " \"dataset_mode\": \(jstr(memhard == nil ? "closed-form" : "memory-hard")),\n"
|
|
s += " \"seed\": \(jstr(p.seedString)),\n"
|
|
s += " \"seed_words\": [\(p.seed.map(jhex).joined(separator: ", "))],\n"
|
|
s += " \"seed_derivation\": \"FNV-1a 64 over UTF-8 of seed, basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32\",\n"
|
|
s += " \"lanes\": 32,\n"
|
|
s += " \"registers\": 8,\n"
|
|
s += " \"iterations\": \(Program.iterations),\n"
|
|
s += " \"instruction_count\": \(Program.count),\n"
|
|
s += " \"loads_per_hash\": \(p.loadsPerHash),\n"
|
|
s += " \"op_mix\": {\(p.histogram.map { "\(jstr($0.0)): \($0.1)" }.joined(separator: ", "))},\n"
|
|
s += " \"register_init\": \"for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]\",\n"
|
|
s += " \"splitmix32\": \"x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16\",\n"
|
|
s += " \"iteration\": \"sel = r0 sampled once at the top of each iteration, then all instructions in order\",\n"
|
|
s += " \"output\": \"lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo\",\n"
|
|
s += " \"op_semantics\": {\n"
|
|
s += " \"add\": \"dst = dst + src + (bit `bit` of sel ? imm2 : imm)\",\n"
|
|
s += " \"sub\": \"dst = dst - src\",\n"
|
|
s += " \"mul\": \"dst = dst * src (low 32)\",\n"
|
|
s += " \"mulhi\": \"dst = high 32 bits of dst * src\",\n"
|
|
s += " \"xor\": \"dst = dst ^ src\",\n"
|
|
s += " \"or\": \"dst = dst | src\",\n"
|
|
s += " \"rotl\": \"dst = rotl(dst, rot), rot in 1..31\",\n"
|
|
s += " \"rotr\": \"dst = rotr(dst, src & 31)\",\n"
|
|
s += " \"mad\": \"dst = src * src2 + dst\",\n"
|
|
s += " \"shfl\": \"dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp\",\n"
|
|
s += " \"load\": \"dst = dst ^ dataset[src & dataset.mask]\",\n"
|
|
s += " \"wload\": \"base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)\"\n"
|
|
s += " },\n"
|
|
s += " \"dataset\": {\n"
|
|
s += " \"log2_words\": \(datasetLog2),\n"
|
|
s += " \"bytes\": \(UInt64(1) << UInt64(datasetLog2 + 2)),\n"
|
|
s += " \"mask\": \(jhex(mask)),\n"
|
|
s += " \"day\": \(jstr(dayString)),\n"
|
|
s += " \"day_words_from\": \(jstr("day/" + dayString)),\n"
|
|
s += " \"d0\": \(jhex(day.0)),\n"
|
|
s += " \"d1\": \(jhex(day.1)),\n"
|
|
if let mp = memhard {
|
|
s += " \"mode\": \"memory-hard\",\n"
|
|
s += " \"spec\": \"proto-metal/MEMHARD.md\",\n"
|
|
s += " \"key\": [\(mp.keyWords.map(jhex).joined(separator: ", "))],\n"
|
|
s += " \"key_derivation\": \"the 8 words of seedWords(\\\"day/\\\" + day); d0, d1 are key[0], key[1]\",\n"
|
|
s += " \"cache\": {\"log2_words\": \(cacheLog2Words), \"bytes\": \(UInt64(cacheWords) * 4), \"line_words\": 16, \"segment_lines\": \(cacheLinesPerSegment), \"segments\": \(cacheSegments), \"block\": \"ChaCha\(chachaRounds) core + feed-forward, rotations 16 12 8 7\", \"sigma\": [\(chachaSigma.map(jhex).joined(separator: ", "))], \"tag\": [\(cacheTag.map(jhex).joined(separator: ", "))], \"chain\": \"in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0\"},\n"
|
|
s += " \"mixer\": {\"draw\": \"SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())\", \"rot\": [\(mp.rotWords.map { "\($0)" }.joined(separator: ", "))], \"mul\": [\(mp.mulWords.map(jhex).joined(separator: ", "))], \"rc\": [\(mp.rcWords.map(jhex).joined(separator: ", "))], \"round\": \"for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]\", \"quarter_round\": \"a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)\"},\n"
|
|
s += " \"item\": \"s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..\(itemRounds - 1): s = M_r(s); line = s[0] & \(jhex(cacheLineMask)); s[i] ^= cache[line * 16 + i]; then s = M_\(itemRounds)(s); item(t) = s\",\n"
|
|
s += " \"word\": \"dataset[w] = item(w >> 4)[w & 15]\"\n"
|
|
} else {
|
|
s += " \"mode\": \"closed-form\",\n"
|
|
s += " \"formula\": \"x = i ^ d0; x *= 0x9E3779B1; x ^= x>>15; x += d1; x *= 0x85EBCA77; x ^= x>>13; x *= 0xC2B2AE3D; x ^= x>>16 (all mod 2^32)\"\n"
|
|
}
|
|
s += " },\n"
|
|
s += " \"instructions\": [\n"
|
|
for (k, ins) in p.instrs.enumerated() {
|
|
s += " {\"i\": \(k), \"op\": \(jstr(ins.op.rawValue)), \"dst\": \(ins.dst), \"src\": \(ins.a), \"src2\": \(ins.b), \"imm\": \(jhex(ins.imm)), \"imm2\": \(jhex(ins.imm2)), \"rot\": \(ins.rot), \"bit\": \(ins.bit), \"mask\": \(ins.mask)}"
|
|
s += k == p.instrs.count - 1 ? "\n" : ",\n"
|
|
}
|
|
s += " ]\n}\n"
|
|
return s
|
|
}
|
|
|
|
func generateVectorsJSON(_ p: Program, dayString: String, datasetLog2: Int, bases: [UInt32], outs: [[UInt64]], v: PackVectors, mask: UInt32, source: String, memhard: Bool) -> String {
|
|
var s = "{\n"
|
|
s += " \"seed\": \(jstr(p.seedString)),\n"
|
|
s += " \"day\": \(jstr(dayString)),\n"
|
|
s += " \"dataset_mode\": \(jstr(memhard ? "memory-hard" : "closed-form")),\n"
|
|
s += " \"dataset_log2_words\": \(datasetLog2),\n"
|
|
s += " \"mask\": \(jhex(mask)),\n"
|
|
s += " \"lanes\": 32,\n"
|
|
s += " \"source\": \(jstr(source)),\n"
|
|
s += " \"warps\": [\n"
|
|
for (i, o) in outs.enumerated() {
|
|
s += " {\"base_nonce\": \(bases[i]), \"expected\": [\n"
|
|
for row in 0..<4 {
|
|
s += " " + (0..<8).map { jhex64(o[row * 8 + $0]) }.joined(separator: ", ") + (row == 3 ? "\n" : ",\n")
|
|
}
|
|
s += i == outs.count - 1 ? " ]}\n" : " ]},\n"
|
|
}
|
|
s += " ],\n"
|
|
s += " \"dataset_head\": [\(v.head.map(jhex).joined(separator: ", "))],\n"
|
|
s += " \"dataset_last_index\": \(mask),\n"
|
|
s += " \"dataset_last\": \(jhex(v.last)),\n"
|
|
s += " \"dataset_samples\": [\(zip(v.sampleIdx, v.sampleVal).map { "{\"index\": \($0.0), \"value\": \(jhex($0.1))}" }.joined(separator: ", "))]"
|
|
if memhard {
|
|
s += ",\n \"cache_head\": [\(v.cacheHead.map(jhex).joined(separator: ", "))],\n"
|
|
s += " \"cache_last_line\": [\(v.cacheLast.map(jhex).joined(separator: ", "))],\n"
|
|
s += " \"cache_fnv1a64\": \(jhex64(v.cacheFNV))\n"
|
|
} else { s += "\n" }
|
|
s += "}\n"
|
|
return s
|
|
}
|
|
|
|
// Runs the Metal kernel for each base nonce (one 32-thread threadgroup each) and compares with `expected`.
|
|
// The dataset comes from the context (closed form or memory-hard), so the GPU build path is covered too.
|
|
func metalCrossCheck(_ ctx: DatasetContext, _ p: Program, datasetLog2: Int, bases: [UInt32], expected: [[UInt64]]) -> (ok: Bool, detail: String) {
|
|
let dataset = ctx.makeDataset(log2: datasetLog2)
|
|
do {
|
|
let k = try compileHash(ctx.gpu, msl: generateMSL(p, datasetLog2: datasetLog2))
|
|
guard let got = gpuWarps(ctx.gpu, k, dataset: dataset, bases: bases) else { return (false, "GPU run failed") }
|
|
var bad = [String]()
|
|
for (i, base) in bases.enumerated() where got[i] != expected[i] { bad.append("base \(base)") }
|
|
return (bad.isEmpty, bad.isEmpty ? "Metal GPU cross-check PASS \(bases.count)/\(bases.count) warps" : "Metal GPU cross-check FAIL: \(bad.joined(separator: ", "))")
|
|
} catch {
|
|
return (false, "Metal compile error \(error)")
|
|
}
|
|
}
|
|
|
|
func exportPack(_ opts: Options) -> Never {
|
|
let dir = opts.exportPack!
|
|
let program = generateProgram(seedString: opts.seed)
|
|
let gpu = GPU()
|
|
let ctx = DatasetContext(gpu: gpu, closedForm: opts.closedForm, dayString: opts.day)
|
|
let day = ctx.day
|
|
let mask = UInt32((1 << opts.datasetLog2) - 1)
|
|
print("igneum-bench --export-pack \(dir)")
|
|
print("seed \"\(opts.seed)\", day \"\(opts.day)\", dataset 2^\(opts.datasetLog2) words (\(ctx.modeName)), loads/hash \(program.loadsPerHash), wide loads/hash \(program.wideLoadsPerHash)")
|
|
print("op mix: " + program.histogram.map { "\($0.0)=\($0.1)" }.joined(separator: " "))
|
|
if !ctx.closed { print("cache: GPU fill \(fmt(ctx.cacheFillGPUms, 2)) ms GPU time; CPU fill \(fmt(ctx.cpuSide()!.fillMs, 1)) ms one core") }
|
|
|
|
let ds = ctx.source(log2: opts.datasetLog2)
|
|
let outs = packVectorBases.map { cpuWarp(program, baseNonce: $0, ds: ds) }
|
|
var v = PackVectors(head: (0..<16).map { ds.word(UInt32($0)) }, last: ds.word(mask), sampleIdx: [], sampleVal: [],
|
|
cacheHead: [], cacheLast: [], cacheFNV: 0)
|
|
var sr = SplitMix64(s: 0x6d68_7361_6d70_6c65) // "mhsample"
|
|
for _ in 0..<64 { let i = UInt32(truncatingIfNeeded: sr.next()) & mask; v.sampleIdx.append(i); v.sampleVal.append(ds.word(i)) }
|
|
if let cpu = ctx.cpuSide() {
|
|
v.cacheHead = (0..<16).map { cpu.cache[$0] }
|
|
v.cacheLast = (0..<16).map { cpu.cache[cacheWords - 16 + $0] }
|
|
v.cacheFNV = fnv64(UnsafeRawPointer(cpu.cache), cacheWords * 4)
|
|
let cc = ctx.cacheCheck()
|
|
print(cc.detail)
|
|
if !cc.ok { print("FAIL: GPU cache differs from the CPU cache, pack not written"); exit(1) }
|
|
}
|
|
|
|
var source = "proto-metal CPU interpreter (cpuWarp, \(ctx.modeName) dataset) on Apple M5 Max"
|
|
let check = metalCrossCheck(ctx, program, datasetLog2: opts.datasetLog2, bases: packVectorBases, expected: outs)
|
|
print(check.detail)
|
|
source += "; " + check.detail
|
|
if !check.ok { print("FAIL: vectors do not match the Metal GPU, pack not written"); exit(1) }
|
|
// The sampled dataset words against the GPU-built dataset as well.
|
|
let sampleCheck = ctx.sampleCheck(log2: opts.datasetLog2, indices: v.sampleIdx + [0, mask])
|
|
print(sampleCheck.detail)
|
|
if !sampleCheck.ok { print("FAIL: GPU dataset words differ from the CPU derivation, pack not written"); exit(1) }
|
|
|
|
var files: [(String, String)] = [
|
|
("program.json", generateProgramJSON(program, dayString: opts.day, day: day, datasetLog2: opts.datasetLog2, memhard: ctx.mp)),
|
|
("vectors.json", generateVectorsJSON(program, dayString: opts.day, datasetLog2: opts.datasetLog2, bases: packVectorBases, outs: outs, v: v, mask: mask, source: source, memhard: !ctx.closed)),
|
|
("kernel.cu", generateCUDA(program, memhard: ctx.mp)),
|
|
("kernel.cl", generateOpenCL(program, memhard: ctx.mp)),
|
|
("program.h", generateProgramHeader(program, dayString: opts.day, day: day, datasetLog2: opts.datasetLog2, memhard: ctx.mp)),
|
|
("vectors.h", generateVectorsHeader(program, bases: packVectorBases, outs: outs, v: v, mask: mask, source: source, memhard: !ctx.closed)),
|
|
("program.metal", generateMSL(program, datasetLog2: opts.datasetLog2)),
|
|
]
|
|
if let mp = ctx.mp {
|
|
files.append(("memhard.h", generateMemhardHeader(program, mp)))
|
|
files.append(("memhard.metal", memhardMSL(mp)))
|
|
}
|
|
do {
|
|
try FileManager.default.createDirectory(atPath: dir, withIntermediateDirectories: true)
|
|
for (name, text) in files {
|
|
try text.write(toFile: "\(dir)/\(name)", atomically: true, encoding: .utf8)
|
|
print("wrote \(dir)/\(name) (\(text.utf8.count) bytes)")
|
|
}
|
|
} catch {
|
|
print("FAIL: write error \(error)"); exit(1)
|
|
}
|
|
for (i, b) in packVectorBases.enumerated() {
|
|
print("vector warp base \(b): lane0 \(String(format: "%016llx", outs[i][0])) lane31 \(String(format: "%016llx", outs[i][31]))")
|
|
}
|
|
print("OVERALL: PASS (pack written)")
|
|
exit(0)
|
|
}
|
|
|
|
// MARK: - Timing
|
|
|
|
@inline(__always) func nowNs() -> UInt64 { clock_gettime_nsec_np(CLOCK_UPTIME_RAW) }
|
|
func ms(_ a: UInt64, _ b: UInt64) -> Double { Double(b - a) / 1e6 }
|
|
func fmt(_ v: Double, _ digits: Int = 2) -> String { String(format: "%.\(digits)f", v) }
|
|
|
|
// MARK: - GPU context
|
|
|
|
final class GPU {
|
|
let device: MTLDevice
|
|
let queue: MTLCommandQueue
|
|
init() {
|
|
guard let d = MTLCreateSystemDefaultDevice(), let q = d.makeCommandQueue() else {
|
|
print("FAIL: no Metal device"); exit(1)
|
|
}
|
|
device = d; queue = q
|
|
}
|
|
}
|
|
|
|
// Everything about the dataset for one day: which construction, the GPU kernels that build it, the GPU cache
|
|
// (memory-hard mode), and the CPU side the verifier uses. Tests and the bench share one of these.
|
|
final class DatasetContext {
|
|
let gpu: GPU
|
|
let closed: Bool
|
|
let dayString: String
|
|
let key: [UInt32] // the 8 words of seedWords("day/" + day)
|
|
var day: (UInt32, UInt32) { (key[0], key[1]) }
|
|
let mp: MixParams? // memory-hard mixer parameters (nil in closed-form mode)
|
|
var modeName: String { closed ? "closed-form" : "memory-hard" }
|
|
private var closedFillPipe: MTLComputePipelineState?
|
|
private var cacheFillPipe: MTLComputePipelineState?
|
|
private var buildPipe: MTLComputePipelineState?
|
|
var compileMs = 0.0
|
|
var gpuCache: MTLBuffer? // 2^26 words, private
|
|
var cacheFillGPUms = 0.0, cacheFillWallMs = 0.0
|
|
var lastBuildGPUms = 0.0, lastBuildWallMs = 0.0
|
|
private var cpu: MemhardCPU?
|
|
|
|
init(gpu: GPU, closedForm: Bool, dayString: String) {
|
|
self.gpu = gpu; closed = closedForm; self.dayString = dayString
|
|
key = seedWords("day/" + dayString)
|
|
mp = closedForm ? nil : MixParams(key: key)
|
|
let t0 = nowNs()
|
|
do {
|
|
if closedForm {
|
|
let lib = try gpu.device.makeLibrary(source: fillMSL, options: MTLCompileOptions())
|
|
closedFillPipe = try gpu.device.makeComputePipelineState(function: lib.makeFunction(name: "igneum_fill")!)
|
|
} else {
|
|
let lib = try gpu.device.makeLibrary(source: memhardMSL(mp!), options: MTLCompileOptions())
|
|
cacheFillPipe = try gpu.device.makeComputePipelineState(function: lib.makeFunction(name: "igneum_cache_fill")!)
|
|
buildPipe = try gpu.device.makeComputePipelineState(function: lib.makeFunction(name: "igneum_build")!)
|
|
}
|
|
} catch { print("FAIL: dataset kernel compile: \(error)"); exit(1) }
|
|
compileMs = ms(t0, nowNs())
|
|
if !closedForm {
|
|
guard let c = gpu.device.makeBuffer(length: cacheWords * 4, options: .storageModePrivate) else { print("FAIL: cannot allocate the cache"); exit(1) }
|
|
gpuCache = c
|
|
let cb = gpu.queue.makeCommandBuffer()!
|
|
let enc = cb.makeComputeCommandEncoder()!
|
|
enc.setComputePipelineState(cacheFillPipe!)
|
|
enc.setBuffer(c, offset: 0, index: 0)
|
|
enc.dispatchThreadgroups(MTLSize(width: cacheSegments / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
|
|
enc.endEncoding()
|
|
let w0 = nowNs()
|
|
cb.commit(); cb.waitUntilCompleted()
|
|
cacheFillWallMs = ms(w0, nowNs())
|
|
if let e = cb.error { print("FAIL: cache fill error \(e)"); exit(1) }
|
|
cacheFillGPUms = (cb.gpuEndTime - cb.gpuStartTime) * 1000
|
|
}
|
|
}
|
|
|
|
// Fills the GPU cache again (for repeat timings). Returns GPU ms.
|
|
func refillCache() -> Double {
|
|
guard let c = gpuCache else { return 0 }
|
|
let cb = gpu.queue.makeCommandBuffer()!
|
|
let enc = cb.makeComputeCommandEncoder()!
|
|
enc.setComputePipelineState(cacheFillPipe!)
|
|
enc.setBuffer(c, offset: 0, index: 0)
|
|
enc.dispatchThreadgroups(MTLSize(width: cacheSegments / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
|
|
enc.endEncoding()
|
|
cb.commit(); cb.waitUntilCompleted()
|
|
return (cb.gpuEndTime - cb.gpuStartTime) * 1000
|
|
}
|
|
|
|
// Allocates a private 2^log2-word dataset and builds it on the GPU. Timing lands in lastBuild*.
|
|
func makeDataset(log2: Int) -> MTLBuffer {
|
|
let words = 1 << log2
|
|
guard let buf = gpu.device.makeBuffer(length: words * 4, options: .storageModePrivate) else {
|
|
print("FAIL: cannot allocate 2^\(log2) word dataset"); exit(1)
|
|
}
|
|
build(into: buf, words: words)
|
|
return buf
|
|
}
|
|
|
|
func build(into buf: MTLBuffer, words: Int) {
|
|
let cb = gpu.queue.makeCommandBuffer()!
|
|
let enc = cb.makeComputeCommandEncoder()!
|
|
if closed {
|
|
enc.setComputePipelineState(closedFillPipe!)
|
|
enc.setBuffer(buf, offset: 0, index: 0)
|
|
var d = day
|
|
enc.setBytes(&d, length: 8, index: 1)
|
|
enc.dispatchThreadgroups(MTLSize(width: words / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
|
|
} else {
|
|
enc.setComputePipelineState(buildPipe!)
|
|
enc.setBuffer(gpuCache!, offset: 0, index: 0)
|
|
enc.setBuffer(buf, offset: 0, index: 1)
|
|
let items = words / 16
|
|
enc.dispatchThreadgroups(MTLSize(width: max(items / 256, 1), height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: min(items, 256), height: 1, depth: 1))
|
|
}
|
|
enc.endEncoding()
|
|
let w0 = nowNs()
|
|
cb.commit(); cb.waitUntilCompleted()
|
|
lastBuildWallMs = ms(w0, nowNs())
|
|
if let e = cb.error { print("FAIL: dataset build error \(e)"); exit(1) }
|
|
lastBuildGPUms = (cb.gpuEndTime - cb.gpuStartTime) * 1000
|
|
}
|
|
|
|
// The CPU verifier's side: the 256 MiB cache computed on one core, once per process. Prints the fill time.
|
|
func cpuSide() -> MemhardCPU? {
|
|
if closed { return nil }
|
|
if let c = cpu { return c }
|
|
let c = MemhardCPU(key: key)
|
|
print("cache: CPU fill \(fmt(c.fillMs, 1)) ms on one core (2^\(cacheLog2Words) words, \(cacheSegments) chains of \(cacheLinesPerSegment) ChaCha\(chachaRounds) blocks)")
|
|
cpu = c
|
|
return c
|
|
}
|
|
|
|
func source(log2: Int) -> DatasetSource {
|
|
DatasetSource(mask: UInt32((1 << log2) - 1), day: day, memhard: cpuSide())
|
|
}
|
|
|
|
// The buffer the inline (shortcut) kernel binds at index 0: the cache in memory-hard mode, the dataset otherwise.
|
|
func inlineBuffer(dataset: MTLBuffer) -> MTLBuffer { closed ? dataset : gpuCache! }
|
|
var inlineSource: LoadSource { closed ? .inlineClosed(day.0, day.1) : .inlineMemhard(mp!) }
|
|
|
|
// Blit the GPU cache to shared memory and compare every word with the CPU cache.
|
|
func cacheCheck() -> (ok: Bool, detail: String) {
|
|
guard let gc = gpuCache, let c = cpuSide() else { return (true, "cache check: not applicable (closed form)") }
|
|
guard let shared = gpu.device.makeBuffer(length: cacheWords * 4, options: .storageModeShared) else { return (false, "cache check: no shared buffer") }
|
|
let cb = gpu.queue.makeCommandBuffer()!
|
|
let blit = cb.makeBlitCommandEncoder()!
|
|
blit.copy(from: gc, sourceOffset: 0, to: shared, destinationOffset: 0, size: cacheWords * 4)
|
|
blit.endEncoding()
|
|
cb.commit(); cb.waitUntilCompleted()
|
|
let same = memcmp(shared.contents(), c.cache, cacheWords * 4) == 0
|
|
let fp = fnv64(shared.contents(), cacheWords * 4)
|
|
return (same, "cache check: GPU cache \(same ? "==" : "!=") CPU cache, all \(cacheWords) words compared, FNV-1a 64 \(h64(fp))")
|
|
}
|
|
|
|
// Reads the given words of a GPU-built dataset back and compares with the CPU derivation.
|
|
func sampleCheck(log2: Int, indices: [UInt32]) -> (ok: Bool, detail: String) {
|
|
let words = 1 << log2
|
|
let dataset = makeDataset(log2: log2)
|
|
guard let shared = gpu.device.makeBuffer(length: words * 4, options: .storageModeShared) else { return (false, "sample check: no shared buffer") }
|
|
let cb = gpu.queue.makeCommandBuffer()!
|
|
let blit = cb.makeBlitCommandEncoder()!
|
|
blit.copy(from: dataset, sourceOffset: 0, to: shared, destinationOffset: 0, size: words * 4)
|
|
blit.endEncoding()
|
|
cb.commit(); cb.waitUntilCompleted()
|
|
let p = shared.contents().bindMemory(to: UInt32.self, capacity: words)
|
|
let ds = source(log2: log2)
|
|
var bad = 0
|
|
for i in indices where p[Int(i)] != ds.word(i) { bad += 1 }
|
|
return (bad == 0, "dataset sample check (\(modeName), 2^\(log2) words): \(indices.count) GPU words vs CPU derivation, \(bad) mismatches")
|
|
}
|
|
}
|
|
|
|
struct EpochResult {
|
|
var seed: String
|
|
var libraryMs: Double
|
|
var pipelineMs: Double
|
|
var hashesPerSecWall: Double
|
|
var hashesPerSecGPU: Double
|
|
var gbpsWall: Double
|
|
var gbpsGPU: Double
|
|
var loadsPerHash: Int
|
|
var itemsPerWarp: Int
|
|
var verify: [(warp: Int, pass: Bool, ms: Double, repMs: Double)]
|
|
var allPass: Bool { verify.allSatisfy { $0.pass } }
|
|
}
|
|
|
|
func runEpoch(gpu: GPU, opts: Options, seedString: String, dataset: MTLBuffer, ctx: DatasetContext) -> EpochResult {
|
|
let program = generateProgram(seedString: seedString)
|
|
let msl = generateMSL(program, datasetLog2: opts.datasetLog2, source: opts.inlineDataset ? ctx.inlineSource : .stored)
|
|
if opts.inlineDataset {
|
|
print(ctx.closed ? "\nNOTE: --inline-dataset: loads compute ds_elem inline, the dataset buffer is never read"
|
|
: "\nNOTE: --inline-dataset: loads derive the item from the 256 MiB cache (8 dependent 64-byte reads + 9 mixers), the dataset buffer is never read")
|
|
}
|
|
if let dir = opts.dumpDir {
|
|
try? FileManager.default.createDirectory(atPath: dir, withIntermediateDirectories: true)
|
|
let safe = seedString.replacingOccurrences(of: "/", with: "_")
|
|
try? msl.write(toFile: "\(dir)/program-\(safe).metal", atomically: true, encoding: .utf8)
|
|
}
|
|
|
|
print("\n=== epoch seed \"\(seedString)\" ===")
|
|
print("program: \(Program.count) instructions x \(Program.iterations) iterations, loads/hash = \(program.loadsPerHash) (wide \(program.wideLoadsPerHash)), distinct items per warp = \(program.itemsPerWarp)")
|
|
print("op mix: " + program.histogram.map { "\($0.0)=\($0.1)" }.joined(separator: " "))
|
|
|
|
// Runtime compile
|
|
let t0 = nowNs()
|
|
let library: MTLLibrary
|
|
do {
|
|
let copts = MTLCompileOptions()
|
|
library = try gpu.device.makeLibrary(source: msl, options: copts)
|
|
} catch {
|
|
print("FAIL: Metal compile error:\n\(error)")
|
|
exit(1)
|
|
}
|
|
let t1 = nowNs()
|
|
guard let fn = library.makeFunction(name: "igneum_hash") else { print("FAIL: no igneum_hash"); exit(1) }
|
|
let pipeline: MTLComputePipelineState
|
|
do { pipeline = try gpu.device.makeComputePipelineState(function: fn) } catch {
|
|
print("FAIL: pipeline error: \(error)"); exit(1)
|
|
}
|
|
let t2 = nowNs()
|
|
let libMs = ms(t0, t1), pipeMs = ms(t1, t2)
|
|
print("compile: library \(fmt(libMs)) ms, pipeline \(fmt(pipeMs)) ms, total \(fmt(libMs + pipeMs)) ms")
|
|
print("threadExecutionWidth = \(pipeline.threadExecutionWidth), maxTotalThreadsPerThreadgroup = \(pipeline.maxTotalThreadsPerThreadgroup)")
|
|
if pipeline.threadExecutionWidth != 32 {
|
|
print("WARNING: threadExecutionWidth is not 32; the one-warp-per-threadgroup assumption does not hold on this device")
|
|
}
|
|
|
|
// Buffers
|
|
let n = 1 << opts.batchLog2
|
|
let groups = n / 32
|
|
guard let outBuf = gpu.device.makeBuffer(length: n * 8, options: .storageModeShared) else { print("FAIL: out buffer"); exit(1) }
|
|
|
|
let buffer0 = opts.inlineDataset ? ctx.inlineBuffer(dataset: dataset) : dataset
|
|
func encodeBatch(_ cb: MTLCommandBuffer, base: UInt32) {
|
|
let enc = cb.makeComputeCommandEncoder()!
|
|
enc.setComputePipelineState(pipeline)
|
|
enc.setBuffer(buffer0, offset: 0, index: 0)
|
|
enc.setBuffer(outBuf, offset: 0, index: 1)
|
|
var b = base
|
|
enc.setBytes(&b, length: 4, index: 2)
|
|
enc.dispatchThreadgroups(MTLSize(width: groups, height: 1, depth: 1),
|
|
threadsPerThreadgroup: MTLSize(width: 32, height: 1, depth: 1))
|
|
enc.endEncoding()
|
|
}
|
|
|
|
// Batch 0: warm-up and verification source (baseNonce 0)
|
|
do {
|
|
let cb = gpu.queue.makeCommandBuffer()!
|
|
encodeBatch(cb, base: 0)
|
|
let w0 = nowNs()
|
|
cb.commit(); cb.waitUntilCompleted()
|
|
let w1 = nowNs()
|
|
if let e = cb.error { print("FAIL: batch 0 error \(e)"); exit(1) }
|
|
print("warm-up batch: \(n) hashes in \(fmt(ms(w0, w1))) ms wall, \(fmt((cb.gpuEndTime - cb.gpuStartTime) * 1000)) ms GPU")
|
|
}
|
|
// Pick warps to verify from batch 0
|
|
var warps = [0, groups / 2 + 1, groups - 1]
|
|
var vr = SplitMix64(s: UInt64(program.seed[4]) | (UInt64(program.seed[5]) << 32))
|
|
while warps.count < opts.verifyWarps { warps.append(vr.below(groups)) }
|
|
warps = Array(warps.prefix(max(opts.verifyWarps, 1)))
|
|
let outPtr = outBuf.contents().bindMemory(to: UInt64.self, capacity: n)
|
|
var gpuOutputs = [[UInt64]]()
|
|
for w in warps { gpuOutputs.append((0..<32).map { outPtr[w * 32 + $0] }) }
|
|
|
|
// Timed batches
|
|
var cbs = [MTLCommandBuffer]()
|
|
for b in 0..<opts.batches {
|
|
let cb = gpu.queue.makeCommandBuffer()!
|
|
encodeBatch(cb, base: UInt32(truncatingIfNeeded: (b + 1) * n))
|
|
cbs.append(cb)
|
|
}
|
|
let s0 = nowNs()
|
|
for cb in cbs { cb.commit() }
|
|
cbs.last!.waitUntilCompleted()
|
|
let s1 = nowNs()
|
|
for cb in cbs { if let e = cb.error { print("FAIL: batch error \(e)"); exit(1) } }
|
|
let gpuSeconds = cbs.reduce(0.0) { $0 + ($1.gpuEndTime - $1.gpuStartTime) }
|
|
let wallSeconds = Double(s1 - s0) / 1e9
|
|
let totalHashes = Double(n * opts.batches)
|
|
let hpsWall = totalHashes / wallSeconds
|
|
let hpsGPU = totalHashes / gpuSeconds
|
|
let bytesPerHash = Double(program.loadsPerHash * 4)
|
|
let gbpsWall = hpsWall * bytesPerHash / 1e9
|
|
let gbpsGPU = hpsGPU * bytesPerHash / 1e9
|
|
print("timed: \(opts.batches) batches x \(n) hashes = \(Int(totalHashes)) hashes")
|
|
print(" wall \(fmt(wallSeconds * 1000)) ms -> \(fmt(hpsWall / 1e6, 3)) Mhash/s, \(fmt(gbpsWall)) GB/s useful (loads x 4 B)")
|
|
print(" GPU \(fmt(gpuSeconds * 1000)) ms -> \(fmt(hpsGPU / 1e6, 3)) Mhash/s, \(fmt(gbpsGPU)) GB/s useful (loads x 4 B)")
|
|
|
|
// CPU verification. The verifier holds the 256 MiB cache (memory-hard mode) or nothing (closed form), never
|
|
// the dataset; every dataset word a load needs is derived on demand inside cpuWarp.
|
|
let ds = ctx.source(log2: opts.datasetLog2)
|
|
var verify = [(warp: Int, pass: Bool, ms: Double, repMs: Double)]()
|
|
for (i, w) in warps.enumerated() {
|
|
let base = UInt32(w * 32)
|
|
let d0 = ds.memhard?.derivations ?? 0
|
|
let c0 = nowNs()
|
|
let cpu = cpuWarp(program, baseNonce: base, ds: ds)
|
|
let c1 = nowNs()
|
|
let derived = (ds.memhard?.derivations ?? 0) - d0
|
|
// repeated runs for a steadier figure
|
|
let reps = 20
|
|
let r0 = nowNs()
|
|
var sink: UInt64 = 0
|
|
for _ in 0..<reps { sink ^= cpuWarp(program, baseNonce: base, ds: ds)[0] }
|
|
let r1 = nowNs()
|
|
let pass = cpu == gpuOutputs[i] && sink != 1
|
|
let single = ms(c0, c1), rep = ms(r0, r1) / Double(reps)
|
|
verify.append((w, pass, single, rep))
|
|
var detail = ""
|
|
if !pass {
|
|
let bad = (0..<32).filter { cpu[$0] != gpuOutputs[i][$0] }
|
|
detail = " mismatched lanes: \(bad) first: cpu=\(String(format: "%016llx", cpu[bad.first ?? 0])) gpu=\(String(format: "%016llx", gpuOutputs[i][bad.first ?? 0]))"
|
|
}
|
|
let items = ds.memhard == nil ? "" : ", \(derived) items derived"
|
|
print("verify warp \(w) (nonces \(base)..\(base + 31)): \(pass ? "PASS" : "FAIL") cpu \(fmt(single, 3)) ms single, \(fmt(rep, 3)) ms avg of \(reps)\(items)\(detail)")
|
|
}
|
|
|
|
return EpochResult(seed: seedString, libraryMs: libMs, pipelineMs: pipeMs,
|
|
hashesPerSecWall: hpsWall, hashesPerSecGPU: hpsGPU, gbpsWall: gbpsWall, gbpsGPU: gbpsGPU,
|
|
loadsPerHash: program.loadsPerHash, itemsPerWarp: program.itemsPerWarp, verify: verify)
|
|
}
|
|
|
|
// MARK: - Hardening tests: shared helpers
|
|
//
|
|
// Added 3 October 2026. Everything below reuses generateProgram, generateMSL, fillMSL and cpuWarp
|
|
// unchanged; the helpers only wrap compile, fill and dispatch so the tests can run many programs
|
|
// and many warps cheaply. Nothing in the bench path (runEpoch) or the pack exporter calls these.
|
|
|
|
struct CompiledHash {
|
|
let pipeline: MTLComputePipelineState
|
|
let libraryMs: Double
|
|
let pipelineMs: Double
|
|
var totalMs: Double { libraryMs + pipelineMs }
|
|
}
|
|
|
|
struct IgneumError: Error, CustomStringConvertible {
|
|
let description: String
|
|
init(_ s: String) { description = s }
|
|
}
|
|
|
|
func compileHash(_ gpu: GPU, msl: String) throws -> CompiledHash {
|
|
let t0 = nowNs()
|
|
let lib = try gpu.device.makeLibrary(source: msl, options: MTLCompileOptions())
|
|
let t1 = nowNs()
|
|
guard let fn = lib.makeFunction(name: "igneum_hash") else { throw IgneumError("no igneum_hash function in library") }
|
|
let pipe = try gpu.device.makeComputePipelineState(function: fn)
|
|
let t2 = nowNs()
|
|
return CompiledHash(pipeline: pipe, libraryMs: ms(t0, t1), pipelineMs: ms(t1, t2))
|
|
}
|
|
|
|
// Dataset allocation and build moved into DatasetContext.makeDataset (3 October 2026), which handles both constructions.
|
|
|
|
// One 32-thread threadgroup per base nonce, all dispatched from one encoder. Warp i lands at byte offset
|
|
// i * 256 of the output buffer, which is pre-filled with a sentinel so an unwritten lane is visible.
|
|
func gpuWarps(_ gpu: GPU, _ k: CompiledHash, dataset: MTLBuffer, bases: [UInt32]) -> [[UInt64]]? {
|
|
guard !bases.isEmpty, let outBuf = gpu.device.makeBuffer(length: bases.count * 256, options: .storageModeShared) else { return nil }
|
|
memset(outBuf.contents(), 0xAA, outBuf.length)
|
|
let cb = gpu.queue.makeCommandBuffer()!
|
|
let enc = cb.makeComputeCommandEncoder()!
|
|
enc.setComputePipelineState(k.pipeline)
|
|
enc.setBuffer(dataset, offset: 0, index: 0)
|
|
for (i, base) in bases.enumerated() {
|
|
enc.setBuffer(outBuf, offset: i * 256, index: 1)
|
|
var b = base
|
|
enc.setBytes(&b, length: 4, index: 2)
|
|
enc.dispatchThreadgroups(MTLSize(width: 1, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 32, height: 1, depth: 1))
|
|
}
|
|
enc.endEncoding()
|
|
cb.commit(); cb.waitUntilCompleted()
|
|
if cb.error != nil { return nil }
|
|
let p = outBuf.contents().bindMemory(to: UInt64.self, capacity: bases.count * 32)
|
|
return (0..<bases.count).map { i in (0..<32).map { p[i * 32 + $0] } }
|
|
}
|
|
|
|
// `count` consecutive nonces from `base` (count a multiple of 32) into `out`. Returns GPU time in ms.
|
|
func gpuRange(_ gpu: GPU, _ k: CompiledHash, dataset: MTLBuffer, base: UInt32, count: Int, out: MTLBuffer) -> Double? {
|
|
let cb = gpu.queue.makeCommandBuffer()!
|
|
let enc = cb.makeComputeCommandEncoder()!
|
|
enc.setComputePipelineState(k.pipeline)
|
|
enc.setBuffer(dataset, offset: 0, index: 0)
|
|
enc.setBuffer(out, offset: 0, index: 1)
|
|
var b = base
|
|
enc.setBytes(&b, length: 4, index: 2)
|
|
enc.dispatchThreadgroups(MTLSize(width: count / 32, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 32, height: 1, depth: 1))
|
|
enc.endEncoding()
|
|
cb.commit(); cb.waitUntilCompleted()
|
|
if cb.error != nil { return nil }
|
|
return (cb.gpuEndTime - cb.gpuStartTime) * 1000
|
|
}
|
|
|
|
func h64(_ v: UInt64) -> String { String(format: "%016llx", v) }
|
|
func pad(_ s: String, _ n: Int) -> String { s.count >= n ? s : s + String(repeating: " ", count: n - s.count) }
|
|
|
|
func describeProgram(_ p: Program) -> String {
|
|
var s = " program seed \"\(p.seedString)\" words [\(p.seed.map(hex).joined(separator: ", "))], \(p.instrs.count) instructions\n"
|
|
for (k, i) in p.instrs.enumerated() {
|
|
s += " \(pad(String(k), 3)) \(pad(i.op.rawValue, 5)) dst=r\(i.dst) src=r\(i.a) src2=r\(i.b) imm=\(hex(i.imm)) imm2=\(hex(i.imm2)) rot=\(i.rot) bit=\(i.bit) mask=\(i.mask)\n"
|
|
}
|
|
return s
|
|
}
|
|
|
|
func fnv64(_ ptr: UnsafeRawPointer, _ count: Int) -> UInt64 {
|
|
var h: UInt64 = 0xcbf29ce484222325
|
|
let b = ptr.bindMemory(to: UInt8.self, capacity: count)
|
|
for i in 0..<count { h ^= UInt64(b[i]); h &*= 0x100000001b3 }
|
|
return h
|
|
}
|
|
|
|
func regexCount(_ pattern: String, in text: String) -> Int {
|
|
let re = try! NSRegularExpression(pattern: pattern)
|
|
return re.numberOfMatches(in: text, range: NSRange(text.startIndex..., in: text))
|
|
}
|
|
|
|
// Static check: every dataset access in the generated MSL is `dataset[rN & MASK]`, and the identifier
|
|
// `dataset` appears nowhere else except the kernel parameter.
|
|
// A wide load (lever b) is `dataset[(simd_broadcast(rN, 0) & WMASK) + lane]` with WMASK = MASK & ~31 and lane < 32,
|
|
// so its index is at most MASK as well.
|
|
func maskCheckMSL(_ msl: String) -> (ok: Bool, detail: String) {
|
|
let total = regexCount("dataset\\[", in: msl)
|
|
let masked = regexCount("dataset\\[r[0-7] & MASK\\]", in: msl)
|
|
let wide = regexCount("dataset\\[\\(simd_broadcast\\(r[0-7], 0\\) & WMASK\\) \\+ lane\\]", in: msl)
|
|
let words = regexCount("\\bdataset\\b", in: msl)
|
|
let ok = total == masked + wide && words == total + 1
|
|
return (ok, "MSL: \(total) dataset[ accesses, \(masked) of the form dataset[rN & MASK], \(wide) wide loads dataset[(simd_broadcast(rN, 0) & WMASK) + lane], identifier appears \(words) times (expected \(total + 1))")
|
|
}
|
|
|
|
// Same for the CUDA twin: hash accesses are `ds[rN & mask]`; the fill kernel's one write is guarded by `if (i < n)`.
|
|
// The closed-form pack has one fill write guarded by `if (i < n)`; the memory-hard pack writes the dataset through a
|
|
// `d` pointer in igneum_build and has no `ds[` write at all.
|
|
func maskCheckCUDA(_ cu: String, memhard: Bool) -> (ok: Bool, detail: String) {
|
|
let total = regexCount("\\bds\\[", in: cu)
|
|
let masked = regexCount("\\bds\\[r[0-7] & mask\\]", in: cu)
|
|
let wide = regexCount("\\bds\\[\\(__shfl_sync\\(0xffffffffu, r[0-7], 0\\) & wmask\\) \\+ lane\\]", in: cu)
|
|
let fill = regexCount("if \\(i < n\\) ds\\[i\\] = ds_elem", in: cu)
|
|
let ok = total == masked + wide + fill && fill == (memhard ? 0 : 1)
|
|
return (ok, "CUDA: \(total) ds[ accesses, \(masked) of the form ds[rN & mask], \(wide) wide, \(fill) guarded fill write (expected \(memhard ? 0 : 1))")
|
|
}
|
|
|
|
// MARK: - --fuzz
|
|
|
|
func runFuzz(_ opts: Options, ctx: DatasetContext) -> Bool {
|
|
let gpu = ctx.gpu
|
|
let n = max(opts.fuzz ?? 200, 1)
|
|
let master = opts.fuzzSeed
|
|
print("\n=== fuzz: \(n) random programs, master seed \"\(master)\", 4 random warps each ===")
|
|
let sizes = [24, 26, 28]
|
|
let d0 = nowNs()
|
|
var datasets = [Int: MTLBuffer]()
|
|
for s in sizes { datasets[s] = ctx.makeDataset(log2: s) }
|
|
print("datasets " + sizes.map { "2^\($0) (\((1 << $0) * 4 / (1 << 20)) MiB)" }.joined(separator: ", ") + " filled in \(fmt(ms(d0, nowNs()), 1)) ms")
|
|
|
|
let mw = seedWords("fuzz/" + master)
|
|
var rng = SplitMix64(s: UInt64(mw[0]) | (UInt64(mw[1]) << 32))
|
|
var pass = 0, fail = 0, compileFail = 0, staticFail = 0, contractFail = 0
|
|
var perSize = [Int: (pass: Int, fail: Int)]()
|
|
var compileMs = [Double]()
|
|
var cpuNs: UInt64 = 0, gpuNs: UInt64 = 0
|
|
var warps = 0
|
|
var opCount = [String: Int]()
|
|
var loadsMin = Int.max, loadsMax = 0
|
|
let t0 = nowNs()
|
|
for i in 0..<n {
|
|
let seedString = "\(master)/\(i)/\(h64(rng.next()))"
|
|
let log2 = sizes[rng.below(sizes.count)]
|
|
let bases = (0..<4).map { _ in UInt32(truncatingIfNeeded: rng.next()) }
|
|
let program = generateProgram(seedString: seedString)
|
|
for ins in program.instrs {
|
|
opCount[ins.op.rawValue, default: 0] += 1
|
|
// Generator contract, relied on by the MSL emitter: rotl amount 1..31, shuffle mask a power of two <= 16,
|
|
// source register never the destination.
|
|
if ins.rot < 1 || ins.rot > 31 || ![1, 2, 4, 8, 16].contains(ins.mask) || ins.a == ins.dst || ins.dst > 7 || ins.a > 7 || ins.b > 7 {
|
|
contractFail += 1
|
|
print("CONTRACT FAIL seed \"\(seedString)\": \(ins)")
|
|
}
|
|
}
|
|
loadsMin = min(loadsMin, program.loadsPerHash); loadsMax = max(loadsMax, program.loadsPerHash)
|
|
let msl = generateMSL(program, datasetLog2: log2)
|
|
let sc = maskCheckMSL(msl)
|
|
if !sc.ok { staticFail += 1; print("STATIC MASK FAIL seed \"\(seedString)\": \(sc.detail)") }
|
|
let k: CompiledHash
|
|
do { k = try compileHash(gpu, msl: msl) } catch {
|
|
compileFail += 1
|
|
print("COMPILE FAIL seed \"\(seedString)\" dataset 2^\(log2):\n\(error)\n\(describeProgram(program))")
|
|
continue
|
|
}
|
|
compileMs.append(k.totalMs)
|
|
let g0 = nowNs()
|
|
guard let gpuOut = gpuWarps(gpu, k, dataset: datasets[log2]!, bases: bases) else {
|
|
fail += 1; print("GPU RUN FAIL seed \"\(seedString)\" dataset 2^\(log2)"); continue
|
|
}
|
|
let g1 = nowNs()
|
|
let ds = ctx.source(log2: log2)
|
|
var ok = true
|
|
for (w, base) in bases.enumerated() {
|
|
let cpu = cpuWarp(program, baseNonce: base, ds: ds)
|
|
warps += 1
|
|
if cpu != gpuOut[w] {
|
|
ok = false
|
|
let bad = (0..<32).filter { cpu[$0] != gpuOut[w][$0] }
|
|
print("MISMATCH seed \"\(seedString)\" dataset 2^\(log2) warp \(w) base nonce \(base) (\(hex(base))) lanes \(bad)")
|
|
for l in bad { print(" lane \(l) nonce \(base &+ UInt32(l)): gpu \(h64(gpuOut[w][l])) cpu \(h64(cpu[l]))") }
|
|
print(describeProgram(program))
|
|
}
|
|
}
|
|
let g2 = nowNs()
|
|
gpuNs += g1 - g0; cpuNs += g2 - g1
|
|
if ok { pass += 1 } else { fail += 1 }
|
|
var ps = perSize[log2] ?? (0, 0)
|
|
if ok { ps.pass += 1 } else { ps.fail += 1 }
|
|
perSize[log2] = ps
|
|
if (i + 1) % 100 == 0 || i + 1 == n {
|
|
print(" \(i + 1)/\(n): pass \(pass) fail \(fail) compile-fail \(compileFail), \(fmt(Double(nowNs() - t0) / 1e9, 1)) s elapsed")
|
|
}
|
|
}
|
|
let total = Double(nowNs() - t0) / 1e9
|
|
let cAvg = compileMs.isEmpty ? 0 : compileMs.reduce(0, +) / Double(compileMs.count)
|
|
print("\n| Dataset | Programs | Pass | Fail |")
|
|
print("|---|---|---|---|")
|
|
for s in sizes {
|
|
let ps = perSize[s] ?? (0, 0)
|
|
print("| 2^\(s) words (\((1 << s) * 4 / (1 << 20)) MiB) | \(ps.pass + ps.fail) | \(ps.pass) | \(ps.fail) |")
|
|
}
|
|
print("| all | \(pass + fail) | \(pass) | \(fail) |")
|
|
print("programs \(n): pass \(pass), mismatch \(fail), compile failures \(compileFail), static mask failures \(staticFail), generator contract failures \(contractFail)")
|
|
print("warps compared \(warps) (\(warps * 32) hashes), loads/hash range \(loadsMin)..\(loadsMax)")
|
|
print("op totals over all programs: " + opCount.sorted { $0.value != $1.value ? $0.value > $1.value : $0.key < $1.key }.map { "\($0.key)=\($0.value)" }.joined(separator: " "))
|
|
print("compile ms (library+pipeline): min \(fmt(compileMs.min() ?? 0, 1)) avg \(fmt(cAvg, 1)) max \(fmt(compileMs.max() ?? 0, 1)); GPU dispatch total \(fmt(Double(gpuNs) / 1e6, 1)) ms; CPU interpreter total \(fmt(Double(cpuNs) / 1e6, 1)) ms; wall \(fmt(total, 1)) s")
|
|
let ok = fail == 0 && compileFail == 0 && staticFail == 0 && contractFail == 0 && pass == n
|
|
print("FUZZ: \(ok ? "PASS" : "FAIL")")
|
|
return ok
|
|
}
|
|
|
|
// MARK: - --edge
|
|
|
|
struct EdgeCase {
|
|
let name: String
|
|
let instrs: [Instr]
|
|
// (instruction index, what must hold, check on lane-0 registers as they are just before that instruction)
|
|
let pre: [(Int, String, ([UInt32]) -> Bool)]
|
|
let informational: Bool // reported but not counted: exercises something the generator never emits
|
|
}
|
|
|
|
// Instruction builder for hand-made programs. imm2 = imm so the `add` is a constant regardless of the selector bit.
|
|
func I(_ op: Op, _ dst: Int, _ a: Int, b: Int = 0, imm: UInt32 = 0, rot: UInt32 = 1, bit: Int = 0, mask: Int = 1) -> Instr {
|
|
Instr(op: op, dst: dst, a: a, b: b, imm: imm, imm2: imm, rot: rot, bit: bit, mask: mask)
|
|
}
|
|
func zero(_ r: Int) -> Instr { I(.sub, r, r) } // r = r - r = 0 (src == dst, never generated, legal MSL)
|
|
func set(_ r: Int, _ v: UInt32) -> [Instr] { [zero(r), I(.add, r, 7, imm: v)] } // needs r7 == 0
|
|
|
|
func edgeCases(mask: UInt32) -> [EdgeCase] {
|
|
let M = mask
|
|
var c = [EdgeCase]()
|
|
c.append(EdgeCase(name: "rotl immediate by 1 and by 31",
|
|
instrs: [I(.rotl, 1, 0, rot: 1), I(.rotl, 2, 0, rot: 31), I(.xor, 3, 1), I(.xor, 4, 2), I(.rotl, 5, 0, rot: 1), I(.rotl, 6, 0, rot: 31)],
|
|
pre: [], informational: false))
|
|
c.append(EdgeCase(name: "rotr by register == 0",
|
|
instrs: [zero(7), I(.rotr, 3, 7), I(.xor, 4, 3)],
|
|
pre: [(1, "r7 == 0", { $0[7] == 0 })], informational: false))
|
|
c.append(EdgeCase(name: "rotr by register == 32 (32 mod 32 = 0)",
|
|
instrs: [zero(7)] + (set(1, 32) + [I(.rotr, 3, 1), I(.xor, 4, 3)]),
|
|
pre: [(3, "r1 == 32", { $0[1] == 32 })], informational: false))
|
|
c.append(EdgeCase(name: "rotr by register == 0xFFFFFFE0 (-32, 0 mod 32)",
|
|
instrs: [zero(7)] + (set(1, 0xFFFFFFE0) + [I(.rotr, 3, 1), I(.xor, 4, 3)]),
|
|
pre: [(3, "r1 == 0xFFFFFFE0", { $0[1] == 0xFFFFFFE0 })], informational: false))
|
|
var rr: [Instr] = [zero(7)]
|
|
rr += set(1, 31); rr += [I(.rotr, 3, 1), I(.add, 1, 7, imm: 32), I(.rotr, 4, 1)]
|
|
rr += set(2, 1); rr += [I(.rotr, 5, 2), I(.xor, 6, 5)]
|
|
c.append(EdgeCase(name: "rotr by register == 31 and == 63 and == 1",
|
|
instrs: rr,
|
|
pre: [(3, "r1 == 31", { $0[1] == 31 }), (5, "r1 == 63", { $0[1] == 63 }), (8, "r2 == 1", { $0[2] == 1 })], informational: false))
|
|
var mh: [Instr] = [zero(7)]
|
|
mh += set(1, 0xFFFFFFFF); mh += set(2, 0xFFFFFFFF)
|
|
mh += [I(.mulhi, 1, 2), I(.xor, 3, 1)]
|
|
mh += set(4, 0x80000000); mh += set(5, 2)
|
|
mh += [I(.mulhi, 4, 5), I(.xor, 3, 4), I(.mulhi, 6, 7), I(.xor, 0, 6)]
|
|
c.append(EdgeCase(name: "mulhi 0xFFFFFFFF x 0xFFFFFFFF, 0x80000000 x 2, x 0",
|
|
instrs: mh,
|
|
pre: [(5, "r1 == r2 == 0xFFFFFFFF", { $0[1] == 0xFFFFFFFF && $0[2] == 0xFFFFFFFF }),
|
|
(6, "mulhi result r1 == 0xFFFFFFFE", { $0[1] == 0xFFFFFFFE }),
|
|
(11, "r4 == 0x80000000, r5 == 2", { $0[4] == 0x80000000 && $0[5] == 2 }),
|
|
(12, "mulhi result r4 == 1", { $0[4] == 1 }),
|
|
(13, "r7 == 0", { $0[7] == 0 }),
|
|
(14, "mulhi by 0 gives r6 == 0", { $0[6] == 0 })], informational: false))
|
|
c.append(EdgeCase(name: "shfl_xor every mask 1..16 in sequence (generator uses only 1,2,4,8,16)",
|
|
instrs: (1...16).map { I(.shfl, $0 % 8, ($0 + 1) % 8, mask: $0) },
|
|
pre: [], informational: false))
|
|
var l0: [Instr] = [zero(7), I(.load, 3, 7)]
|
|
l0 += set(1, M &+ 1); l0 += [I(.load, 4, 1), I(.xor, 5, 4)]
|
|
c.append(EdgeCase(name: "load at index 0 (register 0, and register MASK+1 which masks to 0)",
|
|
instrs: l0,
|
|
pre: [(1, "r7 & MASK == 0", { $0[7] & M == 0 }),
|
|
(4, "r1 == MASK+1, so unmasked index is out of range and masked index is 0", { $0[1] == M &+ 1 && ($0[1] & M) == 0 })],
|
|
informational: false))
|
|
var lm: [Instr] = [zero(7)]
|
|
lm += set(1, M); lm += [I(.load, 3, 1)]
|
|
lm += set(2, 0xFFFFFFFF); lm += [I(.load, 4, 2), I(.xor, 5, 4)]
|
|
c.append(EdgeCase(name: "load at index MASK (register MASK, and register 0xFFFFFFFF which masks to MASK)",
|
|
instrs: lm,
|
|
pre: [(3, "r1 == MASK", { $0[1] == M }),
|
|
(6, "r2 == 0xFFFFFFFF, masked index == MASK", { $0[2] == 0xFFFFFFFF && ($0[2] & M) == M })],
|
|
informational: false))
|
|
c.append(EdgeCase(name: "add wraparound 0xFFFFFFFF + 1",
|
|
instrs: [zero(7)] + (set(1, 0xFFFFFFFF) + [I(.add, 1, 7, imm: 1), I(.xor, 2, 1)]),
|
|
pre: [(3, "r1 == 0xFFFFFFFF", { $0[1] == 0xFFFFFFFF }), (4, "r1 == 0 after add", { $0[1] == 0 })], informational: false))
|
|
c.append(EdgeCase(name: "sub wraparound 0 - 1",
|
|
instrs: [zero(7), zero(1)] + (set(2, 1) + [I(.sub, 1, 2), I(.xor, 3, 1)]),
|
|
pre: [(4, "r1 == 0, r2 == 1", { $0[1] == 0 && $0[2] == 1 }), (5, "r1 == 0xFFFFFFFF after sub", { $0[1] == 0xFFFFFFFF })], informational: false))
|
|
var mm: [Instr] = [zero(7)]
|
|
mm += set(1, 0xFFFFFFFF); mm += set(2, 0xFFFFFFFF)
|
|
mm += [I(.mul, 1, 2), I(.xor, 3, 1)]
|
|
mm += set(4, 0xFFFFFFFF); mm += set(5, 0xFFFFFFFF); mm += set(6, 5)
|
|
mm += [I(.mad, 6, 4, b: 5), I(.xor, 0, 6)]
|
|
c.append(EdgeCase(name: "mul and mad wraparound 0xFFFFFFFF x 0xFFFFFFFF",
|
|
instrs: mm,
|
|
pre: [(5, "r1 == r2 == 0xFFFFFFFF", { $0[1] == 0xFFFFFFFF && $0[2] == 0xFFFFFFFF }),
|
|
(6, "mul low result r1 == 1", { $0[1] == 1 }),
|
|
(13, "r4 == r5 == 0xFFFFFFFF, r6 == 5", { $0[4] == 0xFFFFFFFF && $0[5] == 0xFFFFFFFF && $0[6] == 5 }),
|
|
(14, "mad result r6 == 6", { $0[6] == 6 })], informational: false))
|
|
let z = generateProgram(seedString: "edge/zero-loads")
|
|
c.append(EdgeCase(name: "generated program with every load replaced by xor (zero loads)",
|
|
instrs: z.instrs.map { ins in var m = ins; if m.op == .load { m.op = .xor }; return m },
|
|
pre: [], informational: false))
|
|
c.append(EdgeCase(name: "64 loads and nothing else",
|
|
instrs: (0..<64).map { I(.load, $0 % 8, ($0 + 3) % 8) },
|
|
pre: [], informational: false))
|
|
c.append(EdgeCase(name: "rotl immediate by 0 (outside the generator's 1..31 contract; MSL shifts by 32)",
|
|
instrs: [I(.rotl, 1, 0, rot: 0), I(.xor, 2, 1)],
|
|
pre: [], informational: true))
|
|
return c
|
|
}
|
|
|
|
func runEdge(_ opts: Options, ctx: DatasetContext) -> Bool {
|
|
let gpu = ctx.gpu
|
|
let log2 = opts.datasetLog2
|
|
let mask = UInt32((1 << log2) - 1)
|
|
let ds = ctx.source(log2: log2)
|
|
print("\n=== edge cases, dataset 2^\(log2) words, MASK \(hex(mask)) ===")
|
|
let dataset = ctx.makeDataset(log2: log2)
|
|
let bases: [UInt32] = [0, 1 << 20, 0x7FFFFFF0, 0xFFFFFFE0]
|
|
print("warps: base nonces " + bases.map { hex($0) }.joined(separator: ", ") + " (the last two straddle 2^31 and wrap past 2^32)")
|
|
var allOk = true
|
|
var rows = [String]()
|
|
for ec in edgeCases(mask: mask) {
|
|
let program = Program(seedString: "edge/\(ec.name)", seed: seedWords("edge/\(ec.name)"), instrs: ec.instrs)
|
|
let msl = generateMSL(program, datasetLog2: log2)
|
|
var status = "", detail = ""
|
|
var ok = true
|
|
// Preconditions, checked on lane 0 of every warp in every iteration.
|
|
var preOk = true
|
|
var preNotes = [String]()
|
|
if !ec.pre.isEmpty {
|
|
for base in bases {
|
|
var hits = [Int: Int]()
|
|
var misses = [Int: Int]()
|
|
_ = cpuWarpTraced(program, baseNonce: base, ds: ds) { _, k, regs in
|
|
for (idx, _, check) in ec.pre where idx == k {
|
|
if check(regs) { hits[idx, default: 0] += 1 } else { misses[idx, default: 0] += 1 }
|
|
}
|
|
}
|
|
for (idx, what, _) in ec.pre {
|
|
if (misses[idx] ?? 0) > 0 || (hits[idx] ?? 0) != Program.iterations {
|
|
preOk = false
|
|
preNotes.append("base \(hex(base)) instr \(idx) '\(what)' held \(hits[idx] ?? 0)/\(Program.iterations) iterations")
|
|
}
|
|
}
|
|
}
|
|
if preOk { preNotes = ec.pre.map { "instr \($0.0): \($0.1)" } }
|
|
}
|
|
do {
|
|
let k = try compileHash(gpu, msl: msl)
|
|
guard let gpuOut = gpuWarps(gpu, k, dataset: dataset, bases: bases) else { throw IgneumError("GPU run failed") }
|
|
var badLanes = 0
|
|
var first = ""
|
|
for (w, base) in bases.enumerated() {
|
|
let cpu = cpuWarp(program, baseNonce: base, ds: ds)
|
|
for l in 0..<32 where cpu[l] != gpuOut[w][l] {
|
|
badLanes += 1
|
|
if first.isEmpty { first = "first: base \(hex(base)) lane \(l) gpu \(h64(gpuOut[w][l])) cpu \(h64(cpu[l]))" }
|
|
}
|
|
}
|
|
ok = badLanes == 0 && preOk
|
|
status = badLanes == 0 ? "GPU == CPU 128/128 lanes" : "MISMATCH \(badLanes)/128 lanes, \(first)"
|
|
detail = "compile \(fmt(k.totalMs, 1)) ms"
|
|
if badLanes > 0 { print(describeProgram(program)) }
|
|
} catch {
|
|
ok = false
|
|
status = "COMPILE FAIL: \(error)"
|
|
}
|
|
let pre = ec.pre.isEmpty ? "none needed" : (preOk ? "held (all 8 iterations, lane 0, 4 warps)" : "NOT HELD")
|
|
let verdict = ec.informational ? (ok ? "info: agrees" : "info: differs") : (ok ? "PASS" : "FAIL")
|
|
if !ec.informational && !ok { allOk = false }
|
|
print("\(verdict): \(ec.name)")
|
|
print(" \(ec.instrs.count) instructions, loads/hash \(program.loadsPerHash), \(status), \(detail)")
|
|
for n in preNotes { print(" precondition \(n)") }
|
|
rows.append("| \(ec.name) | \(ec.instrs.count) | \(program.loadsPerHash) | \(pre) | \(status) | \(verdict) |")
|
|
}
|
|
print("\n| Case | Instrs | Loads/hash | Preconditions | GPU vs CPU | Result |")
|
|
print("|---|---|---|---|---|---|")
|
|
for r in rows { print(r) }
|
|
print("EDGE: \(allOk ? "PASS" : "FAIL")")
|
|
return allOk
|
|
}
|
|
|
|
// MARK: - --stats
|
|
|
|
func popcount64(_ v: UInt64) -> Int { v.nonzeroBitCount }
|
|
|
|
func runStats(_ opts: Options, ctx: DatasetContext) -> Bool {
|
|
let gpu = ctx.gpu
|
|
let log2 = opts.datasetLog2
|
|
let ds = ctx.source(log2: log2)
|
|
let n = 1 << 20
|
|
print("\n=== output statistics, 2^20 consecutive nonces per seed, dataset 2^\(log2) words ===")
|
|
print("This is a sanity check for obvious structural bias. It is not a proof of cryptographic strength.")
|
|
let dataset = ctx.makeDataset(log2: log2)
|
|
guard let outBuf = gpu.device.makeBuffer(length: n * 8, options: .storageModeShared) else { print("FAIL: out buffer"); return false }
|
|
let seeds = [opts.seed, "\(opts.seed)/stats1", "\(opts.seed)/stats2"]
|
|
var allOk = true
|
|
var rows = [String]()
|
|
for seedString in seeds {
|
|
let program = generateProgram(seedString: seedString)
|
|
let k: CompiledHash
|
|
do { k = try compileHash(gpu, msl: generateMSL(program, datasetLog2: log2)) } catch { print("FAIL: compile \(error)"); return false }
|
|
memset(outBuf.contents(), 0, n * 8)
|
|
guard let gms = gpuRange(gpu, k, dataset: dataset, base: 0, count: n, out: outBuf) else { print("FAIL: GPU run"); return false }
|
|
let p = outBuf.contents().bindMemory(to: UInt64.self, capacity: n)
|
|
let outs = (0..<n).map { p[$0] }
|
|
// Spot check 2 warps against the CPU so the statistics are known to describe the verified function.
|
|
var spot = true
|
|
for w in [0, (n / 32) - 1] {
|
|
let cpu = cpuWarp(program, baseNonce: UInt32(w * 32), ds: ds)
|
|
if cpu != Array(outs[(w * 32)..<(w * 32 + 32)]) { spot = false }
|
|
}
|
|
|
|
// (a) bit frequency per output bit position
|
|
var ones = [Int](repeating: 0, count: 64)
|
|
for v in outs { var x = v; var b = 0; while x != 0 { if x & 1 == 1 { ones[b] += 1 }; x >>= 1; b += 1 } }
|
|
let expected = Double(n) / 2, sigma = (Double(n) * 0.25).squareRoot()
|
|
var maxDev = 0.0, maxBit = 0
|
|
for b in 0..<64 { let d = abs(Double(ones[b]) - expected); if d > maxDev { maxDev = d; maxBit = b } }
|
|
let maxZ = maxDev / sigma
|
|
let minFreq = Double(ones.min()!) / Double(n), maxFreq = Double(ones.max()!) / Double(n)
|
|
|
|
// (c) chi-square over 65536 buckets for each 16-bit window of the output
|
|
var chiRows = [String]()
|
|
var chiWorstZ = 0.0
|
|
for shift in [0, 16, 32, 48] {
|
|
var buckets = [Int](repeating: 0, count: 65536)
|
|
for v in outs { buckets[Int((v >> UInt64(shift)) & 0xFFFF)] += 1 }
|
|
let e = Double(n) / 65536
|
|
var chi = 0.0
|
|
for c in buckets { let d = Double(c) - e; chi += d * d / e }
|
|
let df = 65535.0
|
|
let z = (chi - df) / (2 * df).squareRoot()
|
|
chiWorstZ = max(chiWorstZ, abs(z))
|
|
chiRows.append("bits \(shift)..\(shift + 15): chi2 \(fmt(chi, 0)) (df 65535, z \(fmt(z, 2)))")
|
|
}
|
|
|
|
// (d) duplicates
|
|
let sorted = outs.sorted()
|
|
var dups = 0
|
|
for i in 1..<n where sorted[i] == sorted[i - 1] { dups += 1 }
|
|
|
|
// (b) avalanche: random nonces, flip bit (t mod 32), count changed output bits. Run on the GPU.
|
|
// The brief asks for 1,000 trials; a 16,000-trial pass (500 per input bit) is added because the
|
|
// standard error of the mean at 1,000 trials is 0.13 bits, too coarse to see a small bias.
|
|
func avalanche(_ trials: Int, salt: String) -> (mean: Double, std: Double, minD: Int, maxD: Int, perBitMin: Double, perBitMax: Double, perOutMin: Double, perOutMax: Double)? {
|
|
let sw = seedWords("avalanche/\(salt)/" + seedString)
|
|
var rng = SplitMix64(s: UInt64(sw[0]) | (UInt64(sw[1]) << 32))
|
|
var bases = [UInt32]()
|
|
var flipped = [Int]()
|
|
for t in 0..<trials {
|
|
let nonce = UInt32(truncatingIfNeeded: rng.next())
|
|
let bit = t % 32
|
|
bases.append(nonce); bases.append(nonce ^ (1 << UInt32(bit)))
|
|
flipped.append(bit)
|
|
}
|
|
guard let av = gpuWarps(gpu, k, dataset: dataset, bases: bases) else { return nil }
|
|
var diffs = [Int]()
|
|
var perBitSum = [Int](repeating: 0, count: 32), perBitN = [Int](repeating: 0, count: 32)
|
|
var perOut = [Int](repeating: 0, count: 64)
|
|
var minDiff = 64, maxDiff = 0
|
|
for t in 0..<trials {
|
|
let x = av[2 * t][0] ^ av[2 * t + 1][0]
|
|
let d = popcount64(x)
|
|
diffs.append(d)
|
|
perBitSum[flipped[t]] += d; perBitN[flipped[t]] += 1
|
|
for b in 0..<64 where (x >> UInt64(b)) & 1 == 1 { perOut[b] += 1 }
|
|
minDiff = min(minDiff, d); maxDiff = max(maxDiff, d)
|
|
}
|
|
let mean = Double(diffs.reduce(0, +)) / Double(trials)
|
|
let variance = diffs.reduce(0.0) { $0 + (Double($1) - mean) * (Double($1) - mean) } / Double(trials - 1)
|
|
var pbMin = 64.0, pbMax = 0.0
|
|
for b in 0..<32 where perBitN[b] > 0 { let m = Double(perBitSum[b]) / Double(perBitN[b]); pbMin = min(pbMin, m); pbMax = max(pbMax, m) }
|
|
let poMin = Double(perOut.min()!) / Double(trials), poMax = Double(perOut.max()!) / Double(trials)
|
|
return (mean, variance.squareRoot(), minDiff, maxDiff, pbMin, pbMax, poMin, poMax)
|
|
}
|
|
guard let a1 = avalanche(1000, salt: "small"), let a2 = avalanche(16000, salt: "large") else { print("FAIL: avalanche GPU run"); return false }
|
|
// Expected for an ideal function: mean 32, std 4 (binomial 64 x 0.5). Standard error of the mean:
|
|
// 0.13 bits at 1,000 trials, 0.03 bits at 16,000. Thresholds are about 4.5 standard errors.
|
|
let avOk = abs(a1.mean - 32) < 0.6 && a1.std > 3.3 && a1.std < 4.7 && abs(a2.mean - 32) < 0.15 && a2.std > 3.6 && a2.std < 4.4
|
|
let freqOk = maxZ < 4.5
|
|
let chiOk = chiWorstZ < 4.5
|
|
let dupOk = dups == 0
|
|
let ok = avOk && freqOk && chiOk && dupOk && spot
|
|
if !ok { allOk = false }
|
|
|
|
print("\nseed \"\(seedString)\": loads/hash \(program.loadsPerHash), GPU \(fmt(gms, 1)) ms for 2^20 hashes, CPU spot check 2 warps \(spot ? "PASS" : "FAIL")")
|
|
print(" (a) bit frequency: min \(fmt(minFreq, 4)) max \(fmt(maxFreq, 4)); largest deviation \(fmt(maxDev, 0)) counts at bit \(maxBit) = \(fmt(maxZ, 2)) sigma (sigma \(fmt(sigma, 0)), 64 bits, expect max under about 3.5)")
|
|
print(" (b) avalanche, 1000 single-bit nonce flips: mean \(fmt(a1.mean, 2)) std \(fmt(a1.std, 2)) min \(a1.minD) max \(a1.maxD) of 64 bits (expect mean 32, std 4); per-input-bit mean range \(fmt(a1.perBitMin, 1))..\(fmt(a1.perBitMax, 1))")
|
|
print(" (b) avalanche, 16000 flips (500 per input bit): mean \(fmt(a2.mean, 3)) std \(fmt(a2.std, 2)) min \(a2.minD) max \(a2.maxD); per-input-bit mean range \(fmt(a2.perBitMin, 2))..\(fmt(a2.perBitMax, 2)); per-output-bit flip probability range \(fmt(a2.perOutMin, 3))..\(fmt(a2.perOutMax, 3)) (expect 0.5, sigma 0.004)")
|
|
for r in chiRows { print(" (c) \(r)") }
|
|
print(" (d) duplicate 64-bit outputs among 2^20: \(dups) (expected about 3e-8)")
|
|
print(" verdict: \(ok ? "no obvious bias" : "SUSPECT")")
|
|
rows.append("| \(seedString) | \(program.loadsPerHash) | \(fmt(minFreq, 4))..\(fmt(maxFreq, 4)) | \(fmt(maxZ, 2)) | \(fmt(a1.mean, 2)) / \(fmt(a1.std, 2)) | \(fmt(a2.mean, 3)) / \(fmt(a2.std, 2)) | \(fmt(a2.perOutMin, 3))..\(fmt(a2.perOutMax, 3)) | \(fmt(chiWorstZ, 2)) | \(dups) | \(ok ? "uniform-looking" : "SUSPECT") |")
|
|
}
|
|
print("\n| Seed | Loads/hash | Bit freq min..max | Max bit z | Avalanche 1k mean / std | Avalanche 16k mean / std | Per-output-bit flip prob | Worst chi2 z (4 windows) | Dups | Verdict |")
|
|
print("|---|---|---|---|---|---|---|---|---|---|")
|
|
for r in rows { print(r) }
|
|
print("STATS: \(allOk ? "PASS (no obvious structural bias; not a security proof)" : "FAIL (something looks biased)")")
|
|
return allOk
|
|
}
|
|
|
|
// MARK: - --determinism
|
|
|
|
func runDeterminism(_ opts: Options, ctx: DatasetContext) -> Bool {
|
|
let gpu = ctx.gpu
|
|
let log2 = opts.datasetLog2
|
|
let mask = UInt32((1 << log2) - 1)
|
|
let ds = ctx.source(log2: log2)
|
|
let n = 1 << 20
|
|
print("\n=== determinism, seed \"\(opts.seed)\", 2^20 nonces from base 0, dataset 2^\(log2) words ===")
|
|
let dataset = ctx.makeDataset(log2: log2)
|
|
guard let outBuf = gpu.device.makeBuffer(length: n * 8, options: .storageModeShared) else { print("FAIL: out buffer"); return false }
|
|
var ok = true
|
|
|
|
// Generator and emitter determinism: two independent generations give identical MSL text.
|
|
let p1 = generateProgram(seedString: opts.seed), p2 = generateProgram(seedString: opts.seed)
|
|
let msl1 = generateMSL(p1, datasetLog2: log2), msl2 = generateMSL(p2, datasetLog2: log2)
|
|
let sameSource = msl1 == msl2
|
|
print("generator: two generations of the program give identical MSL source: \(sameSource ? "yes" : "NO") (\(msl1.utf8.count) bytes)")
|
|
if !sameSource { ok = false }
|
|
|
|
// Three compiles: the same source twice (Metal's shader cache may serve the second), and once more with a
|
|
// comment tag appended so the cache misses and the compiler really runs again on an identical kernel.
|
|
let tag = "\n// recompile tag \(h64(nowNs()))\n"
|
|
let k1: CompiledHash, k2: CompiledHash, k3: CompiledHash
|
|
do {
|
|
k1 = try compileHash(gpu, msl: msl1); k2 = try compileHash(gpu, msl: msl2); k3 = try compileHash(gpu, msl: msl2 + tag)
|
|
} catch { print("FAIL: compile \(error)"); return false }
|
|
print("compiled three times: identical source \(fmt(k1.totalMs, 1)) ms and \(fmt(k2.totalMs, 1)) ms (a sub-millisecond second figure means the system shader cache answered), tagged source \(fmt(k3.totalMs, 1)) ms (forced recompile)")
|
|
|
|
func runOnce(_ k: CompiledHash) -> (fp: UInt64, sentinels: Int, ms: Double)? {
|
|
memset(outBuf.contents(), 0xAA, n * 8)
|
|
guard let gms = gpuRange(gpu, k, dataset: dataset, base: 0, count: n, out: outBuf) else { return nil }
|
|
let p = outBuf.contents().bindMemory(to: UInt64.self, capacity: n)
|
|
var s = 0
|
|
for i in 0..<n where p[i] == 0xAAAAAAAAAAAAAAAA { s += 1 }
|
|
return (fnv64(outBuf.contents(), n * 8), s, gms)
|
|
}
|
|
var reference = [UInt64]()
|
|
var fps = [UInt64]()
|
|
for run in 0..<5 {
|
|
guard let r = runOnce(k1) else { print("FAIL: GPU run \(run)"); return false }
|
|
let p = outBuf.contents().bindMemory(to: UInt64.self, capacity: n)
|
|
if run == 0 { reference = (0..<n).map { p[$0] } }
|
|
var differ = 0
|
|
for i in 0..<n where p[i] != reference[i] { differ += 1 }
|
|
fps.append(r.fp)
|
|
let same = differ == 0 && r.sentinels == 0
|
|
if !same { ok = false }
|
|
print("run \(run + 1)/5 (compile 1): fingerprint \(h64(r.fp)), \(differ) of \(n) outputs differ from run 1, \(r.sentinels) unwritten lanes, GPU \(fmt(r.ms, 1)) ms: \(same ? "identical" : "DIFFERENT")")
|
|
}
|
|
guard let r2 = runOnce(k2) else { print("FAIL: GPU run on compile 2"); return false }
|
|
let p = outBuf.contents().bindMemory(to: UInt64.self, capacity: n)
|
|
var differ2 = 0
|
|
for i in 0..<n where p[i] != reference[i] { differ2 += 1 }
|
|
if differ2 != 0 || r2.sentinels != 0 { ok = false }
|
|
print("run on compile 2 (identical source): fingerprint \(h64(r2.fp)), \(differ2) outputs differ from compile 1 run 1, \(r2.sentinels) unwritten lanes: \(differ2 == 0 ? "identical" : "DIFFERENT")")
|
|
guard let r3 = runOnce(k3) else { print("FAIL: GPU run on compile 3"); return false }
|
|
var differ3 = 0
|
|
for i in 0..<n where p[i] != reference[i] { differ3 += 1 }
|
|
if differ3 != 0 || r3.sentinels != 0 { ok = false }
|
|
print("run on compile 3 (forced recompile): fingerprint \(h64(r3.fp)), \(differ3) outputs differ from compile 1 run 1, \(r3.sentinels) unwritten lanes: \(differ3 == 0 ? "identical" : "DIFFERENT")")
|
|
|
|
// CPU reference on the first and last warp and 6 others, so the fingerprint is tied to the verified function.
|
|
var cpuBad = 0
|
|
var vr = SplitMix64(s: 0x1234_5678_9abc_def0)
|
|
var warps = [0, n / 32 - 1]
|
|
while warps.count < 8 { warps.append(vr.below(n / 32)) }
|
|
for w in warps {
|
|
let cpu = cpuWarp(p1, baseNonce: UInt32(w * 32), ds: ds)
|
|
if cpu != Array(reference[(w * 32)..<(w * 32 + 32)]) { cpuBad += 1 }
|
|
}
|
|
if cpuBad != 0 { ok = false }
|
|
print("CPU interpreter on \(warps.count) warps of the reference run: \(cpuBad == 0 ? "all match" : "\(cpuBad) MISMATCH")")
|
|
|
|
// Dataset fill determinism and GPU-vs-CPU agreement of the dataset itself: fill a second buffer, blit both
|
|
// to shared memory, fingerprint, and compare sampled words (including 0 and MASK) with datasetElem.
|
|
let words = 1 << log2
|
|
let dataset2 = ctx.makeDataset(log2: log2)
|
|
var fillFps = [UInt64]()
|
|
var sampleBad = 0
|
|
if let shared = gpu.device.makeBuffer(length: words * 4, options: .storageModeShared) {
|
|
for (idx, dsBuf) in [dataset, dataset2].enumerated() {
|
|
let cb = gpu.queue.makeCommandBuffer()!
|
|
let blit = cb.makeBlitCommandEncoder()!
|
|
blit.copy(from: dsBuf, sourceOffset: 0, to: shared, destinationOffset: 0, size: words * 4)
|
|
blit.endEncoding()
|
|
cb.commit(); cb.waitUntilCompleted()
|
|
fillFps.append(fnv64(shared.contents(), words * 4))
|
|
if idx == 0 {
|
|
let dp = shared.contents().bindMemory(to: UInt32.self, capacity: words)
|
|
var sr = SplitMix64(s: 0xfeed_beef)
|
|
var idxs: [UInt32] = [0, 1, mask - 1, mask]
|
|
while idxs.count < 4096 { idxs.append(UInt32(sr.below(words))) }
|
|
for i in idxs where dp[Int(i)] != ds.word(i) { sampleBad += 1 }
|
|
}
|
|
}
|
|
let sameFill = fillFps[0] == fillFps[1]
|
|
if !sameFill || sampleBad != 0 { ok = false }
|
|
print("dataset fill: two fills fingerprint \(h64(fillFps[0])) and \(h64(fillFps[1])): \(sameFill ? "identical" : "DIFFERENT"); 4096 sampled words (incl. 0, 1, MASK-1, MASK) vs CPU dataset word (\(ds.modeName)): \(sampleBad == 0 ? "all match" : "\(sampleBad) MISMATCH")")
|
|
} else {
|
|
print("dataset fill check skipped: could not allocate a shared copy")
|
|
}
|
|
print("DETERMINISM: \(ok ? "PASS" : "FAIL")")
|
|
return ok
|
|
}
|
|
|
|
// MARK: - --memcheck
|
|
|
|
func runMemcheck(_ opts: Options, ctx: DatasetContext) -> Bool {
|
|
let gpu = ctx.gpu
|
|
print("\n=== memcheck, seed \"\(opts.seed)\" ===")
|
|
var ok = true
|
|
let program = generateProgram(seedString: opts.seed)
|
|
// Static: every dataset index in the generated sources is masked. Checked at three dataset sizes because
|
|
// the MASK literal changes with size.
|
|
for log2 in [20, 24, 28] {
|
|
let msl = generateMSL(program, datasetLog2: log2)
|
|
let r = maskCheckMSL(msl)
|
|
if !r.ok { ok = false }
|
|
print("static 2^\(log2): \(r.ok ? "PASS" : "FAIL") \(r.detail)")
|
|
}
|
|
let cu = maskCheckCUDA(generateCUDA(program, memhard: ctx.mp), memhard: !ctx.closed)
|
|
if !cu.ok { ok = false }
|
|
print("static CUDA twin: \(cu.ok ? "PASS" : "FAIL") \(cu.detail)")
|
|
print("program has \(program.instrs.filter { $0.op == .load }.count) load instructions (\(program.loadsPerHash) loads/hash)")
|
|
|
|
// Dynamic: a 4 MiB dataset with nonces at the top of the 32-bit range (they wrap to 0 inside the batch),
|
|
// one full batch of 2^20 nonces plus 4 warps verified against the CPU. Metal does not bounds-check device
|
|
// buffers, so "no crash" is weak evidence by itself; the static check above is the real guarantee.
|
|
let log2 = 20
|
|
let mask = UInt32((1 << log2) - 1)
|
|
let ds = ctx.source(log2: log2)
|
|
let dataset = ctx.makeDataset(log2: log2)
|
|
let k: CompiledHash
|
|
do { k = try compileHash(gpu, msl: generateMSL(program, datasetLog2: log2)) } catch { print("FAIL: compile \(error)"); return false }
|
|
let n = 1 << 20
|
|
guard let outBuf = gpu.device.makeBuffer(length: n * 8, options: .storageModeShared) else { print("FAIL: out buffer"); return false }
|
|
for base: UInt32 in [0xFFF00000, 0xFFFFFFE0, 0x80000000, 0] {
|
|
if let gms = gpuRange(gpu, k, dataset: dataset, base: base, count: n, out: outBuf) {
|
|
print("dynamic 4 MiB: 2^20 nonces from base \(hex(base)) (last nonce \(hex(base &+ UInt32(n - 1)))): completed, GPU \(fmt(gms, 1)) ms")
|
|
} else { ok = false; print("dynamic 4 MiB: base \(hex(base)): GPU ERROR") }
|
|
}
|
|
let bases: [UInt32] = [0xFFFFFFE0, 0xFFFFFFFF, 0x80000000, 0xFFF00000]
|
|
guard let gpuOut = gpuWarps(gpu, k, dataset: dataset, bases: bases) else { print("FAIL: GPU warps"); return false }
|
|
var overMask = 0, loads = 0
|
|
for (w, base) in bases.enumerated() {
|
|
let cpu = cpuWarpTraced(program, baseNonce: base, ds: ds) { _, kk, regs in
|
|
let ins = program.instrs[kk]
|
|
if ins.op == .load { loads += 1; if regs[ins.a] > mask { overMask += 1 } }
|
|
}
|
|
let match = cpu == gpuOut[w]
|
|
if !match { ok = false }
|
|
print("verify warp base \(hex(base)): GPU vs CPU \(match ? "PASS" : "FAIL")")
|
|
}
|
|
print("in those 4 warps (lane 0, all iterations) \(overMask) of \(loads) load indices were above MASK before masking, so the mask was exercised")
|
|
print("MEMCHECK: \(ok ? "PASS" : "FAIL")")
|
|
return ok
|
|
}
|
|
|
|
// MARK: - Test dispatcher
|
|
|
|
func runTests(_ opts: Options) -> Never {
|
|
let gpu = GPU()
|
|
print("igneum-bench hardening tests")
|
|
print("GPU: \(gpu.device.name) (maxBufferLength \(gpu.device.maxBufferLength / (1 << 20)) MiB, unified memory \(gpu.device.hasUnifiedMemory)), day \"\(opts.day)\"")
|
|
let ctx = DatasetContext(gpu: gpu, closedForm: opts.closedForm, dayString: opts.day)
|
|
print("dataset construction: \(ctx.modeName)" + (ctx.closed ? "" : "; cache GPU fill \(fmt(ctx.cacheFillGPUms, 2)) ms GPU time"))
|
|
if generatorConfig.loadWeight != 25 || generatorConfig.wideFrac != 0 {
|
|
print("generator levers: load weight \(generatorConfig.loadWeight), wide fraction \(generatorConfig.wideFrac) percent (NOT the default generator)")
|
|
}
|
|
var results = [(String, Bool)]()
|
|
let t0 = nowNs()
|
|
if !ctx.closed {
|
|
let cc = ctx.cacheCheck()
|
|
print(cc.detail)
|
|
results.append(("cache", cc.ok))
|
|
}
|
|
if opts.fuzz != nil { results.append(("fuzz", runFuzz(opts, ctx: ctx))) }
|
|
if opts.edge { results.append(("edge", runEdge(opts, ctx: ctx))) }
|
|
if opts.stats { results.append(("stats", runStats(opts, ctx: ctx))) }
|
|
if opts.determinism { results.append(("determinism", runDeterminism(opts, ctx: ctx))) }
|
|
if opts.memcheck { results.append(("memcheck", runMemcheck(opts, ctx: ctx))) }
|
|
print("\n=== tests summary (\(fmt(Double(nowNs() - t0) / 1e9, 1)) s) ===")
|
|
for (name, ok) in results { print("\(pad(name, 12)) \(ok ? "PASS" : "FAIL")") }
|
|
let all = results.allSatisfy { $0.1 }
|
|
print("OVERALL: \(all ? "PASS" : "FAIL")")
|
|
exit(all ? 0 : 1)
|
|
}
|
|
|
|
// MARK: - Main
|
|
|
|
let opts = parseArgs()
|
|
generatorConfig = GeneratorConfig(loadWeight: opts.loadWeight, wideFrac: opts.wideFrac)
|
|
if opts.exportPack != nil { exportPack(opts) }
|
|
if opts.anyTest { runTests(opts) }
|
|
let gpu = GPU()
|
|
print("igneum-bench")
|
|
print("GPU: \(gpu.device.name) (maxBufferLength \(gpu.device.maxBufferLength / (1 << 20)) MiB, unified memory \(gpu.device.hasUnifiedMemory))")
|
|
print("dataset: 2^\(opts.datasetLog2) uint32 = \(fmt(Double(1 << opts.datasetLog2) * 4 / Double(1 << 20), 0)) MiB, day \"\(opts.day)\", construction \(opts.closedForm ? "closed-form (--closed-form)" : "memory-hard (default; MEMHARD.md)")")
|
|
if generatorConfig.loadWeight != 25 || generatorConfig.wideFrac != 0 {
|
|
print("generator levers: load weight \(generatorConfig.loadWeight) percent, wide-load fraction \(generatorConfig.wideFrac) percent (NOT the default generator)")
|
|
print("generator weights: " + generatorConfig.weights.map { "\($0.0.rawValue)=\($0.1)" }.joined(separator: " "))
|
|
}
|
|
|
|
// Dataset construction, timed. Memory-hard: compile the cache fill + build kernels, fill the 256 MiB cache on the
|
|
// GPU, build the dataset from it. Closed form: the original fill kernel. The CPU cache (verifier side) is computed
|
|
// afterwards on one core and compared word for word with the GPU cache.
|
|
let ctx = DatasetContext(gpu: gpu, closedForm: opts.closedForm, dayString: opts.day)
|
|
let datasetWords = 1 << opts.datasetLog2
|
|
guard let dataset = gpu.device.makeBuffer(length: datasetWords * 4, options: .storageModePrivate) else {
|
|
print("FAIL: cannot allocate dataset buffer"); exit(1)
|
|
}
|
|
print("dataset kernels: compile \(fmt(ctx.compileMs)) ms")
|
|
var cacheRefillMs = 0.0
|
|
if !ctx.closed {
|
|
cacheRefillMs = ctx.refillCache()
|
|
print("cache fill (GPU): \(fmt(ctx.cacheFillGPUms)) ms GPU time first (\(fmt(ctx.cacheFillWallMs)) ms wall), \(fmt(cacheRefillMs)) ms GPU time second; \(cacheSegments) chains x \(cacheLinesPerSegment) ChaCha\(chachaRounds) blocks, 256 MiB written")
|
|
}
|
|
ctx.build(into: dataset, words: datasetWords)
|
|
let build1GPU = ctx.lastBuildGPUms, build1Wall = ctx.lastBuildWallMs
|
|
ctx.build(into: dataset, words: datasetWords)
|
|
let build2GPU = ctx.lastBuildGPUms
|
|
let gib = Double(datasetWords * 4) / Double(1 << 30)
|
|
if ctx.closed {
|
|
print("dataset fill (closed form): \(fmt(build1GPU)) ms GPU time first (\(fmt(build1Wall)) ms wall), \(fmt(build2GPU)) ms second -> \(fmt(gib / (build2GPU / 1000))) GB/s write (second, GPU time)")
|
|
} else {
|
|
let items = Double(datasetWords / 16)
|
|
print("dataset build (memory-hard): \(fmt(build1GPU)) ms GPU time first (\(fmt(build1Wall)) ms wall), \(fmt(build2GPU)) ms second; \(Int(items)) items, \(fmt(items / (build2GPU / 1000) / 1e6, 1)) M items/s, \(fmt(items * 8 / (build2GPU / 1000) / 1e9, 2)) G cache-line reads/s (second)")
|
|
let cc = ctx.cacheCheck()
|
|
print(cc.detail)
|
|
if !cc.ok { print("FAIL: GPU and CPU cache differ"); exit(1) }
|
|
let sc = ctx.sampleCheck(log2: min(opts.datasetLog2, 24), indices: [0, 1, 15, 16, UInt32((1 << min(opts.datasetLog2, 24)) - 1)] + (0..<1019).map { _ in UInt32.random(in: 0..<UInt32(1 << min(opts.datasetLog2, 24))) })
|
|
print(sc.detail)
|
|
if !sc.ok { print("FAIL: GPU dataset words differ from the CPU derivation"); exit(1) }
|
|
}
|
|
|
|
var results = [EpochResult]()
|
|
for epoch in 0..<max(opts.hours, 1) {
|
|
let seedString = epoch == 0 ? opts.seed : "\(opts.seed)/epoch\(epoch)"
|
|
results.append(runEpoch(gpu: gpu, opts: opts, seedString: seedString, dataset: dataset, ctx: ctx))
|
|
}
|
|
|
|
// Summary table
|
|
print("\n=== summary (\(gpu.device.name), dataset 2^\(opts.datasetLog2) words \(ctx.modeName)\(opts.inlineDataset ? " INLINE shortcut kernel" : ""), batch 2^\(opts.batchLog2) x \(opts.batches)) ===")
|
|
print("| seed | compile ms (lib+pipe) | Mhash/s (wall) | Mhash/s (GPU) | GB/s useful (wall) | loads/hash | items/warp | CPU verify ms/warp (avg of 20) | verify |")
|
|
print("|---|---|---|---|---|---|---|---|---|")
|
|
for r in results {
|
|
let avg = r.verify.map { $0.repMs }.reduce(0, +) / Double(max(r.verify.count, 1))
|
|
print("| \(r.seed) | \(fmt(r.libraryMs + r.pipelineMs, 1)) | \(fmt(r.hashesPerSecWall / 1e6, 3)) | \(fmt(r.hashesPerSecGPU / 1e6, 3)) | \(fmt(r.gbpsWall)) | \(r.loadsPerHash) | \(r.itemsPerWarp) | \(fmt(avg, 3)) | \(r.allPass ? "PASS" : "FAIL") (\(r.verify.count) warps) |")
|
|
}
|
|
let overall = results.allSatisfy { $0.allPass }
|
|
if ctx.closed {
|
|
print("dataset fill: \(fmt(build2GPU)) ms GPU time for \(datasetWords * 4 / (1 << 20)) MiB (closed form)")
|
|
} else {
|
|
print("cache fill: \(fmt(cacheRefillMs)) ms GPU, \(fmt(ctx.cpuSide()!.fillMs, 1)) ms one CPU core; dataset build: \(fmt(build2GPU)) ms GPU for \(datasetWords * 4 / (1 << 20)) MiB (memory-hard)")
|
|
}
|
|
print("OVERALL: \(overall ? "PASS" : "FAIL")")
|
|
exit(overall ? 0 : 1)
|