# Conflicts: # docs/bench-log.md # docs/plans/read-width.md # igneum-pow/src/emit.rs # proto-metal/packbench.swift
248 lines
16 KiB
Swift
248 lines
16 KiB
Swift
// packbench: runs a program pack (igneum-pow export) on the Mac's Metal GPU from its files alone: memhard.metal (cache
|
|
// fill, dataset build), program.metal (igneum_hash), program.h (constants), vectors.json (the CPU reference's vectors).
|
|
// Read-width experiment, 5 October 2026 (docs/plans/read-width.md): the Swift bench generates its own programs and does
|
|
// not know the experiment's load classes; this harness runs whatever text the Rust emitter wrote, so Metal is checked
|
|
// against the Rust CPU reference and timed without a Swift mirror of the generator. One file, no packages.
|
|
//
|
|
// swiftc -O -target arm64-apple-macos11 -o packbench packbench.swift -framework Metal
|
|
// ./packbench --pack <dir> [--batches 5] [--batch-log2 24] [--group 256] [--warps 2048] [--batch-base 0]
|
|
//
|
|
// Prints one RESULT line per run: vectors, cache and dataset checks, the batch fingerprint (FNV-1a 64 over the 2^B
|
|
// outputs at base nonce 0) and MH/s by wall and by GPU time. Variant 5 packs (IGNEUM_PERSISTENT_WARPS) are launched
|
|
// as --warps persistent warps with a 1 MiB scratch each; the batch is rounded to a multiple of 32 x warps.
|
|
import Foundation
|
|
import Metal
|
|
|
|
func nowMs() -> Double { return Double(DispatchTime.now().uptimeNanoseconds) / 1e6 }
|
|
func fail(_ m: String) -> Never { print("FAIL: \(m)"); exit(1) }
|
|
|
|
struct Opts { var pack = ""; var batches = 5; var batchLog2 = 24; var group = 256; var warps = 2048; var batchBase: UInt32 = 0 }
|
|
var opts = Opts()
|
|
var args = Array(CommandLine.arguments.dropFirst())
|
|
while !args.isEmpty {
|
|
let a = args.removeFirst()
|
|
func next() -> String { if args.isEmpty { fail("missing value for \(a)") }; return args.removeFirst() }
|
|
switch a {
|
|
case "--pack": opts.pack = next()
|
|
case "--batches": opts.batches = Int(next())!
|
|
case "--batch-log2": opts.batchLog2 = Int(next())!
|
|
case "--group": opts.group = Int(next())!
|
|
case "--warps": opts.warps = Int(next())!
|
|
case "--batch-base": opts.batchBase = UInt32(next())! // base nonce of the fingerprint batch (default 0; a base near 2^32 makes the persistent unit sequence wrap inside the launch)
|
|
default: fail("unknown argument \(a)")
|
|
}
|
|
}
|
|
if opts.pack.isEmpty { fail("--pack <dir> is required") }
|
|
|
|
func readText(_ name: String) -> String {
|
|
guard let s = try? String(contentsOfFile: opts.pack + "/" + name, encoding: .utf8) else { fail("cannot read \(opts.pack)/\(name)") }
|
|
return s
|
|
}
|
|
let programH = readText("program.h")
|
|
func defineU32(_ name: String) -> UInt32? {
|
|
let pat = "#define \(name) ([0-9a-fA-Fx]+)"
|
|
guard let re = try? NSRegularExpression(pattern: pat), let m = re.firstMatch(in: programH, range: NSRange(programH.startIndex..., in: programH)) else { return nil }
|
|
let v = String(programH[Range(m.range(at: 1), in: programH)!]).replacingOccurrences(of: "u", with: "")
|
|
if v.hasPrefix("0x") { return UInt32(v.dropFirst(2), radix: 16) }
|
|
return UInt32(v)
|
|
}
|
|
func defineStr(_ name: String) -> String? {
|
|
let pat = "#define \(name) \"([^\"]*)\""
|
|
guard let re = try? NSRegularExpression(pattern: pat), let m = re.firstMatch(in: programH, range: NSRange(programH.startIndex..., in: programH)) else { return nil }
|
|
return String(programH[Range(m.range(at: 1), in: programH)!])
|
|
}
|
|
let datasetLog2 = Int(defineU32("IGNEUM_DATASET_LOG2") ?? 28)
|
|
let datasetMode = defineU32("IGNEUM_DATASET_MODE") ?? 1
|
|
if datasetMode != 1 { fail("packbench runs memory-hard packs only") }
|
|
let cacheLog2 = Int(defineU32("IGNEUM_CACHE_LOG2_WORDS") ?? 26)
|
|
let cacheSegments = Int(defineU32("IGNEUM_CACHE_SEGMENTS") ?? 65536)
|
|
let loadsPerHash = Int(defineU32("IGNEUM_LOADS_PER_HASH") ?? 128)
|
|
let bytesPerHash = Int(defineU32("IGNEUM_BYTES_PER_HASH") ?? UInt32(loadsPerHash * 4))
|
|
let scratchOps = Int(defineU32("IGNEUM_SCRATCH_OPS") ?? 0)
|
|
let persistent = (defineU32("IGNEUM_PERSISTENT_WARPS") ?? 0) == 1
|
|
let scratchWordsPerLane = Int(defineU32("IGNEUM_SCRATCH_WORDS_PER_LANE") ?? 8192)
|
|
let className = defineStr("IGNEUM_LOAD_CLASS") ?? "v2"
|
|
// Counter ASIC 2.0 (5 October 2026): the mixer multiplier of the item derivation, 1 when absent (version 2), 4 under class v3
|
|
let mixerMult = Int(defineU32("IGNEUM_MIXER_MULT") ?? 1)
|
|
// hot-table experiment (docs/plans/hot-table.md): the epoch table, filled on the device from memhard.metal's igneum_hot_fill
|
|
let hotMb = Int(defineU32("IGNEUM_HOT_MB") ?? 0)
|
|
let hotWords = Int(defineU32("IGNEUM_HOT_WORDS") ?? 0)
|
|
let hotSegments = Int(defineU32("IGNEUM_HOT_SEGMENTS") ?? 0)
|
|
let hotSlots = Int(defineU32("IGNEUM_HOT_SLOTS") ?? 0)
|
|
if hotMb > 0 && (hotWords != hotMb << 18 || hotSegments != hotMb * 256) { fail("program.h hot table sizes disagree") }
|
|
let seedString = defineStr("IGNEUM_SEED_STRING") ?? "?"
|
|
let programId = defineStr("IGNEUM_PROGRAM_ID") ?? ""
|
|
|
|
// vectors.json: bases, expected outputs, dataset head / last, cache fingerprint
|
|
let vj = try! JSONSerialization.jsonObject(with: Data(contentsOf: URL(fileURLWithPath: opts.pack + "/vectors.json"))) as! [String: Any]
|
|
func hex64(_ s: String) -> UInt64 { return UInt64(s.dropFirst(2), radix: 16)! }
|
|
func hex32(_ s: String) -> UInt32 { return UInt32(s.dropFirst(2), radix: 16)! }
|
|
let warpsJ = vj["warps"] as! [[String: Any]]
|
|
let vecBases = warpsJ.map { UInt32(($0["base_nonce"] as! NSNumber).uint64Value) }
|
|
let vecOuts = warpsJ.map { ($0["expected"] as! [String]).map(hex64) }
|
|
let dsHead = (vj["dataset_head"] as! [String]).map(hex32)
|
|
let dsLastIndex = UInt32((vj["dataset_last_index"] as! NSNumber).uint64Value)
|
|
let dsLast = hex32(vj["dataset_last"] as! String)
|
|
let cacheFnvWant = hex64(vj["cache_fnv1a64"] as! String)
|
|
let hotFnvWant: UInt64? = (vj["hot_fnv1a64"] as? String).map(hex64)
|
|
|
|
guard let device = MTLCreateSystemDefaultDevice(), let queue = device.makeCommandQueue() else { fail("no Metal device") }
|
|
let words = 1 << datasetLog2
|
|
let mask = UInt32(words - 1)
|
|
let cacheWords = 1 << cacheLog2
|
|
|
|
func compile(_ file: String) -> MTLLibrary {
|
|
do { return try device.makeLibrary(source: readText(file), options: MTLCompileOptions()) } catch { fail("Metal compile of \(file): \(error)") }
|
|
}
|
|
let t0 = nowMs()
|
|
let mhLib = compile("memhard.metal")
|
|
let progLib = compile("program.metal")
|
|
guard let fillFn = mhLib.makeFunction(name: "igneum_cache_fill"), let buildFn = mhLib.makeFunction(name: "igneum_build"), let hashFn = progLib.makeFunction(name: "igneum_hash") else { fail("kernel functions missing") }
|
|
let fillPipe = try! device.makeComputePipelineState(function: fillFn)
|
|
let buildPipe = try! device.makeComputePipelineState(function: buildFn)
|
|
let hashPipe = try! device.makeComputePipelineState(function: hashFn)
|
|
var hotPipe: MTLComputePipelineState? = nil
|
|
if hotMb > 0 {
|
|
guard let hotFn = mhLib.makeFunction(name: "igneum_hot_fill") else { fail("program.h says IGNEUM_HOT_MB \(hotMb) but memhard.metal has no igneum_hot_fill") }
|
|
hotPipe = try! device.makeComputePipelineState(function: hotFn)
|
|
}
|
|
let compileMs = nowMs() - t0
|
|
if hashPipe.threadExecutionWidth != 32 { print("WARNING: threadExecutionWidth \(hashPipe.threadExecutionWidth), not 32") }
|
|
|
|
guard let cache = device.makeBuffer(length: cacheWords * 4, options: .storageModePrivate) else { fail("cache alloc") }
|
|
guard let dataset = device.makeBuffer(length: words * 4, options: .storageModePrivate) else { fail("dataset alloc") }
|
|
|
|
func run(_ body: (MTLComputeCommandEncoder) -> Void) -> (Double, Double) {
|
|
let cb = queue.makeCommandBuffer()!
|
|
let enc = cb.makeComputeCommandEncoder()!
|
|
body(enc)
|
|
enc.endEncoding()
|
|
let w0 = nowMs()
|
|
cb.commit(); cb.waitUntilCompleted()
|
|
if let e = cb.error { fail("command buffer: \(e)") }
|
|
return (nowMs() - w0, (cb.gpuEndTime - cb.gpuStartTime) * 1000)
|
|
}
|
|
let (cacheWall, cacheGpu) = run { enc in
|
|
enc.setComputePipelineState(fillPipe); enc.setBuffer(cache, offset: 0, index: 0)
|
|
enc.dispatchThreadgroups(MTLSize(width: cacheSegments / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
|
|
}
|
|
let items = words / 16
|
|
let (buildWall, buildGpu) = run { enc in
|
|
enc.setComputePipelineState(buildPipe); enc.setBuffer(cache, offset: 0, index: 0); enc.setBuffer(dataset, offset: 0, index: 1)
|
|
enc.dispatchThreadgroups(MTLSize(width: items / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
|
|
}
|
|
// cache fingerprint and dataset head/last through a blit to shared memory
|
|
func blit(_ src: MTLBuffer, _ offset: Int, _ n: Int) -> MTLBuffer {
|
|
let dst = device.makeBuffer(length: n, options: .storageModeShared)!
|
|
let cb = queue.makeCommandBuffer()!; let b = cb.makeBlitCommandEncoder()!
|
|
b.copy(from: src, sourceOffset: offset, to: dst, destinationOffset: 0, size: n); b.endEncoding(); cb.commit(); cb.waitUntilCompleted()
|
|
return dst
|
|
}
|
|
func fnv1a64(_ p: UnsafeRawPointer, _ n: Int) -> UInt64 {
|
|
var h: UInt64 = 0xcbf29ce484222325
|
|
let b = p.bindMemory(to: UInt8.self, capacity: n)
|
|
for i in 0..<n { h ^= UInt64(b[i]); h = h &* 0x100000001b3 }
|
|
return h
|
|
}
|
|
let cacheCopy = blit(cache, 0, cacheWords * 4)
|
|
let cacheFnv = fnv1a64(cacheCopy.contents(), cacheWords * 4)
|
|
let cacheOk = cacheFnv == cacheFnvWant
|
|
// hot table: fill, then fingerprint against vectors.json
|
|
var hot: MTLBuffer? = nil
|
|
var hotGpu = 0.0, hotWall = 0.0
|
|
var hotOk = true
|
|
var hotFnv: UInt64 = 0
|
|
if hotMb > 0 {
|
|
guard let h = device.makeBuffer(length: hotWords * 4, options: .storageModePrivate) else { fail("hot table alloc of \(hotMb) MiB") }
|
|
hot = h
|
|
(hotWall, hotGpu) = run { enc in
|
|
enc.setComputePipelineState(hotPipe!); enc.setBuffer(h, offset: 0, index: 0)
|
|
enc.dispatchThreadgroups(MTLSize(width: hotSegments / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
|
|
}
|
|
let hotCopy = blit(h, 0, hotWords * 4)
|
|
hotFnv = fnv1a64(hotCopy.contents(), hotWords * 4)
|
|
if let want = hotFnvWant { hotOk = hotFnv == want }
|
|
}
|
|
let headCopy = blit(dataset, 0, 64)
|
|
let headPtr = headCopy.contents().bindMemory(to: UInt32.self, capacity: 16)
|
|
var dsOk = (0..<16).allSatisfy { headPtr[$0] == dsHead[$0] }
|
|
let lastCopy = blit(dataset, Int(dsLastIndex) * 4, 4)
|
|
dsOk = dsOk && lastCopy.contents().bindMemory(to: UInt32.self, capacity: 1)[0] == dsLast
|
|
|
|
// scratch arena (variant 5)
|
|
var scratch: MTLBuffer? = nil
|
|
var salt: UInt32 = 1
|
|
let warpsN = persistent ? opts.warps : 0
|
|
if persistent {
|
|
let bytes = warpsN * 32 * scratchWordsPerLane * 4
|
|
guard let s = device.makeBuffer(length: bytes, options: .storageModePrivate) else { fail("scratch alloc of \(bytes >> 20) MiB") }
|
|
scratch = s
|
|
}
|
|
// One hash launch: `nonces` outputs from `base`. Persistent: warpsN warps loop over nonces / 32 units.
|
|
func encodeHash(_ enc: MTLComputeCommandEncoder, out: MTLBuffer, base: UInt32, nonces: Int, group: Int) {
|
|
enc.setComputePipelineState(hashPipe)
|
|
enc.setBuffer(dataset, offset: 0, index: 0)
|
|
enc.setBuffer(out, offset: 0, index: 1)
|
|
var b = base; enc.setBytes(&b, length: 4, index: 2)
|
|
// the hot table is buffer 3 (the scratch triple moves up by one when both are present)
|
|
let next = hot == nil ? 3 : 4
|
|
if let h = hot { enc.setBuffer(h, offset: 0, index: 3) }
|
|
if persistent {
|
|
let units = nonces / 32
|
|
let nw = min(warpsN, units)
|
|
if units % nw != 0 { fail("nonces \(nonces) is not a multiple of 32 x \(nw) warps") }
|
|
enc.setBuffer(scratch!, offset: 0, index: next)
|
|
var g = UInt32(units); enc.setBytes(&g, length: 4, index: next + 1)
|
|
var s = salt; enc.setBytes(&s, length: 4, index: next + 2)
|
|
salt = salt &+ UInt32(units)
|
|
let threads = nw * 32
|
|
let tg = min(group, threads)
|
|
enc.dispatchThreadgroups(MTLSize(width: threads / tg, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: tg, height: 1, depth: 1))
|
|
} else {
|
|
let tg = min(group, nonces)
|
|
enc.dispatchThreadgroups(MTLSize(width: nonces / tg, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: tg, height: 1, depth: 1))
|
|
}
|
|
}
|
|
// Vectors, standalone (one unit per launch)
|
|
var vecPass = 0
|
|
let vecOut = device.makeBuffer(length: 32 * 8, options: .storageModeShared)!
|
|
for (i, base) in vecBases.enumerated() {
|
|
_ = run { enc in encodeHash(enc, out: vecOut, base: base, nonces: 32, group: 32) }
|
|
let p = vecOut.contents().bindMemory(to: UInt64.self, capacity: 32)
|
|
var ok = true
|
|
for l in 0..<32 where p[l] != vecOuts[i][l] { ok = false; print("vector warp base \(base) lane \(l): GPU \(String(format: "%016llx", p[l])) expected \(String(format: "%016llx", vecOuts[i][l]))"); break }
|
|
if ok { vecPass += 1 }
|
|
}
|
|
// Batch at base 0: fingerprint and the vectors inside the batch
|
|
var nonces = 1 << opts.batchLog2
|
|
if persistent { let unit = 32 * min(warpsN, nonces / 32); nonces = (nonces / unit) * unit }
|
|
let out = device.makeBuffer(length: nonces * 8, options: .storageModeShared)!
|
|
let (warmWall, warmGpu) = run { enc in encodeHash(enc, out: out, base: opts.batchBase, nonces: nonces, group: opts.group) }
|
|
let outPtr = out.contents().bindMemory(to: UInt64.self, capacity: nonces)
|
|
var batchVecPass = 0, batchVecN = 0
|
|
for (i, base) in vecBases.enumerated() where Int(base &- opts.batchBase) + 32 <= nonces { // the batch window, wrapping past 2^32
|
|
let off = Int(base &- opts.batchBase)
|
|
batchVecN += 1
|
|
if (0..<32).allSatisfy({ outPtr[off + $0] == vecOuts[i][$0] }) { batchVecPass += 1 }
|
|
}
|
|
let fingerprint = fnv1a64(out.contents(), nonces * 8)
|
|
// Timed batches
|
|
var wallSum = 0.0, gpuSum = 0.0
|
|
for b in 0..<opts.batches {
|
|
let (w, g) = run { enc in encodeHash(enc, out: out, base: UInt32(truncatingIfNeeded: (b + 1) * nonces), nonces: nonces, group: opts.group) }
|
|
wallSum += w; gpuSum += g
|
|
}
|
|
let hashes = Double(nonces) * Double(opts.batches)
|
|
let mhsWall = hashes / wallSum / 1e3, mhsGpu = hashes / gpuSum / 1e3
|
|
let packName = (opts.pack as NSString).lastPathComponent
|
|
print("pack \(packName) seed \"\(seedString)\" id \(programId) class \(className): loads/hash \(loadsPerHash), dataset bytes/hash \(bytesPerHash), scratch ops/hash \(scratchOps * 8), mixer x\(mixerMult), cache 2^\(cacheLog2) words")
|
|
print("device \(device.name); compile \(String(format: "%.0f", compileMs)) ms; cache fill \(String(format: "%.1f", cacheGpu)) ms GPU (\(String(format: "%.1f", cacheWall)) wall); dataset build \(String(format: "%.1f", buildGpu)) ms GPU (\(String(format: "%.1f", buildWall)) wall)")
|
|
print("cache FNV-1a 64 \(String(format: "%016llx", cacheFnv)) \(cacheOk ? "PASS" : "FAIL"); dataset head and last \(dsOk ? "PASS" : "FAIL"); vectors standalone \(vecPass)/\(vecBases.count), in batch \(batchVecPass)/\(batchVecN)")
|
|
if hotMb > 0 { print("hot table \(hotMb) MiB (\(hotSlots) of 16 load slots): fill \(String(format: "%.2f", hotGpu)) ms GPU (\(String(format: "%.2f", hotWall)) wall), FNV-1a 64 \(String(format: "%016llx", hotFnv)) \(hotFnvWant == nil ? "(not in vectors.json)" : (hotOk ? "PASS" : "FAIL"))") }
|
|
print("warm-up batch \(nonces) hashes: \(String(format: "%.1f", warmGpu)) ms GPU, \(String(format: "%.1f", warmWall)) ms wall")
|
|
let footprintMiB = device.currentAllocatedSize / 1048576
|
|
let recommendedMiB = device.recommendedMaxWorkingSetSize / 1048576
|
|
print("resident footprint after the timed batches: currentAllocatedSize \(footprintMiB) MiB (recommendedMaxWorkingSetSize \(recommendedMiB) MiB)")
|
|
let overall = cacheOk && dsOk && hotOk && vecPass == vecBases.count && batchVecPass == batchVecN
|
|
print("RESULT pack=\(packName) class=\(className) device=\(device.name.replacingOccurrences(of: " ", with: "_")) group=\(opts.group) warps=\(warpsN) batch_base=\(opts.batchBase) arena_mib=\(persistent ? warpsN * 32 * scratchWordsPerLane * 4 / 1048576 : 0) hot_mib=\(hotMb) hot_slots=\(hotSlots) hot_fill_ms=\(String(format: "%.2f", hotGpu)) hot=\(hotMb > 0 ? (hotOk ? "PASS" : "FAIL") : "none") nonces=\(nonces) batches=\(opts.batches) vectors=\(vecPass)/\(vecBases.count) batch_vectors=\(batchVecPass)/\(batchVecN) cache=\(cacheOk ? "PASS" : "FAIL") dataset=\(dsOk ? "PASS" : "FAIL") fingerprint=\(String(format: "%016llx", fingerprint)) mhs_gpu=\(String(format: "%.3f", mhsGpu)) mhs_wall=\(String(format: "%.3f", mhsWall)) loads=\(loadsPerHash) bytes=\(bytesPerHash) footprint_mib=\(footprintMiB) recommended_mib=\(recommendedMiB) scratch_ops=\(scratchOps * 8) overall=\(overall ? "PASS" : "FAIL")")
|
|
exit(overall ? 0 : 1)
|