igneum/proto-metal/packbench.swift
igneum-labs 47ae818b2f Merge branch 'readwidth' into ca2-v3
# Conflicts:
#	docs/bench-log.md
#	docs/plans/read-width.md
#	igneum-pow/src/emit.rs
#	proto-metal/packbench.swift
2026-10-05 22:33:39 +00:00

248 lines
16 KiB
Swift

// packbench: runs a program pack (igneum-pow export) on the Mac's Metal GPU from its files alone: memhard.metal (cache
// fill, dataset build), program.metal (igneum_hash), program.h (constants), vectors.json (the CPU reference's vectors).
// Read-width experiment, 5 October 2026 (docs/plans/read-width.md): the Swift bench generates its own programs and does
// not know the experiment's load classes; this harness runs whatever text the Rust emitter wrote, so Metal is checked
// against the Rust CPU reference and timed without a Swift mirror of the generator. One file, no packages.
//
// swiftc -O -target arm64-apple-macos11 -o packbench packbench.swift -framework Metal
// ./packbench --pack <dir> [--batches 5] [--batch-log2 24] [--group 256] [--warps 2048] [--batch-base 0]
//
// Prints one RESULT line per run: vectors, cache and dataset checks, the batch fingerprint (FNV-1a 64 over the 2^B
// outputs at base nonce 0) and MH/s by wall and by GPU time. Variant 5 packs (IGNEUM_PERSISTENT_WARPS) are launched
// as --warps persistent warps with a 1 MiB scratch each; the batch is rounded to a multiple of 32 x warps.
import Foundation
import Metal
func nowMs() -> Double { return Double(DispatchTime.now().uptimeNanoseconds) / 1e6 }
func fail(_ m: String) -> Never { print("FAIL: \(m)"); exit(1) }
struct Opts { var pack = ""; var batches = 5; var batchLog2 = 24; var group = 256; var warps = 2048; var batchBase: UInt32 = 0 }
var opts = Opts()
var args = Array(CommandLine.arguments.dropFirst())
while !args.isEmpty {
let a = args.removeFirst()
func next() -> String { if args.isEmpty { fail("missing value for \(a)") }; return args.removeFirst() }
switch a {
case "--pack": opts.pack = next()
case "--batches": opts.batches = Int(next())!
case "--batch-log2": opts.batchLog2 = Int(next())!
case "--group": opts.group = Int(next())!
case "--warps": opts.warps = Int(next())!
case "--batch-base": opts.batchBase = UInt32(next())! // base nonce of the fingerprint batch (default 0; a base near 2^32 makes the persistent unit sequence wrap inside the launch)
default: fail("unknown argument \(a)")
}
}
if opts.pack.isEmpty { fail("--pack <dir> is required") }
func readText(_ name: String) -> String {
guard let s = try? String(contentsOfFile: opts.pack + "/" + name, encoding: .utf8) else { fail("cannot read \(opts.pack)/\(name)") }
return s
}
let programH = readText("program.h")
func defineU32(_ name: String) -> UInt32? {
let pat = "#define \(name) ([0-9a-fA-Fx]+)"
guard let re = try? NSRegularExpression(pattern: pat), let m = re.firstMatch(in: programH, range: NSRange(programH.startIndex..., in: programH)) else { return nil }
let v = String(programH[Range(m.range(at: 1), in: programH)!]).replacingOccurrences(of: "u", with: "")
if v.hasPrefix("0x") { return UInt32(v.dropFirst(2), radix: 16) }
return UInt32(v)
}
func defineStr(_ name: String) -> String? {
let pat = "#define \(name) \"([^\"]*)\""
guard let re = try? NSRegularExpression(pattern: pat), let m = re.firstMatch(in: programH, range: NSRange(programH.startIndex..., in: programH)) else { return nil }
return String(programH[Range(m.range(at: 1), in: programH)!])
}
let datasetLog2 = Int(defineU32("IGNEUM_DATASET_LOG2") ?? 28)
let datasetMode = defineU32("IGNEUM_DATASET_MODE") ?? 1
if datasetMode != 1 { fail("packbench runs memory-hard packs only") }
let cacheLog2 = Int(defineU32("IGNEUM_CACHE_LOG2_WORDS") ?? 26)
let cacheSegments = Int(defineU32("IGNEUM_CACHE_SEGMENTS") ?? 65536)
let loadsPerHash = Int(defineU32("IGNEUM_LOADS_PER_HASH") ?? 128)
let bytesPerHash = Int(defineU32("IGNEUM_BYTES_PER_HASH") ?? UInt32(loadsPerHash * 4))
let scratchOps = Int(defineU32("IGNEUM_SCRATCH_OPS") ?? 0)
let persistent = (defineU32("IGNEUM_PERSISTENT_WARPS") ?? 0) == 1
let scratchWordsPerLane = Int(defineU32("IGNEUM_SCRATCH_WORDS_PER_LANE") ?? 8192)
let className = defineStr("IGNEUM_LOAD_CLASS") ?? "v2"
// Counter ASIC 2.0 (5 October 2026): the mixer multiplier of the item derivation, 1 when absent (version 2), 4 under class v3
let mixerMult = Int(defineU32("IGNEUM_MIXER_MULT") ?? 1)
// hot-table experiment (docs/plans/hot-table.md): the epoch table, filled on the device from memhard.metal's igneum_hot_fill
let hotMb = Int(defineU32("IGNEUM_HOT_MB") ?? 0)
let hotWords = Int(defineU32("IGNEUM_HOT_WORDS") ?? 0)
let hotSegments = Int(defineU32("IGNEUM_HOT_SEGMENTS") ?? 0)
let hotSlots = Int(defineU32("IGNEUM_HOT_SLOTS") ?? 0)
if hotMb > 0 && (hotWords != hotMb << 18 || hotSegments != hotMb * 256) { fail("program.h hot table sizes disagree") }
let seedString = defineStr("IGNEUM_SEED_STRING") ?? "?"
let programId = defineStr("IGNEUM_PROGRAM_ID") ?? ""
// vectors.json: bases, expected outputs, dataset head / last, cache fingerprint
let vj = try! JSONSerialization.jsonObject(with: Data(contentsOf: URL(fileURLWithPath: opts.pack + "/vectors.json"))) as! [String: Any]
func hex64(_ s: String) -> UInt64 { return UInt64(s.dropFirst(2), radix: 16)! }
func hex32(_ s: String) -> UInt32 { return UInt32(s.dropFirst(2), radix: 16)! }
let warpsJ = vj["warps"] as! [[String: Any]]
let vecBases = warpsJ.map { UInt32(($0["base_nonce"] as! NSNumber).uint64Value) }
let vecOuts = warpsJ.map { ($0["expected"] as! [String]).map(hex64) }
let dsHead = (vj["dataset_head"] as! [String]).map(hex32)
let dsLastIndex = UInt32((vj["dataset_last_index"] as! NSNumber).uint64Value)
let dsLast = hex32(vj["dataset_last"] as! String)
let cacheFnvWant = hex64(vj["cache_fnv1a64"] as! String)
let hotFnvWant: UInt64? = (vj["hot_fnv1a64"] as? String).map(hex64)
guard let device = MTLCreateSystemDefaultDevice(), let queue = device.makeCommandQueue() else { fail("no Metal device") }
let words = 1 << datasetLog2
let mask = UInt32(words - 1)
let cacheWords = 1 << cacheLog2
func compile(_ file: String) -> MTLLibrary {
do { return try device.makeLibrary(source: readText(file), options: MTLCompileOptions()) } catch { fail("Metal compile of \(file): \(error)") }
}
let t0 = nowMs()
let mhLib = compile("memhard.metal")
let progLib = compile("program.metal")
guard let fillFn = mhLib.makeFunction(name: "igneum_cache_fill"), let buildFn = mhLib.makeFunction(name: "igneum_build"), let hashFn = progLib.makeFunction(name: "igneum_hash") else { fail("kernel functions missing") }
let fillPipe = try! device.makeComputePipelineState(function: fillFn)
let buildPipe = try! device.makeComputePipelineState(function: buildFn)
let hashPipe = try! device.makeComputePipelineState(function: hashFn)
var hotPipe: MTLComputePipelineState? = nil
if hotMb > 0 {
guard let hotFn = mhLib.makeFunction(name: "igneum_hot_fill") else { fail("program.h says IGNEUM_HOT_MB \(hotMb) but memhard.metal has no igneum_hot_fill") }
hotPipe = try! device.makeComputePipelineState(function: hotFn)
}
let compileMs = nowMs() - t0
if hashPipe.threadExecutionWidth != 32 { print("WARNING: threadExecutionWidth \(hashPipe.threadExecutionWidth), not 32") }
guard let cache = device.makeBuffer(length: cacheWords * 4, options: .storageModePrivate) else { fail("cache alloc") }
guard let dataset = device.makeBuffer(length: words * 4, options: .storageModePrivate) else { fail("dataset alloc") }
func run(_ body: (MTLComputeCommandEncoder) -> Void) -> (Double, Double) {
let cb = queue.makeCommandBuffer()!
let enc = cb.makeComputeCommandEncoder()!
body(enc)
enc.endEncoding()
let w0 = nowMs()
cb.commit(); cb.waitUntilCompleted()
if let e = cb.error { fail("command buffer: \(e)") }
return (nowMs() - w0, (cb.gpuEndTime - cb.gpuStartTime) * 1000)
}
let (cacheWall, cacheGpu) = run { enc in
enc.setComputePipelineState(fillPipe); enc.setBuffer(cache, offset: 0, index: 0)
enc.dispatchThreadgroups(MTLSize(width: cacheSegments / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
}
let items = words / 16
let (buildWall, buildGpu) = run { enc in
enc.setComputePipelineState(buildPipe); enc.setBuffer(cache, offset: 0, index: 0); enc.setBuffer(dataset, offset: 0, index: 1)
enc.dispatchThreadgroups(MTLSize(width: items / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
}
// cache fingerprint and dataset head/last through a blit to shared memory
func blit(_ src: MTLBuffer, _ offset: Int, _ n: Int) -> MTLBuffer {
let dst = device.makeBuffer(length: n, options: .storageModeShared)!
let cb = queue.makeCommandBuffer()!; let b = cb.makeBlitCommandEncoder()!
b.copy(from: src, sourceOffset: offset, to: dst, destinationOffset: 0, size: n); b.endEncoding(); cb.commit(); cb.waitUntilCompleted()
return dst
}
func fnv1a64(_ p: UnsafeRawPointer, _ n: Int) -> UInt64 {
var h: UInt64 = 0xcbf29ce484222325
let b = p.bindMemory(to: UInt8.self, capacity: n)
for i in 0..<n { h ^= UInt64(b[i]); h = h &* 0x100000001b3 }
return h
}
let cacheCopy = blit(cache, 0, cacheWords * 4)
let cacheFnv = fnv1a64(cacheCopy.contents(), cacheWords * 4)
let cacheOk = cacheFnv == cacheFnvWant
// hot table: fill, then fingerprint against vectors.json
var hot: MTLBuffer? = nil
var hotGpu = 0.0, hotWall = 0.0
var hotOk = true
var hotFnv: UInt64 = 0
if hotMb > 0 {
guard let h = device.makeBuffer(length: hotWords * 4, options: .storageModePrivate) else { fail("hot table alloc of \(hotMb) MiB") }
hot = h
(hotWall, hotGpu) = run { enc in
enc.setComputePipelineState(hotPipe!); enc.setBuffer(h, offset: 0, index: 0)
enc.dispatchThreadgroups(MTLSize(width: hotSegments / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
}
let hotCopy = blit(h, 0, hotWords * 4)
hotFnv = fnv1a64(hotCopy.contents(), hotWords * 4)
if let want = hotFnvWant { hotOk = hotFnv == want }
}
let headCopy = blit(dataset, 0, 64)
let headPtr = headCopy.contents().bindMemory(to: UInt32.self, capacity: 16)
var dsOk = (0..<16).allSatisfy { headPtr[$0] == dsHead[$0] }
let lastCopy = blit(dataset, Int(dsLastIndex) * 4, 4)
dsOk = dsOk && lastCopy.contents().bindMemory(to: UInt32.self, capacity: 1)[0] == dsLast
// scratch arena (variant 5)
var scratch: MTLBuffer? = nil
var salt: UInt32 = 1
let warpsN = persistent ? opts.warps : 0
if persistent {
let bytes = warpsN * 32 * scratchWordsPerLane * 4
guard let s = device.makeBuffer(length: bytes, options: .storageModePrivate) else { fail("scratch alloc of \(bytes >> 20) MiB") }
scratch = s
}
// One hash launch: `nonces` outputs from `base`. Persistent: warpsN warps loop over nonces / 32 units.
func encodeHash(_ enc: MTLComputeCommandEncoder, out: MTLBuffer, base: UInt32, nonces: Int, group: Int) {
enc.setComputePipelineState(hashPipe)
enc.setBuffer(dataset, offset: 0, index: 0)
enc.setBuffer(out, offset: 0, index: 1)
var b = base; enc.setBytes(&b, length: 4, index: 2)
// the hot table is buffer 3 (the scratch triple moves up by one when both are present)
let next = hot == nil ? 3 : 4
if let h = hot { enc.setBuffer(h, offset: 0, index: 3) }
if persistent {
let units = nonces / 32
let nw = min(warpsN, units)
if units % nw != 0 { fail("nonces \(nonces) is not a multiple of 32 x \(nw) warps") }
enc.setBuffer(scratch!, offset: 0, index: next)
var g = UInt32(units); enc.setBytes(&g, length: 4, index: next + 1)
var s = salt; enc.setBytes(&s, length: 4, index: next + 2)
salt = salt &+ UInt32(units)
let threads = nw * 32
let tg = min(group, threads)
enc.dispatchThreadgroups(MTLSize(width: threads / tg, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: tg, height: 1, depth: 1))
} else {
let tg = min(group, nonces)
enc.dispatchThreadgroups(MTLSize(width: nonces / tg, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: tg, height: 1, depth: 1))
}
}
// Vectors, standalone (one unit per launch)
var vecPass = 0
let vecOut = device.makeBuffer(length: 32 * 8, options: .storageModeShared)!
for (i, base) in vecBases.enumerated() {
_ = run { enc in encodeHash(enc, out: vecOut, base: base, nonces: 32, group: 32) }
let p = vecOut.contents().bindMemory(to: UInt64.self, capacity: 32)
var ok = true
for l in 0..<32 where p[l] != vecOuts[i][l] { ok = false; print("vector warp base \(base) lane \(l): GPU \(String(format: "%016llx", p[l])) expected \(String(format: "%016llx", vecOuts[i][l]))"); break }
if ok { vecPass += 1 }
}
// Batch at base 0: fingerprint and the vectors inside the batch
var nonces = 1 << opts.batchLog2
if persistent { let unit = 32 * min(warpsN, nonces / 32); nonces = (nonces / unit) * unit }
let out = device.makeBuffer(length: nonces * 8, options: .storageModeShared)!
let (warmWall, warmGpu) = run { enc in encodeHash(enc, out: out, base: opts.batchBase, nonces: nonces, group: opts.group) }
let outPtr = out.contents().bindMemory(to: UInt64.self, capacity: nonces)
var batchVecPass = 0, batchVecN = 0
for (i, base) in vecBases.enumerated() where Int(base &- opts.batchBase) + 32 <= nonces { // the batch window, wrapping past 2^32
let off = Int(base &- opts.batchBase)
batchVecN += 1
if (0..<32).allSatisfy({ outPtr[off + $0] == vecOuts[i][$0] }) { batchVecPass += 1 }
}
let fingerprint = fnv1a64(out.contents(), nonces * 8)
// Timed batches
var wallSum = 0.0, gpuSum = 0.0
for b in 0..<opts.batches {
let (w, g) = run { enc in encodeHash(enc, out: out, base: UInt32(truncatingIfNeeded: (b + 1) * nonces), nonces: nonces, group: opts.group) }
wallSum += w; gpuSum += g
}
let hashes = Double(nonces) * Double(opts.batches)
let mhsWall = hashes / wallSum / 1e3, mhsGpu = hashes / gpuSum / 1e3
let packName = (opts.pack as NSString).lastPathComponent
print("pack \(packName) seed \"\(seedString)\" id \(programId) class \(className): loads/hash \(loadsPerHash), dataset bytes/hash \(bytesPerHash), scratch ops/hash \(scratchOps * 8), mixer x\(mixerMult), cache 2^\(cacheLog2) words")
print("device \(device.name); compile \(String(format: "%.0f", compileMs)) ms; cache fill \(String(format: "%.1f", cacheGpu)) ms GPU (\(String(format: "%.1f", cacheWall)) wall); dataset build \(String(format: "%.1f", buildGpu)) ms GPU (\(String(format: "%.1f", buildWall)) wall)")
print("cache FNV-1a 64 \(String(format: "%016llx", cacheFnv)) \(cacheOk ? "PASS" : "FAIL"); dataset head and last \(dsOk ? "PASS" : "FAIL"); vectors standalone \(vecPass)/\(vecBases.count), in batch \(batchVecPass)/\(batchVecN)")
if hotMb > 0 { print("hot table \(hotMb) MiB (\(hotSlots) of 16 load slots): fill \(String(format: "%.2f", hotGpu)) ms GPU (\(String(format: "%.2f", hotWall)) wall), FNV-1a 64 \(String(format: "%016llx", hotFnv)) \(hotFnvWant == nil ? "(not in vectors.json)" : (hotOk ? "PASS" : "FAIL"))") }
print("warm-up batch \(nonces) hashes: \(String(format: "%.1f", warmGpu)) ms GPU, \(String(format: "%.1f", warmWall)) ms wall")
let footprintMiB = device.currentAllocatedSize / 1048576
let recommendedMiB = device.recommendedMaxWorkingSetSize / 1048576
print("resident footprint after the timed batches: currentAllocatedSize \(footprintMiB) MiB (recommendedMaxWorkingSetSize \(recommendedMiB) MiB)")
let overall = cacheOk && dsOk && hotOk && vecPass == vecBases.count && batchVecPass == batchVecN
print("RESULT pack=\(packName) class=\(className) device=\(device.name.replacingOccurrences(of: " ", with: "_")) group=\(opts.group) warps=\(warpsN) batch_base=\(opts.batchBase) arena_mib=\(persistent ? warpsN * 32 * scratchWordsPerLane * 4 / 1048576 : 0) hot_mib=\(hotMb) hot_slots=\(hotSlots) hot_fill_ms=\(String(format: "%.2f", hotGpu)) hot=\(hotMb > 0 ? (hotOk ? "PASS" : "FAIL") : "none") nonces=\(nonces) batches=\(opts.batches) vectors=\(vecPass)/\(vecBases.count) batch_vectors=\(batchVecPass)/\(batchVecN) cache=\(cacheOk ? "PASS" : "FAIL") dataset=\(dsOk ? "PASS" : "FAIL") fingerprint=\(String(format: "%016llx", fingerprint)) mhs_gpu=\(String(format: "%.3f", mhsGpu)) mhs_wall=\(String(format: "%.3f", mhsWall)) loads=\(loadsPerHash) bytes=\(bytesPerHash) footprint_mib=\(footprintMiB) recommended_mib=\(recommendedMiB) scratch_ops=\(scratchOps * 8) overall=\(overall ? "PASS" : "FAIL")")
exit(overall ? 0 : 1)