Site: DAG fits any width, hero graph rebuilt with fade mask, locks and proof glow

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
igneum-josh 2026-10-03 16:56:18 +01:00
parent 9b880890c4
commit 945b329dd2
4 changed files with 1438 additions and 183 deletions

View file

@ -906,10 +906,10 @@ func generateCUDA(_ p: Program, memhard: MixParams?) -> String {
return s
}
func generateProgramHeader(_ p: Program, dayString: String, day: (UInt32, UInt32), datasetLog2: Int) -> String {
func generateProgramHeader(_ p: Program, dayString: String, day: (UInt32, UInt32), datasetLog2: Int, memhard: MixParams?) -> String {
let mask = UInt32((1 << datasetLog2) - 1)
let mix = p.histogram.map { "\($0.0)=\($0.1)" }.joined(separator: " ")
return """
var s = """
// Generated by proto-metal/igneum-bench --export-pack for seed "\(p.seedString)". Do not edit by hand.
// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.
#pragma once
@ -926,20 +926,74 @@ func generateProgramHeader(_ p: Program, dayString: String, day: (UInt32, UInt32
#define IGNEUM_ITERATIONS \(Program.iterations)
#define IGNEUM_INSTR_COUNT \(Program.count)
#define IGNEUM_LOADS_PER_HASH \(p.loadsPerHash)
#define IGNEUM_WIDE_LOADS_PER_HASH \(p.wideLoadsPerHash)
#define IGNEUM_OP_MIX \(jstr(mix))
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
#define IGNEUM_DATASET_MODE \(memhard == nil ? 0 : 1)
#define IGNEUM_SEEDW_INIT { \(p.seed.map(hex).joined(separator: ", ")) }
// Defined in kernel.cu. Both launch on the default stream and return cudaGetLastError().
cudaError_t igneum_launch_fill(uint32_t* ds, uint32_t nWords, uint32_t d0, uint32_t d1);
"""
if let mp = memhard {
s += """
#define IGNEUM_KEY_INIT { \(mp.keyWords.map(hex).joined(separator: ", ")) }
#define IGNEUM_CACHE_LOG2_WORDS \(cacheLog2Words)
#define IGNEUM_CACHE_SEGMENT_LOG2_LINES \(cacheSegmentLog2Lines)
#define IGNEUM_CACHE_SEGMENTS \(cacheSegments)u
#define IGNEUM_ITEM_ROUNDS \(itemRounds)
#define IGNEUM_MIX_ROT_INIT { \(mp.rotWords.map { "\($0)u" }.joined(separator: ", ")) }
#define IGNEUM_MIX_MUL_INIT { \(mp.mulWords.map(hex).joined(separator: ", ")) }
#define IGNEUM_MIX_RC_INIT { \(mp.rcWords.map(hex).joined(separator: ", ")) }
// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);
"""
} else {
s += """
// Defined in kernel.cu. Both launch on the default stream and return cudaGetLastError().
cudaError_t igneum_launch_fill(uint32_t* ds, uint32_t nWords, uint32_t d0, uint32_t d1);
"""
}
s += """
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
uint32_t nonces, uint32_t blockWarps);
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
"""
return s
}
func generateVectorsHeader(_ p: Program, bases: [UInt32], outs: [[UInt64]], head: [UInt32], last: UInt32, mask: UInt32, source: String) -> String {
// memhard.h for a memory-hard pack: the core in CUDA C++, compiled for host and device.
func generateMemhardHeader(_ p: Program, _ mp: MixParams) -> String {
return """
// Generated by proto-metal/igneum-bench --export-pack for seed "\(p.seedString)". Do not edit by hand.
// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against.
// Included by kernel.cu (device) and host.cu (host reference). See proto-metal/MEMHARD.md for the construction.
#pragma once
#include <cstdint>
#if defined(__CUDACC__)
#define IGNEUM_HD __host__ __device__ __forceinline__
#else
#define IGNEUM_HD static inline
#endif
\(emitMemhardCore(mp, cuda: true))
"""
}
struct PackVectors {
var head: [UInt32] // dataset[0..15]
var last: UInt32 // dataset[MASK]
var sampleIdx: [UInt32] // 64 dataset indices
var sampleVal: [UInt32]
var cacheHead: [UInt32] // cache[0..15] (memhard only)
var cacheLast: [UInt32] // last cache line (memhard only)
var cacheFNV: UInt64 // FNV-1a 64 over the whole cache (memhard only)
}
func generateVectorsHeader(_ p: Program, bases: [UInt32], outs: [[UInt64]], v: PackVectors, mask: UInt32, source: String, memhard: Bool) -> String {
var s = """
// Generated by proto-metal/igneum-bench --export-pack for seed "\(p.seedString)". Do not edit by hand.
// Expected outputs: \(source)
@ -963,20 +1017,44 @@ func generateVectorsHeader(_ p: Program, bases: [UInt32], outs: [[UInt64]], head
// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (\(mask)).
static const uint32_t IGNEUM_DS_HEAD[16] = {
\((0..<8).map { hex(head[$0]) }.joined(separator: ", ")),
\((8..<16).map { hex(head[$0]) }.joined(separator: ", "))
\((0..<8).map { hex(v.head[$0]) }.joined(separator: ", ")),
\((8..<16).map { hex(v.head[$0]) }.joined(separator: ", "))
};
static const uint32_t IGNEUM_DS_LAST_INDEX = \(mask)u;
static const uint32_t IGNEUM_DS_LAST = \(hex(last));
static const uint32_t IGNEUM_DS_LAST = \(hex(v.last));
// 64 sampled dataset words (index, value) computed on the Mac.
#define IGNEUM_DS_SAMPLES \(v.sampleIdx.count)
static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = {
\(v.sampleIdx.map { "\($0)u" }.joined(separator: ", "))
};
static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = {
\(v.sampleVal.map(hex).joined(separator: ", "))
};
"""
if memhard {
s += """
// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^\(cacheLog2Words) words.
static const uint32_t IGNEUM_CACHE_HEAD[16] = {
\((0..<8).map { hex(v.cacheHead[$0]) }.joined(separator: ", ")),
\((8..<16).map { hex(v.cacheHead[$0]) }.joined(separator: ", "))
};
static const uint32_t IGNEUM_CACHE_LAST[16] = {
\((0..<8).map { hex(v.cacheLast[$0]) }.joined(separator: ", ")),
\((8..<16).map { hex(v.cacheLast[$0]) }.joined(separator: ", "))
};
static const uint64_t IGNEUM_CACHE_FNV64 = \(hex64(v.cacheFNV));
"""
}
return s
}
func generateProgramJSON(_ p: Program, dayString: String, day: (UInt32, UInt32), datasetLog2: Int) -> String {
func generateProgramJSON(_ p: Program, dayString: String, day: (UInt32, UInt32), datasetLog2: Int, memhard: MixParams?) -> String {
let mask = UInt32((1 << datasetLog2) - 1)
var s = "{\n"
s += " \"format\": \"igneum-program-pack-1\",\n"
s += " \"format\": \"igneum-program-pack-2\",\n"
s += " \"dataset_mode\": \(jstr(memhard == nil ? "closed-form" : "memory-hard")),\n"
s += " \"seed\": \(jstr(p.seedString)),\n"
s += " \"seed_words\": [\(p.seed.map(jhex).joined(separator: ", "))],\n"
s += " \"seed_derivation\": \"FNV-1a 64 over UTF-8 of seed, basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32\",\n"
@ -1001,7 +1079,8 @@ func generateProgramJSON(_ p: Program, dayString: String, day: (UInt32, UInt32),
s += " \"rotr\": \"dst = rotr(dst, src & 31)\",\n"
s += " \"mad\": \"dst = src * src2 + dst\",\n"
s += " \"shfl\": \"dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp\",\n"
s += " \"load\": \"dst = dst ^ dataset[src & dataset.mask]\"\n"
s += " \"load\": \"dst = dst ^ dataset[src & dataset.mask]\",\n"
s += " \"wload\": \"base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)\"\n"
s += " },\n"
s += " \"dataset\": {\n"
s += " \"log2_words\": \(datasetLog2),\n"
@ -1011,7 +1090,19 @@ func generateProgramJSON(_ p: Program, dayString: String, day: (UInt32, UInt32),
s += " \"day_words_from\": \(jstr("day/" + dayString)),\n"
s += " \"d0\": \(jhex(day.0)),\n"
s += " \"d1\": \(jhex(day.1)),\n"
s += " \"formula\": \"x = i ^ d0; x *= 0x9E3779B1; x ^= x>>15; x += d1; x *= 0x85EBCA77; x ^= x>>13; x *= 0xC2B2AE3D; x ^= x>>16 (all mod 2^32)\"\n"
if let mp = memhard {
s += " \"mode\": \"memory-hard\",\n"
s += " \"spec\": \"proto-metal/MEMHARD.md\",\n"
s += " \"key\": [\(mp.keyWords.map(jhex).joined(separator: ", "))],\n"
s += " \"key_derivation\": \"the 8 words of seedWords(\\\"day/\\\" + day); d0, d1 are key[0], key[1]\",\n"
s += " \"cache\": {\"log2_words\": \(cacheLog2Words), \"bytes\": \(UInt64(cacheWords) * 4), \"line_words\": 16, \"segment_lines\": \(cacheLinesPerSegment), \"segments\": \(cacheSegments), \"block\": \"ChaCha\(chachaRounds) core + feed-forward, rotations 16 12 8 7\", \"sigma\": [\(chachaSigma.map(jhex).joined(separator: ", "))], \"tag\": [\(cacheTag.map(jhex).joined(separator: ", "))], \"chain\": \"in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0\"},\n"
s += " \"mixer\": {\"draw\": \"SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())\", \"rot\": [\(mp.rotWords.map { "\($0)" }.joined(separator: ", "))], \"mul\": [\(mp.mulWords.map(jhex).joined(separator: ", "))], \"rc\": [\(mp.rcWords.map(jhex).joined(separator: ", "))], \"round\": \"for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]\", \"quarter_round\": \"a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)\"},\n"
s += " \"item\": \"s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..\(itemRounds - 1): s = M_r(s); line = s[0] & \(jhex(cacheLineMask)); s[i] ^= cache[line * 16 + i]; then s = M_\(itemRounds)(s); item(t) = s\",\n"
s += " \"word\": \"dataset[w] = item(w >> 4)[w & 15]\"\n"
} else {
s += " \"mode\": \"closed-form\",\n"
s += " \"formula\": \"x = i ^ d0; x *= 0x9E3779B1; x ^= x>>15; x += d1; x *= 0x85EBCA77; x ^= x>>13; x *= 0xC2B2AE3D; x ^= x>>16 (all mod 2^32)\"\n"
}
s += " },\n"
s += " \"instructions\": [\n"
for (k, ins) in p.instrs.enumerated() {
@ -1022,10 +1113,11 @@ func generateProgramJSON(_ p: Program, dayString: String, day: (UInt32, UInt32),
return s
}
func generateVectorsJSON(_ p: Program, dayString: String, datasetLog2: Int, bases: [UInt32], outs: [[UInt64]], head: [UInt32], last: UInt32, mask: UInt32, source: String) -> String {
func generateVectorsJSON(_ p: Program, dayString: String, datasetLog2: Int, bases: [UInt32], outs: [[UInt64]], v: PackVectors, mask: UInt32, source: String, memhard: Bool) -> String {
var s = "{\n"
s += " \"seed\": \(jstr(p.seedString)),\n"
s += " \"day\": \(jstr(dayString)),\n"
s += " \"dataset_mode\": \(jstr(memhard ? "memory-hard" : "closed-form")),\n"
s += " \"dataset_log2_words\": \(datasetLog2),\n"
s += " \"mask\": \(jhex(mask)),\n"
s += " \"lanes\": 32,\n"
@ -1039,51 +1131,28 @@ func generateVectorsJSON(_ p: Program, dayString: String, datasetLog2: Int, base
s += i == outs.count - 1 ? " ]}\n" : " ]},\n"
}
s += " ],\n"
s += " \"dataset_head\": [\(head.map(jhex).joined(separator: ", "))],\n"
s += " \"dataset_head\": [\(v.head.map(jhex).joined(separator: ", "))],\n"
s += " \"dataset_last_index\": \(mask),\n"
s += " \"dataset_last\": \(jhex(last))\n"
s += " \"dataset_last\": \(jhex(v.last)),\n"
s += " \"dataset_samples\": [\(zip(v.sampleIdx, v.sampleVal).map { "{\"index\": \($0.0), \"value\": \(jhex($0.1))}" }.joined(separator: ", "))]"
if memhard {
s += ",\n \"cache_head\": [\(v.cacheHead.map(jhex).joined(separator: ", "))],\n"
s += " \"cache_last_line\": [\(v.cacheLast.map(jhex).joined(separator: ", "))],\n"
s += " \"cache_fnv1a64\": \(jhex64(v.cacheFNV))\n"
} else { s += "\n" }
s += "}\n"
return s
}
// Runs the Metal kernel for each base nonce (one 32-thread threadgroup each) and compares with `expected`.
func metalCrossCheck(_ p: Program, datasetLog2: Int, day: (UInt32, UInt32), bases: [UInt32], expected: [[UInt64]]) -> (ok: Bool, detail: String) {
guard let device = MTLCreateSystemDefaultDevice(), let queue = device.makeCommandQueue() else { return (false, "no Metal device") }
let words = 1 << datasetLog2
guard let dataset = device.makeBuffer(length: words * 4, options: .storageModePrivate),
let outBuf = device.makeBuffer(length: 32 * 8, options: .storageModeShared) else { return (false, "buffer allocation failed") }
// The dataset comes from the context (closed form or memory-hard), so the GPU build path is covered too.
func metalCrossCheck(_ ctx: DatasetContext, _ p: Program, datasetLog2: Int, bases: [UInt32], expected: [[UInt64]]) -> (ok: Bool, detail: String) {
let dataset = ctx.makeDataset(log2: datasetLog2)
do {
let flib = try device.makeLibrary(source: fillMSL, options: MTLCompileOptions())
let fpipe = try device.makeComputePipelineState(function: flib.makeFunction(name: "igneum_fill")!)
let hlib = try device.makeLibrary(source: generateMSL(p, datasetLog2: datasetLog2), options: MTLCompileOptions())
let hpipe = try device.makeComputePipelineState(function: hlib.makeFunction(name: "igneum_hash")!)
let cb = queue.makeCommandBuffer()!
let enc = cb.makeComputeCommandEncoder()!
enc.setComputePipelineState(fpipe)
enc.setBuffer(dataset, offset: 0, index: 0)
var d = (day.0, day.1)
enc.setBytes(&d, length: 8, index: 1)
enc.dispatchThreadgroups(MTLSize(width: words / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
enc.endEncoding()
cb.commit(); cb.waitUntilCompleted()
if let e = cb.error { return (false, "fill error \(e)") }
let k = try compileHash(ctx.gpu, msl: generateMSL(p, datasetLog2: datasetLog2))
guard let got = gpuWarps(ctx.gpu, k, dataset: dataset, bases: bases) else { return (false, "GPU run failed") }
var bad = [String]()
for (i, base) in bases.enumerated() {
let cb2 = queue.makeCommandBuffer()!
let e2 = cb2.makeComputeCommandEncoder()!
e2.setComputePipelineState(hpipe)
e2.setBuffer(dataset, offset: 0, index: 0)
e2.setBuffer(outBuf, offset: 0, index: 1)
var b = base
e2.setBytes(&b, length: 4, index: 2)
e2.dispatchThreadgroups(MTLSize(width: 1, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 32, height: 1, depth: 1))
e2.endEncoding()
cb2.commit(); cb2.waitUntilCompleted()
if let e = cb2.error { return (false, "hash error \(e)") }
let ptr = outBuf.contents().bindMemory(to: UInt64.self, capacity: 32)
let got = (0..<32).map { ptr[$0] }
if got != expected[i] { bad.append("base \(base)") }
}
for (i, base) in bases.enumerated() where got[i] != expected[i] { bad.append("base \(base)") }
return (bad.isEmpty, bad.isEmpty ? "Metal GPU cross-check PASS \(bases.count)/\(bases.count) warps" : "Metal GPU cross-check FAIL: \(bad.joined(separator: ", "))")
} catch {
return (false, "Metal compile error \(error)")
@ -1093,31 +1162,52 @@ func metalCrossCheck(_ p: Program, datasetLog2: Int, day: (UInt32, UInt32), base
func exportPack(_ opts: Options) -> Never {
let dir = opts.exportPack!
let program = generateProgram(seedString: opts.seed)
let dayWords = seedWords("day/" + opts.day)
let day = (dayWords[0], dayWords[1])
let gpu = GPU()
let ctx = DatasetContext(gpu: gpu, closedForm: opts.closedForm, dayString: opts.day)
let day = ctx.day
let mask = UInt32((1 << opts.datasetLog2) - 1)
print("igneum-bench --export-pack \(dir)")
print("seed \"\(opts.seed)\", day \"\(opts.day)\", dataset 2^\(opts.datasetLog2) words, loads/hash \(program.loadsPerHash)")
print("seed \"\(opts.seed)\", day \"\(opts.day)\", dataset 2^\(opts.datasetLog2) words (\(ctx.modeName)), loads/hash \(program.loadsPerHash), wide loads/hash \(program.wideLoadsPerHash)")
print("op mix: " + program.histogram.map { "\($0.0)=\($0.1)" }.joined(separator: " "))
if !ctx.closed { print("cache: GPU fill \(fmt(ctx.cacheFillGPUms, 2)) ms GPU time; CPU fill \(fmt(ctx.cpuSide()!.fillMs, 1)) ms one core") }
let outs = packVectorBases.map { cpuWarp(program, baseNonce: $0, day: day, mask: mask) }
let head = (0..<16).map { datasetElem(UInt32($0), day.0, day.1) }
let last = datasetElem(mask, day.0, day.1)
let ds = ctx.source(log2: opts.datasetLog2)
let outs = packVectorBases.map { cpuWarp(program, baseNonce: $0, ds: ds) }
var v = PackVectors(head: (0..<16).map { ds.word(UInt32($0)) }, last: ds.word(mask), sampleIdx: [], sampleVal: [],
cacheHead: [], cacheLast: [], cacheFNV: 0)
var sr = SplitMix64(s: 0x6d68_7361_6d70_6c65) // "mhsample"
for _ in 0..<64 { let i = UInt32(truncatingIfNeeded: sr.next()) & mask; v.sampleIdx.append(i); v.sampleVal.append(ds.word(i)) }
if let cpu = ctx.cpuSide() {
v.cacheHead = (0..<16).map { cpu.cache[$0] }
v.cacheLast = (0..<16).map { cpu.cache[cacheWords - 16 + $0] }
v.cacheFNV = fnv64(UnsafeRawPointer(cpu.cache), cacheWords * 4)
let cc = ctx.cacheCheck()
print(cc.detail)
if !cc.ok { print("FAIL: GPU cache differs from the CPU cache, pack not written"); exit(1) }
}
var source = "proto-metal CPU interpreter (cpuWarp) on Apple M5 Max"
let check = metalCrossCheck(program, datasetLog2: opts.datasetLog2, day: day, bases: packVectorBases, expected: outs)
var source = "proto-metal CPU interpreter (cpuWarp, \(ctx.modeName) dataset) on Apple M5 Max"
let check = metalCrossCheck(ctx, program, datasetLog2: opts.datasetLog2, bases: packVectorBases, expected: outs)
print(check.detail)
source += "; " + check.detail
if !check.ok { print("FAIL: vectors do not match the Metal GPU, pack not written"); exit(1) }
// The sampled dataset words against the GPU-built dataset as well.
let sampleCheck = ctx.sampleCheck(log2: opts.datasetLog2, indices: v.sampleIdx + [0, mask])
print(sampleCheck.detail)
if !sampleCheck.ok { print("FAIL: GPU dataset words differ from the CPU derivation, pack not written"); exit(1) }
let files: [(String, String)] = [
("program.json", generateProgramJSON(program, dayString: opts.day, day: day, datasetLog2: opts.datasetLog2)),
("vectors.json", generateVectorsJSON(program, dayString: opts.day, datasetLog2: opts.datasetLog2, bases: packVectorBases, outs: outs, head: head, last: last, mask: mask, source: source)),
("kernel.cu", generateCUDA(program)),
("program.h", generateProgramHeader(program, dayString: opts.day, day: day, datasetLog2: opts.datasetLog2)),
("vectors.h", generateVectorsHeader(program, bases: packVectorBases, outs: outs, head: head, last: last, mask: mask, source: source)),
var files: [(String, String)] = [
("program.json", generateProgramJSON(program, dayString: opts.day, day: day, datasetLog2: opts.datasetLog2, memhard: ctx.mp)),
("vectors.json", generateVectorsJSON(program, dayString: opts.day, datasetLog2: opts.datasetLog2, bases: packVectorBases, outs: outs, v: v, mask: mask, source: source, memhard: !ctx.closed)),
("kernel.cu", generateCUDA(program, memhard: ctx.mp)),
("program.h", generateProgramHeader(program, dayString: opts.day, day: day, datasetLog2: opts.datasetLog2, memhard: ctx.mp)),
("vectors.h", generateVectorsHeader(program, bases: packVectorBases, outs: outs, v: v, mask: mask, source: source, memhard: !ctx.closed)),
("program.metal", generateMSL(program, datasetLog2: opts.datasetLog2)),
]
if let mp = ctx.mp {
files.append(("memhard.h", generateMemhardHeader(program, mp)))
files.append(("memhard.metal", memhardMSL(mp)))
}
do {
try FileManager.default.createDirectory(atPath: dir, withIntermediateDirectories: true)
for (name, text) in files {
@ -1153,6 +1243,155 @@ final class GPU {
}
}
// Everything about the dataset for one day: which construction, the GPU kernels that build it, the GPU cache
// (memory-hard mode), and the CPU side the verifier uses. Tests and the bench share one of these.
final class DatasetContext {
let gpu: GPU
let closed: Bool
let dayString: String
let key: [UInt32] // the 8 words of seedWords("day/" + day)
var day: (UInt32, UInt32) { (key[0], key[1]) }
let mp: MixParams? // memory-hard mixer parameters (nil in closed-form mode)
var modeName: String { closed ? "closed-form" : "memory-hard" }
private var closedFillPipe: MTLComputePipelineState?
private var cacheFillPipe: MTLComputePipelineState?
private var buildPipe: MTLComputePipelineState?
var compileMs = 0.0
var gpuCache: MTLBuffer? // 2^26 words, private
var cacheFillGPUms = 0.0, cacheFillWallMs = 0.0
var lastBuildGPUms = 0.0, lastBuildWallMs = 0.0
private var cpu: MemhardCPU?
init(gpu: GPU, closedForm: Bool, dayString: String) {
self.gpu = gpu; closed = closedForm; self.dayString = dayString
key = seedWords("day/" + dayString)
mp = closedForm ? nil : MixParams(key: key)
let t0 = nowNs()
do {
if closedForm {
let lib = try gpu.device.makeLibrary(source: fillMSL, options: MTLCompileOptions())
closedFillPipe = try gpu.device.makeComputePipelineState(function: lib.makeFunction(name: "igneum_fill")!)
} else {
let lib = try gpu.device.makeLibrary(source: memhardMSL(mp!), options: MTLCompileOptions())
cacheFillPipe = try gpu.device.makeComputePipelineState(function: lib.makeFunction(name: "igneum_cache_fill")!)
buildPipe = try gpu.device.makeComputePipelineState(function: lib.makeFunction(name: "igneum_build")!)
}
} catch { print("FAIL: dataset kernel compile: \(error)"); exit(1) }
compileMs = ms(t0, nowNs())
if !closedForm {
guard let c = gpu.device.makeBuffer(length: cacheWords * 4, options: .storageModePrivate) else { print("FAIL: cannot allocate the cache"); exit(1) }
gpuCache = c
let cb = gpu.queue.makeCommandBuffer()!
let enc = cb.makeComputeCommandEncoder()!
enc.setComputePipelineState(cacheFillPipe!)
enc.setBuffer(c, offset: 0, index: 0)
enc.dispatchThreadgroups(MTLSize(width: cacheSegments / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
enc.endEncoding()
let w0 = nowNs()
cb.commit(); cb.waitUntilCompleted()
cacheFillWallMs = ms(w0, nowNs())
if let e = cb.error { print("FAIL: cache fill error \(e)"); exit(1) }
cacheFillGPUms = (cb.gpuEndTime - cb.gpuStartTime) * 1000
}
}
// Fills the GPU cache again (for repeat timings). Returns GPU ms.
func refillCache() -> Double {
guard let c = gpuCache else { return 0 }
let cb = gpu.queue.makeCommandBuffer()!
let enc = cb.makeComputeCommandEncoder()!
enc.setComputePipelineState(cacheFillPipe!)
enc.setBuffer(c, offset: 0, index: 0)
enc.dispatchThreadgroups(MTLSize(width: cacheSegments / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
enc.endEncoding()
cb.commit(); cb.waitUntilCompleted()
return (cb.gpuEndTime - cb.gpuStartTime) * 1000
}
// Allocates a private 2^log2-word dataset and builds it on the GPU. Timing lands in lastBuild*.
func makeDataset(log2: Int) -> MTLBuffer {
let words = 1 << log2
guard let buf = gpu.device.makeBuffer(length: words * 4, options: .storageModePrivate) else {
print("FAIL: cannot allocate 2^\(log2) word dataset"); exit(1)
}
build(into: buf, words: words)
return buf
}
func build(into buf: MTLBuffer, words: Int) {
let cb = gpu.queue.makeCommandBuffer()!
let enc = cb.makeComputeCommandEncoder()!
if closed {
enc.setComputePipelineState(closedFillPipe!)
enc.setBuffer(buf, offset: 0, index: 0)
var d = day
enc.setBytes(&d, length: 8, index: 1)
enc.dispatchThreadgroups(MTLSize(width: words / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
} else {
enc.setComputePipelineState(buildPipe!)
enc.setBuffer(gpuCache!, offset: 0, index: 0)
enc.setBuffer(buf, offset: 0, index: 1)
let items = words / 16
enc.dispatchThreadgroups(MTLSize(width: max(items / 256, 1), height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: min(items, 256), height: 1, depth: 1))
}
enc.endEncoding()
let w0 = nowNs()
cb.commit(); cb.waitUntilCompleted()
lastBuildWallMs = ms(w0, nowNs())
if let e = cb.error { print("FAIL: dataset build error \(e)"); exit(1) }
lastBuildGPUms = (cb.gpuEndTime - cb.gpuStartTime) * 1000
}
// The CPU verifier's side: the 256 MiB cache computed on one core, once per process. Prints the fill time.
func cpuSide() -> MemhardCPU? {
if closed { return nil }
if let c = cpu { return c }
let c = MemhardCPU(key: key)
print("cache: CPU fill \(fmt(c.fillMs, 1)) ms on one core (2^\(cacheLog2Words) words, \(cacheSegments) chains of \(cacheLinesPerSegment) ChaCha\(chachaRounds) blocks)")
cpu = c
return c
}
func source(log2: Int) -> DatasetSource {
DatasetSource(mask: UInt32((1 << log2) - 1), day: day, memhard: cpuSide())
}
// The buffer the inline (shortcut) kernel binds at index 0: the cache in memory-hard mode, the dataset otherwise.
func inlineBuffer(dataset: MTLBuffer) -> MTLBuffer { closed ? dataset : gpuCache! }
var inlineSource: LoadSource { closed ? .inlineClosed(day.0, day.1) : .inlineMemhard(mp!) }
// Blit the GPU cache to shared memory and compare every word with the CPU cache.
func cacheCheck() -> (ok: Bool, detail: String) {
guard let gc = gpuCache, let c = cpuSide() else { return (true, "cache check: not applicable (closed form)") }
guard let shared = gpu.device.makeBuffer(length: cacheWords * 4, options: .storageModeShared) else { return (false, "cache check: no shared buffer") }
let cb = gpu.queue.makeCommandBuffer()!
let blit = cb.makeBlitCommandEncoder()!
blit.copy(from: gc, sourceOffset: 0, to: shared, destinationOffset: 0, size: cacheWords * 4)
blit.endEncoding()
cb.commit(); cb.waitUntilCompleted()
let same = memcmp(shared.contents(), c.cache, cacheWords * 4) == 0
let fp = fnv64(shared.contents(), cacheWords * 4)
return (same, "cache check: GPU cache \(same ? "==" : "!=") CPU cache, all \(cacheWords) words compared, FNV-1a 64 \(h64(fp))")
}
// Reads the given words of a GPU-built dataset back and compares with the CPU derivation.
func sampleCheck(log2: Int, indices: [UInt32]) -> (ok: Bool, detail: String) {
let words = 1 << log2
let dataset = makeDataset(log2: log2)
guard let shared = gpu.device.makeBuffer(length: words * 4, options: .storageModeShared) else { return (false, "sample check: no shared buffer") }
let cb = gpu.queue.makeCommandBuffer()!
let blit = cb.makeBlitCommandEncoder()!
blit.copy(from: dataset, sourceOffset: 0, to: shared, destinationOffset: 0, size: words * 4)
blit.endEncoding()
cb.commit(); cb.waitUntilCompleted()
let p = shared.contents().bindMemory(to: UInt32.self, capacity: words)
let ds = source(log2: log2)
var bad = 0
for i in indices where p[Int(i)] != ds.word(i) { bad += 1 }
return (bad == 0, "dataset sample check (\(modeName), 2^\(log2) words): \(indices.count) GPU words vs CPU derivation, \(bad) mismatches")
}
}
struct EpochResult {
var seed: String
var libraryMs: Double
@ -1162,14 +1401,18 @@ struct EpochResult {
var gbpsWall: Double
var gbpsGPU: Double
var loadsPerHash: Int
var itemsPerWarp: Int
var verify: [(warp: Int, pass: Bool, ms: Double, repMs: Double)]
var allPass: Bool { verify.allSatisfy { $0.pass } }
}
func runEpoch(gpu: GPU, opts: Options, seedString: String, dataset: MTLBuffer, day: (UInt32, UInt32)) -> EpochResult {
func runEpoch(gpu: GPU, opts: Options, seedString: String, dataset: MTLBuffer, ctx: DatasetContext) -> EpochResult {
let program = generateProgram(seedString: seedString)
let msl = generateMSL(program, datasetLog2: opts.datasetLog2, inlineDay: opts.inlineDataset ? day : nil)
if opts.inlineDataset { print("\nNOTE: --inline-dataset: loads compute ds_elem inline, the dataset buffer is never read") }
let msl = generateMSL(program, datasetLog2: opts.datasetLog2, source: opts.inlineDataset ? ctx.inlineSource : .stored)
if opts.inlineDataset {
print(ctx.closed ? "\nNOTE: --inline-dataset: loads compute ds_elem inline, the dataset buffer is never read"
: "\nNOTE: --inline-dataset: loads derive the item from the 256 MiB cache (8 dependent 64-byte reads + 9 mixers), the dataset buffer is never read")
}
if let dir = opts.dumpDir {
try? FileManager.default.createDirectory(atPath: dir, withIntermediateDirectories: true)
let safe = seedString.replacingOccurrences(of: "/", with: "_")
@ -1177,7 +1420,7 @@ func runEpoch(gpu: GPU, opts: Options, seedString: String, dataset: MTLBuffer, d
}
print("\n=== epoch seed \"\(seedString)\" ===")
print("program: \(Program.count) instructions x \(Program.iterations) iterations, loads/hash = \(program.loadsPerHash)")
print("program: \(Program.count) instructions x \(Program.iterations) iterations, loads/hash = \(program.loadsPerHash) (wide \(program.wideLoadsPerHash)), distinct items per warp = \(program.itemsPerWarp)")
print("op mix: " + program.histogram.map { "\($0.0)=\($0.1)" }.joined(separator: " "))
// Runtime compile
@ -1209,10 +1452,11 @@ func runEpoch(gpu: GPU, opts: Options, seedString: String, dataset: MTLBuffer, d
let groups = n / 32
guard let outBuf = gpu.device.makeBuffer(length: n * 8, options: .storageModeShared) else { print("FAIL: out buffer"); exit(1) }
let buffer0 = opts.inlineDataset ? ctx.inlineBuffer(dataset: dataset) : dataset
func encodeBatch(_ cb: MTLCommandBuffer, base: UInt32) {
let enc = cb.makeComputeCommandEncoder()!
enc.setComputePipelineState(pipeline)
enc.setBuffer(dataset, offset: 0, index: 0)
enc.setBuffer(buffer0, offset: 0, index: 0)
enc.setBuffer(outBuf, offset: 0, index: 1)
var b = base
enc.setBytes(&b, length: 4, index: 2)
@ -1264,19 +1508,22 @@ func runEpoch(gpu: GPU, opts: Options, seedString: String, dataset: MTLBuffer, d
print(" wall \(fmt(wallSeconds * 1000)) ms -> \(fmt(hpsWall / 1e6, 3)) Mhash/s, \(fmt(gbpsWall)) GB/s useful (loads x 4 B)")
print(" GPU \(fmt(gpuSeconds * 1000)) ms -> \(fmt(hpsGPU / 1e6, 3)) Mhash/s, \(fmt(gbpsGPU)) GB/s useful (loads x 4 B)")
// CPU verification
let mask = UInt32((1 << opts.datasetLog2) - 1)
// CPU verification. The verifier holds the 256 MiB cache (memory-hard mode) or nothing (closed form), never
// the dataset; every dataset word a load needs is derived on demand inside cpuWarp.
let ds = ctx.source(log2: opts.datasetLog2)
var verify = [(warp: Int, pass: Bool, ms: Double, repMs: Double)]()
for (i, w) in warps.enumerated() {
let base = UInt32(w * 32)
let d0 = ds.memhard?.derivations ?? 0
let c0 = nowNs()
let cpu = cpuWarp(program, baseNonce: base, day: day, mask: mask)
let cpu = cpuWarp(program, baseNonce: base, ds: ds)
let c1 = nowNs()
let derived = (ds.memhard?.derivations ?? 0) - d0
// repeated runs for a steadier figure
let reps = 20
let r0 = nowNs()
var sink: UInt64 = 0
for _ in 0..<reps { sink ^= cpuWarp(program, baseNonce: base, day: day, mask: mask)[0] }
for _ in 0..<reps { sink ^= cpuWarp(program, baseNonce: base, ds: ds)[0] }
let r1 = nowNs()
let pass = cpu == gpuOutputs[i] && sink != 1
let single = ms(c0, c1), rep = ms(r0, r1) / Double(reps)
@ -1286,12 +1533,13 @@ func runEpoch(gpu: GPU, opts: Options, seedString: String, dataset: MTLBuffer, d
let bad = (0..<32).filter { cpu[$0] != gpuOutputs[i][$0] }
detail = " mismatched lanes: \(bad) first: cpu=\(String(format: "%016llx", cpu[bad.first ?? 0])) gpu=\(String(format: "%016llx", gpuOutputs[i][bad.first ?? 0]))"
}
print("verify warp \(w) (nonces \(base)..\(base + 31)): \(pass ? "PASS" : "FAIL") cpu \(fmt(single, 3)) ms single, \(fmt(rep, 3)) ms avg of \(reps)\(detail)")
let items = ds.memhard == nil ? "" : ", \(derived) items derived"
print("verify warp \(w) (nonces \(base)..\(base + 31)): \(pass ? "PASS" : "FAIL") cpu \(fmt(single, 3)) ms single, \(fmt(rep, 3)) ms avg of \(reps)\(items)\(detail)")
}
return EpochResult(seed: seedString, libraryMs: libMs, pipelineMs: pipeMs,
hashesPerSecWall: hpsWall, hashesPerSecGPU: hpsGPU, gbpsWall: gbpsWall, gbpsGPU: gbpsGPU,
loadsPerHash: program.loadsPerHash, verify: verify)
loadsPerHash: program.loadsPerHash, itemsPerWarp: program.itemsPerWarp, verify: verify)
}
// MARK: - Hardening tests: shared helpers
@ -1322,31 +1570,7 @@ func compileHash(_ gpu: GPU, msl: String) throws -> CompiledHash {
return CompiledHash(pipeline: pipe, libraryMs: ms(t0, t1), pipelineMs: ms(t1, t2))
}
// Allocates a private 2^log2-word dataset and fills it on the GPU with the closed form for `day`.
func makeDataset(_ gpu: GPU, log2: Int, day: (UInt32, UInt32)) -> MTLBuffer {
let words = 1 << log2
guard let buf = gpu.device.makeBuffer(length: words * 4, options: .storageModePrivate) else {
print("FAIL: cannot allocate 2^\(log2) word dataset"); exit(1)
}
do {
let lib = try gpu.device.makeLibrary(source: fillMSL, options: MTLCompileOptions())
let pipe = try gpu.device.makeComputePipelineState(function: lib.makeFunction(name: "igneum_fill")!)
let cb = gpu.queue.makeCommandBuffer()!
let enc = cb.makeComputeCommandEncoder()!
enc.setComputePipelineState(pipe)
enc.setBuffer(buf, offset: 0, index: 0)
var d = (day.0, day.1)
enc.setBytes(&d, length: 8, index: 1)
enc.dispatchThreadgroups(MTLSize(width: words / 256, height: 1, depth: 1),
threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
enc.endEncoding()
cb.commit(); cb.waitUntilCompleted()
if let e = cb.error { print("FAIL: fill error \(e)"); exit(1) }
} catch {
print("FAIL: fill kernel \(error)"); exit(1)
}
return buf
}
// Dataset allocation and build moved into DatasetContext.makeDataset (3 October 2026), which handles both constructions.
// One 32-thread threadgroup per base nonce, all dispatched from one encoder. Warp i lands at byte offset
// i * 256 of the output buffer, which is pre-filled with a sentinel so an unwritten lane is visible.
@ -1430,14 +1654,15 @@ func maskCheckCUDA(_ cu: String) -> (ok: Bool, detail: String) {
// MARK: - --fuzz
func runFuzz(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
func runFuzz(_ opts: Options, ctx: DatasetContext) -> Bool {
let gpu = ctx.gpu
let n = max(opts.fuzz ?? 200, 1)
let master = opts.fuzzSeed
print("\n=== fuzz: \(n) random programs, master seed \"\(master)\", 4 random warps each ===")
let sizes = [24, 26, 28]
let d0 = nowNs()
var datasets = [Int: MTLBuffer]()
for s in sizes { datasets[s] = makeDataset(gpu, log2: s, day: day) }
for s in sizes { datasets[s] = ctx.makeDataset(log2: s) }
print("datasets " + sizes.map { "2^\($0) (\((1 << $0) * 4 / (1 << 20)) MiB)" }.joined(separator: ", ") + " filled in \(fmt(ms(d0, nowNs()), 1)) ms")
let mw = seedWords("fuzz/" + master)
@ -1481,9 +1706,10 @@ func runFuzz(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
}
let g1 = nowNs()
let mask = UInt32((1 << log2) - 1)
let ds = ctx.source(log2: log2)
var ok = true
for (w, base) in bases.enumerated() {
let cpu = cpuWarp(program, baseNonce: base, day: day, mask: mask)
let cpu = cpuWarp(program, baseNonce: base, ds: ds)
warps += 1
if cpu != gpuOut[w] {
ok = false
@ -1620,11 +1846,13 @@ func edgeCases(mask: UInt32) -> [EdgeCase] {
return c
}
func runEdge(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
func runEdge(_ opts: Options, ctx: DatasetContext) -> Bool {
let gpu = ctx.gpu
let log2 = opts.datasetLog2
let mask = UInt32((1 << log2) - 1)
let ds = ctx.source(log2: log2)
print("\n=== edge cases, dataset 2^\(log2) words, MASK \(hex(mask)) ===")
let dataset = makeDataset(gpu, log2: log2, day: day)
let dataset = ctx.makeDataset(log2: log2)
let bases: [UInt32] = [0, 1 << 20, 0x7FFFFFF0, 0xFFFFFFE0]
print("warps: base nonces " + bases.map { hex($0) }.joined(separator: ", ") + " (the last two straddle 2^31 and wrap past 2^32)")
var allOk = true
@ -1641,7 +1869,7 @@ func runEdge(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
for base in bases {
var hits = [Int: Int]()
var misses = [Int: Int]()
_ = cpuWarpTraced(program, baseNonce: base, day: day, mask: mask) { _, k, regs in
_ = cpuWarpTraced(program, baseNonce: base, ds: ds) { _, k, regs in
for (idx, _, check) in ec.pre where idx == k {
if check(regs) { hits[idx, default: 0] += 1 } else { misses[idx, default: 0] += 1 }
}
@ -1661,7 +1889,7 @@ func runEdge(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
var badLanes = 0
var first = ""
for (w, base) in bases.enumerated() {
let cpu = cpuWarp(program, baseNonce: base, day: day, mask: mask)
let cpu = cpuWarp(program, baseNonce: base, ds: ds)
for l in 0..<32 where cpu[l] != gpuOut[w][l] {
badLanes += 1
if first.isEmpty { first = "first: base \(hex(base)) lane \(l) gpu \(h64(gpuOut[w][l])) cpu \(h64(cpu[l]))" }
@ -1694,13 +1922,15 @@ func runEdge(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
func popcount64(_ v: UInt64) -> Int { v.nonzeroBitCount }
func runStats(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
func runStats(_ opts: Options, ctx: DatasetContext) -> Bool {
let gpu = ctx.gpu
let log2 = opts.datasetLog2
let mask = UInt32((1 << log2) - 1)
let ds = ctx.source(log2: log2)
let n = 1 << 20
print("\n=== output statistics, 2^20 consecutive nonces per seed, dataset 2^\(log2) words ===")
print("This is a sanity check for obvious structural bias. It is not a proof of cryptographic strength.")
let dataset = makeDataset(gpu, log2: log2, day: day)
let dataset = ctx.makeDataset(log2: log2)
guard let outBuf = gpu.device.makeBuffer(length: n * 8, options: .storageModeShared) else { print("FAIL: out buffer"); return false }
let seeds = [opts.seed, "\(opts.seed)/stats1", "\(opts.seed)/stats2"]
var allOk = true
@ -1716,7 +1946,7 @@ func runStats(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
// Spot check 2 warps against the CPU so the statistics are known to describe the verified function.
var spot = true
for w in [0, (n / 32) - 1] {
let cpu = cpuWarp(program, baseNonce: UInt32(w * 32), day: day, mask: mask)
let cpu = cpuWarp(program, baseNonce: UInt32(w * 32), ds: ds)
if cpu != Array(outs[(w * 32)..<(w * 32 + 32)]) { spot = false }
}
@ -1811,12 +2041,14 @@ func runStats(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
// MARK: - --determinism
func runDeterminism(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
func runDeterminism(_ opts: Options, ctx: DatasetContext) -> Bool {
let gpu = ctx.gpu
let log2 = opts.datasetLog2
let mask = UInt32((1 << log2) - 1)
let ds = ctx.source(log2: log2)
let n = 1 << 20
print("\n=== determinism, seed \"\(opts.seed)\", 2^20 nonces from base 0, dataset 2^\(log2) words ===")
let dataset = makeDataset(gpu, log2: log2, day: day)
let dataset = ctx.makeDataset(log2: log2)
guard let outBuf = gpu.device.makeBuffer(length: n * 8, options: .storageModeShared) else { print("FAIL: out buffer"); return false }
var ok = true
@ -1875,7 +2107,7 @@ func runDeterminism(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
var warps = [0, n / 32 - 1]
while warps.count < 8 { warps.append(vr.below(n / 32)) }
for w in warps {
let cpu = cpuWarp(p1, baseNonce: UInt32(w * 32), day: day, mask: mask)
let cpu = cpuWarp(p1, baseNonce: UInt32(w * 32), ds: ds)
if cpu != Array(reference[(w * 32)..<(w * 32 + 32)]) { cpuBad += 1 }
}
if cpuBad != 0 { ok = false }
@ -1884,7 +2116,7 @@ func runDeterminism(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
// Dataset fill determinism and GPU-vs-CPU agreement of the dataset itself: fill a second buffer, blit both
// to shared memory, fingerprint, and compare sampled words (including 0 and MASK) with datasetElem.
let words = 1 << log2
let dataset2 = makeDataset(gpu, log2: log2, day: day)
let dataset2 = ctx.makeDataset(log2: log2)
var fillFps = [UInt64]()
var sampleBad = 0
if let shared = gpu.device.makeBuffer(length: words * 4, options: .storageModeShared) {
@ -1900,7 +2132,7 @@ func runDeterminism(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
var sr = SplitMix64(s: 0xfeed_beef)
var idxs: [UInt32] = [0, 1, mask - 1, mask]
while idxs.count < 4096 { idxs.append(UInt32(sr.below(words))) }
for i in idxs where dp[Int(i)] != datasetElem(i, day.0, day.1) { sampleBad += 1 }
for i in idxs where dp[Int(i)] != ds.word(i) { sampleBad += 1 }
}
}
let sameFill = fillFps[0] == fillFps[1]
@ -1915,7 +2147,8 @@ func runDeterminism(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
// MARK: - --memcheck
func runMemcheck(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
func runMemcheck(_ opts: Options, ctx: DatasetContext) -> Bool {
let gpu = ctx.gpu
print("\n=== memcheck, seed \"\(opts.seed)\" ===")
var ok = true
let program = generateProgram(seedString: opts.seed)
@ -1937,7 +2170,8 @@ func runMemcheck(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
// buffers, so "no crash" is weak evidence by itself; the static check above is the real guarantee.
let log2 = 20
let mask = UInt32((1 << log2) - 1)
let dataset = makeDataset(gpu, log2: log2, day: day)
let ds = ctx.source(log2: log2)
let dataset = ctx.makeDataset(log2: log2)
let k: CompiledHash
do { k = try compileHash(gpu, msl: generateMSL(program, datasetLog2: log2)) } catch { print("FAIL: compile \(error)"); return false }
let n = 1 << 20
@ -1951,7 +2185,7 @@ func runMemcheck(_ opts: Options, gpu: GPU, day: (UInt32, UInt32)) -> Bool {
guard let gpuOut = gpuWarps(gpu, k, dataset: dataset, bases: bases) else { print("FAIL: GPU warps"); return false }
var overMask = 0, loads = 0
for (w, base) in bases.enumerated() {
let cpu = cpuWarpTraced(program, baseNonce: base, day: day, mask: mask) { _, kk, regs in
let cpu = cpuWarpTraced(program, baseNonce: base, ds: ds) { _, kk, regs in
let ins = program.instrs[kk]
if ins.op == .load { loads += 1; if regs[ins.a] > mask { overMask += 1 } }
}
@ -1970,15 +2204,23 @@ func runTests(_ opts: Options) -> Never {
let gpu = GPU()
print("igneum-bench hardening tests")
print("GPU: \(gpu.device.name) (maxBufferLength \(gpu.device.maxBufferLength / (1 << 20)) MiB, unified memory \(gpu.device.hasUnifiedMemory)), day \"\(opts.day)\"")
let dayWords = seedWords("day/" + opts.day)
let day = (dayWords[0], dayWords[1])
let ctx = DatasetContext(gpu: gpu, closedForm: opts.closedForm, dayString: opts.day)
print("dataset construction: \(ctx.modeName)" + (ctx.closed ? "" : "; cache GPU fill \(fmt(ctx.cacheFillGPUms, 2)) ms GPU time"))
if generatorConfig.loadWeight != 25 || generatorConfig.wideFrac != 0 {
print("generator levers: load weight \(generatorConfig.loadWeight), wide fraction \(generatorConfig.wideFrac) percent (NOT the default generator)")
}
var results = [(String, Bool)]()
let t0 = nowNs()
if opts.fuzz != nil { results.append(("fuzz", runFuzz(opts, gpu: gpu, day: day))) }
if opts.edge { results.append(("edge", runEdge(opts, gpu: gpu, day: day))) }
if opts.stats { results.append(("stats", runStats(opts, gpu: gpu, day: day))) }
if opts.determinism { results.append(("determinism", runDeterminism(opts, gpu: gpu, day: day))) }
if opts.memcheck { results.append(("memcheck", runMemcheck(opts, gpu: gpu, day: day))) }
if !ctx.closed {
let cc = ctx.cacheCheck()
print(cc.detail)
results.append(("cache", cc.ok))
}
if opts.fuzz != nil { results.append(("fuzz", runFuzz(opts, ctx: ctx))) }
if opts.edge { results.append(("edge", runEdge(opts, ctx: ctx))) }
if opts.stats { results.append(("stats", runStats(opts, ctx: ctx))) }
if opts.determinism { results.append(("determinism", runDeterminism(opts, ctx: ctx))) }
if opts.memcheck { results.append(("memcheck", runMemcheck(opts, ctx: ctx))) }
print("\n=== tests summary (\(fmt(Double(nowNs() - t0) / 1e9, 1)) s) ===")
for (name, ok) in results { print("\(pad(name, 12)) \(ok ? "PASS" : "FAIL")") }
let all = results.allSatisfy { $0.1 }
@ -1989,64 +2231,69 @@ func runTests(_ opts: Options) -> Never {
// MARK: - Main
let opts = parseArgs()
generatorConfig = GeneratorConfig(loadWeight: opts.loadWeight, wideFrac: opts.wideFrac)
if opts.exportPack != nil { exportPack(opts) }
if opts.anyTest { runTests(opts) }
let gpu = GPU()
print("igneum-bench")
print("GPU: \(gpu.device.name) (maxBufferLength \(gpu.device.maxBufferLength / (1 << 20)) MiB, unified memory \(gpu.device.hasUnifiedMemory))")
print("dataset: 2^\(opts.datasetLog2) uint32 = \(fmt(Double(1 << opts.datasetLog2) * 4 / Double(1 << 20), 0)) MiB, day \"\(opts.day)\"")
print("dataset: 2^\(opts.datasetLog2) uint32 = \(fmt(Double(1 << opts.datasetLog2) * 4 / Double(1 << 20), 0)) MiB, day \"\(opts.day)\", construction \(opts.closedForm ? "closed-form (--closed-form)" : "memory-hard (default; MEMHARD.md)")")
if generatorConfig.loadWeight != 25 || generatorConfig.wideFrac != 0 {
print("generator levers: load weight \(generatorConfig.loadWeight) percent, wide-load fraction \(generatorConfig.wideFrac) percent (NOT the default generator)")
print("generator weights: " + generatorConfig.weights.map { "\($0.0.rawValue)=\($0.1)" }.joined(separator: " "))
}
let dayWords = seedWords("day/" + opts.day)
let day = (dayWords[0], dayWords[1])
// Dataset construction, timed. Memory-hard: compile the cache fill + build kernels, fill the 256 MiB cache on the
// GPU, build the dataset from it. Closed form: the original fill kernel. The CPU cache (verifier side) is computed
// afterwards on one core and compared word for word with the GPU cache.
let ctx = DatasetContext(gpu: gpu, closedForm: opts.closedForm, dayString: opts.day)
let datasetWords = 1 << opts.datasetLog2
guard let dataset = gpu.device.makeBuffer(length: datasetWords * 4, options: .storageModePrivate) else {
print("FAIL: cannot allocate dataset buffer"); exit(1)
}
// Fill the dataset on the GPU, timed.
var fillMsWall = 0.0, fillMsGPU = 0.0
do {
let f0 = nowNs()
let lib = try gpu.device.makeLibrary(source: fillMSL, options: MTLCompileOptions())
let pipe = try gpu.device.makeComputePipelineState(function: lib.makeFunction(name: "igneum_fill")!)
let f1 = nowNs()
let cb = gpu.queue.makeCommandBuffer()!
let enc = cb.makeComputeCommandEncoder()!
enc.setComputePipelineState(pipe)
enc.setBuffer(dataset, offset: 0, index: 0)
var d = (day.0, day.1)
enc.setBytes(&d, length: 8, index: 1)
let tg = 256
enc.dispatchThreadgroups(MTLSize(width: datasetWords / tg, height: 1, depth: 1),
threadsPerThreadgroup: MTLSize(width: tg, height: 1, depth: 1))
enc.endEncoding()
let f2 = nowNs()
cb.commit(); cb.waitUntilCompleted()
let f3 = nowNs()
if let e = cb.error { print("FAIL: fill error \(e)"); exit(1) }
fillMsWall = ms(f2, f3)
fillMsGPU = (cb.gpuEndTime - cb.gpuStartTime) * 1000
let gib = Double(datasetWords * 4) / Double(1 << 30)
print("dataset fill: compile \(fmt(ms(f0, f1))) ms; fill \(fmt(fillMsWall)) ms wall, \(fmt(fillMsGPU)) ms GPU -> \(fmt(gib / (fillMsGPU / 1000))) GB/s write (GPU time)")
} catch {
print("FAIL: fill kernel: \(error)"); exit(1)
print("dataset kernels: compile \(fmt(ctx.compileMs)) ms")
var cacheRefillMs = 0.0
if !ctx.closed {
cacheRefillMs = ctx.refillCache()
print("cache fill (GPU): \(fmt(ctx.cacheFillGPUms)) ms GPU time first (\(fmt(ctx.cacheFillWallMs)) ms wall), \(fmt(cacheRefillMs)) ms GPU time second; \(cacheSegments) chains x \(cacheLinesPerSegment) ChaCha\(chachaRounds) blocks, 256 MiB written")
}
ctx.build(into: dataset, words: datasetWords)
let build1GPU = ctx.lastBuildGPUms, build1Wall = ctx.lastBuildWallMs
ctx.build(into: dataset, words: datasetWords)
let build2GPU = ctx.lastBuildGPUms
let gib = Double(datasetWords * 4) / Double(1 << 30)
if ctx.closed {
print("dataset fill (closed form): \(fmt(build1GPU)) ms GPU time first (\(fmt(build1Wall)) ms wall), \(fmt(build2GPU)) ms second -> \(fmt(gib / (build2GPU / 1000))) GB/s write (second, GPU time)")
} else {
let items = Double(datasetWords / 16)
print("dataset build (memory-hard): \(fmt(build1GPU)) ms GPU time first (\(fmt(build1Wall)) ms wall), \(fmt(build2GPU)) ms second; \(Int(items)) items, \(fmt(items / (build2GPU / 1000) / 1e6, 1)) M items/s, \(fmt(items * 8 / (build2GPU / 1000) / 1e9, 2)) G cache-line reads/s (second)")
let cc = ctx.cacheCheck()
print(cc.detail)
if !cc.ok { print("FAIL: GPU and CPU cache differ"); exit(1) }
let sc = ctx.sampleCheck(log2: min(opts.datasetLog2, 24), indices: [0, 1, 15, 16, UInt32((1 << min(opts.datasetLog2, 24)) - 1)] + (0..<1019).map { _ in UInt32.random(in: 0..<UInt32(1 << min(opts.datasetLog2, 24))) })
print(sc.detail)
if !sc.ok { print("FAIL: GPU dataset words differ from the CPU derivation"); exit(1) }
}
var results = [EpochResult]()
for epoch in 0..<max(opts.hours, 1) {
let seedString = epoch == 0 ? opts.seed : "\(opts.seed)/epoch\(epoch)"
results.append(runEpoch(gpu: gpu, opts: opts, seedString: seedString, dataset: dataset, day: day))
results.append(runEpoch(gpu: gpu, opts: opts, seedString: seedString, dataset: dataset, ctx: ctx))
}
// Summary table
print("\n=== summary (\(gpu.device.name), dataset 2^\(opts.datasetLog2) words, batch 2^\(opts.batchLog2) x \(opts.batches)) ===")
print("| seed | compile ms (lib+pipe) | Mhash/s (wall) | GB/s useful (wall) | loads/hash | CPU verify ms/warp (avg) | verify |")
print("|---|---|---|---|---|---|---|")
print("\n=== summary (\(gpu.device.name), dataset 2^\(opts.datasetLog2) words \(ctx.modeName)\(opts.inlineDataset ? " INLINE shortcut kernel" : ""), batch 2^\(opts.batchLog2) x \(opts.batches)) ===")
print("| seed | compile ms (lib+pipe) | Mhash/s (wall) | Mhash/s (GPU) | GB/s useful (wall) | loads/hash | items/warp | CPU verify ms/warp (avg of 20) | verify |")
print("|---|---|---|---|---|---|---|---|---|")
for r in results {
let avg = r.verify.map { $0.repMs }.reduce(0, +) / Double(max(r.verify.count, 1))
print("| \(r.seed) | \(fmt(r.libraryMs + r.pipelineMs, 1)) | \(fmt(r.hashesPerSecWall / 1e6, 3)) | \(fmt(r.gbpsWall)) | \(r.loadsPerHash) | \(fmt(avg, 3)) | \(r.allPass ? "PASS" : "FAIL") (\(r.verify.count) warps) |")
print("| \(r.seed) | \(fmt(r.libraryMs + r.pipelineMs, 1)) | \(fmt(r.hashesPerSecWall / 1e6, 3)) | \(fmt(r.hashesPerSecGPU / 1e6, 3)) | \(fmt(r.gbpsWall)) | \(r.loadsPerHash) | \(r.itemsPerWarp) | \(fmt(avg, 3)) | \(r.allPass ? "PASS" : "FAIL") (\(r.verify.count) warps) |")
}
let overall = results.allSatisfy { $0.allPass }
print("dataset fill: \(fmt(fillMsGPU)) ms GPU time for \(datasetWords * 4 / (1 << 20)) MiB")
if ctx.closed {
print("dataset fill: \(fmt(build2GPU)) ms GPU time for \(datasetWords * 4 / (1 << 20)) MiB (closed form)")
} else {
print("cache fill: \(fmt(cacheRefillMs)) ms GPU, \(fmt(ctx.cpuSide()!.fillMs, 1)) ms one CPU core; dataset build: \(fmt(build2GPU)) ms GPU for \(datasetWords * 4 / (1 << 20)) MiB (memory-hard)")
}
print("OVERALL: \(overall ? "PASS" : "FAIL")")
exit(overall ? 0 : 1)

Binary file not shown.

996
sim/finality_v2.py Normal file
View file

@ -0,0 +1,996 @@
#!/usr/bin/env python3
"""Igneum finality rule V2: checkpoint-level simulation with latency, partitions and eclipses.
Rule under test (CLAUDE.md, FINALITY RULE V2, review round 2, 3 October 2026):
* Vote weight of a key = its blue blocks over a flat trailing 30-day window (DAA time,
86,400 blocks a day). No damping. Dust threshold DUST blocks: a key below it is not a
voter and is not counted in any denominator.
* A checkpoint forms every 30 blocks (blue score 30i). Every voter signs every checkpoint
it sees while online. Aggregators collect votes; a certificate (a lock) exists once the
collected signatures reach QUORUM (2/3) of the denominator.
* ACTIVE denominator: sum over eligible keys of weight x participation, where participation
= min(1, certified votes of that key over the last PRESENCE (240) checkpoint indices / 240).
A key whose first block is less than 240 checkpoints old counts as participation 1.
* TOTAL denominator (the alternative): sum of weight over every eligible key, no factor.
* Equivocation (one key signing two different checkpoint blocks at one index) zeroes the
key's weight for the rest of the window once the evidence is seen.
Participation bookkeeping has three readings, selected with --pmode. The difference only shows
when a checkpoint index gets no certificate:
cert (the brief, literal) an uncertified index contributes 0 certified votes to EVERY key,
so a stall decays everyone's participation and the denominator shrinks until the
present signers reach 2/3 of it. Self-healing, and the partition hazard.
seen an uncertified index credits the keys whose votes the node observed on the wire.
Silent keys decay, present keys do not. Not objective: each node counts its own view.
frozen the window is over the last 240 CERTIFIED indices, so a stall freezes participation.
Fails safe and never recovers liveness within the window.
Model (all assumptions, repeated in results_v2.md):
* Time step = one 30-s slot. The network mines Poisson(30) blocks per slot, split per key by
hashrate share (perfect difficulty retarget). Every block is blue (GHOSTDAG abstracted).
* Weight window = 720 hourly buckets of blocks per key (30 days), rolled every hour.
* A checkpoint index forms on a side each time that side's blue score passes a multiple of
30, so a partition side mining a fraction s of the hashrate forms checkpoints at s per slot.
Checkpoint blocks on different sides are different blocks at the same index.
* Regions: a one-way delay matrix, DELAY between regions, INTRA inside one, lognormal jitter
per hop. A vote for checkpoint i issued by a key in region r reaches the aggregator in
region a at t0 + d(src, r) + d(r, a) (+ a per-key extra delay for an eclipsed key).
One aggregator per region on the side; the certificate forms at the first one to reach
quorum; its signer set is every vote that arrived there by max(threshold time, t0 + GRACE).
* Honest keys follow a two-state uptime chain: availability UPTIME (UPTIME_BIG for keys at or
above 1% of hashrate, a pool with redundant infrastructure), mean outage OUTAGE slots.
* A partition splits regions into sides, each with its own view (certificates, participation
ring, blue score). At the heal the views merge: certificates are unioned, two certificates
at one index with different blocks count as a CONFLICTING LOCK, participation rings are
OR-merged by index, and any key that signed on two sides at a common index is stripped.
* Weights are global (a side does not see the other side's blocks for the partition's
duration; at most 150 minutes of a 30-day window, ignored).
Standard library plus numpy. Deterministic for a given --seed.
Usage:
python3 finality_v2.py # every scenario, markdown on stdout
python3 finality_v2.py --scenarios A,E --delay 5
python3 finality_v2.py --quick # shortened runs for development
"""
import argparse
import sys
import time
import numpy as np
SLOT_S = 30.0
BLOCKS_PER_CP = 30
SLOTS_PER_HOUR = 120
SLOTS_PER_DAY = 2_880
WINDOW_HOURS = 720
WINDOW_BLOCKS = 86_400 * 30
N_HONEST = 1_000
PARETO_SHAPE = 1.0
GEOGRAPHY = (0.45, 0.35, 0.20) # honest hashrate per region, model assumption
TWO_THIRDS = 2.0 / 3.0
class P:
"""Parameters. Keyword overrides."""
def __init__(self, **kw):
self.delay = 2.0 # one-way inter-region delay, seconds
self.intra = 0.1 # one-way intra-region delay, seconds
self.jitter = 0.25 # lognormal sigma per hop
self.grace = 15.0 # seconds after t0 during which late votes still enter the certificate
self.uptime = 0.97 # availability of an honest key under 1% of hashrate
self.uptime_big = 0.995 # availability of a key at or above 1% of hashrate (redundant pool infrastructure)
self.big_share = 0.01
self.outage = 20 # mean outage length, slots (10 minutes)
self.denom = "active" # "active" or "total"
self.pmode = "cert" # "cert", "seen", "frozen"
self.presence = 240 # checkpoints in the participation window
self.dust = 100 # blocks
self.quorum = TWO_THIRDS
self.__dict__.update(kw)
def label(self):
return self.denom if self.denom == "total" else "active/" + self.pmode
class View:
"""One side's view: participation ring, checkpoint counter, certificates."""
def __init__(self, sid, regions, n, presence, next_idx):
self.sid = sid
self.regions = list(regions)
self.ring = np.zeros((presence, n), dtype=np.int8)
self.ring_idx = np.full(presence, -1, dtype=np.int64)
self.pcount = np.zeros(n, dtype=np.int32)
self.next_idx = int(next_idx)
self.blue = 0.0
self.certs = {} # idx -> (sid, slot, latency_s)
self.stalls = [] # (idx, slot)
self.key_mask = None # keys whose region is on this side
self.mine_mask = None
def clone_ring_from(self, other):
self.ring[:] = other.ring
self.ring_idx[:] = other.ring_idx
self.pcount[:] = other.pcount
class Sim:
def __init__(self, p, rng, n_regions=3):
self.p = p
self.rng = rng
self.R = int(n_regions)
self.D = np.full((self.R, self.R), p.delay)
np.fill_diagonal(self.D, p.intra)
self.N = 0
self.hash = np.zeros(0)
self.region = np.zeros(0, dtype=np.int64)
self.online = np.zeros(0, dtype=bool)
self.flaky = np.zeros(0, dtype=bool)
self.signs = np.zeros(0, dtype=bool)
self.equiv = np.zeros(0, dtype=bool)
self.stripped = np.zeros(0, dtype=bool)
self.extra_delay = np.zeros(0)
self.first_idx = np.zeros(0, dtype=np.int64)
self.buckets = np.zeros((WINDOW_HOURS, 0))
self.weight = np.zeros(0)
self.slot = 0
self.views = []
self.next_sid = 1
self.records = [] # (slot, idx, sid, latency_s or -1, signed/denom, denom/total)
self.conflicts = [] # (idx, slot the second certificate existed)
self.q_on = 1.0 / p.outage
self.q_off = np.zeros(0)
# ------------------------------------------------------------- keys
def add_keys(self, hashes, regions, flaky=True, equiv=False, signs=True):
hashes = np.asarray(hashes, dtype=float)
regions = np.asarray(regions, dtype=np.int64)
n = hashes.size
self.hash = np.concatenate([self.hash, hashes])
self.region = np.concatenate([self.region, regions])
self.online = np.concatenate([self.online, np.ones(n, dtype=bool)])
self.flaky = np.concatenate([self.flaky, np.full(n, flaky, dtype=bool)])
self.signs = np.concatenate([self.signs, np.full(n, signs, dtype=bool)])
self.equiv = np.concatenate([self.equiv, np.full(n, equiv, dtype=bool)])
self.stripped = np.concatenate([self.stripped, np.zeros(n, dtype=bool)])
self.extra_delay = np.concatenate([self.extra_delay, np.zeros(n)])
self.first_idx = np.concatenate([self.first_idx, np.full(n, -1, dtype=np.int64)])
self.buckets = np.concatenate([self.buckets, np.zeros((WINDOW_HOURS, n))], axis=1)
self.weight = np.concatenate([self.weight, np.zeros(n)])
first = self.N
self.N += n
self.q_off = np.concatenate([self.q_off, np.zeros(n)])
self.set_uptime()
return np.arange(first, self.N)
def set_uptime(self):
"""Per-key off-switch probability per slot from the size-dependent availability. Call after hashrate changes."""
tot = self.hash.sum()
share = self.hash / tot if tot > 0 else np.zeros(self.N)
u = np.where(share >= self.p.big_share, self.p.uptime_big, self.p.uptime)
self.q_off = self.q_on * (1.0 - u) / u
def warm_start(self):
"""Fill the 30-day window as if the mining keys had been mining at their share for 30 days."""
share = self.hash / self.hash.sum()
lam = 3_600.0 * share
for h in range(WINDOW_HOURS):
self.buckets[h] = self.rng.poisson(lam)
self.weight = self.buckets.sum(axis=0)
self.first_idx[self.hash > 0] = -10 ** 9
self.slot = 0
def init_views(self, warm):
v = View(0, range(self.R), self.N, self.p.presence, next_idx=(WINDOW_BLOCKS // BLOCKS_PER_CP if warm else 0))
self._set_masks(v)
if warm:
elig = (self.weight >= self.p.dust) & self.signs
v.ring[:] = elig.astype(np.int8)[None, :]
v.ring_idx[:] = np.arange(v.next_idx - self.p.presence, v.next_idx)
v.pcount[:] = v.ring.sum(axis=0)
self.views = [v]
self.next_sid = 1
def _set_masks(self, v):
v.key_mask = np.isin(self.region, v.regions)
v.mine_mask = v.key_mask.copy()
# ------------------------------------------------------------- partitions
def split(self, groups):
base = self.views[0]
new = []
for g in groups:
v = View(self.next_sid, g, self.N, self.p.presence, next_idx=base.next_idx)
self.next_sid += 1
v.clone_ring_from(base)
self._set_masks(v)
new.append(v)
self.split_idx = base.next_idx
self.views = new
def heal(self):
views = self.views
P_ = self.p.presence
nxt = max(v.next_idx for v in views)
m = View(self.next_sid, range(self.R), self.N, P_, next_idx=nxt)
self.next_sid += 1
self._set_masks(m)
for i in range(max(0, nxt - P_), nxt):
row = i % P_
acc = np.zeros(self.N, dtype=np.int8)
hit = False
for v in views:
if v.ring_idx[row] == i:
acc |= v.ring[row]
hit = True
if hit:
m.ring[row] = acc
m.ring_idx[row] = i
m.pcount[:] = m.ring.sum(axis=0)
# certificates and conflicts
for v in views:
for idx, (sid, slot, lat) in v.certs.items():
if idx in m.certs and m.certs[idx][0] != sid:
self.conflicts.append((idx, max(slot, m.certs[idx][1])))
else:
m.certs.setdefault(idx, (sid, slot, lat))
m.stalls.extend(v.stalls)
# equivocation evidence: an equivocating key signed every side's checkpoint at every common index
common = min(v.next_idx for v in views) > self.split_idx
if common and self.equiv.any():
self.stripped |= self.equiv
self.strip_slot = self.slot
self.views = [m]
# ------------------------------------------------------------- stepping
def run(self, n_slots, snap=None):
"""snap = (every_n_slots, fn(sim)) called before the step on matching slots."""
for _ in range(int(n_slots)):
if snap is not None and self.slot % snap[0] == 0:
snap[1](self)
self.step()
def step(self):
p = self.p
slot = self.slot
tot = self.hash.sum()
blocks = self.rng.poisson(BLOCKS_PER_CP * self.hash / tot) if tot > 0 else np.zeros(self.N)
hp = (slot // SLOTS_PER_HOUR) % WINDOW_HOURS
if slot % SLOTS_PER_HOUR == 0:
self.weight -= self.buckets[hp]
self.buckets[hp] = 0.0
self.buckets[hp] += blocks
self.weight += blocks
gidx = max(v.next_idx for v in self.views)
newly = (blocks > 0) & (self.first_idx == -1)
if newly.any():
self.first_idx[newly] = gidx
r = self.rng.random(self.N)
flip = np.where(self.online, r < self.q_off, r < self.q_on) & self.flaky
self.online ^= flip
for v in self.views:
v.blue += blocks[v.mine_mask].sum()
while v.blue >= BLOCKS_PER_CP:
v.blue -= BLOCKS_PER_CP
self.checkpoint(v, blocks)
self.slot += 1
def participation(self, v, idx):
part = np.minimum(1.0, v.pcount / float(self.p.presence))
new = (self.first_idx > idx - self.p.presence) & (self.first_idx >= 0)
part[new] = 1.0
return part
def checkpoint(self, v, blocks):
p = self.p
idx = v.next_idx
v.next_idx += 1
t0 = self.slot * SLOT_S
# miner of the checkpoint block: a key on this side weighted by its blocks this slot
bm = blocks * v.mine_mask
cum = np.cumsum(bm)
if cum[-1] > 0:
j = int(np.searchsorted(cum, self.rng.random() * cum[-1], side="right"))
src = int(self.region[min(j, self.N - 1)])
else:
src = v.regions[0]
jit1 = np.exp(self.rng.normal(0.0, p.jitter, self.R))
jit2 = np.exp(self.rng.normal(0.0, p.jitter, (self.R, self.R)))
elig = (self.weight >= p.dust) & ~self.stripped
voters = elig & self.online & self.signs & (v.key_mask | self.equiv)
if p.denom == "active":
denom = float((self.weight * self.participation(v, idx))[elig].sum())
else:
denom = float(self.weight[elig].sum())
total = float(self.weight[elig].sum())
need = p.quorum * denom
vi = np.flatnonzero(voters)
signed_w = float(self.weight[vi].sum())
ratio = signed_w / denom if denom > 0 else 0.0
best = None
if denom > 0 and vi.size > 0:
w = self.weight[vi]
reg = self.region[vi]
hop1 = np.where(self.equiv[vi], p.intra, self.D[src, reg] * jit1[reg])
issue = t0 + hop1 + self.extra_delay[vi]
for a in v.regions:
hop2 = np.where(self.equiv[vi], p.intra, self.D[reg, a] * jit2[reg, a])
arr = issue + hop2
order = np.argsort(arr, kind="stable")
cw = np.cumsum(w[order])
k = int(np.searchsorted(cw, need))
if k < cw.size:
thr = float(arr[order[k]])
if best is None or thr < best[0]:
best = (thr, arr)
if best is not None:
thr, arr = best
lat = thr - t0
seal = max(thr, t0 + p.grace)
mask = np.zeros(self.N, dtype=np.int8)
mask[vi[arr <= seal]] = 1
v.certs[idx] = (v.sid, self.slot, lat)
self._push(v, idx, mask)
self.records.append((self.slot, idx, v.sid, lat, ratio, denom / total if total > 0 else 0.0))
else:
v.stalls.append((idx, self.slot))
if p.pmode == "cert":
self._push(v, idx, np.zeros(self.N, dtype=np.int8))
elif p.pmode == "seen":
self._push(v, idx, voters.astype(np.int8))
# frozen: no ring update
self.records.append((self.slot, idx, v.sid, -1.0, ratio, denom / total if total > 0 else 0.0))
def _push(self, v, idx, mask):
row = idx % self.p.presence
v.pcount -= v.ring[row]
v.ring[row] = mask
v.pcount += mask
v.ring_idx[row] = idx
# ------------------------------------------------------------- queries
def recs(self):
return np.array(self.records, dtype=float).reshape(-1, 6)
def elig_mask(self):
return (self.weight >= self.p.dust) & ~self.stripped
def share(self, keys):
e = self.elig_mask()
tot = self.weight[e].sum()
if tot <= 0:
return 0.0
m = np.zeros(self.N, dtype=bool)
m[keys] = True
return float(self.weight[m & e].sum() / tot)
# ---------------------------------------------------------------- helpers
def pareto_hashrates(rng, n, total=1.0):
h = rng.pareto(PARETO_SHAPE, n) + 1.0
return h * (total / h.sum())
def assign_regions(hashes, targets):
"""Largest key first, each to the region with the largest remaining hashrate deficit."""
targets = np.asarray(targets, dtype=float)
tot = hashes.sum()
filled = np.zeros(targets.size)
reg = np.zeros(hashes.size, dtype=np.int64)
for i in np.argsort(-hashes):
r = int(np.argmax(targets * tot - filled))
reg[i] = r
filled[r] += hashes[i]
return reg
def pick_weight_subset(rng, weights, frac):
"""Random keys whose weight sums to about frac of total, never over."""
target = frac * weights.sum()
mask = np.zeros(weights.size, dtype=bool)
acc = 0.0
for i in rng.permutation(weights.size):
if acc + weights[i] <= target:
mask[i] = True
acc += weights[i]
return mask, acc / weights.sum()
def gini(x):
x = np.sort(np.asarray(x, dtype=float))
n = x.size
if n == 0 or x.sum() == 0:
return 0.0
cum = np.cumsum(x)
return (n + 1 - 2.0 * cum.sum() / cum[-1]) / n
def pct(x, d=1):
return ("%%.%df%%%%" % d) % (100.0 * x)
def md_table(headers, rows):
out = ["| " + " | ".join(str(h) for h in headers) + " |", "|" + "---|" * len(headers)]
for r in rows:
out.append("| " + " | ".join(str(c) for c in r) + " |")
return "\n".join(out)
def lat_stats(recs, s_from=0, s_to=None):
if s_to is None:
s_to = np.inf
sel = recs[(recs[:, 0] >= s_from) & (recs[:, 0] < s_to)]
lat = sel[:, 3]
ok = lat >= 0
n = int(sel.shape[0])
st = int((~ok).sum())
cps = np.floor(lat[ok] / SLOT_S)
d = dict(n=n, stalls=st, c0=int((cps == 0).sum()), c1=int((cps == 1).sum()), c2=int((cps >= 2).sum()))
if ok.any():
d.update(med=float(np.median(lat[ok])), p99=float(np.percentile(lat[ok], 99)), mx=float(lat[ok].max()),
ratio_min=float(sel[ok, 4].min()), ratio_med=float(np.median(sel[ok, 4])))
else:
d.update(med=float("nan"), p99=float("nan"), mx=float("nan"), ratio_min=float("nan"), ratio_med=float("nan"))
return d
def lat_row(label, d):
return [label, d["n"], d["c0"], d["c1"], d["c2"], d["stalls"], "%.1f" % d["med"], "%.1f" % d["p99"], "%.1f" % d["mx"],
"%.3f" % d["ratio_min"]]
LAT_HEADERS = ["run", "checkpoints", "locked in slot 0", "slot 1", "slot 2+", "stalled", "median s", "p99 s", "max s",
"min signed/denominator"]
def first_lock_after(recs, slot, sid=None):
sel = recs[(recs[:, 0] >= slot) & (recs[:, 3] >= 0)]
if sid is not None:
sel = sel[sel[:, 2] == sid]
if sel.shape[0] == 0:
return None
return int(sel[0, 0])
def stalls_between(recs, s_from, s_to, sid=None):
sel = recs[(recs[:, 0] >= s_from) & (recs[:, 0] < s_to) & (recs[:, 3] < 0)]
if sid is not None:
sel = sel[sel[:, 2] == sid]
return int(sel.shape[0])
def fmt_min(slots):
if slots is None:
return "never"
m = slots * SLOT_S / 60.0
if m < 120:
return "%.0f min" % m
if m < 48 * 60:
return "%.1f h" % (m / 60.0)
return "%.1f d" % (m / 1440.0)
def fmt_days(slots):
return "never" if slots is None else "%.1f" % (slots / SLOTS_PER_DAY)
def build_honest(p, seed, n_regions=3, geography=GEOGRAPHY, big_pool=None):
rng = np.random.default_rng(seed)
sim = Sim(p, rng, n_regions=n_regions)
if big_pool is None:
h = pareto_hashrates(rng, N_HONEST)
reg = assign_regions(h, geography)
sim.add_keys(h, reg, flaky=True)
pool = None
else:
h = pareto_hashrates(rng, N_HONEST - 1, total=1.0 - big_pool)
reg = assign_regions(h, geography)
sim.add_keys(h, reg, flaky=True)
pool = int(sim.add_keys([big_pool], [3], flaky=False)[0])
return sim, rng, pool
# ---------------------------------------------------------------- scenarios
def scenario_a(args):
out = ["### A. Steady state from zero history, 1,000 honest keys, 3 regions %s, %d days" % (
"/".join(pct(g, 0) for g in GEOGRAPHY), args.days_a), ""]
days = args.days_a
snaps = {}
lat_rows = []
prop_rows = []
for denom in ("active", "total"):
p = P(denom=denom, delay=args.delay)
sim, rng, _ = build_honest(p, args.seed)
sim.init_views(warm=False)
series = []
def snap(s, series=series):
e = s.elig_mask()
series.append((s.slot // SLOTS_PER_DAY, s.weight.sum() / WINDOW_BLOCKS, int((~e).sum())))
sim.run(days * SLOTS_PER_DAY, snap=(SLOTS_PER_DAY, snap))
snap(sim)
recs = sim.recs()
snaps[denom] = series
lat_rows.append(lat_row("%s, days 0 to 1" % denom, lat_stats(recs, 0, SLOTS_PER_DAY)))
lat_rows.append(lat_row("%s, days 1 to 30" % denom, lat_stats(recs, SLOTS_PER_DAY, 30 * SLOTS_PER_DAY)))
lat_rows.append(lat_row("%s, days 30 to %d" % (denom, days), lat_stats(recs, 30 * SLOTS_PER_DAY, None)))
w = sim.weight
ws = w / w.sum()
hs = sim.hash / sim.hash.sum()
order = np.argsort(-hs)
rel = ws / hs
prop_rows.append([denom, "%.5f" % np.corrcoef(hs, ws)[0, 1], "%.3f / %.3f" % (gini(hs), gini(ws)),
pct(hs[order[0]]) + " / " + pct(ws[order[0]]),
pct(hs[order[:10]].sum()) + " / " + pct(ws[order[:10]].sum()),
pct(hs[order[500:]].sum()) + " / " + pct(ws[order[500:]].sum()),
"%.3f / %.3f" % (rel.min(), rel.max()), int((rel < 0.9).sum()), int((~sim.elig_mask()).sum())])
# genesis stall run
first = first_lock_after(recs, 0)
snaps[denom + "_first"] = first
s = snaps["active"]
ramp = []
for d in (1, 2, 5, 10, 20, 30, 31, 45, days):
row = [x for x in s if x[0] == d]
if row:
ramp.append([d, pct(row[0][1]), row[0][2]])
out.append("Weight ramp (active run; the total run mines the same blocks):")
out.append("")
out.append(md_table(["day", "total weight / full window", "keys under dust (100 blocks)"], ramp))
out.append("")
out.append("Weight against hashrate at day %d:" % days)
out.append("")
out.append(md_table(["denominator", "corr(hash, weight)", "Gini hash / weight", "top-1 hash / weight",
"top-10 hash / weight", "bottom-500 hash / weight", "min / max weight:hash", "keys under 0.9x",
"dust keys"], prop_rows))
out.append("")
out.append("Lock latency (seconds from checkpoint block to quorum; slot = 30 s), inter-region delay %.1f s:" % args.delay)
out.append("")
out.append(md_table(LAT_HEADERS, lat_rows))
out.append("")
out.append("First lock from genesis: active at slot %s, total at slot %s (the first checkpoints have no key above dust)." % (
snaps["active_first"], snaps["total_first"]))
out.append("")
# delay comparison, short warm-started runs
rows = []
for delay in (0.5, 2.0, 5.0):
for grace in (args.grace,):
p = P(denom="active", delay=delay, grace=grace)
sim, rng, _ = build_honest(p, args.seed)
sim.warm_start()
sim.init_views(warm=True)
sim.run(args.days_delay * SLOTS_PER_DAY)
recs = sim.recs()
d = lat_stats(recs)
# participation of the slowest region
v = sim.views[0]
part = sim.participation(v, v.next_idx)
by_region = ["%.3f" % np.average(part[sim.region == r], weights=sim.weight[sim.region == r]) for r in range(3)]
rows.append(lat_row("delay %.1f s, grace %.0f s, %d days warm" % (delay, grace, args.days_delay), d) + ["/".join(by_region)])
out.append("Delay sweep, warm-started (steady-state weights), active denominator. Last column: weight-averaged participation per region.")
out.append("")
out.append(md_table(LAT_HEADERS + ["participation r0/r1/r2"], rows))
return "\n".join(out)
def scenario_b(args):
out = ["### B. Rental burst at day 60, one public key with a x honest hashrate, signs every checkpoint", ""]
mults = (1, 2, 4, 9)
series = {}
cross = {}
for a in mults:
p = P(denom="active", delay=args.delay)
sim, rng, _ = build_honest(p, args.seed, n_regions=4)
att = int(sim.add_keys([0.0], [3], flaky=False)[0])
sim.warm_start()
sim.init_views(warm=True)
sim.run(SLOTS_PER_HOUR) # one hour of baseline
t_event = sim.slot
sim.hash[att] = float(a)
s = []
c13 = c23 = None
for day in range(1, args.days_b + 1):
for _ in range(SLOTS_PER_DAY // 24):
sim.run(24)
sh = sim.share([att])
if c13 is None and sh >= 1.0 / 3.0:
c13 = sim.slot - t_event
if c23 is None and sh >= TWO_THIRDS:
c23 = sim.slot - t_event
s.append((day, sim.share([att])))
recs = sim.recs()
series[a] = s
cross[a] = (c13, c23, lat_stats(recs, t_event, None))
pick = [d for d in (1, 5, 10, 15, 20, 25, 30, 35) if d <= args.days_b]
rows = []
for d in pick:
row = ["+%d" % d]
for a in mults:
sim_v = dict(series[a])[d]
formula = min(d / 30.0, 1.0) * a / (1.0 + a)
row.append("%s / %s" % (pct(sim_v), pct(formula)))
rows.append(row)
out.append("Attacker weight share, simulated / formula (t/30) x a/(1+a):")
out.append("")
out.append(md_table(["day after burst"] + ["a=%d (%s of hashrate)" % (a, pct(a / (1.0 + a), 0)) for a in mults], rows))
out.append("")
ev = [
["crosses 1/3 (honest alone can no longer lock), sim day"] + [fmt_days(cross[a][0]) for a in mults],
["crosses 1/3, formula 10(1+a)/a"] + ["%.1f" % (10.0 * (1 + a) / a) for a in mults],
["crosses 2/3 (locks alone), sim day"] + [fmt_days(cross[a][1]) for a in mults],
["crosses 2/3, formula 20(1+a)/a"] + ["%.1f" % (20.0 * (1 + a) / a) if 20.0 * (1 + a) / a <= 30 else "never (ceiling %s)" % pct(a / (1 + a), 0) for a in mults],
["max abs deviation sim vs formula, days 1 to 30, points"] + [
"%.2f" % (100 * max(abs(v - min(d / 30.0, 1.0) * a / (1.0 + a)) for d, v in series[a] if d <= 30)) for a in mults],
["stalled checkpoints after the burst"] + [cross[a][2]["stalls"] for a in mults],
]
out.append(md_table(["event"] + ["a=%d" % a for a in mults], ev))
return "\n".join(out)
def scenario_c(args):
out = ["### C. Silent set: a random set holding x of weight stops signing at day 60 and keeps mining", ""]
fracs = (0.34, 0.40, 0.45, 0.50, 0.55)
configs = [("active", "cert", args.hours_c), ("active", "seen", args.hours_c), ("active", "frozen", args.hours_c),
("total", "cert", args.hours_c_total)]
rows = []
detail = []
for frac in fracs:
row = [pct(frac, 0)]
for denom, pmode, hours in configs:
p = P(denom=denom, pmode=pmode, delay=args.delay)
sim, rng, _ = build_honest(p, args.seed)
sim.warm_start()
sim.init_views(warm=True)
sim.run(SLOTS_PER_HOUR)
silent, got = pick_weight_subset(rng, sim.weight, frac)
sim.signs[silent] = False
t_event = sim.slot
sim.run(int(hours * SLOTS_PER_HOUR))
recs = sim.recs()
fl = first_lock_after(recs, t_event)
st = stalls_between(recs, t_event, sim.slot)
v = sim.views[0]
part = sim.participation(v, v.next_idx)
sil_part = float(np.average(part[silent], weights=sim.weight[silent]))
# stalls after the first lock (does it stay locked?)
later = stalls_between(recs, fl, sim.slot) if fl is not None else st
fl_txt = "never (%s)" % fmt_min(sim.slot - t_event) if fl is None else fmt_min(fl - t_event)
row.append("%s, %d stalled" % (fl_txt, st))
if denom == "active" and pmode == "cert":
tail = recs[(recs[:, 0] >= sim.slot - SLOTS_PER_HOUR) & (recs[:, 3] >= 0)]
detail.append([pct(frac, 0), pct(got), int(silent.sum()), fl_txt, st, later, "%.3f" % sil_part,
"%.3f" % (np.median(tail[:, 4]) if tail.shape[0] else float("nan")),
"%.3f" % (np.median(tail[:, 5]) if tail.shape[0] else float("nan"))])
rows.append(row)
out.append("Time from the event to the first lock, and checkpoints stalled in the run (%d h active runs, %d h total run):" % (
args.hours_c, args.hours_c_total))
out.append("")
out.append(md_table(["silent weight"] + ["%s/%s" % (d, m) if d == "active" else d for d, m, _ in configs], rows))
out.append("")
out.append("Active/cert detail:")
out.append("")
out.append(md_table(["silent weight", "picked", "keys", "first lock", "stalled", "stalled after first lock",
"silent participation at end", "signed/denominator at end (median, last hour)",
"active/total at end"], detail))
return "\n".join(out)
def scenario_d(args):
out = ["### D. Churn: a random set holding x of weight stops mining and signing at day 60", ""]
fracs = (0.35, 0.50)
rows = []
for frac in fracs:
for denom in ("active", "total"):
days = args.days_d35 if frac < 0.4 else args.days_d50
p = P(denom=denom, delay=args.delay)
sim, rng, _ = build_honest(p, args.seed)
sim.warm_start()
sim.init_views(warm=True)
sim.run(SLOTS_PER_HOUR)
gone, got = pick_weight_subset(rng, sim.weight, frac)
sim.signs[gone] = False
sim.online[gone] = False
sim.flaky[gone] = False
sim.hash[gone] = 0.0
t_event = sim.slot
live = ~gone
sim.run(int(days * SLOTS_PER_DAY))
recs = sim.recs()
fl = first_lock_after(recs, t_event)
st = stalls_between(recs, t_event, sim.slot)
later = stalls_between(recs, fl, sim.slot) if fl is not None else 0
live_share_at = None
rows.append([pct(frac, 0), pct(got), int(gone.sum()), denom,
"never in %s" % fmt_min(sim.slot - t_event) if fl is None else fmt_min(fl - t_event),
st, later, pct(sim.share(np.flatnonzero(live)))])
out.append(md_table(["churn weight", "picked", "keys", "denominator", "first lock after the event", "stalled checkpoints",
"stalled after first lock", "live share of total weight at end of run"], rows))
out.append("")
out.append("Analytic (perfect retarget): live share of total weight on day t = 1 - x(30 - t)/30, so the total "
"denominator recovers at t = 30(1 - 1/(3x)): never for x <= 1/3, day 1.4 at 35%, day 10 at 50%.")
return "\n".join(out)
def run_partition(seed, fracs, att_share, dur_min, p, pre_min=60, post_min=180):
rng = np.random.default_rng(seed)
nR = max(3, len(fracs) + 1)
sim = Sim(p, rng, n_regions=nR)
h = pareto_hashrates(rng, N_HONEST)
reg = assign_regions(h, fracs)
sim.add_keys(h, reg, flaky=True)
groups = [[i] for i in range(len(fracs))]
att = None
if att_share > 0:
att = int(sim.add_keys([att_share / (1.0 - att_share)], [len(fracs)], flaky=False, equiv=True)[0])
groups[0].append(len(fracs))
sim.warm_start()
sim.init_views(warm=True)
sim.run(pre_min * 2)
t_split = sim.slot
sim.split(groups)
sides = [v.sid for v in sim.views]
sim.run(dur_min * 2)
t_heal = sim.slot
sim.heal()
sim.run(post_min * 2)
recs = sim.recs()
res = dict(conflicts=len(sim.conflicts), t_split=t_split, t_heal=t_heal)
res["first_conflict_min"] = (min(c[1] for c in sim.conflicts) - t_split) / 2.0 if sim.conflicts else None
res["side_first_lock"] = []
res["side_locks"] = []
res["side_stalls"] = []
for sid in sides:
fl = first_lock_after(recs[recs[:, 0] < t_heal], t_split, sid=sid)
res["side_first_lock"].append(None if fl is None else (fl - t_split) / 2.0)
sel = recs[(recs[:, 0] >= t_split) & (recs[:, 0] < t_heal) & (recs[:, 2] == sid)]
res["side_locks"].append(int((sel[:, 3] >= 0).sum()))
res["side_stalls"].append(int((sel[:, 3] < 0).sum()))
res["post_stalls"] = stalls_between(recs, t_heal, sim.slot)
fl = first_lock_after(recs, t_heal)
res["post_first_lock_min"] = None if fl is None else (fl - t_heal) / 2.0
res["att_share"] = sim.share([att]) if att is not None else 0.0
return res
def fm(m):
return "never" if m is None else "%.0f" % m
def scenario_e(args):
out = ["### E. Partition: honest weight split across sides for a set time, then healed", ""]
configs = [P(denom="active", pmode="cert", delay=args.delay), P(denom="active", pmode="seen", delay=args.delay),
P(denom="total", delay=args.delay)]
splits = [("50/50", [0.5, 0.5]), ("33/33/34", [0.33, 0.33, 0.34])]
rows_conf = []
rows_first = []
for name, fr in splits:
for att in (0.0, 0.34):
for dur in (30, 90, 150):
rc = [name, pct(att, 0), dur]
rf = [name, pct(att, 0), dur]
for p in configs:
r = run_partition(args.seed, fr, att, dur, p)
rc.append("%d%s" % (r["conflicts"], "" if r["conflicts"] == 0 else " (first at %s min)" % fm(r["first_conflict_min"])))
rf.append(" / ".join(fm(x) for x in r["side_first_lock"]) + " ; post-heal stalls %d" % r["post_stalls"])
rows_conf.append(rc)
rows_first.append(rf)
labels = [p.label() for p in configs]
out.append("Conflicting locks (two certificates at one index, different blocks). Attacker = equivocating key holding the stated share of total weight, honest weight split as stated. Pass needs 0 at 0%% attacker for 30, 90 and 150 min.")
out.append("")
out.append(md_table(["honest split", "attacker", "partition min"] + labels, rows_conf))
out.append("")
out.append("Minutes after the split until each side's first lock (side order as in the split; attacker mines on the first side) and stalls in the 3 hours after the heal:")
out.append("")
out.append(md_table(["honest split", "attacker", "partition min"] + labels, rows_first))
out.append("")
# supplementary lopsided splits at 0% attacker, 150 min
rows = []
for name, fr in (("60/40", [0.6, 0.4]), ("67/33", [0.67, 0.33]), ("80/20", [0.8, 0.2])):
for dur in (150, 360):
rc = [name, dur]
for p in configs:
r = run_partition(args.seed, fr, 0.0, dur, p, post_min=120)
rc.append("%d conflicts; first locks %s min" % (r["conflicts"], " / ".join(fm(x) for x in r["side_first_lock"])))
rows.append(rc)
out.append("Supplementary, lopsided honest splits, 0% attacker:")
out.append("")
out.append(md_table(["honest split", "partition min"] + labels, rows))
out.append("")
# presence window sweep, active/cert, 0% attacker
rows = []
for presence in (240, 720, 2880):
for name, fr in (("50/50", [0.5, 0.5]), ("60/40", [0.6, 0.4]), ("33/33/34", [0.33, 0.33, 0.34])):
dur = {240: 360, 720: 720, 2880: 1500}[presence]
p = P(denom="active", pmode="cert", presence=presence, delay=args.delay)
r = run_partition(args.seed, fr, 0.0, dur, p, post_min=120)
s = min(fr)
pred = presence * (1.0 - 1.5 * s) / s / 2.0 if s < TWO_THIRDS else None
rows.append([presence, "%.1f h" % (presence / 120.0), name, dur, r["conflicts"],
" / ".join(fm(x) for x in r["side_first_lock"]), "%.0f" % pred if pred is not None else "no"])
out.append("Presence window sweep, active/cert, 0% attacker. Prediction for the smallest side (share s): "
"first lock after presence x (1 - 1.5 s) / s slots, in minutes = that / 2.")
out.append("")
out.append(md_table(["presence (checkpoints)", "presence (hours)", "honest split", "partition min", "conflicts",
"first lock per side, min", "predicted smallest-side lock, min"], rows))
return "\n".join(out)
def run_eclipse(seed, dur_h, poisoned, p, pre_min=60, post_h=5):
rng = np.random.default_rng(seed)
sim = Sim(p, rng, n_regions=5)
# shares of TOTAL weight: pool 20%, attacker 34% when poisoned, the rest honest
others = 0.8 if not poisoned else 0.46
h = pareto_hashrates(rng, N_HONEST - 1, total=others)
reg = assign_regions(h, GEOGRAPHY)
sim.add_keys(h, reg, flaky=True)
K = int(sim.add_keys([0.2], [3], flaky=False)[0])
att = None
if poisoned:
att = int(sim.add_keys([0.34], [4], flaky=False, equiv=True)[0])
sim.warm_start()
sim.init_views(warm=True)
track = []
def snap(s):
v = [x for x in s.views if 0 in x.regions][0]
track.append((s.slot, float(min(1.0, v.pcount[K] / p.presence))))
sim.run(pre_min * 2, snap=(10, snap))
t0 = sim.slot
if poisoned:
sim.split([[0, 1, 2], [3, 4]])
else:
sim.extra_delay[K] = dur_h * 3600.0
sim.run(int(dur_h * SLOTS_PER_HOUR), snap=(10, snap))
t_end = sim.slot
if poisoned:
sim.heal()
else:
sim.extra_delay[K] = 0.0
sim.run(post_h * SLOTS_PER_HOUR, snap=(10, snap))
snap(sim)
recs = sim.recs()
tr = np.array(track)
during = tr[(tr[:, 0] >= t0) & (tr[:, 0] <= t_end)]
after = tr[tr[:, 0] >= t_end]
res = dict(min_part=float(tr[:, 1].min()), part_at_end=float(during[-1, 1]) if during.shape[0] else 1.0)
below = tr[(tr[:, 0] >= t0) & (tr[:, 1] < 0.999)]
res["drop_min"] = None if below.shape[0] == 0 else (below[0, 0] - t0) / 2.0
rec = after[after[:, 1] >= 0.999]
res["recover_min"] = None if rec.shape[0] == 0 else (rec[0, 0] - t_end) / 2.0
res["conflicts"] = len(sim.conflicts)
res["first_conflict_min"] = (min(c[1] for c in sim.conflicts) - t0) / 2.0 if sim.conflicts else None
honest_sid = [v for v in [1]] if poisoned else [0]
res["honest_stalls"] = stalls_between(recs, t0, t_end, sid=(1 if poisoned else 0))
res["ecl_locks"] = int(((recs[:, 2] == 2) & (recs[:, 3] >= 0) & (recs[:, 0] >= t0) & (recs[:, 0] < t_end)).sum()) if poisoned else 0
res["post_stalls"] = stalls_between(recs, t_end, sim.slot)
return res
def scenario_f(args):
out = ["### F. Eclipse of one pool holding 20% of weight", ""]
configs = [P(denom="active", pmode="cert", delay=args.delay), P(denom="active", pmode="seen", delay=args.delay),
P(denom="total", delay=args.delay)]
rows = []
for dur in (1, 2, 4):
for p in configs:
r = run_eclipse(args.seed, dur, False, p)
rows.append([dur, p.label(), "%.3f" % r["min_part"], fm(r["drop_min"]), fm(r["recover_min"]), r["conflicts"],
r["honest_stalls"], r["post_stalls"]])
out.append("F1. Delayed view: the pool receives every block and vote %s late, votes for the right blocks, late. Participation as the rest of the network computes it." % "D hours")
out.append("")
out.append(md_table(["eclipse h", "denominator", "pool participation, minimum", "first drop below 1, min after start",
"back to 1, min after end", "conflicting locks", "stalls during", "stalls after"], rows))
out.append("")
rows = []
for dur in (1, 2, 4):
for p in configs:
r = run_eclipse(args.seed, dur, True, p)
rows.append([dur, p.label(), "%.3f" % r["min_part"], fm(r["recover_min"]), r["conflicts"], fm(r["first_conflict_min"]),
r["ecl_locks"], r["honest_stalls"], r["post_stalls"]])
out.append("F2. Poisoned view: an attacker holding 34% of weight feeds the pool a private fork for the eclipse, signs both forks, and is stripped at the heal. The pool votes for the attacker's checkpoints. Honest side = the other 46%.")
out.append("")
out.append(md_table(["eclipse h", "denominator", "pool participation, minimum (honest view)", "back to 1, min after end",
"conflicting locks", "first conflict, min after start", "locks on the eclipsed side",
"honest-side stalls during", "stalls after heal"], rows))
return "\n".join(out)
def scenario_g(args):
out = ["### G. Honest doubling overnight at day 60: 1,000 new keys, fresh Pareto draw, same total hashrate as the old 1,000", ""]
rows_share = []
rows_lat = []
ev = []
for denom in ("active", "total"):
p = P(denom=denom, delay=args.delay)
sim, rng, _ = build_honest(p, args.seed)
h_new = pareto_hashrates(rng, N_HONEST, total=1.0)
new = sim.add_keys(np.zeros(N_HONEST), assign_regions(h_new, GEOGRAPHY), flaky=True)
old = np.arange(N_HONEST)
sim.warm_start()
sim.init_views(warm=True)
sim.run(SLOTS_PER_HOUR)
t_event = sim.slot
sim.hash[new] = h_new
sim.set_uptime()
series = []
last_23 = None
for day in range(1, args.days_g + 1):
sim.run(SLOTS_PER_DAY)
so = sim.share(old)
sn = sim.share(new)
dust_new = int((sim.weight[new] < p.dust).sum())
series.append((day, so, sn, dust_new))
if so >= TWO_THIRDS:
last_23 = day
recs = sim.recs()
if denom == "active":
for d in (1, 5, 10, 15, 20, 21, 25):
r = [x for x in series if x[0] == d]
if r:
rows_share.append(["+%d" % d, pct(r[0][1]), pct(r[0][2]), pct(min(d / 60.0, 0.5)), r[0][3]])
rows_lat.append(lat_row("%s, day before" % denom, lat_stats(recs, 0, t_event)))
rows_lat.append(lat_row("%s, days 1 to 5 after" % denom, lat_stats(recs, t_event, t_event + 5 * SLOTS_PER_DAY)))
rows_lat.append(lat_row("%s, days 5 to %d after" % (denom, args.days_g), lat_stats(recs, t_event + 5 * SLOTS_PER_DAY, None)))
ev.append([denom, last_23, next((d for d, so, sn, _ in series if sn >= 0.45), None),
next((d for d, so, sn, _ in series if sn >= 0.49), None), stalls_between(recs, t_event, sim.slot)])
out.append("Weight shares (active run):")
out.append("")
out.append(md_table(["day after doubling", "old cohort", "new cohort", "new cohort formula t/60", "new keys under dust"], rows_share))
out.append("")
out.append(md_table(["denominator", "last day old cohort holds 2/3 (locks alone)", "new cohort reaches 45%", "new cohort reaches 49%",
"stalled checkpoints"], ev))
out.append("")
out.append(md_table(LAT_HEADERS, rows_lat))
return "\n".join(out)
SCENARIOS = {"A": scenario_a, "B": scenario_b, "C": scenario_c, "D": scenario_d, "E": scenario_e, "F": scenario_f, "G": scenario_g}
def main(argv=None):
ap = argparse.ArgumentParser(description="Igneum finality rule V2 simulation")
ap.add_argument("--seed", type=int, default=7)
ap.add_argument("--scenarios", default="A,B,C,D,E,F,G")
ap.add_argument("--delay", type=float, default=2.0, help="one-way inter-region delay in seconds for the main runs")
ap.add_argument("--grace", type=float, default=15.0)
ap.add_argument("--quick", action="store_true", help="shortened runs for development")
args = ap.parse_args(argv)
q = args.quick
args.days_a = 3 if q else 60
args.days_delay = 1 if q else 3
args.days_b = 4 if q else 35
args.hours_c = 3 if q else 6
args.hours_c_total = 3 if q else 72
args.days_d35 = 1 if q else 3
args.days_d50 = 1 if q else 12
args.days_g = 3 if q else 25
print("# finality_v2 output")
print()
p = P()
print("seed %d, slot %.0f s, %d blocks per checkpoint, window %d h, presence %d checkpoints, dust %d, quorum 2/3, "
"inter-region delay %.1f s (intra %.1f s, jitter sigma %.2f), grace %.0f s, uptime %.2f (mean outage %d slots), "
"uptime %.3f for keys at or above %s of hashrate, honest keys %d Pareto %.1f, geography %s%s" % (
args.seed, SLOT_S, BLOCKS_PER_CP, WINDOW_HOURS, p.presence, p.dust, args.delay, p.intra, p.jitter,
args.grace, p.uptime, p.outage, p.uptime_big, pct(p.big_share, 0), N_HONEST, PARETO_SHAPE, "/".join(pct(g, 0) for g in GEOGRAPHY),
", QUICK" if q else ""))
print()
for s in args.scenarios.split(","):
s = s.strip().upper()
if s not in SCENARIOS:
print("unknown scenario %s" % s, file=sys.stderr)
return 2
t = time.time()
print(SCENARIOS[s](args))
print()
print("(%s took %.0f s)" % (s, time.time() - t), file=sys.stderr)
return 0
if __name__ == "__main__":
sys.exit(main())

View file

@ -62,7 +62,8 @@ p{margin:0}
/* hero */
.hero{position:relative;padding-block:clamp(56px,9vw,112px) clamp(40px,6vw,72px);overflow:hidden}
.hero canvas{position:absolute;inset:0;width:100%;height:100%;opacity:.55;pointer-events:none}
.hero canvas{position:absolute;inset:0;width:100%;height:100%;pointer-events:none;-webkit-mask-image:linear-gradient(90deg,rgba(0,0,0,.08) 0,rgba(0,0,0,.35) 45%,rgba(0,0,0,.9) 100%);mask-image:linear-gradient(90deg,rgba(0,0,0,.08) 0,rgba(0,0,0,.35) 45%,rgba(0,0,0,.9) 100%)}
@media (max-width:979px){.hero canvas{-webkit-mask-image:linear-gradient(180deg,rgba(0,0,0,.12) 0,rgba(0,0,0,.12) 55%,rgba(0,0,0,.8) 100%);mask-image:linear-gradient(180deg,rgba(0,0,0,.12) 0,rgba(0,0,0,.12) 55%,rgba(0,0,0,.8) 100%)}}
.hero .wrap{position:relative;display:grid;gap:40px;grid-template-columns:1fr;align-items:center}
@media (min-width:980px){.hero .wrap{grid-template-columns:1.15fr .85fr;gap:56px}}
.hero-copy{display:flex;flex-direction:column;gap:24px}
@ -93,13 +94,15 @@ section{padding-block:clamp(56px,8vw,96px)}
.sec-head p{font-size:16px;color:var(--ash);max-width:46ch}
/* dag */
.dag{overflow-x:auto;padding:32px}
.dag-rows{display:flex;flex-direction:column;gap:14px;min-width:640px}
.dag-row{display:flex;gap:14px;align-items:center}
.blk{width:34px;height:34px;border-radius:8px;border:2px solid var(--line-2);transition:background .5s ease,border-color .5s ease,transform .3s ease}
.dag{padding:clamp(18px,3vw,32px);overflow:hidden}
.dag-rows{display:flex;flex-direction:column;gap:var(--dg,14px);--bs:34px;--dg:14px}
@media (max-width:640px){.dag-rows{--bs:26px;--dg:9px}}
.dag-row{display:flex;gap:var(--dg);align-items:center;overflow:hidden}
.blk{width:var(--bs);height:var(--bs);flex:0 0 var(--bs);border-radius:calc(var(--bs) * .24);border:2px solid var(--line-2);transition:background .5s ease,border-color .5s ease,transform .3s ease}
.blk.proving{background:var(--molten);border-color:var(--molten)}
.blk.proven{background:var(--ember);border-color:var(--ember)}
.blk.lock{border-color:var(--ember);background:transparent;display:flex;align-items:center;justify-content:center}
.blk.lock svg{width:50%;height:50%}
.blk.new{animation:pop .4s ease}
@keyframes pop{from{transform:scale(.4);opacity:0}to{transform:scale(1);opacity:1}}
.legend{display:flex;flex-wrap:wrap;gap:20px;margin-top:24px;font-size:14px;color:var(--ink-2)}
@ -408,41 +411,50 @@ footer .fl{display:flex;flex-wrap:wrap;gap:20px}
var bio=new IntersectionObserver(function(es){es.forEach(function(e){if(e.isIntersecting){bar.querySelectorAll('div').forEach(function(d){d.style.width=d.getAttribute('data-w');});bio.disconnect();}});},{threshold:.4});
bio.observe(bar);
// hero background: drifting DAG
var cv=document.getElementById('bg'),ctx=cv.getContext('2d'),nodes=[],W,H,dpr=Math.min(window.devicePixelRatio||1,2);
// hero background: a block graph that grows left to right, proves, and locks
var cv=document.getElementById('bg'),ctx=cv.getContext('2d'),nodes=[],W,H,dpr=Math.min(window.devicePixelRatio||1,2),lanes=5,count=0;
function size(){W=cv.clientWidth;H=cv.clientHeight;cv.width=W*dpr;cv.height=H*dpr;ctx.setTransform(dpr,0,0,dpr,0,0);}
size();window.addEventListener('resize',size);
function spawn(){nodes.push({x:-20,y:H*(0.15+Math.random()*0.7),vx:0.25+Math.random()*0.35,age:0,proven:false,parents:nodes.slice(-3).filter(function(){return Math.random()<0.7;})});if(nodes.length>70)nodes.shift();}
var last=0;
var spacing=function(){return W<700?74:92;};
function spawn(x){count++;var lane=Math.floor(Math.random()*lanes);var y=H*(0.18+lane*(0.64/(lanes-1)))+(Math.random()*18-9);
var cands=nodes.slice(-7).filter(function(p){return p.x<x-30;});var parents=[];
cands.sort(function(a,b){return b.x-a.x;});if(cands.length)parents.push(cands[0]);
cands.slice(1).forEach(function(p){if(Math.random()<0.45&&parents.length<3)parents.push(p);});
nodes.push({x:x,y:y,born:performance.now(),state:0,lock:count%10===0,parents:parents});if(nodes.length>60)nodes.shift();}
var last=0,speed=0.22;
function frame(t){
if(!reduce){
if(t-last>900){spawn();last=t;}
ctx.clearRect(0,0,W,H);
nodes.forEach(function(n){n.x+=n.vx;n.age++;if(n.age>140&&!n.proven)n.proven=true;});
ctx.lineWidth=1;
nodes.forEach(function(n){n.parents.forEach(function(p){if(nodes.indexOf(p)<0)return;ctx.strokeStyle='rgba(154,154,158,0.25)';ctx.beginPath();ctx.moveTo(n.x,n.y);ctx.lineTo(p.x,p.y);ctx.stroke();});});
nodes.forEach(function(n){ctx.beginPath();ctx.arc(n.x,n.y,n.proven?5:4,0,Math.PI*2);ctx.fillStyle=n.proven?'rgba(242,84,27,0.9)':(n.age>60?'rgba(255,179,92,0.85)':'rgba(12,12,14,1)');ctx.fill();if(!n.proven&&n.age<=60){ctx.strokeStyle='rgba(154,154,158,0.7)';ctx.stroke();}});
requestAnimationFrame(frame);
}
if(t-last>1000){spawn(W+16);last=t;}
ctx.clearRect(0,0,W,H);
nodes.forEach(function(n){n.x-=speed;var age=t-n.born;if(age>3500)n.state=2;else if(age>1800)n.state=1;});
ctx.lineWidth=1;
nodes.forEach(function(n){n.parents.forEach(function(p){if(nodes.indexOf(p)<0)return;var proven=n.state===2&&p.state===2;ctx.strokeStyle=proven?'rgba(242,84,27,0.35)':'rgba(154,154,158,0.18)';ctx.beginPath();ctx.moveTo(n.x,n.y);var mx=(n.x+p.x)/2;ctx.bezierCurveTo(mx,n.y,mx,p.y,p.x,p.y);ctx.stroke();});});
nodes.forEach(function(n){var r=n.state===2?5:4;
if(n.state===2&&n.lock){ctx.beginPath();ctx.arc(n.x,n.y,11,0,Math.PI*2);ctx.strokeStyle='rgba(242,84,27,0.9)';ctx.lineWidth=1.5;ctx.stroke();ctx.lineWidth=1;}
if(n.state===2){var g=ctx.createRadialGradient(n.x,n.y,0,n.x,n.y,16);g.addColorStop(0,'rgba(242,84,27,0.35)');g.addColorStop(1,'rgba(242,84,27,0)');ctx.fillStyle=g;ctx.beginPath();ctx.arc(n.x,n.y,16,0,Math.PI*2);ctx.fill();}
ctx.beginPath();ctx.arc(n.x,n.y,r,0,Math.PI*2);ctx.fillStyle=n.state===2?'#F2541B':(n.state===1?'#FFB35C':'#0C0C0E');ctx.fill();
if(n.state===0){ctx.strokeStyle='rgba(154,154,158,0.8)';ctx.stroke();}});
requestAnimationFrame(frame);
}
if(!reduce){for(var i=0;i<30;i++){spawn();nodes[nodes.length-1].x=Math.random()*W;nodes[nodes.length-1].age=Math.floor(Math.random()*200);}requestAnimationFrame(frame);}
if(!reduce){var sp=spacing();for(var i=0;i<Math.ceil(W/sp)+2;i++){spawn(W-i*sp);nodes[nodes.length-1].born=performance.now()-i*1000;}requestAnimationFrame(frame);}
// DAG animation in the prove section
var dag=document.getElementById('dag');
var rows=[[],[],[]];
function render(){dag.innerHTML='';rows.forEach(function(r,i){var d=document.createElement('div');d.className='dag-row';d.style.paddingLeft=(i*24)+'px';r.forEach(function(b){var e=document.createElement('div');e.className='blk '+b.state+(b.fresh?' new':'');if(b.state==='lock'){e.innerHTML='<svg viewBox="0 0 24 24" width="16" height="16" fill="none" stroke="#F4F1EC" stroke-width="2.4" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true"><rect x="5" y="11" width="14" height="9" rx="2"></rect><path d="M8 11V7a4 4 0 0 1 8 0v4"></path></svg>';}b.fresh=false;d.appendChild(e);});dag.appendChild(d);});}
function cap(){var cs=getComputedStyle(dag);var bs=parseFloat(cs.getPropertyValue('--bs'))||34,gp=parseFloat(cs.getPropertyValue('--dg'))||14;var off=bs*0.7;return Math.max(4,Math.floor((dag.clientWidth-2*off+gp)/(bs+gp)));}
function render(){dag.innerHTML='';var cs=getComputedStyle(dag);var bs=parseFloat(cs.getPropertyValue('--bs'))||34;rows.forEach(function(r,i){var d=document.createElement('div');d.className='dag-row';d.style.paddingLeft=Math.round(i*bs*0.7)+'px';r.forEach(function(b){var e=document.createElement('div');e.className='blk '+b.state+(b.fresh?' new':'');if(b.state==='lock'){e.innerHTML='<svg viewBox="0 0 24 24" width="16" height="16" fill="none" stroke="#F4F1EC" stroke-width="2.4" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true"><rect x="5" y="11" width="14" height="9" rx="2"></rect><path d="M8 11V7a4 4 0 0 1 8 0v4"></path></svg>';}b.fresh=false;d.appendChild(e);});dag.appendChild(d);});}
var tick=0;
function step(){
tick++;
rows.forEach(function(r){r.forEach(function(b){b.age++;if(b.state==='mined'&&b.age>3)b.state='proving';else if(b.state==='proving'&&b.age>7)b.state='proven';});});
var ri=tick%3;rows[ri].push({state:'mined',age:0,fresh:true});
if(tick%10===0){var r=rows[1];for(var i=r.length-1;i>=0;i--){if(r[i].state==='proven'){r[i].state='lock';break;}}}
rows.forEach(function(r){while(r.length>9)r.shift();});
var c=cap();rows.forEach(function(r){while(r.length>c)r.shift();});
render();
}
for(var k=0;k<24;k++){step();}
rows.forEach(function(r){r.forEach(function(b){b.fresh=false;});});render();
if(!reduce)setInterval(step,1000);
window.addEventListener('resize',function(){var c=cap();rows.forEach(function(r){while(r.length>c)r.shift();});render();});
// hourly program ticker
var ops=['rotl','rotr','xor','add','sub','mul','mulhi','mad','or'];