diff --git a/proto-cuda/host.cu b/proto-cuda/host.cu index 7011bdc38..d31ce6177 100644 --- a/proto-cuda/host.cu +++ b/proto-cuda/host.cu @@ -134,6 +134,7 @@ struct Options { bool sweep = false; int device = 0; bool serve = false; // --serve: GPU worker for igneum-miner --worker (jobs on stdin), 3 October 2026 + bool noPrepare = false; // --no-prepare: serve without the prepare command (ready line says "prepare 0"), to test the miner's fallback }; static int packMib() { return (int)(((1ull << IGNEUM_DATASET_LOG2) * 4ull) >> 20); } @@ -148,7 +149,8 @@ static void usage() { " --block-warps W warps per thread block, 1..32 (default 1 = one warp per block, like the Metal run)\n" " --device D CUDA device index (default 0)\n" " --serve GPU worker for igneum-miner --worker: reads \"job ...\" lines on stdin, prints found/done lines\n" - " (needs the pack's kernel_bound.cu compiled in: build.bat adds it when the pack has one)\n", packMib()); + " (needs the pack's kernel_bound.cu compiled in: build.bat adds it when the pack has one)\n" + " --no-prepare with --serve: no prepare support (the miner then falls back to exit 42 at a seed change)\n", packMib()); } static bool isPow2(long long v) { return v > 0 && (v & (v - 1)) == 0; } @@ -168,6 +170,7 @@ static Options parseArgs(int argc, char** argv) { else if (a == "--device") next(o.device); else if (a == "--sweep") o.sweep = true; else if (a == "--serve") o.serve = true; + else if (a == "--no-prepare") o.noPrepare = true; else if (a == "-h" || a == "--help") { usage(); std::exit(0); } else { std::printf("unknown argument %s\n", argv[i]); usage(); std::exit(2); } } @@ -401,8 +404,21 @@ static SizeResult runSize(const Options& o, int mib, uint64_t* dOut, uint32_t no // found every nonce whose 64-bit hash is <= target // done end of the job (wall ms) // error -// The program is compiled ahead of time from the pack (no NVRTC), so this worker serves exactly one epoch seed and -// one day seed: the pack's. A job for other seeds is answered with an error naming both; re-export the pack with +// prepare build /kernel.cu and kernel_bound.cu (the pack the +// miner wrote for those seeds) to cubins with nvcc in the background, +// then load them and build that pair's cache and dataset +// prepared ... the pair is resident; a job on it switches instantly +// prepare-failed +// The program is compiled ahead of time from the pack (no NVRTC), so at start this worker serves exactly one epoch +// seed and one day seed: the pack's. The next pair arrives through `prepare`: the miner writes the pack for the +// prepared seeds (igneum-miner --prepare-packs ) and names its directory; a background thread runs nvcc on that +// pack's kernel.cu (cache fill and build kernels, memhard.h for its day) and kernel_bound.cu (the bound hash kernel) +// to two cubins for this device's architecture, and the main loop loads them through the driver API +// (cudaGetDriverEntryPoint, so nothing new is linked), fills the cache and builds the dataset while jobs on the current +// pair keep running (at most two pairs resident; the old one is released after the first job on the new one). nvcc +// must be on PATH with a host compiler, as build.bat needs it; the ready line says "prepare 1" only when `nvcc --version` +// answers. A prepared pair's cache is not cross-checked against the host fill (memhard.h is compiled in for the +// original day); the miner's CPU re-check of every found nonce covers it. Without prepare, re-export the pack with // `igneum-miner export-pack ` and rebuild. The init words of a dispatch are // seed_words_from_bytes("igneum-block/" || prehash || nonce_hi_le32), passed by value to igneum_hash_bound // (kernel_bound.cu); the lane nonce is baseNonce + gid as in the bench kernel. @@ -438,6 +454,162 @@ static bool unhexStr(const std::string& s, std::vector& out) { return true; } +#if defined(IGNEUM_BOUND) && IGNEUM_DATASET_MODE == 1 && defined(__has_include) +#if __has_include() +#define IGNEUM_CUDA_PREPARE 1 +#include +#include +#include +#include +#endif +#endif + +#ifdef IGNEUM_CUDA_PREPARE +// The few driver API entry points the hot swap needs, fetched through the runtime so the link line is unchanged. +struct DriverApi { + CUresult (*moduleLoad)(CUmodule*, const char*) = nullptr; + CUresult (*moduleUnload)(CUmodule) = nullptr; + CUresult (*moduleGetFunction)(CUfunction*, CUmodule, const char*) = nullptr; + CUresult (*moduleGetFunctionCount)(unsigned int*, CUmodule) = nullptr; // CUDA 12.4 and newer + CUresult (*moduleEnumerateFunctions)(CUfunction*, unsigned int, CUmodule) = nullptr; + CUresult (*funcGetName)(const char**, CUfunction) = nullptr; // CUDA 12.3 and newer + CUresult (*launchKernel)(CUfunction, unsigned, unsigned, unsigned, unsigned, unsigned, unsigned, unsigned, CUstream, void**, void**) = nullptr; + CUresult (*getErrorString)(CUresult, const char**) = nullptr; + bool ok = false; + std::string why; + template bool get(const char* name, F& fn, bool required) { + void* p = nullptr; + cudaError_t e = cudaGetDriverEntryPoint(name, &p, cudaEnableDefault); + if (e != cudaSuccess || !p) { if (required) { why = std::string("no driver entry point ") + name; } return false; } + fn = reinterpret_cast(p); + return true; + } + void load() { + ok = get("cuModuleLoad", moduleLoad, true) && get("cuModuleUnload", moduleUnload, true) && get("cuModuleGetFunction", moduleGetFunction, true) && + get("cuLaunchKernel", launchKernel, true) && get("cuGetErrorString", getErrorString, true); + get("cuModuleGetFunctionCount", moduleGetFunctionCount, false); + get("cuModuleEnumerateFunctions", moduleEnumerateFunctions, false); + get("cuFuncGetName", funcGetName, false); + } + std::string err(CUresult r) { const char* s = nullptr; if (getErrorString) getErrorString(r, &s); return s ? s : "CUDA driver error"; } + // A kernel by its plain name: the Itanium mangling nvcc gives device code first, then the enumeration (12.4+). + bool find(CUmodule m, const char* plain, const char* mangled, CUfunction* out) { + if (moduleGetFunction(out, m, mangled) == CUDA_SUCCESS) return true; + if (moduleGetFunction(out, m, plain) == CUDA_SUCCESS) return true; + if (!moduleGetFunctionCount || !moduleEnumerateFunctions || !funcGetName) return false; + unsigned int n = 0; + if (moduleGetFunctionCount(&n, m) != CUDA_SUCCESS || n == 0) return false; + std::vector fns(n); + if (moduleEnumerateFunctions(fns.data(), n, m) != CUDA_SUCCESS) return false; + for (CUfunction f : fns) { + const char* name = nullptr; + if (funcGetName(&name, f) == CUDA_SUCCESS && name && std::strstr(name, plain)) { *out = f; return true; } + } + return false; + } +}; + +// One resident pair: the compiled-in pack (runtime launchers, gCache) or a prepared pack (two cubins, driver launches). +struct CudaPair { + std::string epochHex, dayHex; + uint32_t sw[8] = {0}, kw[8] = {0}; + bool builtIn = false; + CUmodule modKernel = nullptr, modBound = nullptr; + CUfunction fCacheFill = nullptr, fBuild = nullptr, fHashBound = nullptr; + uint32_t* cache = nullptr; + uint32_t* ds = nullptr; + double nvccMs = 0, cacheMs = 0, dsMs = 0; +}; + +static void releasePair(DriverApi& drv, CudaPair* p) { + if (!p) return; + if (p->ds) cudaFree(p->ds); + if (p->cache) cudaFree(p->cache); + if (p->modBound) drv.moduleUnload(p->modBound); + if (p->modKernel) drv.moduleUnload(p->modKernel); + delete p; +} + +// The nvcc step of a prepare, on its own thread. Only the compiler runs here; every CUDA call stays on the main thread. +struct PrepareTask { + std::string epochHex, dayHex, packDir, arch, error; + std::atomic done{false}; + bool ok = false; + double t0 = 0, nvccMs = 0; + std::thread thread; +}; + +static bool nvccAvailable() { +#ifdef _WIN32 + return std::system("nvcc --version >NUL 2>&1") == 0; +#else + return std::system("nvcc --version >/dev/null 2>&1") == 0; +#endif +} + +static void prepareCompile(PrepareTask* t) { + double c0 = wallMs(); + const char* files[2] = { "kernel", "kernel_bound" }; + for (const char* f : files) { + std::string cmd = "nvcc -cubin -O3 -std=c++17 -arch=" + t->arch + " -allow-unsupported-compiler -I \"" + t->packDir + "\" -o \"" + t->packDir + "/" + f + + ".cubin\" \"" + t->packDir + "/" + f + ".cu\" > \"" + t->packDir + "/nvcc-" + f + ".log\" 2>&1"; + int rc = std::system(cmd.c_str()); + if (rc != 0) { + std::string log; + std::ifstream in(t->packDir + "/nvcc-" + f + ".log"); + std::string line; + while (std::getline(in, line) && log.size() < 300) { log += line; log += " | "; } + t->error = std::string("nvcc failed on ") + f + ".cu (exit " + std::to_string(rc) + "): " + log; + t->done = true; + return; + } + } + t->nvccMs = wallMs() - c0; + t->ok = true; + t->done = true; +} + +// Loads the two cubins, builds the cache and dataset (main thread). Returns the pair or null with `error` set. +static CudaPair* prepareLoad(DriverApi& drv, const PrepareTask& t, uint32_t words, std::string& error) { + CudaPair* p = new CudaPair(); + p->epochHex = t.epochHex; p->dayHex = t.dayHex; p->nvccMs = t.nvccMs; + { + std::vector eb, db; + if (unhexStr(t.epochHex, eb) && eb.size() == 32) seedWordsFromBytes(eb.data(), 32, p->sw); + if (unhexStr(t.dayHex, db)) seedWordsFromBytes(db.data(), db.size(), p->kw); + } + CUresult r = drv.moduleLoad(&p->modKernel, (t.packDir + "/kernel.cubin").c_str()); + if (r != CUDA_SUCCESS) { error = "cuModuleLoad kernel.cubin: " + drv.err(r); releasePair(drv, p); return nullptr; } + r = drv.moduleLoad(&p->modBound, (t.packDir + "/kernel_bound.cubin").c_str()); + if (r != CUDA_SUCCESS) { error = "cuModuleLoad kernel_bound.cubin: " + drv.err(r); releasePair(drv, p); return nullptr; } + if (!drv.find(p->modKernel, "igneum_cache_fill", "_Z17igneum_cache_fillPjj", &p->fCacheFill)) { error = "igneum_cache_fill not found in kernel.cubin"; releasePair(drv, p); return nullptr; } + if (!drv.find(p->modKernel, "igneum_build", "_Z12igneum_buildPjPKjj", &p->fBuild)) { error = "igneum_build not found in kernel.cubin"; releasePair(drv, p); return nullptr; } + if (!drv.find(p->modBound, "igneum_hash_bound", "_Z17igneum_hash_boundPKjPyjj15IgneumInitWords", &p->fHashBound)) { error = "igneum_hash_bound not found in kernel_bound.cubin"; releasePair(drv, p); return nullptr; } + // Cache (the same segment count as the compiled-in pack: the dataset schedule is a network constant) + double c0 = wallMs(); + size_t cacheBytes = (size_t)CACHE_WORDS_HOST * 4u; + if (cudaMalloc((void**)&p->cache, cacheBytes) != cudaSuccess) { error = "cudaMalloc cache"; p->cache = nullptr; releasePair(drv, p); return nullptr; } + { + uint32_t nSeg = IGNEUM_CACHE_SEGMENTS, block = 256u, grid = (nSeg + block - 1u) / block; + void* args[2] = { &p->cache, &nSeg }; + r = drv.launchKernel(p->fCacheFill, grid, 1, 1, block, 1, 1, 0, nullptr, args, nullptr); + if (r != CUDA_SUCCESS || cudaDeviceSynchronize() != cudaSuccess) { error = "cache fill launch: " + drv.err(r); releasePair(drv, p); return nullptr; } + } + p->cacheMs = wallMs() - c0; + // Dataset + c0 = wallMs(); + if (cudaMalloc((void**)&p->ds, (size_t)words * 4u) != cudaSuccess) { error = "cudaMalloc dataset"; p->ds = nullptr; releasePair(drv, p); return nullptr; } + { + uint32_t nItems = words / 16u, block = 256u, grid = (nItems + block - 1u) / block; + void* args[3] = { &p->ds, &p->cache, &nItems }; + r = drv.launchKernel(p->fBuild, grid, 1, 1, block, 1, 1, 0, nullptr, args, nullptr); + if (r != CUDA_SUCCESS || cudaDeviceSynchronize() != cudaSuccess) { error = "dataset build launch: " + drv.err(r); releasePair(drv, p); return nullptr; } + } + p->dsMs = wallMs() - c0; + return p; +} +#endif + static int runServe(const Options& o) { #if !defined(IGNEUM_BOUND) || IGNEUM_DATASET_MODE != 1 (void)o; @@ -465,7 +637,27 @@ static int runServe(const Options& o) { std::vector hOut(batch); int regs = 0, bps = 0; igneum_hash_bound_info(®s, &bps, (uint32_t)o.blockWarps); - std::printf("ready cuda %s pack %s dataset-log2 %d batch %u regs %d\n", devName.c_str(), IGNEUM_SEED_STRING, IGNEUM_DATASET_LOG2, batch, regs); + // Prepare support: the driver entry points and nvcc on PATH + int prepareOk = 0; +#ifdef IGNEUM_CUDA_PREPARE + DriverApi drv; + drv.load(); + std::string arch = "sm_" + std::to_string(prop.major) + std::to_string(prop.minor); + bool haveNvcc = !o.noPrepare && nvccAvailable(); + prepareOk = (!o.noPrepare && drv.ok && haveNvcc) ? 1 : 0; + CudaPair* cur = new CudaPair(); + cur->builtIn = true; cur->cache = gCache; cur->ds = dDs; + std::memcpy(cur->sw, SEEDW, 32); std::memcpy(cur->kw, KEYW, 32); + CudaPair* prepared = nullptr; + CudaPair* old = nullptr; + PrepareTask* task = nullptr; +#endif + std::printf("ready cuda %s pack %s dataset-log2 %d batch %u regs %d prepare %d\n", devName.c_str(), IGNEUM_SEED_STRING, IGNEUM_DATASET_LOG2, batch, regs, prepareOk); +#ifdef IGNEUM_CUDA_PREPARE + if (!o.noPrepare && !prepareOk) std::printf("info prepare unavailable: %s\n", !drv.ok ? drv.why.c_str() : "nvcc is not on PATH (open the build prompt, or install the CUDA Toolkit)"); +#else + std::printf("info prepare unavailable: this binary was built without cuda.h (CPU emulation or an old toolkit)\n"); +#endif std::fflush(stdout); std::string line; @@ -474,6 +666,41 @@ static int runServe(const Options& o) { std::vector f; { size_t i = 0; while (i < line.size()) { while (i < line.size() && line[i] == ' ') ++i; size_t j = i; while (j < line.size() && line[j] != ' ') ++j; if (j > i) f.push_back(line.substr(i, j - i)); i = j; } } if (f.empty()) continue; +#ifdef IGNEUM_CUDA_PREPARE + // A finished nvcc step is loaded here, between lines, on this thread + if (task && task->done) { + task->thread.join(); + if (task->ok) { + std::string error; + CudaPair* p = prepareLoad(drv, *task, words, error); + if (p) { + if (prepared) releasePair(drv, prepared); + prepared = p; + std::printf("prepared %s %s %.1f nvcc %.1f cache %.1f dataset %.1f resident 2 programs 2 datasets\n", p->epochHex.c_str(), p->dayHex.c_str(), wallMs() - task->t0, p->nvccMs, p->cacheMs, p->dsMs); + } else { + std::printf("prepare-failed %s %s %s\n", task->epochHex.c_str(), task->dayHex.c_str(), error.c_str()); + } + } else { + std::printf("prepare-failed %s %s %s\n", task->epochHex.c_str(), task->dayHex.c_str(), task->error.c_str()); + } + std::fflush(stdout); + delete task; task = nullptr; + } + if (f[0] == "prepare") { + if (!prepareOk) { std::printf("info ignored (no prepare support): %s\n", line.c_str()); std::fflush(stdout); continue; } + if (f.size() < 4) { std::printf("prepare-failed %s %s this ahead-of-time worker needs a pack directory as the third field (igneum-miner --prepare-packs )\n", f.size() > 1 ? f[1].c_str() : "0", f.size() > 2 ? f[2].c_str() : "0"); std::fflush(stdout); continue; } + if (f[1].size() != 64) { std::printf("prepare-failed %s %s bad field (epoch_seed 64 hex, day_seed hex)\n", f[1].c_str(), f[2].c_str()); std::fflush(stdout); continue; } + if (task) { std::printf("prepare-failed %s %s a prepare is still running\n", f[1].c_str(), f[2].c_str()); std::fflush(stdout); continue; } + if (prepared && prepared->epochHex == f[1] && prepared->dayHex == f[2]) { std::printf("prepared %s %s 0 (already resident)\n", f[1].c_str(), f[2].c_str()); std::fflush(stdout); continue; } + task = new PrepareTask(); + task->epochHex = f[1]; task->dayHex = f[2]; task->packDir = f[3]; task->arch = arch; task->t0 = wallMs(); + task->thread = std::thread(prepareCompile, task); + std::printf("info prepare started for epoch %.16s day %s from %s (nvcc -arch=%s in the background)\n", f[1].c_str(), f[2].c_str(), f[3].c_str(), arch.c_str()); std::fflush(stdout); + continue; + } +#else + if (f[0] == "prepare") { std::printf("info ignored (no prepare support): %s\n", line.c_str()); std::fflush(stdout); continue; } +#endif if (f[0] != "job") { std::printf("info ignored: %s\n", line.c_str()); std::fflush(stdout); continue; } std::string jobId = f.size() > 1 ? f[1] : "0"; if (f.size() < 8) { std::printf("error %s malformed job line (need 7 fields after job)\n", jobId.c_str()); std::fflush(stdout); continue; } @@ -488,6 +715,26 @@ static int runServe(const Options& o) { uint32_t sw[8], kw[8]; seedWordsFromBytes(epochSeed.data(), epochSeed.size(), sw); seedWordsFromBytes(daySeed.data(), daySeed.size(), kw); + double t0 = wallMs(); +#ifdef IGNEUM_CUDA_PREPARE + bool switched = false; + if (std::memcmp(sw, cur->sw, 32) != 0 || std::memcmp(kw, cur->kw, 32) != 0) { + if (prepared && std::memcmp(sw, prepared->sw, 32) == 0 && std::memcmp(kw, prepared->kw, 32) == 0) { + if (old) releasePair(drv, old); + old = cur; cur = prepared; prepared = nullptr; switched = true; + std::printf("info switched to the prepared pair epoch %.16s day %s in %.2f ms\n", cur->epochHex.c_str(), cur->dayHex.c_str(), wallMs() - t0); std::fflush(stdout); + } else if (std::memcmp(sw, cur->sw, 32) != 0) { + std::printf("error %s epoch seed mismatch: this worker holds %s%s (seed words %08x %08x ...)%s, the job's epoch seed %s gives %08x %08x ...; send prepare with a pack directory, or run igneum-miner export-pack and rebuild\n", + jobId.c_str(), cur->builtIn ? "pack \"" IGNEUM_SEED_STRING "\"" : "prepared epoch ", cur->builtIn ? "" : cur->epochHex.c_str(), cur->sw[0], cur->sw[1], prepared ? " plus one prepared pair" : "", f[6].substr(0, 16).c_str(), sw[0], sw[1]); + std::fflush(stdout); continue; + } else { + std::printf("error %s day seed mismatch: this worker's cache is for key %08x %08x ..., the job's day seed %s gives %08x %08x ...; send prepare with a pack directory, or run igneum-miner export-pack and rebuild\n", + jobId.c_str(), cur->kw[0], cur->kw[1], f[7].c_str(), kw[0], kw[1]); + std::fflush(stdout); continue; + } + } + const uint32_t* jobDs = cur->ds; +#else if (std::memcmp(sw, SEEDW, 32) != 0) { std::printf("error %s epoch seed mismatch: this worker was built for pack \"%s\" (seed words %08x %08x ...), the job's epoch seed %s gives %08x %08x ...; run igneum-miner export-pack and rebuild\n", jobId.c_str(), IGNEUM_SEED_STRING, SEEDW[0], SEEDW[1], f[6].substr(0, 16).c_str(), sw[0], sw[1]); @@ -498,7 +745,8 @@ static int runServe(const Options& o) { jobId.c_str(), KEYW[0], KEYW[1], f[7].c_str(), kw[0], kw[1]); std::fflush(stdout); continue; } - double t0 = wallMs(); + const uint32_t* jobDs = dDs; +#endif uint64_t remaining = nonceCount, hashes = 0; uint32_t hi = (uint32_t)(nonceStart >> 32), lo = (uint32_t)nonceStart; bool failed = false; @@ -515,7 +763,18 @@ static int runServe(const Options& o) { b[45] = (uint8_t)hi; b[46] = (uint8_t)(hi >> 8); b[47] = (uint8_t)(hi >> 16); b[48] = (uint8_t)(hi >> 24); seedWordsFromBytes(b, 49, iw.w); } - cudaError_t e = igneum_launch_hash_bound(dDs, dOut, lo, mask, iw, chunk, (uint32_t)o.blockWarps); + cudaError_t e = cudaSuccess; +#ifdef IGNEUM_CUDA_PREPARE + if (!cur->builtIn) { + // A prepared pair: the same launch shape as igneum_launch_hash_bound, through the driver API + uint32_t block = 32u * (uint32_t)o.blockWarps; + uint32_t baseNonce = lo, maskArg = mask; + void* args[5] = { (void*)&jobDs, (void*)&dOut, &baseNonce, &maskArg, &iw }; + CUresult r = drv.launchKernel(cur->fHashBound, chunk / block, 1, 1, block, 1, 1, 0, nullptr, args, nullptr); + if (r != CUDA_SUCCESS) { std::printf("error %s dispatch failed: %s\n", jobId.c_str(), drv.err(r).c_str()); std::fflush(stdout); failed = true; break; } + } else +#endif + e = igneum_launch_hash_bound(jobDs, dOut, lo, mask, iw, chunk, (uint32_t)o.blockWarps); if (e == cudaSuccess) e = cudaDeviceSynchronize(); if (e == cudaSuccess) e = cudaMemcpy(hOut.data(), dOut, (size_t)chunk * sizeof(uint64_t), cudaMemcpyDeviceToHost); if (e != cudaSuccess) { std::printf("error %s dispatch failed: %s\n", jobId.c_str(), cudaGetErrorString(e)); std::fflush(stdout); failed = true; break; } @@ -530,11 +789,21 @@ static int runServe(const Options& o) { } if (failed) continue; std::printf("done %s %llu %.2f\n", jobId.c_str(), (unsigned long long)hashes, wallMs() - t0); +#ifdef IGNEUM_CUDA_PREPARE + if (switched && old) { releasePair(drv, old); old = nullptr; std::printf("info dropped the previous pair (its program, cache and dataset)\n"); } +#endif std::fflush(stdout); } cudaFree(dOut); +#ifdef IGNEUM_CUDA_PREPARE + if (task) { task->thread.join(); delete task; } + if (old) releasePair(drv, old); + if (prepared) releasePair(drv, prepared); + releasePair(drv, cur); // frees gCache and dDs when the built-in pair is still current +#else cudaFree(dDs); cudaFree(gCache); +#endif return 0; #endif } @@ -551,6 +820,12 @@ int main(int argc, char** argv) { if (count == 0) { std::printf("FAIL: no CUDA device\n"); return 2; } if (o.device < 0 || o.device >= count) { std::printf("FAIL: device %d out of range (%d devices)\n", o.device, count); return 2; } CUDA_CHECK(cudaSetDevice(o.device)); + // Blocking sync, set before the context exists: the host thread sleeps in cudaDeviceSynchronize instead of + // spinning (one full core per worker process at the default spin schedule, measured on the RTX 5090 with + // eight workers, 3 Oct 2026). The microseconds of wake-up latency are nothing against a 100 ms dispatch. +#ifdef cudaDeviceScheduleBlockingSync + CUDA_CHECK(cudaSetDeviceFlags(cudaDeviceScheduleBlockingSync)); +#endif if (o.serve) { if (o.batchLog2 == 24) { Options s2 = o; s2.batchLog2 = 22; return runServe(s2); } // 2^22 nonces per dispatch by default return runServe(o); diff --git a/proto-cuda/windows-miner/START-MINING.bat b/proto-cuda/windows-miner/START-MINING.bat index d4c334e16..934492f90 100644 --- a/proto-cuda/windows-miner/START-MINING.bat +++ b/proto-cuda/windows-miner/START-MINING.bat @@ -5,7 +5,8 @@ rem one igneum-miner per worker against the Mac node. Logs land next to this fil rem The window shows a dashboard: hash rate, blocks found and one line per miner. Stop with Ctrl+C; the window stays open at the end. rem ---- settings: edit these lines only ---------------------------------------------------------- -set "NODE_HOST=192.168.68.64" +rem auto = a node on this PC (START-NODE.bat, 127.0.0.1) when one is running, else the Mac node at 192.168.68.64. +set "NODE_HOST=auto" set "NODE_PORT=26610" rem Devnet payout address (igneumdev:...). Leave empty for the miner's fixed test address. set "PAYOUT_ADDRESS=" diff --git a/proto-cuda/windows-miner/start-mining.ps1 b/proto-cuda/windows-miner/start-mining.ps1 index 2247ee69a..b52d08a7a 100644 --- a/proto-cuda/windows-miner/start-mining.ps1 +++ b/proto-cuda/windows-miner/start-mining.ps1 @@ -70,6 +70,9 @@ $vcvars = if ($env:VCVARS) { $env:VCVARS } else { 'C:\Program Files\Microsoft Vi $vcvarsVer = if ($env:VCVARS_VER) { $env:VCVARS_VER } else { '14.30' } $minersPerVendor = 1 if ($env:MINERS -and [int]$env:MINERS -ge 1) { $minersPerVendor = [int]$env:MINERS } +# One worker process per card (3 Oct 2026): the MINERS identities run through ONE igneum-miner with --identities N and one +# worker (one context, one cache and dataset, one host thread). ONE_WORKER_PER_CARD=0 restores a worker per identity. +$oneWorkerPerCard = -not ($env:ONE_WORKER_PER_CARD -eq '0') # Nonces per job, per vendor (a multiple of 32). Empty = the miner's default of 16,777,216. A job is one template, so a # job should take well under a minute: a slow GPU with a long job mines a stale template and reports late. $nvidiaJobNonces = $env:NVIDIA_JOB_NONCES @@ -321,8 +324,9 @@ foreach ($name in $plan) { $exe = Build-Worker $name $seeds if (-not $exe) { Log "$name skipped: no worker binary"; continue } $instances = @() - for ($i = 1; $i -le $minersPerVendor; $i++) { - $suffix = if ($minersPerVendor -gt 1) { "-$i" } else { '' } + $processes = if ($oneWorkerPerCard) { 1 } else { $minersPerVendor } + for ($i = 1; $i -le $processes; $i++) { + $suffix = if ($processes -gt 1) { "-$i" } else { '' } $instances += @{ vendor = $name; index = $i; label = "$name-$machine$suffix"; file = "$name$suffix-$stamp"; log = $null; err = $null; runId = "$name-$machine$suffix-$stamp"; proc = $null; restarts = 0; starts = 0; startedAt = $null; restartAt = $null; exitCode = $null; logPos = [long]0; logRem = ''; errPos = [long]0; errRem = ''; accepted = 0; found = (New-Object System.Collections.ArrayList); @@ -332,7 +336,8 @@ foreach ($name in $plan) { $vendors[$name] = @{ name = $name; card = (Get-CardName $name); exe = $exe; instances = $instances; rebuilds = 0; building = $false } } if ($vendors.Count -eq 0) { Log 'no worker could be built; see the launcher log'; exit 1 } -Log ("identities per vendor: $minersPerVendor (each identity runs its own worker: about 1.3 GiB of GPU memory per instance)") +if ($oneWorkerPerCard) { Log ("identities per vendor: $minersPerVendor through ONE worker process per card (labels --1..$minersPerVendor; about 1.3 GiB of GPU memory per card)") } +else { Log ("identities per vendor: $minersPerVendor (each identity runs its own worker: about 1.3 GiB of GPU memory per instance)") } Log ("job nonces: nvidia " + $(if ($nvidiaJobNonces) { $nvidiaJobNonces } else { 'miner default (16,777,216)' }) + ", amd $amdJobNonces") function Start-Miner($v, $inst) { @@ -349,6 +354,7 @@ function Start-Miner($v, $inst) { # --exit-on-seed-change is the fallback only: a worker that answers "prepare 1" on its ready line is never exited at a # seed change; the miner writes the next pack under --prepare-packs and the worker builds it in the background. $margs = @('mine', $nodeUrl, '1', '100000000', $inst.label, '--worker', ('"' + $v.exe + '"'), '--status-secs', '30', '--exit-on-seed-change', '--prepare-packs', ('"' + $preparedPacks + '"')) + if ($oneWorkerPerCard -and $minersPerVendor -gt 1) { $margs += @('--identities', $minersPerVendor) } if ($payout) { $margs += @('--address', $payout) } else { $margs += @('--payout-label', $inst.label) } if ($v.name -eq 'nvidia') { if ($nvidiaJobNonces) { $margs += @('--job-nonces', $nvidiaJobNonces) } diff --git a/proto-metal/main.swift b/proto-metal/main.swift index 2a0cbb4c3..7287bff2b 100644 --- a/proto-metal/main.swift +++ b/proto-metal/main.swift @@ -18,6 +18,7 @@ struct Options { var dumpDir: String? = nil var exportPack: String? = nil // write a CUDA program pack for --seed into this directory and exit var serve = false // --serve: GPU worker for igneum-miner (jobs on stdin, results on stdout), 3 October 2026 + var noPrepare = false // --no-prepare: serve without the prepare command (ready line says "prepare 0"), to test the miner's fallback // Hardening tests (added 3 October 2026). Any of these runs instead of the bench. var fuzz: Int? = nil // --fuzz N: N random programs, GPU vs CPU on 4 random warps each var fuzzSeed = "igneum-fuzz-2026-10-03" @@ -54,6 +55,7 @@ func parseArgs() -> Options { case "--dump": o.dumpDir = take() case "--export-pack": o.exportPack = take() case "--serve": o.serve = true + case "--no-prepare": o.noPrepare = true case "--fuzz": o.fuzz = Int(take()) ?? 200 case "--fuzz-seed": o.fuzzSeed = take() case "--edge": o.edge = true @@ -356,6 +358,8 @@ struct Program { let seedString: String let seed: [UInt32] let instrs: [Instr] + var generator = 1 // 2 for every current program (generateProgramV2); 1 for the retired lever generator + var attempt: UInt32 = 0 // attempt index under the acceptance rule (0 = the bare seed) static let iterations = 8 static let count = 64 var loadsPerHash: Int { instrs.filter { $0.op == .load || $0.op == .wload }.count * Program.iterations } @@ -397,10 +401,18 @@ struct GeneratorConfig { } var generatorConfig = GeneratorConfig() -func generateProgram(seedString: String) -> Program { generateProgram(seedString: seedString, words: seedWords(seedString)) } +// The program of a seed string: generator version 2 with the acceptance rule (below), the same program the Rust crate +// derives. The retired version 1 generator is used only when a lever (--load-weight, --wide-frac) is set, for the +// MEMHARD.md section 2.4 measurements; those programs are not the lottery hash. +func generateProgram(seedString: String) -> Program { + if generatorConfig.loadWeight != 25 || generatorConfig.wideFrac != 0 { + return generateProgramV1(seedString: seedString, words: seedWords(seedString)) + } + return generateProgramV2(seedString: seedString, bytes: Array(seedString.utf8)) +} -// The generator from already-derived seed words (what the chain feeds: the epoch block hash on devnet v0, the VDF output later). -func generateProgram(seedString: String, words sw: [UInt32]) -> Program { +// Version 1 (retired 4 October 2026): op rolled per instruction against the 11-family table, load count free. +func generateProgramV1(seedString: String, words sw: [UInt32]) -> Program { var rng = SplitMix64(s: (UInt64(sw[0]) | (UInt64(sw[1]) << 32)) ^ ((UInt64(sw[2]) | (UInt64(sw[3]) << 32)) &* 0x9E3779B97F4A7C15)) var instrs = [Instr]() let weights = generatorConfig.weights @@ -423,6 +435,232 @@ func generateProgram(seedString: String, words sw: [UInt32]) -> Program { return Program(seedString: seedString, seed: sw, instrs: instrs) } +// MARK: - Generator version 2 and the acceptance rule (4 October 2026) +// +// Draw for draw the Rust generator (igneum-pow/src/generator.rs, candidate_from_words) and acceptance rule +// (igneum-pow/src/accept.rs), spec 01 sections 1.4.3 and 1.4.6. Exactly 16 load slots drawn first from instructions +// 1..63; a load's source is drawn from the registers other than dst written by an earlier instruction and not read by +// a load since; a candidate that fails the rule is replaced by attempt k + 1, seedWordsBytes(seed || k_le32). + +let generatorVersion = 2 +let loadSlots = 16 +let maxAttempts: UInt32 = 32 +let nonloadWeights: [(Op, Int)] = [(.add, 12), (.xor, 10), (.mul, 8), (.mad, 8), (.shfl, 8), + (.rotl, 7), (.sub, 6), (.mulhi, 6), (.rotr, 6), (.or, 4)] +let acceptUnits = 64 +let acceptDatasetLog2 = 28 +let acceptMaxSaturated: UInt32 = 164 +let acceptBiasTolerance: UInt32 = 136 +let acceptMinDistinctSum: UInt64 = 245_760 + +func fnv1a64Bytes(_ bytes: [UInt8]) -> UInt64 { + var h: UInt64 = 0xcbf29ce484222325 + for b in bytes { h ^= UInt64(b); h &*= 0x100000001b3 } + return h +} + +func le32(_ v: UInt32) -> [UInt8] { [UInt8(v & 0xff), UInt8((v >> 8) & 0xff), UInt8((v >> 16) & 0xff), UInt8((v >> 24) & 0xff)] } + +// The seed words of attempt k: seedWordsBytes(seed) for k = 0, seedWordsBytes(seed || k_le32) otherwise. +func attemptWords(_ seedBytes: [UInt8], _ attempt: UInt32) -> [UInt32] { + attempt == 0 ? seedWordsBytes(seedBytes) : seedWordsBytes(seedBytes + le32(attempt)) +} + +// FNV-1a 64 over "igneum-program/" || generator_le32 || seed words LE || attempt_le32 (Program::program_id in Rust). +func programId(_ p: Program) -> UInt64 { + var b = Array("igneum-program/".utf8) + le32(UInt32(p.generator)) + for w in p.seed { b += le32(w) } + b += le32(p.attempt) + return fnv1a64Bytes(b) +} + +// One version 2 candidate from its seed words, before the acceptance rule. +func candidateProgram(seedString: String, words sw: [UInt32], attempt: UInt32) -> Program { + var rng = SplitMix64(s: (UInt64(sw[0]) | (UInt64(sw[1]) << 32)) ^ ((UInt64(sw[2]) | (UInt64(sw[3]) << 32)) &* 0x9E3779B97F4A7C15)) + var slots = Array(1..= dst { a += 1 } + } else { + a = eligible[rng.below(eligible.count)] + } + } else { + a = rng.below(7); if a >= dst { a += 1 } + } + let b = rng.below(8) + let imm = UInt32(truncatingIfNeeded: rng.next()) + let imm2 = UInt32(truncatingIfNeeded: rng.next()) + let rot = UInt32(1 + rng.below(31)) + let bit = rng.below(32) + let mask = 1 << rng.below(5) + if op == .load { fresh[a] = false } + fresh[dst] = true + instrs.append(Instr(op: op, dst: dst, a: a, b: b, imm: imm, imm2: imm2, rot: rot, bit: bit, mask: mask)) + } + return Program(seedString: seedString, seed: sw, instrs: instrs, generator: generatorVersion, attempt: attempt) +} + +@inline(__always) func opInjects(_ op: Op) -> Bool { + switch op { case .add, .sub, .xor, .mad, .shfl, .load, .wload: return true; default: return false } +} + +// Parts (a) and (b) of the rule. Returns nil when the program passes, else the reason. +func acceptStatic(_ p: Program) -> String? { + var pending = [Bool](repeating: false, count: 8) + for _ in 0..<2 { + for (k, ins) in p.instrs.enumerated() { + let isLoad = ins.op == .load || ins.op == .wload + if isLoad && pending[ins.a] { return "(a) load at instruction \(k) reads r\(ins.a), unwritten since the previous load from it" } + pending[ins.dst] = false + if isLoad { pending[ins.a] = true } + } + } + var injected = [Bool](repeating: false, count: 8) + for ins in p.instrs where opInjects(ins.op) { injected[ins.dst] = true } + for r in 0..<8 where !injected[r] { return "(b) r\(r) has no add, sub, xor, mad, shfl or load write" } + return nil +} + +// The 64 base nonces of the dynamic test: SplitMix64 seeded with FNV-1a 64("igneum-accept/" || seed words LE). +func acceptBaseNonces(_ seed: [UInt32]) -> [UInt32] { + var b = Array("igneum-accept/".utf8) + for w in seed { b += le32(w) } + var rng = SplitMix64(s: fnv1a64Bytes(b)) + return (0.. String? { + let lanes = 32 + let loads = p.loadsPerHash + let mask: UInt32 = (1 << UInt32(acceptDatasetLog2)) - 1 + let (d0, d1) = (p.seed[0], p.seed[1]) + var andAcc = [UInt32](repeating: 0xffffffff, count: 8) + var orAcc = [UInt32](repeating: 0, count: 8) + var saturated: UInt32 = 0 + var bitOnes = [UInt32](repeating: 0, count: 64) + var distinctSum: UInt64 = 0 + var r = [UInt32](repeating: 0, count: 8 * lanes) // r[reg * 32 + lane] + var laneAddrs = [UInt32](repeating: 0, count: lanes * loads) + var sel = [UInt32](repeating: 0, count: lanes) + var idx = [UInt32](repeating: 0, count: lanes) + var tmp = [UInt32](repeating: 0, count: lanes) + for (unit, base) in acceptBaseNonces(p.seed).enumerated() { + for lane in 0..> UInt32(ins.bit)) & 1 + r[d + lane] = r[d + lane] &+ r[a + lane] &+ (s != 0 ? ins.imm2 : ins.imm) + } + case .sub: for lane in 0..> UInt64(j)) & 1) } + var sl = Array(laneAddrs[lane * loads..<(lane + 1) * loads]) + sl.sort() + var distinct: UInt64 = 0 + for k in 0..= acceptMaxSaturated { return "(c) \(saturated) of 16384 final register values saturated (limit 163)" } + let half = UInt32(acceptUnits * lanes / 2) + for j in 0..<64 { + let d = bitOnes[j] > half ? bitOnes[j] - half : half - bitOnes[j] + if d > acceptBiasTolerance { return "(c) output bit \(j) set in \(bitOnes[j]) of 2048 hashes" } + } + if distinctSum <= acceptMinDistinctSum { return "(c) distinct addresses \(distinctSum) over 2048 hashes (needs above 245760)" } + return nil +} + +func acceptProgram(_ p: Program) -> String? { acceptStatic(p) ?? acceptDynamic(p) } + +// The program of a seed under version 2: the first accepted candidate over attempts 0, 1, 2, ... +func generateProgramV2(seedString: String, bytes: [UInt8]) -> Program { + for attempt in 0.. String { String(format: "0x%08xu", v) } @@ -2452,19 +2690,30 @@ func runTests(_ opts: Options) -> Never { // MARK: - Serve mode (GPU worker for igneum-miner --worker), 3 October 2026 // -// Protocol, one line each. Only "found", "done" and "error" are parsed by the miner; every other line is logged. +// Protocol, one line each. Only "found", "done", "error", "prepared", "prepare-failed" and "ready" are parsed by the +// miner; every other line is logged. // stdin: job +// prepare [] compile that program and build that day's dataset in the +// background while jobs on the current seeds keep running // quit -// stdout: ready metal +// stdout: ready metal dataset-log2 batch prepare 1 // found every nonce whose 64-bit hash is <= target (hash <= target64) // done end of the job (wall ms, dispatch plus scan) // error -// The program for an epoch seed is generateProgram(words: seedWordsBytes(epoch_seed)), compiled once and cached; the -// cache and 1 GiB dataset for a day seed come from seedWordsBytes(day_seed_bytes), built once and cached (two of each). +// prepared program dataset ... the pair is resident +// prepare-failed +// The program for an epoch seed is generateProgramV2(bytes: epoch_seed) (version 2 with the acceptance rule, the same +// derivation as igneum-pow), compiled once and kept; the +// cache and 1 GiB dataset for a day seed come from seedWordsBytes(day_seed_bytes), built once and kept. At most two +// programs and two datasets are resident: the current job's pair and one more (the prepared pair, or the previous +// pair until the first job on the new one is done). A job whose pair is resident switches instantly; a job whose pair +// is not (no prepare, or a pair nobody predicted) compiles inline as before. The pack_dir of a prepare is ignored +// here (Metal compiles from the seed); ahead-of-time workers build their kernel from it. // The init words of a dispatch are seedWordsBytes("igneum-block/" || prehash || nonce_hi_le32) and go to buffer 3 of // igneum_hash_bound; the lane nonce is baseNonce + gid as in the bench kernel. -func emit(_ line: String) { print(line); fflush(stdout) } +let emitLock = NSLock() +func emit(_ line: String) { emitLock.lock(); print(line); fflush(stdout); emitLock.unlock() } func unhex(_ s: String) -> [UInt8]? { let chars = Array(s.utf8) @@ -2498,6 +2747,41 @@ final class ServeDataset { init(dayHex: String, ctx: DatasetContext, buffer: MTLBuffer) { self.dayHex = dayHex; self.ctx = ctx; self.buffer = buffer } } +// The resident programs and datasets, shared by the job loop (main thread) and the prepare queue (background). +final class ServeStore { + private let lock = NSLock() + private var programs = [ServeProgram]() + private var datasets = [ServeDataset]() + /// The pair of the last job (epoch seed hex, day seed hex); never evicted + var current: (String, String)? = nil + + func program(_ seedHex: String) -> ServeProgram? { lock.lock(); defer { lock.unlock() }; return programs.first { $0.seedHex == seedHex } } + func dataset(_ dayHex: String) -> ServeDataset? { lock.lock(); defer { lock.unlock() }; return datasets.first { $0.dayHex == dayHex } } + func counts() -> (Int, Int) { lock.lock(); defer { lock.unlock() }; return (programs.count, datasets.count) } + + /// Adds a program; with more than two resident, the oldest one that is not the current job's goes. + func add(_ p: ServeProgram) { + lock.lock(); defer { lock.unlock() } + if programs.contains(where: { $0.seedHex == p.seedHex }) { return } + programs.append(p) + while programs.count > 2, let i = programs.firstIndex(where: { $0.seedHex != current?.0 && $0.seedHex != p.seedHex }) { programs.remove(at: i) } + } + func add(_ d: ServeDataset) { + lock.lock(); defer { lock.unlock() } + if datasets.contains(where: { $0.dayHex == d.dayHex }) { return } + datasets.append(d) + while datasets.count > 2, let i = datasets.firstIndex(where: { $0.dayHex != current?.1 && $0.dayHex != d.dayHex }) { datasets.remove(at: i) } + } + /// After the first job on a new pair: drop everything but that pair (the old program and dataset are released). + func prune(to pair: (String, String)) -> (Int, Int) { + lock.lock(); defer { lock.unlock() } + let before = (programs.count, datasets.count) + programs.removeAll { $0.seedHex != pair.0 } + datasets.removeAll { $0.dayHex != pair.1 } + return (before.0 - programs.count, before.1 - datasets.count) + } +} + func compileBound(_ gpu: GPU, msl: String) throws -> CompiledHash { let t0 = nowNs() let lib = try gpu.device.makeLibrary(source: msl, options: MTLCompileOptions()) @@ -2508,18 +2792,63 @@ func compileBound(_ gpu: GPU, msl: String) throws -> CompiledHash { return CompiledHash(pipeline: pipe, libraryMs: ms(t0, t1), pipelineMs: ms(t1, t2)) } +// Builds the program for an epoch seed (hex) unless resident. Returns (program, compile ms) or throws. +func serveProgram(_ gpu: GPU, _ store: ServeStore, seedHex: String, seed: [UInt8], datasetLog2: Int) throws -> (ServeProgram, Double) { + if let p = store.program(seedHex) { return (p, 0) } + let t0 = nowNs() + let p = generateProgramV2(seedString: "epoch/" + seedHex, bytes: seed) + let msl = generateMSL(p, datasetLog2: datasetLog2, source: .stored, bound: true) + let c = try compileBound(gpu, msl: msl) + let sp = ServeProgram(seedHex: seedHex, program: p, compiled: c) + store.add(sp) + return (sp, ms(t0, nowNs())) +} + +// Builds the cache and dataset for a day seed (hex) unless resident. Returns (dataset, build ms). +func serveDataset(_ gpu: GPU, _ store: ServeStore, dayHex: String, day: [UInt8], datasetLog2: Int) -> (ServeDataset, Double) { + if let d = store.dataset(dayHex) { return (d, 0) } + let t0 = nowNs() + let key = seedWordsBytes(day) + let ctx = DatasetContext(gpu: gpu, closedForm: false, dayString: "day/" + dayHex, key: key) + let buf = ctx.makeDataset(log2: datasetLog2) + let sd = ServeDataset(dayHex: dayHex, ctx: ctx, buffer: buf) + store.add(sd) + return (sd, ms(t0, nowNs())) +} + func runServe(_ opts: Options) -> Never { let gpu = GPU() let datasetLog2 = opts.datasetLog2 let batch = 1 << opts.batchLog2 // nonces per dispatch - var programs = [ServeProgram]() - var datasets = [ServeDataset]() + let store = ServeStore() + let prepareQueue = DispatchQueue(label: "igneum.prepare") // one prepare at a time, off the job loop guard let outBuf = gpu.device.makeBuffer(length: batch * 8, options: .storageModeShared) else { emit("error 0 cannot allocate the output buffer"); exit(1) } - emit("ready metal \(gpu.device.name.replacingOccurrences(of: " ", with: "_")) dataset-log2 \(datasetLog2) batch \(batch)") + emit("ready metal \(gpu.device.name.replacingOccurrences(of: " ", with: "_")) dataset-log2 \(datasetLog2) batch \(batch) prepare \(opts.noPrepare ? 0 : 1)") + var lastPair: (String, String)? = nil while let line = readLine(strippingNewline: true) { let f = line.split(separator: " ").map(String.init) if f.isEmpty { continue } if f[0] == "quit" { break } + if f[0] == "prepare" { + if opts.noPrepare { emit("info ignored (started with --no-prepare): \(line)"); continue } + if f.count < 3 { emit("prepare-failed 0 0 malformed prepare line (need epoch_seed_hex and day_seed_hex)"); continue } + guard let epochSeed = unhex(f[1]), epochSeed.count == 32, let daySeed = unhex(f[2]) else { + emit("prepare-failed \(f[1]) \(f[2]) bad field (epoch_seed 64 hex, day_seed hex)"); continue + } + let (epochHex, dayHex) = (f[1], f[2]) + let have = (store.program(epochHex) != nil, store.dataset(dayHex) != nil) + if have.0 && have.1 { emit("prepared \(epochHex) \(dayHex) 0 program 0 dataset 0 (already resident)"); continue } + prepareQueue.async { + let t0 = nowNs() + do { + let (sp, progMs) = try serveProgram(gpu, store, seedHex: epochHex, seed: epochSeed, datasetLog2: datasetLog2) + let (sd, dsMs) = serveDataset(gpu, store, dayHex: dayHex, day: daySeed, datasetLog2: datasetLog2) + let (np, nd) = store.counts() + emit("prepared \(epochHex) \(dayHex) \(fmt(ms(t0, nowNs()), 1)) program \(fmt(progMs, 1)) dataset \(fmt(dsMs, 1)) loads/hash \(sp.program.loadsPerHash) cache-fill \(fmt(sd.ctx.cacheFillGPUms, 1)) resident \(np) programs \(nd) datasets") + } catch { emit("prepare-failed \(epochHex) \(dayHex) Metal compile failed: \(error)") } + } + continue + } if f[0] != "job" { emit("info ignored: \(line)"); continue } if f.count < 8 { emit("error \(f.count > 1 ? f[1] : "0") malformed job line (need 7 fields after job)"); continue } let jobId = f[1] @@ -2530,31 +2859,24 @@ func runServe(_ opts: Options) -> Never { } if nonceCount == 0 || nonceCount % 32 != 0 || (nonceStart & 31) != 0 { emit("error \(jobId) nonce_start must be 32-aligned and nonce_count a non-zero multiple of 32"); continue } let t0 = nowNs() - // Program for the epoch seed - var prog = programs.first { $0.seedHex == f[6] } - if prog == nil { - let p = generateProgram(seedString: "epoch/" + f[6], words: seedWordsBytes(epochSeed)) - let msl = generateMSL(p, datasetLog2: datasetLog2, source: .stored, bound: true) - do { - let c = try compileBound(gpu, msl: msl) - emit("info program epoch \(f[6].prefix(16)) loads/hash \(p.loadsPerHash) compiled in \(fmt(c.totalMs, 1)) ms") - let sp = ServeProgram(seedHex: f[6], program: p, compiled: c) - programs.append(sp); if programs.count > 2 { programs.removeFirst() } - prog = sp - } catch { emit("error \(jobId) Metal compile failed: \(error)"); continue } + let pair = (f[6], f[7]) + let switched = lastPair == nil || lastPair! != pair + if switched { store.current = pair } + // Program and dataset for the pair: resident (prepared, or the current pair) or compiled inline now. A prepare + // of the same pair may be in flight on the queue; waiting for the queue makes this a join instead of a double build. + let program: ServeProgram + let dataset: ServeDataset + do { + if store.program(pair.0) == nil || store.dataset(pair.1) == nil { prepareQueue.sync {} } + let (sp, progMs) = try serveProgram(gpu, store, seedHex: pair.0, seed: epochSeed, datasetLog2: datasetLog2) + if progMs > 0 { emit("info program epoch \(pair.0.prefix(16)) loads/hash \(sp.program.loadsPerHash) compiled inline in \(fmt(progMs, 1)) ms (not prepared)") } + let (sd, dsMs) = serveDataset(gpu, store, dayHex: pair.1, day: daySeed, datasetLog2: datasetLog2) + if dsMs > 0 { emit("info dataset day \(pair.1) built inline in \(fmt(dsMs, 1)) ms (cache fill \(fmt(sd.ctx.cacheFillGPUms, 1)) ms, build \(fmt(sd.ctx.lastBuildGPUms, 1)) ms GPU; not prepared)") } + program = sp; dataset = sd + } catch { emit("error \(jobId) Metal compile failed: \(error)"); continue } + if switched, let prev = lastPair { + emit("info switched from epoch \(prev.0.prefix(16)) day \(prev.1) to epoch \(pair.0.prefix(16)) day \(pair.1) in \(fmt(ms(t0, nowNs()), 2)) ms (resident: \(store.counts().0) programs, \(store.counts().1) datasets)") } - // Cache and dataset for the day seed - var ds = datasets.first { $0.dayHex == f[7] } - if ds == nil { - let key = seedWordsBytes(daySeed) - let ctx = DatasetContext(gpu: gpu, closedForm: false, dayString: "day/" + f[7], key: key) - let buf = ctx.makeDataset(log2: datasetLog2) - emit("info dataset day \(f[7]) key \(key.map { String(format: "%08x", $0) }.joined(separator: " ")) cache fill \(fmt(ctx.cacheFillGPUms, 1)) ms build \(fmt(ctx.lastBuildGPUms, 1)) ms GPU") - let sd = ServeDataset(dayHex: f[7], ctx: ctx, buffer: buf) - datasets.append(sd); if datasets.count > 2 { datasets.removeFirst() } - ds = sd - } - guard let program = prog, let dataset = ds else { continue } // Mine: chunks of at most `batch` lane nonces that share one high word var remaining = nonceCount var hi = UInt32(truncatingIfNeeded: nonceStart >> 32) @@ -2592,6 +2914,12 @@ func runServe(_ opts: Options) -> Never { } if failed { continue } emit("done \(jobId) \(hashes) \(fmt(ms(t0, nowNs()), 2))") + if switched, lastPair != nil { + // The first job on the new pair is done: the old pair goes (at most two of each were resident until now) + let (dp, dd) = store.prune(to: pair) + if dp + dd > 0 { emit("info dropped \(dp) program(s) and \(dd) dataset(s) of the previous pair") } + } + lastPair = pair } exit(0) } diff --git a/proto-opencl/host.c b/proto-opencl/host.c index be7a0432b..183cdc70a 100644 --- a/proto-opencl/host.c +++ b/proto-opencl/host.c @@ -31,6 +31,7 @@ #else #include #include +#include #endif #define IGNEUM_NO_CUDA @@ -192,6 +193,7 @@ typedef struct { const char* kernelPath; const char* extraOpts; int serve; // --serve: GPU worker for igneum-miner --worker (jobs on stdin), 3 October 2026 + int noPrepare; // --no-prepare: serve without the prepare command (ready line says "prepare 0"), to test the miner's fallback int kernelGiven; // --kernel was passed const char* vendor; // --vendor S: pick the first GPU whose vendor string contains S (default: first GPU of any vendor) } Options; @@ -217,6 +219,7 @@ static void usage(void) { " Apple's OpenCL runtime reports unusable event timestamps, so wall is the default on the Apple platform.\n" " --vendor S choose the first GPU whose vendor string contains S (for example \"Advanced Micro Devices\"); fails if none\n" " --serve GPU worker for igneum-miner --worker: reads \"job ...\" lines on stdin, prints found/done lines.\n" + " --no-prepare with --serve: no prepare support (the miner then falls back to exit 42 at a seed change).\n" " Builds the pack's kernel_bound.cl (next to the compiled-in kernel.cl) unless --kernel says otherwise.\n", packMib(), IGNEUM_KERNEL_PATH); } @@ -227,7 +230,7 @@ static Options parseArgs(int argc, char** argv) { Options o; int i; o.datasetMib = 1024; o.batchLog2 = 24; o.batches = 5; o.groupWarps = 1; o.sweep = 0; o.device = -1; - o.exchange = 0; o.list = 0; o.timeWall = -1; o.kernelPath = IGNEUM_KERNEL_PATH; o.extraOpts = ""; o.serve = 0; o.kernelGiven = 0; o.vendor = NULL; + o.exchange = 0; o.list = 0; o.timeWall = -1; o.kernelPath = IGNEUM_KERNEL_PATH; o.extraOpts = ""; o.serve = 0; o.noPrepare = 0; o.kernelGiven = 0; o.vendor = NULL; for (i = 1; i < argc; ++i) { const char* a = argv[i]; int needs = (strcmp(a, "--dataset-mib") == 0 || strcmp(a, "--batch-log2") == 0 || strcmp(a, "--batches") == 0 || @@ -241,6 +244,7 @@ static Options parseArgs(int argc, char** argv) { else if (strcmp(a, "--device") == 0) o.device = atoi(argv[++i]); else if (strcmp(a, "--kernel") == 0) { o.kernelPath = argv[++i]; o.kernelGiven = 1; } else if (strcmp(a, "--serve") == 0) o.serve = 1; + else if (strcmp(a, "--no-prepare") == 0) o.noPrepare = 1; else if (strcmp(a, "--vendor") == 0) { if (i + 1 >= argc) { usage(); exit(2); } o.vendor = argv[++i]; } else if (strcmp(a, "--build-opts") == 0) o.extraOpts = argv[++i]; else if (strcmp(a, "--time") == 0) { @@ -848,9 +852,20 @@ static SizeResult runSize(Device* dv, const DeviceInfo* di, const Options* o, in * found every nonce whose 64-bit hash is <= target * done end of the job (wall ms) * error - * The pack's program.h is compiled in and kernel_bound.cl is built at runtime, so this worker serves exactly one - * epoch seed and one day seed: the pack's. A job for other seeds is answered with an error naming both; re-export the - * pack with `igneum-miner export-pack ` and rebuild. The init words of a dispatch are + * prepare build /kernel_bound.cl (the pack the miner wrote + * for those seeds) plus its cache and dataset in the background + * stdout: ready opencl ... prepare 1 + * prepared ... the pair is resident; a job on it switches instantly + * prepare-failed + * The pack's program.h is compiled in and kernel_bound.cl is built at runtime, so at start this worker serves exactly + * one epoch seed and one day seed: the pack's. The next pair arrives through `prepare`: the miner writes the pack for the + * prepared seeds (igneum-miner --prepare-packs ) and names its directory; a background thread builds that pack's + * kernel_bound.cl with the same build options, fills its cache and builds its dataset on a second queue while jobs on + * the current pair keep running (at most two pairs resident: the current one and the prepared one; the old pair is + * released after the first job on the new one). A job for seeds that are neither the current nor the prepared pair is + * answered with an error naming both. Without prepare, re-export the pack with `igneum-miner export-pack ` + * and rebuild. A prepared pair's cache is not cross-checked against a host fill (memhard.h is compiled in for the + * original day); the miner's CPU re-check of every found nonce covers it. The init words of a dispatch are * seed_words_from_bytes("igneum-block/" || prehash || nonce_hi_le32), written to a small buffer that is the fifth * argument of igneum_hash_bound; the lane nonce is baseNonce + gid as in the bench kernel. The exchange rule of * WAVEFRONT.md applies unchanged (the bound kernel has the same body and the same IGNEUM_EXCHANGE build). @@ -881,6 +896,132 @@ static int unhexBuf(const char* s, uint8_t* out, size_t cap, size_t* len) { return 1; } +#if IGNEUM_DATASET_MODE == 1 +/* One resident (program, cache, dataset) triple for a seed pair. The first one is the compiled-in pack on the main + * queue; prepared ones are built from a pack directory on their own queue. */ +typedef struct { + char epochHex[65]; + char dayHex[512]; + uint32_t sw[8], kw[8]; /* seed words and key words, for job matching */ + cl_program prog; + cl_kernel kHashBound, kCacheFill, kBuild; + cl_mem cache, ds; + double buildMs, cacheMs, datasetMs; +} ServePair; + +static void releasePair(ServePair* p) { + if (!p) return; + if (p->ds) clReleaseMemObject(p->ds); + if (p->cache) clReleaseMemObject(p->cache); + if (p->kHashBound) clReleaseKernel(p->kHashBound); + if (p->kCacheFill) clReleaseKernel(p->kCacheFill); + if (p->kBuild) clReleaseKernel(p->kBuild); + if (p->prog) clReleaseProgram(p->prog); + free(p); +} + +/* The prepare request and its result, handed between the main loop and the prepare thread. */ +typedef struct { + Device* dv; + const DeviceInfo* di; + char epochHex[65]; + char dayHex[512]; + char packDir[1024]; + uint32_t words; + char error[512]; + ServePair* result; /* set by the thread on success */ + volatile int done; /* 1 when the thread has finished (success or failure) */ + double t0, doneAt; +} PrepareTask; + +static void prepareFail(PrepareTask* t, const char* what, cl_int err) { + snprintf(t->error, sizeof(t->error), "%s (%s)", what, clErrName(err)); +} + +/* Builds the pair for a prepare request. Runs on its own thread with its own command queue. */ +static void prepareRun(PrepareTask* t) { + cl_int err = 0; + size_t srcLen = 0; + char path[1200]; + char* src; + ServePair* p = (ServePair*)calloc(1, sizeof(ServePair)); + cl_command_queue q = NULL; + double tb; + cl_uint nSeg = IGNEUM_CACHE_SEGMENTS, nItems; + size_t local, bytes = (size_t)CACHE_WORDS_HOST * 4u; + strncpy(p->epochHex, t->epochHex, 64); p->epochHex[64] = 0; + strncpy(p->dayHex, t->dayHex, sizeof(p->dayHex) - 1); + snprintf(path, sizeof(path), "%s/kernel_bound.cl", t->packDir); + src = readFile(path, &srcLen); + if (!src) { snprintf(t->error, sizeof(t->error), "cannot read %s", path); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; } + tb = wallMs(); + p->prog = clCreateProgramWithSource(t->dv->ctx, 1, (const char**)&src, &srcLen, &err); + free(src); + if (err != CL_SUCCESS) { prepareFail(t, "clCreateProgramWithSource", err); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; } + err = clBuildProgram(p->prog, 1, &t->di->device, t->dv->buildOptions, NULL, NULL); + if (err != CL_SUCCESS) { + size_t logLen = 0; + char* log; + clGetProgramBuildInfo(p->prog, t->di->device, CL_PROGRAM_BUILD_LOG, 0, NULL, &logLen); + log = (char*)calloc(logLen + 1, 1); + if (logLen) clGetProgramBuildInfo(p->prog, t->di->device, CL_PROGRAM_BUILD_LOG, logLen, log, NULL); + snprintf(t->error, sizeof(t->error), "clBuildProgram failed (%s): %.300s", clErrName(err), log); + free(log); releasePair(p); t->done = 1; return; + } + p->buildMs = wallMs() - tb; + p->kHashBound = clCreateKernel(p->prog, "igneum_hash_bound", &err); + if (err != CL_SUCCESS) { prepareFail(t, "clCreateKernel igneum_hash_bound (is this a kernel_bound.cl?)", err); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; } + p->kCacheFill = clCreateKernel(p->prog, "igneum_cache_fill", &err); + if (err != CL_SUCCESS) { prepareFail(t, "clCreateKernel igneum_cache_fill", err); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; } + p->kBuild = clCreateKernel(p->prog, "igneum_build", &err); + if (err != CL_SUCCESS) { prepareFail(t, "clCreateKernel igneum_build", err); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; } + q = clCreateCommandQueue(t->dv->ctx, t->di->device, 0, &err); + if (err != CL_SUCCESS) { prepareFail(t, "clCreateCommandQueue (prepare)", err); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; } + /* Cache: the same segment count as the compiled-in pack (the dataset schedule is a network constant) */ + tb = wallMs(); + p->cache = clCreateBuffer(t->dv->ctx, CL_MEM_READ_WRITE, bytes, NULL, &err); + if (err != CL_SUCCESS) { prepareFail(t, "clCreateBuffer cache", err); clReleaseCommandQueue(q); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; } + local = kernelMaxLocal(t->dv, p->kCacheFill, t->di, 256); + { + size_t g = ((nSeg + local - 1) / local) * local; + err = clSetKernelArg(p->kCacheFill, 0, sizeof(cl_mem), &p->cache); + if (err == CL_SUCCESS) err = clSetKernelArg(p->kCacheFill, 1, sizeof(cl_uint), &nSeg); + if (err == CL_SUCCESS) err = clEnqueueNDRangeKernel(q, p->kCacheFill, 1, NULL, &g, &local, 0, NULL, NULL); + if (err == CL_SUCCESS) err = clFinish(q); + } + if (err != CL_SUCCESS) { prepareFail(t, "cache fill", err); clReleaseCommandQueue(q); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; } + p->cacheMs = wallMs() - tb; + /* Dataset */ + tb = wallMs(); + nItems = t->words / 16u; + p->ds = clCreateBuffer(t->dv->ctx, CL_MEM_READ_WRITE, (size_t)t->words * 4u, NULL, &err); + if (err != CL_SUCCESS) { prepareFail(t, "clCreateBuffer dataset", err); clReleaseCommandQueue(q); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; } + local = kernelMaxLocal(t->dv, p->kBuild, t->di, 256); + { + size_t g = ((nItems + local - 1) / local) * local; + err = clSetKernelArg(p->kBuild, 0, sizeof(cl_mem), &p->ds); + if (err == CL_SUCCESS) err = clSetKernelArg(p->kBuild, 1, sizeof(cl_mem), &p->cache); + if (err == CL_SUCCESS) err = clSetKernelArg(p->kBuild, 2, sizeof(cl_uint), &nItems); + if (err == CL_SUCCESS) err = clEnqueueNDRangeKernel(q, p->kBuild, 1, NULL, &g, &local, 0, NULL, NULL); + if (err == CL_SUCCESS) err = clFinish(q); + } + if (err != CL_SUCCESS) { prepareFail(t, "dataset build", err); clReleaseCommandQueue(q); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; } + p->datasetMs = wallMs() - tb; + clReleaseCommandQueue(q); + t->result = p; + t->doneAt = wallMs(); + t->done = 1; +} + +#ifdef _WIN32 +static DWORD WINAPI prepareThreadMain(LPVOID arg) { prepareRun((PrepareTask*)arg); return 0; } +static int startPrepareThread(PrepareTask* t) { HANDLE h = CreateThread(NULL, 0, prepareThreadMain, t, 0, NULL); if (!h) return 0; CloseHandle(h); return 1; } +#else +static void* prepareThreadMain(void* arg) { prepareRun((PrepareTask*)arg); return NULL; } +static int startPrepareThread(PrepareTask* t) { pthread_t th; if (pthread_create(&th, NULL, prepareThreadMain, t) != 0) return 0; pthread_detach(th); return 1; } +#endif +#endif + static int runServe(Device* dv, const DeviceInfo* di, const Options* o) { #if IGNEUM_DATASET_MODE != 1 (void)dv; (void)di; (void)o; @@ -893,26 +1034,35 @@ static int runServe(Device* dv, const DeviceInfo* di, const Options* o) { const uint32_t batch = 1u << (o->batchLog2 == 24 ? 22 : o->batchLog2); /* 2^22 nonces per dispatch by default */ size_t groupSize = 32 * (size_t)o->groupWarps; cl_int err = 0; - cl_mem dDs, dOut, dInit; + cl_mem dOut, dInit; cl_uint nItems = words / 16u; uint64_t* hOut; char devName[256]; - char line[1024]; + char line[2048]; size_t k; + ServePair* cur; /* the pair jobs run on */ + ServePair* prepared = NULL; /* the pair the last prepare built, until a job switches to it */ + ServePair* old = NULL; /* the previous pair, released after the first job on the new one */ + PrepareTask* task = NULL; /* the prepare in flight */ if (!dv->kHashBound) { printf("error 0 the kernel source has no igneum_hash_bound (build from the pack's kernel_bound.cl, or pass --kernel)\n"); fflush(stdout); return 2; } if (!setupCache(dv, di)) { printf("error 0 cache check failed (device cache differs from the host cache or the pack's FNV)\n"); fflush(stdout); return 1; } - dDs = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, (size_t)words * 4u, NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer dataset"); - CL_CHECK(clSetKernelArg(dv->kBuild, 0, sizeof(cl_mem), &dDs)); - CL_CHECK(clSetKernelArg(dv->kBuild, 1, sizeof(cl_mem), &gCache)); - CL_CHECK(clSetKernelArg(dv->kBuild, 2, sizeof(cl_uint), &nItems)); - clReleaseEvent(launch1D(dv, dv->kBuild, nItems, kernelMaxLocal(dv, dv->kBuild, di, 256))); + cur = (ServePair*)calloc(1, sizeof(ServePair)); + memcpy(cur->sw, SEEDW, 32); memcpy(cur->kw, KEYW, 32); + cur->kHashBound = dv->kHashBound; cur->kCacheFill = dv->kCacheFill; cur->kBuild = dv->kBuild; cur->prog = dv->prog; + dv->kHashBound = dv->kCacheFill = dv->kBuild = NULL; dv->prog = NULL; /* owned by the pair now */ + cur->cache = gCache; gCache = NULL; + cur->ds = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, (size_t)words * 4u, NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer dataset"); + CL_CHECK(clSetKernelArg(cur->kBuild, 0, sizeof(cl_mem), &cur->ds)); + CL_CHECK(clSetKernelArg(cur->kBuild, 1, sizeof(cl_mem), &cur->cache)); + CL_CHECK(clSetKernelArg(cur->kBuild, 2, sizeof(cl_uint), &nItems)); + clReleaseEvent(launch1D(dv, cur->kBuild, nItems, kernelMaxLocal(dv, cur->kBuild, di, 256))); CL_CHECK(clFinish(dv->q)); dOut = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, (size_t)batch * sizeof(uint64_t), NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer out"); dInit = clCreateBuffer(dv->ctx, CL_MEM_READ_ONLY, 32, NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer init words"); hOut = (uint64_t*)malloc((size_t)batch * sizeof(uint64_t)); strncpy(devName, di->name, 255); devName[255] = 0; for (k = 0; devName[k]; ++k) if (devName[k] == ' ') devName[k] = '_'; - printf("ready opencl %s platform %s pack %s dataset-log2 %d batch %u exchange %d\n", devName, di->platformName, IGNEUM_SEED_STRING, IGNEUM_DATASET_LOG2, batch, dv->exchange); + printf("ready opencl %s platform %s pack %s dataset-log2 %d batch %u exchange %d prepare %d\n", devName, di->platformName, IGNEUM_SEED_STRING, IGNEUM_DATASET_LOG2, batch, dv->exchange, o->noPrepare ? 0 : 1); fflush(stdout); while (fgets(line, sizeof(line), stdin)) { @@ -928,11 +1078,42 @@ static int runServe(Device* dv, const DeviceInfo* di, const Options* o) { uint64_t remaining, hashes = 0; uint32_t hi, lo; double t0; - int failed = 0; + int failed = 0, switched = 0; line[strcspn(line, "\r\n")] = 0; + /* A finished prepare is reported here, between lines (the thread never prints) */ + if (task && task->done) { + if (task->result) { + ServePair* p = task->result; + { + uint8_t eb[32], db[256]; size_t el = 0, dl = 0; + if (unhexBuf(p->epochHex, eb, 32, &el) && el == 32) seedWordsFromBytes(eb, 32, p->sw); + if (unhexBuf(p->dayHex, db, sizeof(db), &dl)) seedWordsFromBytes(db, dl, p->kw); + } + if (prepared) releasePair(prepared); + prepared = p; + printf("prepared %s %s %.1f build %.1f cache %.1f dataset %.1f resident 2 programs 2 datasets\n", p->epochHex, p->dayHex, task->doneAt - task->t0, p->buildMs, p->cacheMs, p->datasetMs); + } else { + printf("prepare-failed %s %s %s\n", task->epochHex, task->dayHex, task->error); + } + fflush(stdout); + free(task); task = NULL; + } for (tok = strtok_r(line, " ", &save); tok && nf < 9; tok = strtok_r(NULL, " ", &save)) f[nf++] = tok; if (nf == 0) continue; if (strcmp(f[0], "quit") == 0) break; + if (strcmp(f[0], "prepare") == 0) { + if (o->noPrepare) { printf("info ignored (started with --no-prepare): prepare\n"); fflush(stdout); continue; } + if (nf < 4) { printf("prepare-failed %s %s this ahead-of-time worker needs a pack directory as the third field (igneum-miner --prepare-packs )\n", nf > 1 ? f[1] : "0", nf > 2 ? f[2] : "0"); fflush(stdout); continue; } + if (strlen(f[1]) != 64 || strlen(f[2]) >= 500) { printf("prepare-failed %s %s bad field (epoch_seed 64 hex, day_seed hex)\n", f[1], f[2]); fflush(stdout); continue; } + if (task) { printf("prepare-failed %s %s a prepare is still running\n", f[1], f[2]); fflush(stdout); continue; } + if (prepared && strcmp(prepared->epochHex, f[1]) == 0 && strcmp(prepared->dayHex, f[2]) == 0) { printf("prepared %s %s 0 (already resident)\n", f[1], f[2]); fflush(stdout); continue; } + task = (PrepareTask*)calloc(1, sizeof(PrepareTask)); + task->dv = dv; task->di = di; task->words = words; task->t0 = wallMs(); + strncpy(task->epochHex, f[1], 64); strncpy(task->dayHex, f[2], sizeof(task->dayHex) - 1); strncpy(task->packDir, f[3], sizeof(task->packDir) - 1); + if (!startPrepareThread(task)) { printf("prepare-failed %s %s cannot start the prepare thread\n", f[1], f[2]); fflush(stdout); free(task); task = NULL; continue; } + printf("info prepare started for epoch %.16s day %s from %s (builds in the background)\n", f[1], f[2], f[3]); fflush(stdout); + continue; + } if (strcmp(f[0], "job") != 0) { printf("info ignored line\n"); fflush(stdout); continue; } strncpy(jobId, nf > 1 ? f[1] : "0", 63); jobId[63] = 0; if (nf < 8) { printf("error %s malformed job line (need 7 fields after job)\n", jobId); fflush(stdout); continue; } @@ -944,17 +1125,23 @@ static int runServe(Device* dv, const DeviceInfo* di, const Options* o) { if (nonceCount == 0 || nonceCount % 32 != 0 || (nonceStart & 31) != 0) { printf("error %s nonce_start must be 32-aligned and nonce_count a non-zero multiple of 32\n", jobId); fflush(stdout); continue; } seedWordsFromBytes(epochSeed, 32, sw); seedWordsFromBytes(daySeed, dayLen, kw); - if (memcmp(sw, SEEDW, 32) != 0) { - printf("error %s epoch seed mismatch: this worker was built for pack \"%s\" (seed words %08x %08x ...), the job's epoch seed %.16s gives %08x %08x ...; run igneum-miner export-pack and rebuild\n", - jobId, IGNEUM_SEED_STRING, SEEDW[0], SEEDW[1], f[6], sw[0], sw[1]); - fflush(stdout); continue; - } - if (memcmp(kw, KEYW, 32) != 0) { - printf("error %s day seed mismatch: this worker's cache is for key %08x %08x ..., the job's day seed %s gives %08x %08x ...; run igneum-miner export-pack and rebuild\n", - jobId, KEYW[0], KEYW[1], f[7], kw[0], kw[1]); - fflush(stdout); continue; - } t0 = wallMs(); + if (memcmp(sw, cur->sw, 32) != 0 || memcmp(kw, cur->kw, 32) != 0) { + if (prepared && memcmp(sw, prepared->sw, 32) == 0 && memcmp(kw, prepared->kw, 32) == 0) { + /* The prepared pair: switch now, release the old one after this job */ + if (old) releasePair(old); + old = cur; cur = prepared; prepared = NULL; switched = 1; + printf("info switched to the prepared pair epoch %.16s day %s in %.2f ms\n", cur->epochHex, cur->dayHex, wallMs() - t0); fflush(stdout); + } else if (memcmp(sw, cur->sw, 32) != 0) { + printf("error %s epoch seed mismatch: this worker holds %s%s (seed words %08x %08x ...)%s, the job's epoch seed %.16s gives %08x %08x ...; send prepare with a pack directory, or run igneum-miner export-pack and rebuild\n", + jobId, cur->epochHex[0] ? "prepared epoch " : "pack \"" IGNEUM_SEED_STRING "\"", cur->epochHex[0] ? cur->epochHex : "", cur->sw[0], cur->sw[1], prepared ? " plus one prepared pair" : "", f[6], sw[0], sw[1]); + fflush(stdout); continue; + } else { + printf("error %s day seed mismatch: this worker's cache is for key %08x %08x ..., the job's day seed %s gives %08x %08x ...; send prepare with a pack directory, or run igneum-miner export-pack and rebuild\n", + jobId, cur->kw[0], cur->kw[1], f[7], kw[0], kw[1]); + fflush(stdout); continue; + } + } remaining = nonceCount; hi = (uint32_t)(nonceStart >> 32); lo = (uint32_t)nonceStart; while (remaining > 0) { @@ -973,13 +1160,15 @@ static int runServe(Device* dv, const DeviceInfo* di, const Options* o) { seedWordsFromBytes(b, 49, iw); CL_CHECK(clEnqueueWriteBuffer(dv->q, dInit, CL_TRUE, 0, 32, iw, 0, NULL, NULL)); baseNonce = lo; - CL_CHECK(clSetKernelArg(dv->kHashBound, 0, sizeof(cl_mem), &dDs)); - CL_CHECK(clSetKernelArg(dv->kHashBound, 1, sizeof(cl_mem), &dOut)); - CL_CHECK(clSetKernelArg(dv->kHashBound, 2, sizeof(cl_uint), &baseNonce)); - CL_CHECK(clSetKernelArg(dv->kHashBound, 3, sizeof(cl_uint), &maskArg)); - CL_CHECK(clSetKernelArg(dv->kHashBound, 4, sizeof(cl_mem), &dInit)); - ev = launch1D(dv, dv->kHashBound, chunk, groupSize); - err = clFinish(dv->q); + CL_CHECK(clSetKernelArg(cur->kHashBound, 0, sizeof(cl_mem), &cur->ds)); + CL_CHECK(clSetKernelArg(cur->kHashBound, 1, sizeof(cl_mem), &dOut)); + CL_CHECK(clSetKernelArg(cur->kHashBound, 2, sizeof(cl_uint), &baseNonce)); + CL_CHECK(clSetKernelArg(cur->kHashBound, 3, sizeof(cl_uint), &maskArg)); + CL_CHECK(clSetKernelArg(cur->kHashBound, 4, sizeof(cl_mem), &dInit)); + ev = launch1D(dv, cur->kHashBound, chunk, groupSize); + /* Wait on the dispatch event, not clFinish: the runtime can sleep the thread on an event, where clFinish + * on some drivers spins one core for the whole dispatch. */ + err = clWaitForEvents(1, &ev); clReleaseEvent(ev); if (err != CL_SUCCESS) { printf("error %s dispatch failed: %s\n", jobId, clErrName(err)); fflush(stdout); failed = 1; break; } CL_CHECK(clEnqueueReadBuffer(dv->q, dOut, CL_TRUE, 0, (size_t)chunk * sizeof(uint64_t), hOut, 0, NULL, NULL)); @@ -994,12 +1183,15 @@ static int runServe(Device* dv, const DeviceInfo* di, const Options* o) { } if (failed) continue; printf("done %s %llu %.2f\n", jobId, (unsigned long long)hashes, wallMs() - t0); + if (switched && old) { releasePair(old); old = NULL; printf("info dropped the previous pair (its program, cache and dataset)\n"); } fflush(stdout); } free(hOut); clReleaseMemObject(dInit); clReleaseMemObject(dOut); - clReleaseMemObject(dDs); + if (old) releasePair(old); + if (prepared) releasePair(prepared); + releasePair(cur); return 0; #endif } diff --git a/site/bench.html b/site/bench.html index 17ae5a827..24625d7db 100644 --- a/site/bench.html +++ b/site/bench.html @@ -64,11 +64,11 @@ footer{border-top:1px solid var(--line);padding-block:32px 48px;font-size:13px;c
-
IGNEUM
+
IGNEUM

Engineering log

Every measurement the project has made, newest at the bottom, written by the people and agents who ran it, with the commands and hardware. Prototype numbers are not mining numbers and say so.

- +

Igneum bench log

Append-only. Every number here was measured on the machine named, on the date given.

2026-10-03 proto-metal / igneum-bench, first run

@@ -137,7 +137,58 @@ footer{border-top:1px solid var(--line);padding-block:32px 48px;font-size:13px;c

Full JSON per scenario under /tmp/igneum-harness/results and /tmp/igneum-harness/sim. The simulator (igneum/harness-sim in the fork worktree) runs real consensus code in virtual time with PoW skipped, as rusty-kaspa simpa does; the live scenarios (5, 6, 7 Part B) drive real igneumd processes over wRPC and the fork's own p2p (igneum/p2p-probe).

Finality and difficulty-controller scenarios are stubs here: their criteria are written and they run against those branches once merged into the harness worktree (see tools/harness/scenarios/stubs.mjs).

3 October 2026, weak-program census: 400,000 program runs through the CPU reference, the redundant-load finding, and the rules for M5 and M6 (cryptographer)

-

Machine: Apple M5 Max, 8 threads at nice -n 15 while a devnet build and its simulations shared the box (load average 25 to 107), rustc 1.99.0, release build with LTO. New crate igneum-census/ (path dependency on igneum-pow, nothing in igneum-pow changed); the instrumented interpreter is checked against igneum_pow::hash_warp on the first warp of every program and against the igneum-genesis spec vectors at start. Commands: igneum-census run --root igneum-census-2026-10-03 --count 100000 --warps 128 --threads 8 --gen default (memory-hard, 1 GiB, day 2026-10-03; 2,797 s), the same with --gen fixed16-fresh --warps 64 (1,049 s), --gen fixed16-fresh2 --warps 64 (6,650 s, starved to under a core for most of it), and --gen default --closed-form --warps 128 (125.6 s once the machine was quiet); summarise, probe, show. Full tables and the rules in docs/analysis/weak-program-census-2026-10-03.md. Current generator, 100,000 programs x 4,096 nonces: loads per hash 24 to 256 (mean 127.9); distinct addresses per hash 24 to 200 (mean 102.4): 19.9 percent of all loads re-read an address the same hash already read, 94.8 percent of programs have at least one such load, 22.5 percent have a pair that cancels to the identity. The Mac rates of the eight bench seeds vary 1.38x by static loads/s and 1.10x by distinct loads/s (3.51 to 3.85 G/s), so the GPU is bound by the distinct count; igneum-second-seed/epoch1 (144 static, 104 distinct) hashes at the rate of igneum-second-seed (104 and 104). Weak programs, current generator: 2.43 percent have a register with no injecting write (saturates to all ones); 0.73 percent have a register with a nonce-independent bit, 0.03 percent a whole nonce-independent register, 0.03 percent a load site read at one address by all 32 lanes, 0.68 percent more than 1 percent of final registers at 0 or all ones, 6 programs an output bit past 6 sigma (0.03 expected by chance). Avalanche clean on every program (mean 31.85 to 32.15, every output bit flips 0.473 to 0.523). Mechanisms: no injecting write; zero-absorbing register sets closed under mulhi/mul; or or mul as the last write. Rules: G1 exactly 16 load slots drawn first from slots 1..63; G2 a load reads only a register written earlier in the program and not read by a load since; R-a no cyclically redundant load; R-b every register has an injecting write; R-c 64 fixed warps on the seed-keyed closed-form dataset with no constant register bit, no lane-constant site, saturation under 1 percent, no output bit past 6 sigma, more than 120 distinct addresses per hash on average. Rejection: 95.0 percent under the current generator (the redundancy alone), 38.3 percent under the first form of G2 (the iteration wrap), 5.14 percent under the proposed form (R-a or R-b 3.93 percent, R-c 2.05 percent), so 1.054 candidates per epoch on average; the accepted population does 120.05 to 128 distinct loads per hash, median 128.00. Closed-form check: with the same seeds and nonces on the closed-form dataset instead of the memory-hard one, R-c agrees on 99,961 of the 100,000 proposed-generator programs (2,054 rejected memory-hard, 2,055 closed-form; the 39 that differ sit at a threshold edge, one nearly constant bit or a bias near 6 sigma) and the per-program metrics agree to three decimals, so the acceptance test can be a pure function of the program. Hash-rate spread: today 2.7x between the 1st and 99th percentile program by distinct loads (56 to 152 per hash; 321 to 118 Mhash/s projected on the RTX 5090 at 18.0 G distinct loads/s), 8.3x min to max; under G1 + G2 every program does 128 distinct loads, projected 141 Mhash/s on the 5090 and 28 on the M5 Max, with the 1.10x program-shape residual the only spread left, approximate. One 5090 run of igneum-second-seed (predicted 173 Mhash/s if distinct-bound, 228 if static-bound) settles the reading on NVIDIA. Not done: no GPU run of the new generator; the spec text is proposed in the analysis doc, section 9, not written into docs/spec/01-lottery-hash.md; test vectors are re-cut when the generator rule is adopted.

+

Machine: Apple M5 Max, 8 threads at nice -n 15 while a devnet build and its simulations shared the box (load average 25 to 107), rustc 1.99.0, release build with LTO. New crate igneum-census/ (path dependency on igneum-pow, nothing in igneum-pow changed); the instrumented interpreter is checked against igneum_pow::hash_warp on the first warp of every program and against the igneum-genesis spec vectors at start. Commands: igneum-census run --root igneum-census-2026-10-03 --count 100000 --warps 128 --threads 8 --gen default (memory-hard, 1 GiB, day 2026-10-03; 2,797 s), the same with --gen fixed16-fresh --warps 64 (1,049 s), --gen fixed16-fresh2 --warps 64 (6,650 s, starved to under a core for most of it), and --gen default --closed-form --warps 128 (125.6 s once the machine was quiet); summarise, probe, show. Full tables and the rules in docs/analysis/weak-program-census-2026-10-03.md. Current generator, 100,000 programs x 4,096 nonces: loads per hash 24 to 256 (mean 127.9); distinct addresses per hash 24 to 200 (mean 102.4): 19.9 percent of all loads re-read an address the same hash already read, 94.8 percent of programs have at least one such load, 22.5 percent have a pair that cancels to the identity. The Mac rates of the eight bench seeds vary 1.38x by static loads/s and 1.10x by distinct loads/s (3.51 to 3.85 G/s), so the GPU is bound by the distinct count; igneum-second-seed/epoch1 (144 static, 104 distinct) hashes at the rate of igneum-second-seed (104 and 104). Weak programs, current generator: 2.43 percent have a register with no injecting write (saturates to all ones); 0.73 percent have a register with a nonce-independent bit, 0.03 percent a whole nonce-independent register, 0.03 percent a load site read at one address by all 32 lanes, 0.68 percent more than 1 percent of final registers at 0 or all ones, 6 programs an output bit past 6 sigma (0.03 expected by chance). Avalanche clean on every program (mean 31.85 to 32.15, every output bit flips 0.473 to 0.523). Mechanisms: no injecting write; zero-absorbing register sets closed under mulhi/mul; or or mul as the last write. Rules: G1 exactly 16 load slots drawn first from slots 1..63; G2 a load reads only a register written earlier in the program and not read by a load since; R-a no cyclically redundant load; R-b every register has an injecting write; R-c 64 fixed warps on the seed-keyed closed-form dataset with no constant register bit, no lane-constant site, saturation under 1 percent, no output bit past 6 sigma, more than 120 distinct addresses per hash on average. Rejection: 95.0 percent under the current generator (the redundancy alone), 38.3 percent under the first form of G2 (the iteration wrap), 5.14 percent under the proposed form (R-a or R-b 3.93 percent, R-c 2.05 percent), so 1.054 candidates per epoch on average; the accepted population does 120.05 to 128 distinct loads per hash, median 128.00. Closed-form check: with the same seeds and nonces on the closed-form dataset instead of the memory-hard one, R-c agrees on 99,961 of the 100,000 proposed-generator programs (2,054 rejected memory-hard, 2,055 closed-form; the 39 that differ sit at a threshold edge, one nearly constant bit or a bias near 6 sigma) and the per-program metrics agree to three decimals, so the acceptance test can be a pure function of the program. Hash-rate spread: today 2.7x between the 1st and 99th percentile program by distinct loads (56 to 152 per hash; 321 to 118 Mhash/s projected on the RTX 5090 at 18.0 G distinct loads/s), 8.3x min to max; under G1 + G2 every program does 128 distinct loads, projected 141 Mhash/s on the 5090 and 28 on the M5 Max, with the 1.10x program-shape residual the only spread left, approximate. One 5090 run of igneum-second-seed (predicted 173 Mhash/s if distinct-bound, 228 if static-bound) settles the reading on NVIDIA. Not done: no GPU run of the new generator; the spec text is proposed in the analysis doc, section 9, not written into docs/spec/01-lottery-hash.md; test vectors are re-cut when the generator rule is adopted.

+

3 October 2026, proving v0: first SP1 proof of an Igneum block, Apple M5 Max CPU, loaded machine (execution-engineer, proving)

+

Machine: Apple M5 Max (18 cores, 64 GB), macOS Darwin 25.6.0, load average 14 to 45 during the runs (the live devnet, the observer and other agents' builds were running), everything under nice -n 19. Toolchain: SP1 v6.8.1 (sp1up, cargo-prove c84ada1 of 24 Sep 2026, succinct rustc 1.96.0-dev, circuit version v6.1.0), sp1-sdk 6.8.1 CPU prover, revm 43.0.3, alloy-primitives 1.7.3 with SP1's sha3 patch and k256 patch. Code: proving/igneum-prove (guest ELF 2.69 MB, host 54 MB), fixtures cut from tools/evm-smoke/seq.json (the 3-node simnet export, execution-layer commit fb33069) by igneum-prove-export, which replayed all 79 segments from genesis through the ported executor and matched every one of the node's state roots (final root 0x5b18b3a5...). Statement: re-execute one chain block (rewards by rule, nonce-rule skip, two-dimensional gas with the prototype pgas table, fee flows with the developer split, state root over the whole in-memory state) and commit the pre-root, post-root, receipts root, gas, pgas and the executed and skipped counts; the host checks the guest's public values against its own native run before and after each proof.

+
FixtureTxs (executed / skipped)EVM gaspgasPre-state accountsSP1 cyclesProver gasCycles per EVM gasExecute s
block-78-increment (Counter increment(5) plus a duplicate copy skipped by the nonce rule)2 (1 / 1)45,3541,48810626,246843,343140.03
block-56-transfers (three funding transfers)3 (3 / 0)63,0006005549,469733,29790.03
+
block-78-increment, CPU proverProve sProof bytesVerify sVerified
Setup (pk, vk; vk hash 0x00c3a917...)6.9
Core (STARK shards)22.07,317,2170.164yes
Compressed (recursion, one shard)55.71,272,7690.033yes
+

Reading: at 626 k cycles the block is far below one SP1 shard, so these times are fixed overhead (proof system setup and the recursion stack), not throughput; cycles per EVM gas (9 to 14) is the first data point for the pgas table calibration (R1) and is dominated by the state-root computation over every account plus one secp256k1 recovery per transaction (through the patched k256). Nothing here is a 12 GB-card shard time (ledger P1); that is the RTX 5090 run of proving/windows-wsl2/ and then the 3060-class gate. The devnet was not touched. Not done: the Groth16 or Plonk wrapper (ledger P3), MPT witnesses, more than one shard per block, chain recursion, the prover key in the statement (P12).

+

2026-10-03 execution layer attack suite: malformed txs, nonce games, RPC fuzz, pgas exhaustion, reorgs, registry abuse (execution test engineer)

+

Machine: Apple M5 Max (18 cores), shared with other agents' builds (load 9 to 15). Worktree vendor/igneum-node-exec-attacks on branch exec-attacks (from execution-layer fb330692); igneumd, igneum-miner and a new hostile-miner bin igneum-inject built release with CARGO_TARGET_DIR=target nice -n 19 cargo build -j 4 -p kaspad -p igneum-miner --features igneum-pow (stable-aarch64 toolchain; the default cargo on PATH is too old for edition 2024). Tools and the per-scenario commands: tools/exec-attacks/ (README, net.sh, scenario{1,2,3,4,5,6}*.mjs, igneum-inject); raw results under tools/exec-attacks/results/*.json. Network: 3 igneumd --simnet --enable-unsynced-mining --unsaferpc --disable-upnp nodes, PoW skipped, chain id 4463, eth RPC 27690/27691/27692, gRPC 27610/27620/27630, p2p 27611/27621/27631, appdir /tmp/igneum-exec-attacks; one honest stub miner for scenarios 1 to 5 and 4, three miners split into partitions for scenario 6. igneum-inject fetches a block template, replaces the EVM body with arbitrary raw EIP-2718 bytes, recomputes hash_merkle_root and resubmits, so the hostile-miner path reaches body validation and the executor directly. The live devnet (26610, 26611, 26640, 26641, 28640) and other agents' ports (up to 27599) were not touched; every process was stopped at the end.

+

Run in priority order 1, 2, 5, 3, 6, 4. One row per scenario: criterion (from the design), measured result, verdict.

+
#ScenarioCriterionResultVerdict
1Malformed and boundary txs (mempool and hostile block)State-free faults invalidate the block; state-dependent faults skip the tx with no receipt; no panic; memory bounded7 state-free faults (bad RLP, type-3 blob, wrong chain id, intrinsic gas above limit, initcode above 49,152, duplicate hash in block, non-contiguous nonces, invalid signature s=0) each made the hostile block invalid and were rejected by the mempool where decodable; 5 state-dependent faults (nonce far ahead, nonce reuse, zero fee below base, insufficient funds, max fee at 2^120) each landed in an accepted block and were skipped with no receipt; gas limit exactly at B_e executed; node kept producing blocks; node RSS 345 MiB to 348 MiB (x1.01); 0 node panics in any logPASS (30/30 checks)
2Nonce games across parallel blocksExactly one execution per nonce; deterministic; state roots identical on all nodesnonces n..n+3 spread across 3 parallel blocks with heavy duplication executed once each, account nonce advanced to n+4; a conflicting same-nonce pair in two parallel blocks executed exactly once; state roots identical on all 3 nodes at the tip in both roundsPASS (9/9)
5RPC fuzzErrors not crashes; honest latency under 200 ms31 eth_*/igneum_* methods x 9 junk param shapes plus deep nesting (5,000 levels) and broken bodies all returned a JSON-RPC envelope or a handled HTTP error, none dropped the connection or crashed; under a one-client eth_call flood of 4,184 req/s (about 200x honest) honest p95 latency 29.1 ms, max 33.5 ms, 0 flood errors; node kept advancingPASS (5/5)
3Proving-gas (pgas) exhaustionThe per-block pgas budget B_p caps inclusion and the template respects it; measure execution time per blockB_p = 30,000,000. modexp loops: 1,000 iters executed 3.45 M pgas in 1.34 ms; 3,000 -> 10.33 M pgas, 4.56 ms; 6,000 -> 20.64 M pgas, 7.54 ms; 9,000 -> would-be 30.96 M pgas, skipped with BlockProvingBudget after 10.85 ms of native execution; no executed block carried more than B_p (max 20.64 M)PASS (4/4)
6Reorgs under executionState root recomputed deterministically; displaced-tx receipts handled per design; no stuck mempoolPartition P1={node1}/P2={node2,node3} healed via igneum-inject addpeer after 1/3/5/8 s forced selected-chain reorgs of depth 3, 6, 13, 11 on the losing node; all 3 nodes converged to one sink and agreed on the state root at the common height each time; the tx executed on the pre-heal chain re-resolved to one canonical, cross-node-consistent outcome (DAG merges the losing blocks, design 1.2/1.3; it does not orphan them); a fresh tx was mined after every reorg (mempool not stuck)PASS (31/31 checks over 4 cycles)
4Developer registry abuseDesign 4.5: base fees burned, no positive-expectation loop; record the max share a self-dealer recoversregister(someone-else's-contract) and register(unrelated EOA) both revert; a factory's CREATE and CREATE2 children inherit the factory payee; a same-tx creator override sets a different payee; an EOA cannot override a factory child; an unregistered factory's child has no payee (share burns); self-dealer (sender = payee = block miner) recovered 100.0% of the tip but only 56.45% of total fees paid, because both base fees are burned; recovered < paid alwaysPASS (19/19). Max share a self-dealer recovers: 56.45% of fees paid (tip only; base fees always lost)
+

Totals: 98 checks, 0 failures, 0 node panics, memory bounded. Execution time per block under the pgas attack stayed single-digit to low-tens of milliseconds (1.3 to 10.9 ms) at these loop sizes; the whole-account state-root recompute (design 10.3 item 1) dominates and will fall once the incremental trie lands.

+

Findings (not consensus failures; filed for the ledger):

+
  • F-exec-A (low): the EVM mempool admits a transaction whose gas_limit exceeds the block execution limit B_e. igneum/exec/src/pool.rs EvmPool::add checks funds, nonce and fee cap but never bounds gas_limit by BLOCK_EXECUTION_GAS_LIMIT. Reproduction: fund an account, send a type-2 tx with gas=31_000_000 (B_e is 30,000,000) to any node's eth RPC; eth_sendRawTransaction returns a hash (admitted). The transaction can never be selected (EvmPool::select breaks when gas + gas_limit > B_e) nor form a valid block (check_evm_body -> SumGasLimitAboveBlockLimit), so it occupies a queue slot until evicted. Self-limited because admission still reserves gas_limit x max_fee_per_gas in the funds check. Fix: reject gas_limit > B_e in EvmPool::add, as geth rejects gas > block gas limit.
  • F-exec-B (medium, griefing): an over-pgas-budget transaction is executed natively in full before it is skipped, and because it is skipped it pays no fee. A transaction whose own pgas exceeds B_p (for example one large modexp, or the 9,000-iter loop above at 30.96 M pgas) is included, executed (10.85 ms of real work here, more for a bigger input), then dropped with BlockProvingBudget and charged nothing (igneum/exec/src/executor.rs: the skip happens after inspect_one_tx runs and before any fee is taken). Every node re-executes it on every inclusion for free, and because the nonce never advances it also head-of-line-blocks that sender's higher nonces (seen here: the 14,000 and 20,000 loops were never includable behind the stuck 9,000). The funds check at admission does not bound pgas (pgas is not known without execution), so a modestly funded account can force repeated free computation network-wide. Fix options: charge the intrinsic plus consumed pgas on a budget skip, cap single-transaction pgas at admission via eth_estimateGas-style simulation, or drop a sender's queue on a BlockProvingBudget skip rather than retrying.
+

Not covered here (out of scope for this pass, and because the proving layer is not implemented on this branch): proof records, the native-execution veto, sortition, and the finality lock (proven/locked are always false on devnet v3, so only executed was exercised). These need the proving layer and the finality merge (design 10.4) before they can be attacked.

+

4 October 2026, sim/economy: mining versus proving under stress, agent-based (economist; model, not hardware)

+

Machine: Apple M5 Max, shared (load 9 to 25), single process at nice 19, about 28 minutes of compute in total. sim/economy/sim.py, Python 3.10.10, numpy 2.2.6; 1,000 operators, 30 days, 180-s ticks, 13 to 25 s per run. Inputs: RTX 5090 229 MH/s (measured, this log); every other number approximate (docs/analysis/economy-2026-10-04.md, assumptions table). Six scenarios x 5 seeds (sim/economy/results.md): no backlog, no window miss, no hash under 50% of pre-event in any run. Hash troughs: a 0.95, b (price down 70%, external x10) 0.82, c 0.97, d (20% operator leaves) 0.75, e (30% withholder) 0.98, f (2x pool arrives) 0.95 of pre-event; day 30: 1.00 / 0.87 / 1.00 / 0.80 / 1.00 / 1.92. Blocks proven within 60 s: 1.00 in every hour; within 20 s: 0.14 to 0.41. Cards in hybrid mode (mine, answer own assignments) at day 30: 42 to 54%; cards off: 1% (a, c, e) to 10% (b). Profit $ per card-day, baseline: 5090 6.37, 3090 1.66, 3060 0.78, small 0.39; shard share 5090 0.59, 3090 0.26, 3060 0.16. Proving-share 10-90 range over the last 10 days 3 to 9 points (one seed of f at 10.1). Sensitivities on b (2 seeds, sim/economy/levers.md): traffic 3 / 30 / 100 / 300 shards per block gives hash trough 0.81 / 0.82 / 0.63 / 0.06, oldest unproven age 0 / 0 / 85 / permanent, worst day within 60 s 1.000 / 1.000 / 0.994 / 0.825, hours under 50% hash 0 / 0 / 0 / 22. Observation window 20 min to 24 h: score 0.939 to 0.952, churn only. Lever study on b at 100 shards per block (2 seeds): window 5 / 10 / 20 / 30 s gives age max 565 / 325 / 0 / 0 s, hash trough 0.53 / 0.62 / 0.77 / 0.77, score 0.592 / 0.765 / 0.948 / 0.948; pool 0.1 / 0.2 / 0.3 / 0.4 gives age 168 / 325 / 16 / 0 and cards off 0.05 / 0.07 / 0.07 / 0.10; burn 0 to 0.5 and claim timeout 60 to 600 s leave the age at 325 s in every row. Proposal (not applied): window = p90 shard time plus one swap, 25 s at today's targets (O-5.1); B_p tied to the live proving fleet rather than a launch calibration. Not done: DAG and network latency, pool protocol, bonds on jobs beyond a class filter, price feedback from burns, the launch ramp; the age column of the 300-shard sensitivity row predates the age-formula fix.

+

2026-10-04 execution layer attack fixes: F-exec-A (mempool gas-limit bound) and F-exec-B (pgas abort rule, spec 7.5) (execution-engineer)

+

Machine: Apple M5 Max (18 cores), shared with other agents' builds (load 13 to 18). Worktree vendor/igneum-node-exec, branch execution-layer (fix commit on top of fb330692); built release with CARGO_TARGET_DIR=target nice -n 19 cargo build --release -j 4 -p kaspad -p igneum-miner --features igneum-pow (stable-aarch64 toolchain), unit tests with cargo test --release -j 4 -p igneum-exec -p igneum-evm-types. Network: 3 igneumd --simnet --enable-unsynced-mining --unsaferpc --disable-upnp nodes from this worktree, one honest stub miner, eth RPC 27990/27991/27992, gRPC 27910/27920/27930, p2p 27911/27921/27931, appdir /tmp/igneum-exec-fix; hostile blocks through the attack suite's igneum-inject (vendor/igneum-node-exec-attacks/target/release, same wire protocol). The attack scripts of tools/exec-attacks ran unchanged against this network with IGNEUM_RPCS and IGNEUM_GRPC1 pointed at it, from copies outside the repository so results/*.json of the 3 October run stay as recorded; the after-fix reproduction is a separate script (session scratchpad, scenario3_after.mjs, 25 checks) written against the new rule. The live devnet and the attack suite's 276xx ports were not touched; every process was stopped at the end; 0 panics in the three node logs.

+

What changed (spec 7.5, design 10 note): the inspector meters pgas against the including block's remaining B_p and halts the transaction before the opcode or precompile that would cross it (a precompile over the cap is answered with a revert that spends none of the forwarded gas, and the parent halts at its next instruction); the executor charges an aborted transaction as out of gas for the gas and pgas consumed to the abort, status 0, nonce advanced, receipt pgasAborted; the proving charge never takes a sender past the signed budget. The mempool refuses gas_limit > B_e (F-exec-A) and an estimated pgas above B_p (estimate = simulation at the tip under the cap), the template packs by the estimate, eth_estimateGas and eth_call fail naming the pgas when the cap is hit, igneum_estimateGas returns both dimensions. Simulations now read the state through DatabaseRef instead of cloning it per call. igneum-exec-diff treats a pgasAborted transaction as an Igneum-only flow (plain revm would run it to its own end).

+

Unit tests (new, all pass): pool::gas_limit_is_bounded_by_the_block_execution_limit, pool::estimated_proving_gas_is_bounded_by_the_block_proving_limit, pool::template_never_exceeds_the_remaining_proving_budget, executor::over_budget_pgas_is_aborted_charged_and_the_nonce_advances (an SLOAD-loop bomb with a 30 M gas limit, about 54 M pgas if run out, is cut under B_p, charged exactly gas_used x price + pgas_used x f_p, nonce advanced; a second inclusion skips with NonceTooLow in under 100 ms; the next nonce executes), executor::the_cap_is_the_remaining_block_budget (two bombs in one block fill it to within 1,000 pgas of B_p; a third copy skips at its intrinsic pgas), executor::estimate_reports_the_cap. 6 of 6 in igneum-exec, 3 of 3 in igneum-evm-types.

+

Scenario 3 (pgas exhaustion), before (3 October run, tools/exec-attacks/results/scenario3.json) and after, B_p = 30,000,000, modexp loops from one sender:

+
LoopBefore: outcomeBefore: pgas, timeAfter: mempoolAfter: hostile inclusion (igneum-inject)
1,000executed3,449,475 pgas, 1.34 msexecuted, 3,449,475 pgas, 1.46 msnot needed
3,000executed10,327,475 pgas, 4.56 msexecuted, 10,327,475 pgas, 3.94 msnot needed
6,000executed20,644,475 pgas, 7.54 msexecuted, 20,644,475 pgas, 8.57 ms; two of them in one go land in separate chain blocks (70, 72), neither abortednot needed
9,000skipped BlockProvingBudget { would_be: 30,961,475 } after 10.85 ms of execution, nothing charged, nonce stuck200 pgas charged to the blockrefused: "proving gas above the block proving limit: at least 29,998,593 pgas metered before the abort, limit 30,000,000"executed with status 0, pgasAborted, 29,998,593 pgas, 4,680,437 gas (limit 29,000,000), 11.37 ms, block pgas 29,998,593; sender charged 39,359,467,000,000,000 wei = 0.0394 IGN (gas x 2 gwei + pgas x 1 gwei), nonce 1 to 2; a second hostile inclusion skipped NonceTooLow { expected: 2, got: 1 } in 35 us, no second charge
14,000never included (behind the stuck 9,000)nonerefused, same messagenot run
20,000never includednonerefused, same messagenot run
+

After-fix checks: F-exec-A (gas_limit 30,000,001 refused: "gas limit 30000001 above the block execution gas limit 30000000"); eth_estimateGas for the 9,000 loop fails naming 29,998,593 pgas, igneum_estimateGas returns exceedsProvingLimit: true, and for the 6,000 loop pgas 20,644,475 with folded gas 27,465,126; the sender's next nonce (a transfer) executed four blocks after the abort; node RSS 322 MiB to 328 MiB (x1.02); blocks kept coming. 25 of 25. The unchanged scenario3_pgas.mjs now reports 3 of 4: its check "a heavy transaction is skipped with BlockProvingBudget" asserts the old rule and fails by design (the 9,000, 14,000 and 20,000 loops are refused at the mempool), the other three pass (max executed block pgas 20,644,475).

+

Scenario 1 (malformed and boundary, unchanged script): 30 of 30; the single-gas-limit-over-block case now records mempoolAdmitted: false (the observation that filed F-exec-A is gone); the script's RSS probe looks for the attack worktree's binary path and found no process here, so that check was trivial in this run (the after-fix script measured RSS itself, above).

+

Differential: igneum-exec-diff over the test network's export, segments 0 to 176, 17 executed transactions compared (one pgasAborted), 6 skipped copies confirmed, 11 accounts compared, 0 mismatches.

+

Not changed: the pgas table magnitudes (prototype), B_p = 30 M (prototype). Open: the admission estimate runs under the RPC's state read lock, so a flood of heavy eth_sendRawTransaction calls delays the follower by up to B_p of simulation each (same shape as the eth_call flood of scenario 5, which stayed under 34 ms p95); a per-sender or per-second cap on estimates is the next step if the devnet shows it.

+

3 October 2026, per-identity hash rate "decay" on the RTX 5090: diagnosis and Metal reproduction (miner-community-lead)

+

Machine for the reproduction: Apple M5 Max, 64 GiB, Darwin 25.6.0, load average 2 to 147 (other agents' builds and, during R1, another agent's Metal worker on the same GPU); everything at nice -n 19. Binaries: HEAD proto-metal/main.swift built with swiftc -O into the scratchpad (465,529 bytes, the same size as proto-metal/igneum-bench), vendor/igneum-node-diff/target/release/igneumd and igneum-miner (22:38 and 22:17 BST, the difficulty worktree pair; the miner's Seeder and worker protocol are the same code as HEAD and as the Windows build 745d41ef). Private networks on 127.0.0.1 ports 27500 to 27562, appdirs under /tmp/igneum-decay-test, all stopped afterwards. Full write-up: docs/analysis/hashrate-decay-2026-10-03.md; proposed fix: docs/analysis/hashrate-decay-2026-10-03.patch (not applied; git apply --check passes against vendor/igneum-node). PC data (node tools/logs.mjs <run_id> --all, STATUS lines deduplicated by timestamp, per-interval rates from consecutive cumulative figures): segment 22:57 to 23:04 UTC, nvidia-1: 40 jobs in the first 30 s then exactly 32 per 30 s for 12 intervals at 17.5 to 18.4 MH/s wall while the printed cumulative figure fell 22.18 to 18.13; nvidia-8 (started 4.7 s later) printed a rising 16.80 to 17.71. Segment 22:23 to 22:57 UTC (epoch 2, DAA 8,474 to 10,513): per-identity gap between jobs 0.098 s to 0.330 s per 0.68 to 0.81 s job, inside-jobs rate rising 28.7 to 34.7 MH/s, wall falling 24.6 to 20.6 MH/s, card total 197 to about 165 MH/s; at the 22:57 epoch boundary the gap returned to 2% and the difficulty held (84.5M to 83.0M). Code audit: nothing allocated per job survives the job in proto-cuda/host.cu, proto-opencl/host.c or proto-metal/main.swift serve loops (tables in the analysis); the miner's only per-job growth is time in Seeder::seeds_for (memo keyed by (epoch, sink), one getBlock RPC per block from the sink to the epoch start on every miss, 1,274 to 3,313 calls on the PC). cudaDeviceSynchronize at the default schedule spins one thread per worker (the project lead's 6.2% per process); the hot-swap working tree sets cudaDeviceScheduleBlockingSync and swaps clFinish for clWaitForEvents. Metal runs (STATUS every 30 s; "gap" = 1 minus wall over inside, per interval): R1 control, epoch 0, genesis bits 0x1d100000, 308 s (cut by the 22:21:37 UTC SIGTERM of every process of this session): first interval 25.18 MH/s alone on the GPU, then 14.0 to 14.4 MH/s in every interval after another agent's worker joined at 25 s, gap 0 to 2%, worker RSS 56.8 MiB flat. R2 walk reproduction, 900 s: skip_proof_of_work node pumped to DAA 4,000 (one-second timestamps, difficulty held at 76.8M), one identity, pumped blocks at 1/s for 300 s, none for 300 s, 1/s for 300 s: inside 27.0 to 27.7 MH/s in all 29 intervals; wall 22.3 to 24.3 (gap 12 to 18%, walk 400 to 700), 25.9 to 27.3 (gap 0 to 4%), 18.0 to 21.0 (gap 25 to 35%, walk 700 to 1,000); miner CPU 0 to 1% in the quiet phase, 11 to 21% in the last. R3 one worker at difficulty 2^25 (Kaspa sampled rule, genesis bits held), 600 s: 30.51 wall / 30.72 inside, 1,091 jobs, 224 blocks, 53 to 56 jobs per 30 s throughout. R4 eight workers at 2^25: 29.38 / 29.45 summed (3.32 to 4.38 each), 1,053 jobs, 242 blocks, 7 jobs per 100 s per identity in every interval, worker CPU 0.0 to 0.6%, RSS 46 to 57 MiB. R5 one worker at 2^31: 37.01 / 37.75, 1,324 jobs, 4 blocks, flat. R6 eight workers at 2^31: 36.67 / 36.75 summed (4.29 to 5.55 each), 1,314 jobs, 5 blocks, flat. (R5 and R6 ran a different epoch-0 program from R3 and R4, 112 loads per hash, hence 37 against 30.5 MH/s.) Side findings: the difficulty worktree's node panics at consensus/src/processes/difficulty.rs:431 ("Work should not exceed 2**192") when fed 85 blocks/s with wall-clock timestamps under the Igneum dual rule (a pump artefact, logged for the consensus-engineer); skip_proof_of_work nodes still log "PoW rejected ... by igneum-lottery-v1-bound" for every block they accept. Not done: the fix applied and measured on the PC (the acceptance figure is a flat gap at DAA 10,800 with eight identities); the OpenCL event wait checked on the AMD driver; a unit test of seeds_for (the client is concrete).

+

2026-10-04 finality v2 attack harness: seven hostile scenarios on a six-voter private test network (consensus test engineer, cryptographer)

+

Machine: Apple M5 Max, rustc stable, macOS Darwin 25.6.0. Fork: worktree vendor/igneum-node-fin-attacks, branch fin-attacks on master c6d47547 to 2a00ff55 (BLS votes, certificates in coinbase extra data, p2p message 70, the finality RPCs). Build: CARGO_TARGET_DIR=target nice -n 19 cargo build --release -j 4 -p kaspad -p igneum-miner --features kaspad/igneum-pow. Tool: tools/finality-attacks (run.mjs, lib/, README with the proposed fixes). Network: igneum-devnet-800, ports 27800 and up, data /tmp/igneum-fin-attacks, skip_proof_of_work (the hostile miners never hash; each gets its block share from a Poisson clock; every other consensus rule unchanged). Devnet finality parameters: interval 30, depth 20, weight window 7,200 DAA, dust 5, presence 20 indices, 8 aggregators, ban 7,200 DAA, quorum 2/3 of active and 17/30 of total. Durations at SCALE 0.6. The live devnet (26610, 26611, 26640, 26641, 28640) was never touched. Run time 32 min of process time (the machine slept twice during the run, which pauses the monotonic clocks the harness and the miners use, so wall-clock timestamps in the log jump; no result depends on wall time).

+

Hostile pieces are test-only flags of igneum-miner, never honest node or consensus code: vmine (Poisson submit at a chosen hash share, decoupled 0.5-s voter), --equivocate, --sybil b:bb:a:ab (one miner mints many vote keys), --drop-votes (strips the node's finality section from its coinbase so its blocks carry no votes or certificates while it still votes over RPC), --pulse burst:on:period, and fin-rpc-attack (malformed, mis-signed, replayed, oversized and non-hex votes over submitFinalityVote). Six voters throughout; three nodes for the cross-node scenarios, two nodes over a TCP proxy for the partitions.

+
#Scenario (priority order)Criterion (spec 03)MeasuredVerdict
3Dishonest aggregators (6 voters, 3 nodes, every node aggregates)other aggregators' certificates still lock; a sub-quorum certificate cannot lock (Q3); block-carried votes give participation (F3); lock latency under 2 s median35 / 35 / 35 locked per node, identical lock hashes on all three, 0 conflicting certificates, median lock latency 1,018 ms (bounded by the miners' 1-s poll). Sub-quorum rejection is by code review (lock_test needs both integer tests; a certificate below either is Certified, never Locked); injecting one on the wire needs a finality-aware p2p probe (not built)PASS (wire injection not run)
2Sybil dust (one miner mints 200 keys at 4 blocks and 200 at 6, dust 5; 3 honest voters)dust keys zero weight and no voters; above-dust weight = blocks; total weight = voters' blue blocks; sortition by weight not key count (F17)200 dust keys seen, all voter false; 203 voters above dust; total weight 1,240 = sum of voter blocks 1,240; aggregator sortition is PER KEY (is_aggregator(output, voters, 8) counts keys), with 203 voters a real signer's chance to be an aggregator fell to about 8/203 and the last checkpoint named 0 aggregators (zero-aggregator certificates, "anyone MAY aggregate")weights PASS; sortition FAIL (F17)
1Equivocation at scale (2 of 6 keys sign two checkpoints at every index, 3 nodes)both keys stripped within one checkpoint on every node; no conflicting certificate; honest locks continuestripped keys 2 / 2 / 2 on the three nodes, 78 / 8 / 8 detections (node-local on the equivocators' node, block-carried evidence on the others), 0 conflicting certificates, 35 / 35 / 35 locks by the 4 honest keysPASS
6APartition 3/3 for 90 s after a 252-s shared warmup (window 1,439 DAA at the cut), then healzero locks on either side during the split; locks resume after the heal; no conflicting certificatesside 0: no new lock in 90 s; side 1: first new lock at 84 s, 8 locks before the heal; 0 conflicting certificates (side 0 never locked those indices); locks resumed on both sides after the heal. Side 1 crossed the floor because its own fresh blocks raised its share of its window: at the cut each side held 50% of 1,439 DAA of weight; the 3-miner side added about 2.6 blocks/s and by 84 s held (720 + 220) / (1,439 + 220) = 56.7%. The model is share(T) = (F/2 + R T) / (F + R T) with F the window weight at the cut and R the side's block rate, so the floor holds for T* = 2F / (13R): 74 s predicted at F = 1,439 and R = 3, 84 s measured (sibling losses lower R). The simulation used fixed weights and could not see this (spec 3.7 item 8)FAIL (floor is time-bounded)
6BPartition 4/2 for 90 s after a 252-s warmup, then healthe 4 side (66.7% of total) keeps locking; the 2 side (33%) does not; no conflicting certificates4 side locked 47 to 59 (first new lock 15 s after the cut, the active test passes at exactly 2/3); 2 side stayed at 47; 0 conflicting certificates; both resumed after the healPASS
4Vote-dropping block producer (40% of blocks carry no finality section, 2 nodes)participation and locks unaffected because other blocks carry the votes; delay measuredthe node that saw the dropper's blocks only through gossip and the other producers' blocks locked 35 checkpoints; median lock latency 1,019 ms with the dropper vs 1,019 ms control, 0 ms addedPASS
8Malformed votes over the RPC (9 cases, fresh key per case)rejected without a crash; node stays upcontrol vote accepted; replay answered "already known" (deduplicated, not double-counted); bad signature and wrong chain id rejected "invalid vote signature"; a vote for a hash the node does not hold at a known index is recorded and flagged, not certified; 8-byte, 2 MB and non-hex payloads rejected "vote must be 280 bytes" / "vote is not hex" before any processing; node answered getInfo after all 9. The message-70 half (sub-quorum and replayed certificates, oversized bitmaps) needs the p2p probe; by code review Certificate::read bounds the bitmap at 1 MB, the relay bounds a message at 1 MB and a malformed one is a ProtocolError that disconnects the peerPASS (RPC half)
5Pulsed miner (base share 1/6, 10x for 20 s of every 120 s, 5 steady voters, 216 s, Kaspa's DAA rule as master runs it)weight proportional to block share over the window (no retarget amplification, W2 and F14); cannot lock aloneweight share 35.3% vs block share 35.3%, ratio 0.999: W2 counts blocks and the retarget lag bought nothing extra. But checkpoints 1 to 10 were locked by the burster ALONE: its first 20-s burst gave one key 66.7% to 71.7% of a window that held under 300 blocks (cp 5: 98 of 147 signed by 1 of 6 voters; cp 10: 201 of 297), above both Q3 tests. From cp 11 every lock needed 3 or 4 signers as its share decayed to 35%. This is ledger F1 measured live: with no first-month gate (min_daa 0 on devnet, 3,600 DAA on mainnet, spec 3.8 not implemented) a short burst owns a young windowamplification PASS; lock-alone FAIL (F1)
7Eclipse of one voter with an adversarial side chainnot run: needs the finality-aware p2p probe to feed a private forknot measurednot run
+

Failures and the proposed fixes (diffs in tools/finality-attacks/README.md, for gate 3 to ratify; no rule was changed here):

+
  1. F17 (S2): is_aggregator draws the 8 aggregators per key. Liveness only, because any node may aggregate and a certificate must still meet Q3 by weight, which the Sybil split does not change. Proposed: draw by weight, output x total < 8 x weight x 2^64, mirroring the spec 7.2 step-2 fix.
  2. Floor time bound (S6A): the 56.7% floor protects a partition for about T* = 2F / (13R) of DAA time, with F the window weight at the cut and R the majority side's block rate. Approximate, formula only, window slide ignored: on mainnet with a full 30-day window and a 50/50 split at 1 block/s, 9.2 days; a 55/45 split, 2.1 days; 60/40 locks at once (3.3.1 already says so). Minimum fix: state the bound in spec 3.3.1 and 3.9. Rule option for gate 3: evaluate the floor against the weight table of the last locked checkpoint while no newer lock exists, so a stalled side cannot lift its own share by mining; cost: after a permanent loss of weight the floor needs a manual override instead of the 4.1 days of 3.3.1 D.
  3. F1 (S5): no certificate should form before the window holds a full window of history (spec 3.8, O-3.1). Proposed: min_daa = weight_window (2,592,000 on mainnet, 7,200 on devnet) in FinalityParams, one line each.
+

Not demonstrated: certificate injection on the wire (S3, S8 half), the eclipse (S7), the 2-hour presence window at mainnet length, the heal rule of 3.5 (both partitions healed without a conflicting certificate, so it was not exercised).

+

4 October 2026, difficulty rule under attack: pool hopping, pulsed rental, timestamp stretching, short-lane oscillation, epoch games, polluted window, block flood (consensus test engineer)

+

Machine: the same Apple M5 Max, shared with other agents' builds and test networks (load 15 to 30). Simulator sim/difficulty/attacks/attacks.py over sim/difficulty/sim.py (controllers unchanged): several miners with on/off strategies, block attribution by hash share at the solve, hashes per miner, timestamp forging inside the fork's rules (132 s ahead, above the 27-sample past median). Node runs on branch diff-attacks of vendor/igneum-node (worktree vendor/igneum-node-diff-attacks, from difficulty at 3ea7a3e3; adds only the attack variable IGNEUM_ATTACK_TS_OFFSET_MS in the template builder and a reproduction test), ports 27700 to 27721, appdir /tmp/igneum-diff-attacks, genesis bits 0x1f010000. Everything in sim/difficulty/attacks/README.md, raw tables in results.md, headers of the three test-network runs in testnet/. Simulator, seeds 7 to 9, Igneum / Kaspa's rule, criterion, verdict: (1) pool hopper 10 to 100% of the base, on while D is below its 6-hour mean, 24 h: hopper's blocks per hash +1.5% at most / +0.8% at most, under 5% both, Igneum 0.7 points above Kaspa's in every greedy cell (unchanged with the ease clamp at 6% or 3%: the price of a controller that moves inside the hour; a 60 s dwell turns the 50% and 100% hoppers into losers, -1.7% and -4.0%): PASS under 5%, FAIL on "no larger than Kaspa's" by the letter, no change proposed. (2) 50x burst for 10 min every hour: pulser's weight per hash 0.26 / 0.98 of the base's, blocks per hash 3.7% / 85% of the base's; once: 0.13 / 0.87: PASS (no weight amplifier under either rule; Kaspa's makes the burst cheap, Igneum's makes it 27x dearer; the hour after costs the base 37% and a 153 s worst gap under Igneum). (3) forger at 30 or 50% stamping at the latest allowed, the earliest allowed, or alternating: Igneum block rate 0.66 / 0.42 / 0.51 at 30% and 0.56 / 0.12 / 0.23 at 50%, difficulty 1.5x to 9.9x on an unchanged hash rate, worst gap 234 s; Kaspa's rule +5% to +11% easing: FAIL both, Igneum far worse. Cause: the symmetric per-step clamp turns every forged block and the honest block after it into zero measured time (spec 2.3's "the next honest block cancels it" is the bug, not the defence), so the lanes measure 1 - 2a(1 - a) of real time at share a; past-stamping also drags the past median down without bound. (4) 25% miner on and off every 120 blocks: std of the rate ratio 0.160 / 0.147 against 0.045 / 0.010 steady and a 0.143 floor from the attacker's own square wave: FAIL by the letter for both, neither oscillates (12% and 3% above the floor), no change proposed. (5) hold dodger 30%, hold flooders 10x and 30%: 0.0 / +0.7 / +0.3% against 0.0 / 0.0 / -0.1%: PASS. (6) 10x miner leaving at block 600 of an epoch, where the long lane takes over: settled 303 s (292 s with the short lane engaged at the switch), worst gap 17 s, against 334 s for a leave at block 1,200; Kaspa's 2,910 to 3,540 s: PASS. (7) side finding, blocks every 12 ms fed to the rule (another agent's pump): the target passes below 2^64 after 4,142 blocks and calc_work panics at difficulty.rs:431 (should_panic test igneum_flood_at_85_blocks_per_second_drives_the_target_below_block_work_range on diff-attacks): FAIL, floor proposed. Test network, 3 igneumd nodes, honest 4-thread miner A on node 1 for 15 min, forger F (4 threads, about 50%) honest on node 2 for 5 min then on node 3 with the offset: Igneum, earliest allowed stamp: 0.97 blocks/s at difficulty 91,000 before, 0.24 blocks/s at 275,000 during forging and 0.20 at 341,000 in the last 5 min, forger offsets -121 to -566 s, hash unchanged (A 0.078, F 0.069 MH/s). Igneum, latest allowed stamp (+134 s): 0.98 blocks/s at 89,000 before, 0.59 at 170,000 during, 0.50 at 191,000 in the last 5 min (A 0.112, F 0.110 MH/s); the simulator's 50% cases give 0.12 at 9.9x and 0.56 at 1.8x. Kaspa's rule, earliest allowed stamp: 1.73 blocks/s on a 2.5x too easy genesis to block 600 at 210 s, then 1.02 blocks/s at 83,400 through ten minutes of -180 s stamps. 517, 767 and 1,454 blocks, 0 rejected. Proposed (README.md, diffs, not applied): Part A, timestamp rules: 10 s future tolerance (FUTURE_TOLERANCE_MS, a new constant so the past-median window keeps its 27 samples) and a floor at the selected parent's timestamp minus 10 s (BACK_TOLERANCE_MS) beside the past median; Part B, the chain steps of the short and epoch lanes measured on a sanitised running clock c(b) = max(c(p) + clamp(t(b) - c(p), -20 T, +20 T), t(b) - 60 T), step = min(c(b) - c(p), 20 T), stored per header, so forgeries telescope instead of cancelling; the long lane unchanged. Measured, both parts, 3 seeds: the 50% forger drifts the rate +0.7% (past), +0.9% (future), +1.1% (alternating), worst seed +2.7%; base profiles unchanged except down50 762 s against 782 s, warm-up 327 s against 381 s, polluted peak 11.4x against 8.0x; the other attacks identical. Either part alone fails (unchanged rule under the tight rules: -36% and -83% to past-stamping; the clock under the 132 s rules: collapse at 50%, a martingale once the forgery range exceeds half the cap). Part C for the flood: clamp the output at MIN_DIFFICULTY_TARGET = 2^128 beside the existing maximum, in both rules. Not done: the candidate in the node (simulator only; the per-header clock is a store change); DAG effects of forged stamps on red and merged blocks (one chain in the simulator); a rule change for the hopper's 0.7-point excess (none found that keeps the controller fast; README.md, scenario 1); Kaspa's rule under the flood (same hole, 17x slower, not run).

+

2026-10-04 finality fixes F17 and F1: aggregators drawn by weight, first-month gate min_daa = window; attack scenarios 2 and 5 before and after (consensus-engineer)

+

Machine: Apple M5 Max, rustc stable, macOS Darwin 25.6.0. Branch fin-fixes (worktree vendor/igneum-node-fin-fixes, from master 2a00ff55), commit da1eb889. Build: CARGO_TARGET_DIR=target nice -n 19 cargo build --release -j 4 -p kaspad -p igneum-miner --features kaspad/igneum-pow. Tests: cargo test --release -p kaspa-consensus-core -p kaspa-consensus -- finality: 6 of 6 in consensus-core (sortition_threshold, sortition_is_by_weight_not_key_count with 200 dust keys and 6 real ones, first_month_rule_is_the_full_window, the three pre-existing), 1 of 1 in consensus (processes::finality::tests::no_certificate_while_the_window_is_filling, a 150-block TestConsensus chain at a 60-DAA window where one key holds all the weight: nothing certifies under DAA 60, a hand-built early certificate is refused, every checkpoint from DAA 60 locks). The live devnet (26610, 26611, 26640, 26641, 28640) and the other agents' nodes (26680, 27700 to 27720, 28680) were never touched.

+

The two diffs. (1) F17: is_aggregator(output, weight, total_weight, aggregators) is eligible when output x total < aggregators x weight x 2^64 (was output x voters < 8 x 2^64, drawn per key); ingest_vote passes the key's weight and the table total. A key split into n parts holds n thresholds that sum to the one it had; a key without weight never draws; a key at 1/8 of total or more always draws. A node that serves a drawn aggregator aggregates at once; any other node after checkpoint_depth + aggregator_fallback (15) DAA seconds, so anyone MAY aggregate stays the liveness fallback. (2) F1: min_daa = weight_window (mainnet 2,592,000, devnet 7,200); evaluate never locks and ingest_certificate refuses any certificate while the checkpoint's DAA score is below it; the node logs and reports "finality not active, window filling, N of M" (finality_reason, window_filled_daa, window_full_daa on getFinalityCheckpoints).

+

Re-run of scenarios 2 and 5 (tools/finality-attacks, hostile igneum-miner from the fin-attacks worktree, unchanged; it drove the fixed node without modification because the new RPC fields are additive). Six voters on one node, skip_proof_of_work, 600 s per run. The harness copy used for the runs is /tmp/igneum-fin-fixes/harness (lib/net.mjs with ports, data directory, network suffix and the finality override taken from the environment; rerun.mjs with the s2 and s5 measurements below); the repo harness was not edited, and its s2 pass test still reads "voters > 8 means per-key sortition", which is now wrong and needs the by-weight test below. Finality override for every run: interval 30, depth 20, window 1,800 DAA, dust 5, presence 20, 8 aggregators, ban 1,800; min_daa 0 for the before runs (the master default) and 1,800 for the after runs (the fixed rule, min_daa = window), fallback 15. The window was shortened from 7,200 to 1,800 so it fills inside a 10-minute run; the rule under test is the equality, not the number. Before = fin-attacks igneumd (master code, built 3 Oct 23:25) on ports 28100 and 28300, network ids igneum-devnet-801 and 803; after = fin-fixes igneumd on 28500 and 28700, ids 805 and 807. Results in /tmp/igneum-fin-fixes/{before,after}-{s2,s5}/.

+

Scenario 2, Sybil dust (one miner mints 200 keys at 4 blocks and 200 at 6, dust 5; 3 honest voters at 1/3 each; only the honest keys vote, so the measurable is how many honest keys the node records as drawn aggregators per checkpoint). "Crowded" = checkpoints with more than 8 voters above dust (98 of about 127 in each run, mean 125 voters). Expected honest seats per crowded checkpoint: per key, 3 x min(1, 8 / voters); by weight, 3 x min(1, 8 x w / T) with w the honest key's weight.

+
Before (master)After (fin-fixes)
Dust keys with weight or a vote0 of 2000 of 200
Total weight = sum of voter blocks1,798 = 1,7981,798 = 1,798
Honest weight share, crowded checkpoints (mean)32.3%28.8%
Expected honest seats per crowded checkpoint, per-key draw0.330.35
Expected honest seats per crowded checkpoint, by-weight draw1.671.55
Measured honest seats per crowded checkpoint0.321.61
Locks33, first at DAA 629 (a young window held the honest keys alone)25, first at DAA 3,059 (window full at 1,800; the silent 1,200 blocks of sybil weight held the honest keys under the 56.7% floor until they aged out, finality_reason "paused" from DAA 1,800 to 3,059, then "active")
Verdictsortition FAIL (per key: 0.32 against 0.33)sortition PASS (by weight: 1.61 against 1.55)
+

Scenario 5, pulsed miner (five steady voters at 1/6, one burster at 1/6 pulsing 10x for 20 s of every 120 s).

+
Before (master)After (fin-fixes)
Burster weight share vs block share over the run30.9% vs 32.0%, ratio 0.96630.8% vs 32.1%, ratio 0.959
Checkpoints determined / locked148 / 148148 / 88
First lockcheckpoint 1 at DAA 29, built "by 1 of 1 voters, weight 22 (total 22)", the aggregator and sole voter above dust the burster's key (its first 20-s burst); checkpoints 2 and 3 locked at 3 of 3 and 4 of 4 voters as the others cleared dustcheckpoint 61 at DAA 1,829 (the first checkpoint at or above min_daa 1,800), 5 of 6 voters, 84.7% of active and of total
Locks with the checkpoint under min_daa 1,800not gated (checkpoints 1 to 60 all locked)0
Locks carried by one voter above dust1 (checkpoint 1)0
finality_reason over the runfield absent"window filling, 92 of 1800" ... "1448 of 1800" at 180 s, "paused" at 240 s (window full, first lock pending), "active" from 300 s; 59 "window filling" determination lines in the node log
Conflicting certificates00
Verdictamplification PASS, lock-alone FAIL (F1)amplification PASS, lock-alone PASS
+

Notes. The fallback path ("fallback: any node may aggregate") fired 0 times in both after runs: every voter on the single node is local and, with six keys of similar weight, each is drawn at every checkpoint, so every certificate named a drawn aggregator. The "paused" reading between the window filling and the first lock is the report's label for "window full, no lock yet"; it lasted one sample in s5 and five minutes in s2 (the floor against silent sybil weight, 3.3.1). The repo harness tools/finality-attacks/run.mjs keeps its pre-fix s2 and s5 criteria and should adopt the two measurements above; the fin-attacks miner prints no FINALITY line (that is the fin-fixes miner).

+

4 October 2026, difficulty rule: timestamp attack fixed (tight bounds, sanitised clock, target floor), simulator regression, 3-node forger test (consensus-engineer)

+

Machine: the same Apple M5 Max, shared with other agents' simulations (load 6 to 10). Worktree vendor/igneum-node-diff, branch difficulty, commit "Difficulty: timestamp rules 10 s both ways, sanitised clock per header, target floor 2^128" on 3ea7a3e3; built with CARGO_TARGET_DIR=target nice -n 19 cargo ... -j 4 on the rustup toolchain (cargo 1.99; the Homebrew cargo 1.69 in PATH cannot read the 2024 edition). Everything in docs/analysis/difficulty-2026-10-03.md section 11 and spec 2.3; ledger M23. The change, the three parts of sim/difficulty/attacks/README.md as proposed: (A) FUTURE_TOLERANCE_MS 10 s in isolation and BACK_TOLERANCE_MS 10 s behind the selected parent in context (new RuleError::TimeTooFarBehindParent), past-median rule and window unchanged, template floor matched; (B) a sanitised clock per header, c(b) = max(c(p) + clamp(t(b) - c(p), -20 T, +20 T), t(b) - 60 T), in a new store (stores::clock, prefix IgneumClock 62, written in commit_header, deleted at pruning, falling back to the parent's raw stamp when the parent has none), the short and epoch lanes walking clock steps min(c(b) - c(p), 20 T); (C) bound_target floors both rules at 2^128. sim/difficulty/sim.py class Igneum carries the same clock and floor. Unit tests: cargo test --release -p kaspa-consensus --lib difficulty 12 pass (the diff-attacks flood test with should_panic removed: every output at or above 2^128, floor reached between blocks 2,000 and 3,000, work under 2^129; both_rules_share_the_target_floor; igneum_clock_steps_pay_a_forgery_back: three blocks 10 s behind their parents then honest blocks, clock steps sum to the 8 s real span, raw clamped solvetimes to -6 s); cargo test --release -p kaspa-consensus-core --lib igneum 9 pass (sanitised_clock_telescopes_a_forged_stamp). Simulator, seeds 7 to 9 (attacks.py --scenario ts --ts-rules tight), forger at 30 / 50% stamping latest / earliest / alternating, block rate after one hour of forging: 1.008 / 1.008 / 1.004 and 1.009 / 1.007 / 1.011 of target (drift +0.4% to +1.1%, worst seed +2.7%), mean difficulty ratio 1.00, worst gap 10.1 s; the 3 October rule under Kaspa's bounds: 0.66 / 0.42 / 0.51 and 0.56 / 0.12 / 0.23, under the 10 s bounds alone 0.636 and 0.170 on the earliest cells. Flood at 85 blocks/s in the simulator: floor reached at block 2,635, no overflow. Pool hopping (--scenario hop, 24 h): unchanged to three decimals, +1.5% at most greedy, -1.7% and -4.0% with a 60 s dwell at 50 and 100%. Base-profile regression, 3-seed means, 3 October against 4 October: record 102.8 against 91.0 s settled (first within 10% 70.3 against 68.5 s); up50 154.4 against 154.4; down50 782.3 against 762.2 (worst gap 78.3 against 73.9 s); epoch30 87.6 against 87.6; hop10 239.4 against 245.1; polluted 70.1 against 74.9 (seed 9: 78.9 against 92.0), peak 8.0x against 11.4x; steady std unchanged. All means within 10%; the two profiles with an idle gap move, because a gap over 60 T is paid back as three 20 T steps instead of one clamped step. Test network (3 igneumd nodes from the fix plus the diff-attacks template hook on a scratch branch, ports 28500 to 28521, appdir /tmp/igneum-diff-fix, genesis bits 0x1f010000, honest 4-thread miner A on node 1 for 15 min, forger F 4 threads honest on node 2 for 5 min then on node 3 with the offset, two runs): earliest allowed stamp (offset -1e9 ms, floored by the rules; 986 blocks, 823 chain, 0 rejected): chain rate 0.82 blocks/s honest (60 to 300 s) at difficulty 102,226, 0.88 during forging (300 to 900 s) at 100,134, 0.83 in the last 300 s at 106,141, all-blocks rate 1.00 / 1.02 / 0.95, forger offsets -10 to -90 s (mean -23), hash A 0.110 MH/s, F 0.109. Latest allowed stamp (+9,000 ms; 1,028 blocks, 829 chain, 0 rejected): 0.78 honest at 102,382, 0.89 forging at 96,717, 0.89 last 300 s at 102,754, all-blocks 0.95 / 1.09 / 1.11, offsets +9 to +26 s, hash A 0.112, F 0.111. Flat within the CPU miners' noise; the 3 October rule on the same schedule fell to 0.24 blocks/s at 275,135 and 0.59 at 170,222 (bench-log entry above). Records in /tmp/igneum-diff-fix/ts-past-igneum/record.csv and ts-future-igneum/record.csv (not archived into the repo). Not done: the DAG effect of forged stamps on red and merged blocks (one chain in the simulator); the clock of a header whose parent arrived through a pruning proof starts from the raw stamp (one window of exposure after a sync, unmeasured); upstream's timestamp integration tests assume the 132 s bounds and were not re-run; the hopper's 0.7-point excess over Kaspa's rule stays open.

+

4 October 2026, devnet-v4 integration: nine branches merged, 3-node test network on the merged node, Windows cross-build (release engineer)

+

Machine: Apple M5 Max, shared (load 8 to 16, another agent's build and the live devnet running throughout), rustc stable, every build and test at nice 19 with 6 jobs into vendor/igneum-node/target-integration. Branch devnet-v4 of vendor/igneum-node, head dc749905; merge order, conflicts and the cut-over commands in docs/fork-divergence.md, "Integration 4 Oct 2026". The hot swap was not in master (no pow_epoch in 2a00ff55); it was captured from the uncommitted hotswap worktree as a4224689 and merged first. Build times: first release build 3 min 37 s, the execution layer's crates 6 min more, the Windows cross-build 4 min 49 s from a warm dependency cache (proto-cuda/windows-node/cross-build.sh vendor/igneum-node-v4 6). Tests: 628 passed, 0 failed, 24 ignored across the 21 touched crates (--no-fail-fast), then kaspa-p2p-flows 30 of 30 after the estimated_header_size fix (the header's voteKeyHash field was not counted since 815cd00f). Three Kaspa UTXO-body tests are ignored with the reason (the execution layer retires UTXO transactions from bodies); two p2p-lib test modules were brought to the pair-shaped BlockBody. Test network, 02:02:27 to 02:18:53 BST: 3 igneumd on igneum-devnet-880 (gRPC 28800/28810/28820, p2p 28801/28811/28821, wRPC JSON 28802/28812/28822, eth RPC 28803/28813/28823; node 2 and 3 --addpeer node 1, node 3 also node 2), override file genesis_bits 0x1f010000 (2^16) and finality interval 30, depth 20, window 300 DAA, dust 5, presence 20, aggregators 8, ban 300, min_daa 300, fallback 15; IGNEUM_POW_EPOCH_BLOCKS=300, IGNEUM_POW_EPOCH_LEAD=60 (a short epoch so the hourly swap crosses boundaries inside the run; the devnet values are 3,600 and 600). Miners m1, m2, m3 (igneum-miner mine ... 3 960 --engine igneum-pow --payout-label mN --evm-address 0x7099...79C8), 0.078 MH/s each (3 threads on the loaded machine), 375 / 342 / 338 blocks found, 0 rejected.

+
MeasureValue
Blocks accepted per node (PoW accepted lines)1,055 / 1,055 / 1,055, 0 rejected, 0 invalid
Block rate95 to 1,026 blocks between the 0-s and 900-s samples: 1.03 blocks/s; 1,055 in 960 s
Sink identical on all 3 nodes31 of 31 samples; peers 2 on every node at every sample; max tips 2
Difficulty (dual-lane rule)56,268 at 20 s, 155,835 peak at 90 s, 110,533 at 900 s
Epoch boundaries (DAA 300, 600, 900)first block of the new epoch accepted 2.23 / 0.50 / 1.47 s after the last of the old; inter-accept gap over the run median 0.59 s, p90 2.21 s, max 7.27 s
Caches built per node4 (genesis day plus the three epoch seeds); miners' CPU program and cache for a new seed ready in 191 to 212 ms
Finalitywindow filling to DAA 300, paused one sample, active from 270 s (DAA 363); first lock checkpoint 11 (blue 331) by 3 of 3 voters at 100%; 24 locks to checkpoint 34, same index on all 3 nodes at every sample; checkpoint 12 locked at 2 of 3 votes (68.1% of active and of total)
Lock latency (miner, proposed to locked over RPC)median 1,011 / 1,012 / 1,010 ms, max 1,988 / 2,716 / 1,965 ms; 34 of 34 votes accepted per miner
EVM smoke (tools/evm-smoke, copy outside the repo, IGNEUM_RPCS on 28803/28813/28823)84 of 85 checks in 118 s: chain id 4463, miner balance 428.5 IGN at chain block 113 (the IGNA payout address), three accounts funded, 59 transfers executed in 10 chain blocks (max 17 per block, 33 to 91 us execution per block), deploy, call, receipts, logs, revert, developer share, state roots identical across nodes; the failing check is "duplicates landed in parallel blocks within 6 attempts" (identical copies to two nodes landed once in 2 of 6 attempts, the conflicting pair never both), which needs parallel blocks the PoW network did not produce in that 36 s (tips 1 at most samples)
igneum-exec-diff on the smoke exportsegments 0 to 198, 59 transactions compared, 8 accounts, 0 mismatches
Harness scenario 5 (ports 28900+, copy of tools/harness)63 cases (46 RPC, 17 p2p): node up on every case, 0 cache builds (RSS flat, node log 1 build = the honest day), the M15 p2p cases disconnected by the strike guard: PASS
Harness scenario 2 (copy adapted to the merged rules; the repo copy probes 132 s and pmt+1)live: floor-2 and floor-1 rejected, floor and floor+1 accepted where floor = max(pmt + 1, parent - 10 s); future flip between +10.00 and +10.02 s; sim: honest 0.908 b/s, ahead +0.2%, oscillate -0.7% over 1,200 virtual s: PASS
+

Binaries: target-integration/release/{igneumd 40,463,680 B, igneum-miner 7,916,096 B, igneum-exec-diff, igneum-inject, igneum-p2p-probe, igneum-harness-sim}; target-integration/x86_64-pc-windows-gnu/release/{igneumd.exe 50,169,344 B, igneum-miner.exe 10,045,440 B}; package /tmp/igneum-integration/igneum-node-windows-v4.zip (32,946,104 B). Not done: the GPU prepare hot-swap path on the merged miner (CPU miners only here), the cut-over itself, v4 builds for the Mac seed relay and igneum-seed-1, the repo harness's scenario 2 rules and its scenario 5 summary text (stale "built a cache on HEAD" wording while the per-case data says 0 builds).

diff --git a/site/journey.json b/site/journey.json index f794da49e..e6ed90f6c 100644 --- a/site/journey.json +++ b/site/journey.json @@ -1,5 +1,5 @@ { - "updated": "2026-10-03", + "updated": "2026-10-04", "stage": "phase-3", "phases": [ { @@ -50,6 +50,54 @@ } ], "log": [ + { + "date": "2026-10-04", + "text": "Sim/economy: mining versus proving under stress, agent-based" + }, + { + "date": "2026-10-04", + "text": "Difficulty rule under attack: pool hopping, pulsed rental, timestamp stretching, short-lane oscillation, epoch games, polluted window, block flood" + }, + { + "date": "2026-10-04", + "text": "Difficulty rule: timestamp attack fixed , simulator regression, 3-node forger test" + }, + { + "date": "2026-10-04", + "text": "Devnet-v4 integration: nine branches merged, 3-node test network on the merged node, Windows cross-build" + }, + { + "date": "2026-10-03", + "text": "Per-identity hash rate \"decay\" on the RTX 5090: diagnosis and Metal reproduction" + }, + { + "date": "2026-10-03", + "text": "Igneum-node devnet v2: sustained-mining finality rule v2 on a four-miner test network, and as a follower of the live devnet" + }, + { + "date": "2026-10-03", + "text": "Execution layer devnet v3: revm over the selected chain, 3-node simnet, viem smoke test" + }, + { + "date": "2026-10-03", + "text": "Weak-program census: 400,000 program runs through the CPU reference, the redundant-load finding, and the rules for M5 and M6" + }, + { + "date": "2026-10-03", + "text": "Proving v0: first SP1 proof of an Igneum block, Apple M5 Max CPU, loaded machine" + }, + { + "date": "2026-10-03", + "text": "Windows node package: igneumd cross-compiled for x86_64-pc-windows-gnu, two-peer sync test" + }, + { + "date": "2026-10-03", + "text": "R3.26 / M15: PoW checked after the cheap checks, cache-build cap, attack before and after" + }, + { + "date": "2026-10-03", + "text": "Difficulty controller: devnet record, simulator, Igneum dual-lane rule, 3-node CPU test network" + }, { "date": "2026-10-03", "text": "Devnet started four months ahead of plan. Finality, the EVM layer and the Igneum difficulty controller are in build."