Workers: compile-ahead hot swap in the CUDA, OpenCL and Metal hosts, generator v2 port in the Metal host; windows-miner one worker per card; site bench and journey regenerated

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
igneum-labs 2026-10-04 07:55:50 +00:00
parent afaa2ed97e
commit c2120542dd
7 changed files with 981 additions and 80 deletions

View file

@ -134,6 +134,7 @@ struct Options {
bool sweep = false;
int device = 0;
bool serve = false; // --serve: GPU worker for igneum-miner --worker (jobs on stdin), 3 October 2026
bool noPrepare = false; // --no-prepare: serve without the prepare command (ready line says "prepare 0"), to test the miner's fallback
};
static int packMib() { return (int)(((1ull << IGNEUM_DATASET_LOG2) * 4ull) >> 20); }
@ -148,7 +149,8 @@ static void usage() {
" --block-warps W warps per thread block, 1..32 (default 1 = one warp per block, like the Metal run)\n"
" --device D CUDA device index (default 0)\n"
" --serve GPU worker for igneum-miner --worker: reads \"job ...\" lines on stdin, prints found/done lines\n"
" (needs the pack's kernel_bound.cu compiled in: build.bat adds it when the pack has one)\n", packMib());
" (needs the pack's kernel_bound.cu compiled in: build.bat adds it when the pack has one)\n"
" --no-prepare with --serve: no prepare support (the miner then falls back to exit 42 at a seed change)\n", packMib());
}
static bool isPow2(long long v) { return v > 0 && (v & (v - 1)) == 0; }
@ -168,6 +170,7 @@ static Options parseArgs(int argc, char** argv) {
else if (a == "--device") next(o.device);
else if (a == "--sweep") o.sweep = true;
else if (a == "--serve") o.serve = true;
else if (a == "--no-prepare") o.noPrepare = true;
else if (a == "-h" || a == "--help") { usage(); std::exit(0); }
else { std::printf("unknown argument %s\n", argv[i]); usage(); std::exit(2); }
}
@ -401,8 +404,21 @@ static SizeResult runSize(const Options& o, int mib, uint64_t* dOut, uint32_t no
// found <job_id> <nonce u64> <hash_hex 16> every nonce whose 64-bit hash is <= target
// done <job_id> <hashes> <ms> end of the job (wall ms)
// error <job_id> <text>
// The program is compiled ahead of time from the pack (no NVRTC), so this worker serves exactly one epoch seed and
// one day seed: the pack's. A job for other seeds is answered with an error naming both; re-export the pack with
// prepare <epoch_seed_hex 64> <day_seed_hex> <pack_dir> build <pack_dir>/kernel.cu and kernel_bound.cu (the pack the
// miner wrote for those seeds) to cubins with nvcc in the background,
// then load them and build that pair's cache and dataset
// prepared <epoch_seed_hex> <day_seed_hex> <ms> ... the pair is resident; a job on it switches instantly
// prepare-failed <epoch_seed_hex> <day_seed_hex> <text>
// The program is compiled ahead of time from the pack (no NVRTC), so at start this worker serves exactly one epoch
// seed and one day seed: the pack's. The next pair arrives through `prepare`: the miner writes the pack for the
// prepared seeds (igneum-miner --prepare-packs <dir>) and names its directory; a background thread runs nvcc on that
// pack's kernel.cu (cache fill and build kernels, memhard.h for its day) and kernel_bound.cu (the bound hash kernel)
// to two cubins for this device's architecture, and the main loop loads them through the driver API
// (cudaGetDriverEntryPoint, so nothing new is linked), fills the cache and builds the dataset while jobs on the current
// pair keep running (at most two pairs resident; the old one is released after the first job on the new one). nvcc
// must be on PATH with a host compiler, as build.bat needs it; the ready line says "prepare 1" only when `nvcc --version`
// answers. A prepared pair's cache is not cross-checked against the host fill (memhard.h is compiled in for the
// original day); the miner's CPU re-check of every found nonce covers it. Without prepare, re-export the pack with
// `igneum-miner export-pack <node> <dir>` and rebuild. The init words of a dispatch are
// seed_words_from_bytes("igneum-block/" || prehash || nonce_hi_le32), passed by value to igneum_hash_bound
// (kernel_bound.cu); the lane nonce is baseNonce + gid as in the bench kernel.
@ -438,6 +454,162 @@ static bool unhexStr(const std::string& s, std::vector<uint8_t>& out) {
return true;
}
#if defined(IGNEUM_BOUND) && IGNEUM_DATASET_MODE == 1 && defined(__has_include)
#if __has_include(<cuda.h>)
#define IGNEUM_CUDA_PREPARE 1
#include <cuda.h>
#include <thread>
#include <atomic>
#include <fstream>
#endif
#endif
#ifdef IGNEUM_CUDA_PREPARE
// The few driver API entry points the hot swap needs, fetched through the runtime so the link line is unchanged.
struct DriverApi {
CUresult (*moduleLoad)(CUmodule*, const char*) = nullptr;
CUresult (*moduleUnload)(CUmodule) = nullptr;
CUresult (*moduleGetFunction)(CUfunction*, CUmodule, const char*) = nullptr;
CUresult (*moduleGetFunctionCount)(unsigned int*, CUmodule) = nullptr; // CUDA 12.4 and newer
CUresult (*moduleEnumerateFunctions)(CUfunction*, unsigned int, CUmodule) = nullptr;
CUresult (*funcGetName)(const char**, CUfunction) = nullptr; // CUDA 12.3 and newer
CUresult (*launchKernel)(CUfunction, unsigned, unsigned, unsigned, unsigned, unsigned, unsigned, unsigned, CUstream, void**, void**) = nullptr;
CUresult (*getErrorString)(CUresult, const char**) = nullptr;
bool ok = false;
std::string why;
template <typename F> bool get(const char* name, F& fn, bool required) {
void* p = nullptr;
cudaError_t e = cudaGetDriverEntryPoint(name, &p, cudaEnableDefault);
if (e != cudaSuccess || !p) { if (required) { why = std::string("no driver entry point ") + name; } return false; }
fn = reinterpret_cast<F>(p);
return true;
}
void load() {
ok = get("cuModuleLoad", moduleLoad, true) && get("cuModuleUnload", moduleUnload, true) && get("cuModuleGetFunction", moduleGetFunction, true) &&
get("cuLaunchKernel", launchKernel, true) && get("cuGetErrorString", getErrorString, true);
get("cuModuleGetFunctionCount", moduleGetFunctionCount, false);
get("cuModuleEnumerateFunctions", moduleEnumerateFunctions, false);
get("cuFuncGetName", funcGetName, false);
}
std::string err(CUresult r) { const char* s = nullptr; if (getErrorString) getErrorString(r, &s); return s ? s : "CUDA driver error"; }
// A kernel by its plain name: the Itanium mangling nvcc gives device code first, then the enumeration (12.4+).
bool find(CUmodule m, const char* plain, const char* mangled, CUfunction* out) {
if (moduleGetFunction(out, m, mangled) == CUDA_SUCCESS) return true;
if (moduleGetFunction(out, m, plain) == CUDA_SUCCESS) return true;
if (!moduleGetFunctionCount || !moduleEnumerateFunctions || !funcGetName) return false;
unsigned int n = 0;
if (moduleGetFunctionCount(&n, m) != CUDA_SUCCESS || n == 0) return false;
std::vector<CUfunction> fns(n);
if (moduleEnumerateFunctions(fns.data(), n, m) != CUDA_SUCCESS) return false;
for (CUfunction f : fns) {
const char* name = nullptr;
if (funcGetName(&name, f) == CUDA_SUCCESS && name && std::strstr(name, plain)) { *out = f; return true; }
}
return false;
}
};
// One resident pair: the compiled-in pack (runtime launchers, gCache) or a prepared pack (two cubins, driver launches).
struct CudaPair {
std::string epochHex, dayHex;
uint32_t sw[8] = {0}, kw[8] = {0};
bool builtIn = false;
CUmodule modKernel = nullptr, modBound = nullptr;
CUfunction fCacheFill = nullptr, fBuild = nullptr, fHashBound = nullptr;
uint32_t* cache = nullptr;
uint32_t* ds = nullptr;
double nvccMs = 0, cacheMs = 0, dsMs = 0;
};
static void releasePair(DriverApi& drv, CudaPair* p) {
if (!p) return;
if (p->ds) cudaFree(p->ds);
if (p->cache) cudaFree(p->cache);
if (p->modBound) drv.moduleUnload(p->modBound);
if (p->modKernel) drv.moduleUnload(p->modKernel);
delete p;
}
// The nvcc step of a prepare, on its own thread. Only the compiler runs here; every CUDA call stays on the main thread.
struct PrepareTask {
std::string epochHex, dayHex, packDir, arch, error;
std::atomic<bool> done{false};
bool ok = false;
double t0 = 0, nvccMs = 0;
std::thread thread;
};
static bool nvccAvailable() {
#ifdef _WIN32
return std::system("nvcc --version >NUL 2>&1") == 0;
#else
return std::system("nvcc --version >/dev/null 2>&1") == 0;
#endif
}
static void prepareCompile(PrepareTask* t) {
double c0 = wallMs();
const char* files[2] = { "kernel", "kernel_bound" };
for (const char* f : files) {
std::string cmd = "nvcc -cubin -O3 -std=c++17 -arch=" + t->arch + " -allow-unsupported-compiler -I \"" + t->packDir + "\" -o \"" + t->packDir + "/" + f +
".cubin\" \"" + t->packDir + "/" + f + ".cu\" > \"" + t->packDir + "/nvcc-" + f + ".log\" 2>&1";
int rc = std::system(cmd.c_str());
if (rc != 0) {
std::string log;
std::ifstream in(t->packDir + "/nvcc-" + f + ".log");
std::string line;
while (std::getline(in, line) && log.size() < 300) { log += line; log += " | "; }
t->error = std::string("nvcc failed on ") + f + ".cu (exit " + std::to_string(rc) + "): " + log;
t->done = true;
return;
}
}
t->nvccMs = wallMs() - c0;
t->ok = true;
t->done = true;
}
// Loads the two cubins, builds the cache and dataset (main thread). Returns the pair or null with `error` set.
static CudaPair* prepareLoad(DriverApi& drv, const PrepareTask& t, uint32_t words, std::string& error) {
CudaPair* p = new CudaPair();
p->epochHex = t.epochHex; p->dayHex = t.dayHex; p->nvccMs = t.nvccMs;
{
std::vector<uint8_t> eb, db;
if (unhexStr(t.epochHex, eb) && eb.size() == 32) seedWordsFromBytes(eb.data(), 32, p->sw);
if (unhexStr(t.dayHex, db)) seedWordsFromBytes(db.data(), db.size(), p->kw);
}
CUresult r = drv.moduleLoad(&p->modKernel, (t.packDir + "/kernel.cubin").c_str());
if (r != CUDA_SUCCESS) { error = "cuModuleLoad kernel.cubin: " + drv.err(r); releasePair(drv, p); return nullptr; }
r = drv.moduleLoad(&p->modBound, (t.packDir + "/kernel_bound.cubin").c_str());
if (r != CUDA_SUCCESS) { error = "cuModuleLoad kernel_bound.cubin: " + drv.err(r); releasePair(drv, p); return nullptr; }
if (!drv.find(p->modKernel, "igneum_cache_fill", "_Z17igneum_cache_fillPjj", &p->fCacheFill)) { error = "igneum_cache_fill not found in kernel.cubin"; releasePair(drv, p); return nullptr; }
if (!drv.find(p->modKernel, "igneum_build", "_Z12igneum_buildPjPKjj", &p->fBuild)) { error = "igneum_build not found in kernel.cubin"; releasePair(drv, p); return nullptr; }
if (!drv.find(p->modBound, "igneum_hash_bound", "_Z17igneum_hash_boundPKjPyjj15IgneumInitWords", &p->fHashBound)) { error = "igneum_hash_bound not found in kernel_bound.cubin"; releasePair(drv, p); return nullptr; }
// Cache (the same segment count as the compiled-in pack: the dataset schedule is a network constant)
double c0 = wallMs();
size_t cacheBytes = (size_t)CACHE_WORDS_HOST * 4u;
if (cudaMalloc((void**)&p->cache, cacheBytes) != cudaSuccess) { error = "cudaMalloc cache"; p->cache = nullptr; releasePair(drv, p); return nullptr; }
{
uint32_t nSeg = IGNEUM_CACHE_SEGMENTS, block = 256u, grid = (nSeg + block - 1u) / block;
void* args[2] = { &p->cache, &nSeg };
r = drv.launchKernel(p->fCacheFill, grid, 1, 1, block, 1, 1, 0, nullptr, args, nullptr);
if (r != CUDA_SUCCESS || cudaDeviceSynchronize() != cudaSuccess) { error = "cache fill launch: " + drv.err(r); releasePair(drv, p); return nullptr; }
}
p->cacheMs = wallMs() - c0;
// Dataset
c0 = wallMs();
if (cudaMalloc((void**)&p->ds, (size_t)words * 4u) != cudaSuccess) { error = "cudaMalloc dataset"; p->ds = nullptr; releasePair(drv, p); return nullptr; }
{
uint32_t nItems = words / 16u, block = 256u, grid = (nItems + block - 1u) / block;
void* args[3] = { &p->ds, &p->cache, &nItems };
r = drv.launchKernel(p->fBuild, grid, 1, 1, block, 1, 1, 0, nullptr, args, nullptr);
if (r != CUDA_SUCCESS || cudaDeviceSynchronize() != cudaSuccess) { error = "dataset build launch: " + drv.err(r); releasePair(drv, p); return nullptr; }
}
p->dsMs = wallMs() - c0;
return p;
}
#endif
static int runServe(const Options& o) {
#if !defined(IGNEUM_BOUND) || IGNEUM_DATASET_MODE != 1
(void)o;
@ -465,7 +637,27 @@ static int runServe(const Options& o) {
std::vector<uint64_t> hOut(batch);
int regs = 0, bps = 0;
igneum_hash_bound_info(&regs, &bps, (uint32_t)o.blockWarps);
std::printf("ready cuda %s pack %s dataset-log2 %d batch %u regs %d\n", devName.c_str(), IGNEUM_SEED_STRING, IGNEUM_DATASET_LOG2, batch, regs);
// Prepare support: the driver entry points and nvcc on PATH
int prepareOk = 0;
#ifdef IGNEUM_CUDA_PREPARE
DriverApi drv;
drv.load();
std::string arch = "sm_" + std::to_string(prop.major) + std::to_string(prop.minor);
bool haveNvcc = !o.noPrepare && nvccAvailable();
prepareOk = (!o.noPrepare && drv.ok && haveNvcc) ? 1 : 0;
CudaPair* cur = new CudaPair();
cur->builtIn = true; cur->cache = gCache; cur->ds = dDs;
std::memcpy(cur->sw, SEEDW, 32); std::memcpy(cur->kw, KEYW, 32);
CudaPair* prepared = nullptr;
CudaPair* old = nullptr;
PrepareTask* task = nullptr;
#endif
std::printf("ready cuda %s pack %s dataset-log2 %d batch %u regs %d prepare %d\n", devName.c_str(), IGNEUM_SEED_STRING, IGNEUM_DATASET_LOG2, batch, regs, prepareOk);
#ifdef IGNEUM_CUDA_PREPARE
if (!o.noPrepare && !prepareOk) std::printf("info prepare unavailable: %s\n", !drv.ok ? drv.why.c_str() : "nvcc is not on PATH (open the build prompt, or install the CUDA Toolkit)");
#else
std::printf("info prepare unavailable: this binary was built without cuda.h (CPU emulation or an old toolkit)\n");
#endif
std::fflush(stdout);
std::string line;
@ -474,6 +666,41 @@ static int runServe(const Options& o) {
std::vector<std::string> f;
{ size_t i = 0; while (i < line.size()) { while (i < line.size() && line[i] == ' ') ++i; size_t j = i; while (j < line.size() && line[j] != ' ') ++j; if (j > i) f.push_back(line.substr(i, j - i)); i = j; } }
if (f.empty()) continue;
#ifdef IGNEUM_CUDA_PREPARE
// A finished nvcc step is loaded here, between lines, on this thread
if (task && task->done) {
task->thread.join();
if (task->ok) {
std::string error;
CudaPair* p = prepareLoad(drv, *task, words, error);
if (p) {
if (prepared) releasePair(drv, prepared);
prepared = p;
std::printf("prepared %s %s %.1f nvcc %.1f cache %.1f dataset %.1f resident 2 programs 2 datasets\n", p->epochHex.c_str(), p->dayHex.c_str(), wallMs() - task->t0, p->nvccMs, p->cacheMs, p->dsMs);
} else {
std::printf("prepare-failed %s %s %s\n", task->epochHex.c_str(), task->dayHex.c_str(), error.c_str());
}
} else {
std::printf("prepare-failed %s %s %s\n", task->epochHex.c_str(), task->dayHex.c_str(), task->error.c_str());
}
std::fflush(stdout);
delete task; task = nullptr;
}
if (f[0] == "prepare") {
if (!prepareOk) { std::printf("info ignored (no prepare support): %s\n", line.c_str()); std::fflush(stdout); continue; }
if (f.size() < 4) { std::printf("prepare-failed %s %s this ahead-of-time worker needs a pack directory as the third field (igneum-miner --prepare-packs <dir>)\n", f.size() > 1 ? f[1].c_str() : "0", f.size() > 2 ? f[2].c_str() : "0"); std::fflush(stdout); continue; }
if (f[1].size() != 64) { std::printf("prepare-failed %s %s bad field (epoch_seed 64 hex, day_seed hex)\n", f[1].c_str(), f[2].c_str()); std::fflush(stdout); continue; }
if (task) { std::printf("prepare-failed %s %s a prepare is still running\n", f[1].c_str(), f[2].c_str()); std::fflush(stdout); continue; }
if (prepared && prepared->epochHex == f[1] && prepared->dayHex == f[2]) { std::printf("prepared %s %s 0 (already resident)\n", f[1].c_str(), f[2].c_str()); std::fflush(stdout); continue; }
task = new PrepareTask();
task->epochHex = f[1]; task->dayHex = f[2]; task->packDir = f[3]; task->arch = arch; task->t0 = wallMs();
task->thread = std::thread(prepareCompile, task);
std::printf("info prepare started for epoch %.16s day %s from %s (nvcc -arch=%s in the background)\n", f[1].c_str(), f[2].c_str(), f[3].c_str(), arch.c_str()); std::fflush(stdout);
continue;
}
#else
if (f[0] == "prepare") { std::printf("info ignored (no prepare support): %s\n", line.c_str()); std::fflush(stdout); continue; }
#endif
if (f[0] != "job") { std::printf("info ignored: %s\n", line.c_str()); std::fflush(stdout); continue; }
std::string jobId = f.size() > 1 ? f[1] : "0";
if (f.size() < 8) { std::printf("error %s malformed job line (need 7 fields after job)\n", jobId.c_str()); std::fflush(stdout); continue; }
@ -488,6 +715,26 @@ static int runServe(const Options& o) {
uint32_t sw[8], kw[8];
seedWordsFromBytes(epochSeed.data(), epochSeed.size(), sw);
seedWordsFromBytes(daySeed.data(), daySeed.size(), kw);
double t0 = wallMs();
#ifdef IGNEUM_CUDA_PREPARE
bool switched = false;
if (std::memcmp(sw, cur->sw, 32) != 0 || std::memcmp(kw, cur->kw, 32) != 0) {
if (prepared && std::memcmp(sw, prepared->sw, 32) == 0 && std::memcmp(kw, prepared->kw, 32) == 0) {
if (old) releasePair(drv, old);
old = cur; cur = prepared; prepared = nullptr; switched = true;
std::printf("info switched to the prepared pair epoch %.16s day %s in %.2f ms\n", cur->epochHex.c_str(), cur->dayHex.c_str(), wallMs() - t0); std::fflush(stdout);
} else if (std::memcmp(sw, cur->sw, 32) != 0) {
std::printf("error %s epoch seed mismatch: this worker holds %s%s (seed words %08x %08x ...)%s, the job's epoch seed %s gives %08x %08x ...; send prepare with a pack directory, or run igneum-miner export-pack and rebuild\n",
jobId.c_str(), cur->builtIn ? "pack \"" IGNEUM_SEED_STRING "\"" : "prepared epoch ", cur->builtIn ? "" : cur->epochHex.c_str(), cur->sw[0], cur->sw[1], prepared ? " plus one prepared pair" : "", f[6].substr(0, 16).c_str(), sw[0], sw[1]);
std::fflush(stdout); continue;
} else {
std::printf("error %s day seed mismatch: this worker's cache is for key %08x %08x ..., the job's day seed %s gives %08x %08x ...; send prepare with a pack directory, or run igneum-miner export-pack and rebuild\n",
jobId.c_str(), cur->kw[0], cur->kw[1], f[7].c_str(), kw[0], kw[1]);
std::fflush(stdout); continue;
}
}
const uint32_t* jobDs = cur->ds;
#else
if (std::memcmp(sw, SEEDW, 32) != 0) {
std::printf("error %s epoch seed mismatch: this worker was built for pack \"%s\" (seed words %08x %08x ...), the job's epoch seed %s gives %08x %08x ...; run igneum-miner export-pack and rebuild\n",
jobId.c_str(), IGNEUM_SEED_STRING, SEEDW[0], SEEDW[1], f[6].substr(0, 16).c_str(), sw[0], sw[1]);
@ -498,7 +745,8 @@ static int runServe(const Options& o) {
jobId.c_str(), KEYW[0], KEYW[1], f[7].c_str(), kw[0], kw[1]);
std::fflush(stdout); continue;
}
double t0 = wallMs();
const uint32_t* jobDs = dDs;
#endif
uint64_t remaining = nonceCount, hashes = 0;
uint32_t hi = (uint32_t)(nonceStart >> 32), lo = (uint32_t)nonceStart;
bool failed = false;
@ -515,7 +763,18 @@ static int runServe(const Options& o) {
b[45] = (uint8_t)hi; b[46] = (uint8_t)(hi >> 8); b[47] = (uint8_t)(hi >> 16); b[48] = (uint8_t)(hi >> 24);
seedWordsFromBytes(b, 49, iw.w);
}
cudaError_t e = igneum_launch_hash_bound(dDs, dOut, lo, mask, iw, chunk, (uint32_t)o.blockWarps);
cudaError_t e = cudaSuccess;
#ifdef IGNEUM_CUDA_PREPARE
if (!cur->builtIn) {
// A prepared pair: the same launch shape as igneum_launch_hash_bound, through the driver API
uint32_t block = 32u * (uint32_t)o.blockWarps;
uint32_t baseNonce = lo, maskArg = mask;
void* args[5] = { (void*)&jobDs, (void*)&dOut, &baseNonce, &maskArg, &iw };
CUresult r = drv.launchKernel(cur->fHashBound, chunk / block, 1, 1, block, 1, 1, 0, nullptr, args, nullptr);
if (r != CUDA_SUCCESS) { std::printf("error %s dispatch failed: %s\n", jobId.c_str(), drv.err(r).c_str()); std::fflush(stdout); failed = true; break; }
} else
#endif
e = igneum_launch_hash_bound(jobDs, dOut, lo, mask, iw, chunk, (uint32_t)o.blockWarps);
if (e == cudaSuccess) e = cudaDeviceSynchronize();
if (e == cudaSuccess) e = cudaMemcpy(hOut.data(), dOut, (size_t)chunk * sizeof(uint64_t), cudaMemcpyDeviceToHost);
if (e != cudaSuccess) { std::printf("error %s dispatch failed: %s\n", jobId.c_str(), cudaGetErrorString(e)); std::fflush(stdout); failed = true; break; }
@ -530,11 +789,21 @@ static int runServe(const Options& o) {
}
if (failed) continue;
std::printf("done %s %llu %.2f\n", jobId.c_str(), (unsigned long long)hashes, wallMs() - t0);
#ifdef IGNEUM_CUDA_PREPARE
if (switched && old) { releasePair(drv, old); old = nullptr; std::printf("info dropped the previous pair (its program, cache and dataset)\n"); }
#endif
std::fflush(stdout);
}
cudaFree(dOut);
#ifdef IGNEUM_CUDA_PREPARE
if (task) { task->thread.join(); delete task; }
if (old) releasePair(drv, old);
if (prepared) releasePair(drv, prepared);
releasePair(drv, cur); // frees gCache and dDs when the built-in pair is still current
#else
cudaFree(dDs);
cudaFree(gCache);
#endif
return 0;
#endif
}
@ -551,6 +820,12 @@ int main(int argc, char** argv) {
if (count == 0) { std::printf("FAIL: no CUDA device\n"); return 2; }
if (o.device < 0 || o.device >= count) { std::printf("FAIL: device %d out of range (%d devices)\n", o.device, count); return 2; }
CUDA_CHECK(cudaSetDevice(o.device));
// Blocking sync, set before the context exists: the host thread sleeps in cudaDeviceSynchronize instead of
// spinning (one full core per worker process at the default spin schedule, measured on the RTX 5090 with
// eight workers, 3 Oct 2026). The microseconds of wake-up latency are nothing against a 100 ms dispatch.
#ifdef cudaDeviceScheduleBlockingSync
CUDA_CHECK(cudaSetDeviceFlags(cudaDeviceScheduleBlockingSync));
#endif
if (o.serve) {
if (o.batchLog2 == 24) { Options s2 = o; s2.batchLog2 = 22; return runServe(s2); } // 2^22 nonces per dispatch by default
return runServe(o);

View file

@ -5,7 +5,8 @@ rem one igneum-miner per worker against the Mac node. Logs land next to this fil
rem The window shows a dashboard: hash rate, blocks found and one line per miner. Stop with Ctrl+C; the window stays open at the end.
rem ---- settings: edit these lines only ----------------------------------------------------------
set "NODE_HOST=192.168.68.64"
rem auto = a node on this PC (START-NODE.bat, 127.0.0.1) when one is running, else the Mac node at 192.168.68.64.
set "NODE_HOST=auto"
set "NODE_PORT=26610"
rem Devnet payout address (igneumdev:...). Leave empty for the miner's fixed test address.
set "PAYOUT_ADDRESS="

View file

@ -70,6 +70,9 @@ $vcvars = if ($env:VCVARS) { $env:VCVARS } else { 'C:\Program Files\Microsoft Vi
$vcvarsVer = if ($env:VCVARS_VER) { $env:VCVARS_VER } else { '14.30' }
$minersPerVendor = 1
if ($env:MINERS -and [int]$env:MINERS -ge 1) { $minersPerVendor = [int]$env:MINERS }
# One worker process per card (3 Oct 2026): the MINERS identities run through ONE igneum-miner with --identities N and one
# worker (one context, one cache and dataset, one host thread). ONE_WORKER_PER_CARD=0 restores a worker per identity.
$oneWorkerPerCard = -not ($env:ONE_WORKER_PER_CARD -eq '0')
# Nonces per job, per vendor (a multiple of 32). Empty = the miner's default of 16,777,216. A job is one template, so a
# job should take well under a minute: a slow GPU with a long job mines a stale template and reports late.
$nvidiaJobNonces = $env:NVIDIA_JOB_NONCES
@ -321,8 +324,9 @@ foreach ($name in $plan) {
$exe = Build-Worker $name $seeds
if (-not $exe) { Log "$name skipped: no worker binary"; continue }
$instances = @()
for ($i = 1; $i -le $minersPerVendor; $i++) {
$suffix = if ($minersPerVendor -gt 1) { "-$i" } else { '' }
$processes = if ($oneWorkerPerCard) { 1 } else { $minersPerVendor }
for ($i = 1; $i -le $processes; $i++) {
$suffix = if ($processes -gt 1) { "-$i" } else { '' }
$instances += @{ vendor = $name; index = $i; label = "$name-$machine$suffix"; file = "$name$suffix-$stamp"; log = $null; err = $null;
runId = "$name-$machine$suffix-$stamp"; proc = $null; restarts = 0; starts = 0; startedAt = $null; restartAt = $null; exitCode = $null;
logPos = [long]0; logRem = ''; errPos = [long]0; errRem = ''; accepted = 0; found = (New-Object System.Collections.ArrayList);
@ -332,7 +336,8 @@ foreach ($name in $plan) {
$vendors[$name] = @{ name = $name; card = (Get-CardName $name); exe = $exe; instances = $instances; rebuilds = 0; building = $false }
}
if ($vendors.Count -eq 0) { Log 'no worker could be built; see the launcher log'; exit 1 }
Log ("identities per vendor: $minersPerVendor (each identity runs its own worker: about 1.3 GiB of GPU memory per instance)")
if ($oneWorkerPerCard) { Log ("identities per vendor: $minersPerVendor through ONE worker process per card (labels <vendor>-<PC>-1..$minersPerVendor; about 1.3 GiB of GPU memory per card)") }
else { Log ("identities per vendor: $minersPerVendor (each identity runs its own worker: about 1.3 GiB of GPU memory per instance)") }
Log ("job nonces: nvidia " + $(if ($nvidiaJobNonces) { $nvidiaJobNonces } else { 'miner default (16,777,216)' }) + ", amd $amdJobNonces")
function Start-Miner($v, $inst) {
@ -349,6 +354,7 @@ function Start-Miner($v, $inst) {
# --exit-on-seed-change is the fallback only: a worker that answers "prepare 1" on its ready line is never exited at a
# seed change; the miner writes the next pack under --prepare-packs and the worker builds it in the background.
$margs = @('mine', $nodeUrl, '1', '100000000', $inst.label, '--worker', ('"' + $v.exe + '"'), '--status-secs', '30', '--exit-on-seed-change', '--prepare-packs', ('"' + $preparedPacks + '"'))
if ($oneWorkerPerCard -and $minersPerVendor -gt 1) { $margs += @('--identities', $minersPerVendor) }
if ($payout) { $margs += @('--address', $payout) } else { $margs += @('--payout-label', $inst.label) }
if ($v.name -eq 'nvidia') {
if ($nvidiaJobNonces) { $margs += @('--job-nonces', $nvidiaJobNonces) }

View file

@ -18,6 +18,7 @@ struct Options {
var dumpDir: String? = nil
var exportPack: String? = nil // write a CUDA program pack for --seed into this directory and exit
var serve = false // --serve: GPU worker for igneum-miner (jobs on stdin, results on stdout), 3 October 2026
var noPrepare = false // --no-prepare: serve without the prepare command (ready line says "prepare 0"), to test the miner's fallback
// Hardening tests (added 3 October 2026). Any of these runs instead of the bench.
var fuzz: Int? = nil // --fuzz N: N random programs, GPU vs CPU on 4 random warps each
var fuzzSeed = "igneum-fuzz-2026-10-03"
@ -54,6 +55,7 @@ func parseArgs() -> Options {
case "--dump": o.dumpDir = take()
case "--export-pack": o.exportPack = take()
case "--serve": o.serve = true
case "--no-prepare": o.noPrepare = true
case "--fuzz": o.fuzz = Int(take()) ?? 200
case "--fuzz-seed": o.fuzzSeed = take()
case "--edge": o.edge = true
@ -356,6 +358,8 @@ struct Program {
let seedString: String
let seed: [UInt32]
let instrs: [Instr]
var generator = 1 // 2 for every current program (generateProgramV2); 1 for the retired lever generator
var attempt: UInt32 = 0 // attempt index under the acceptance rule (0 = the bare seed)
static let iterations = 8
static let count = 64
var loadsPerHash: Int { instrs.filter { $0.op == .load || $0.op == .wload }.count * Program.iterations }
@ -397,10 +401,18 @@ struct GeneratorConfig {
}
var generatorConfig = GeneratorConfig()
func generateProgram(seedString: String) -> Program { generateProgram(seedString: seedString, words: seedWords(seedString)) }
// The program of a seed string: generator version 2 with the acceptance rule (below), the same program the Rust crate
// derives. The retired version 1 generator is used only when a lever (--load-weight, --wide-frac) is set, for the
// MEMHARD.md section 2.4 measurements; those programs are not the lottery hash.
func generateProgram(seedString: String) -> Program {
if generatorConfig.loadWeight != 25 || generatorConfig.wideFrac != 0 {
return generateProgramV1(seedString: seedString, words: seedWords(seedString))
}
return generateProgramV2(seedString: seedString, bytes: Array(seedString.utf8))
}
// The generator from already-derived seed words (what the chain feeds: the epoch block hash on devnet v0, the VDF output later).
func generateProgram(seedString: String, words sw: [UInt32]) -> Program {
// Version 1 (retired 4 October 2026): op rolled per instruction against the 11-family table, load count free.
func generateProgramV1(seedString: String, words sw: [UInt32]) -> Program {
var rng = SplitMix64(s: (UInt64(sw[0]) | (UInt64(sw[1]) << 32)) ^ ((UInt64(sw[2]) | (UInt64(sw[3]) << 32)) &* 0x9E3779B97F4A7C15))
var instrs = [Instr]()
let weights = generatorConfig.weights
@ -423,6 +435,232 @@ func generateProgram(seedString: String, words sw: [UInt32]) -> Program {
return Program(seedString: seedString, seed: sw, instrs: instrs)
}
// MARK: - Generator version 2 and the acceptance rule (4 October 2026)
//
// Draw for draw the Rust generator (igneum-pow/src/generator.rs, candidate_from_words) and acceptance rule
// (igneum-pow/src/accept.rs), spec 01 sections 1.4.3 and 1.4.6. Exactly 16 load slots drawn first from instructions
// 1..63; a load's source is drawn from the registers other than dst written by an earlier instruction and not read by
// a load since; a candidate that fails the rule is replaced by attempt k + 1, seedWordsBytes(seed || k_le32).
let generatorVersion = 2
let loadSlots = 16
let maxAttempts: UInt32 = 32
let nonloadWeights: [(Op, Int)] = [(.add, 12), (.xor, 10), (.mul, 8), (.mad, 8), (.shfl, 8),
(.rotl, 7), (.sub, 6), (.mulhi, 6), (.rotr, 6), (.or, 4)]
let acceptUnits = 64
let acceptDatasetLog2 = 28
let acceptMaxSaturated: UInt32 = 164
let acceptBiasTolerance: UInt32 = 136
let acceptMinDistinctSum: UInt64 = 245_760
func fnv1a64Bytes(_ bytes: [UInt8]) -> UInt64 {
var h: UInt64 = 0xcbf29ce484222325
for b in bytes { h ^= UInt64(b); h &*= 0x100000001b3 }
return h
}
func le32(_ v: UInt32) -> [UInt8] { [UInt8(v & 0xff), UInt8((v >> 8) & 0xff), UInt8((v >> 16) & 0xff), UInt8((v >> 24) & 0xff)] }
// The seed words of attempt k: seedWordsBytes(seed) for k = 0, seedWordsBytes(seed || k_le32) otherwise.
func attemptWords(_ seedBytes: [UInt8], _ attempt: UInt32) -> [UInt32] {
attempt == 0 ? seedWordsBytes(seedBytes) : seedWordsBytes(seedBytes + le32(attempt))
}
// FNV-1a 64 over "igneum-program/" || generator_le32 || seed words LE || attempt_le32 (Program::program_id in Rust).
func programId(_ p: Program) -> UInt64 {
var b = Array("igneum-program/".utf8) + le32(UInt32(p.generator))
for w in p.seed { b += le32(w) }
b += le32(p.attempt)
return fnv1a64Bytes(b)
}
// One version 2 candidate from its seed words, before the acceptance rule.
func candidateProgram(seedString: String, words sw: [UInt32], attempt: UInt32) -> Program {
var rng = SplitMix64(s: (UInt64(sw[0]) | (UInt64(sw[1]) << 32)) ^ ((UInt64(sw[2]) | (UInt64(sw[3]) << 32)) &* 0x9E3779B97F4A7C15))
var slots = Array(1..<Program.count)
for i in 0..<loadSlots {
let j = i + rng.below(Program.count - 1 - i)
slots.swapAt(i, j)
}
var isLoad = [Bool](repeating: false, count: Program.count)
for i in 0..<loadSlots { isLoad[slots[i]] = true }
var fresh = [Bool](repeating: false, count: 8)
var instrs = [Instr]()
for k in 0..<Program.count {
var roll = rng.below(75)
var op = Op.add
for (o, w) in nonloadWeights { if roll < w { op = o; break }; roll -= w }
if isLoad[k] { op = .load }
let dst = rng.below(8)
var a: Int
if op == .load {
let eligible = (0..<8).filter { $0 != dst && fresh[$0] }
if eligible.isEmpty {
a = rng.below(7); if a >= dst { a += 1 }
} else {
a = eligible[rng.below(eligible.count)]
}
} else {
a = rng.below(7); if a >= dst { a += 1 }
}
let b = rng.below(8)
let imm = UInt32(truncatingIfNeeded: rng.next())
let imm2 = UInt32(truncatingIfNeeded: rng.next())
let rot = UInt32(1 + rng.below(31))
let bit = rng.below(32)
let mask = 1 << rng.below(5)
if op == .load { fresh[a] = false }
fresh[dst] = true
instrs.append(Instr(op: op, dst: dst, a: a, b: b, imm: imm, imm2: imm2, rot: rot, bit: bit, mask: mask))
}
return Program(seedString: seedString, seed: sw, instrs: instrs, generator: generatorVersion, attempt: attempt)
}
@inline(__always) func opInjects(_ op: Op) -> Bool {
switch op { case .add, .sub, .xor, .mad, .shfl, .load, .wload: return true; default: return false }
}
// Parts (a) and (b) of the rule. Returns nil when the program passes, else the reason.
func acceptStatic(_ p: Program) -> String? {
var pending = [Bool](repeating: false, count: 8)
for _ in 0..<2 {
for (k, ins) in p.instrs.enumerated() {
let isLoad = ins.op == .load || ins.op == .wload
if isLoad && pending[ins.a] { return "(a) load at instruction \(k) reads r\(ins.a), unwritten since the previous load from it" }
pending[ins.dst] = false
if isLoad { pending[ins.a] = true }
}
}
var injected = [Bool](repeating: false, count: 8)
for ins in p.instrs where opInjects(ins.op) { injected[ins.dst] = true }
for r in 0..<8 where !injected[r] { return "(b) r\(r) has no add, sub, xor, mad, shfl or load write" }
return nil
}
// The 64 base nonces of the dynamic test: SplitMix64 seeded with FNV-1a 64("igneum-accept/" || seed words LE).
func acceptBaseNonces(_ seed: [UInt32]) -> [UInt32] {
var b = Array("igneum-accept/".utf8)
for w in seed { b += le32(w) }
var rng = SplitMix64(s: fnv1a64Bytes(b))
return (0..<acceptUnits).map { _ in UInt32(truncatingIfNeeded: rng.next()) & ~31 }
}
// Part (c): 64 units on the closed-form dataset keyed by the seed words, init words = seed words. Returns nil
// when the program passes, else the reason. Mirrors run_unit and check_dynamic in igneum-pow/src/accept.rs.
func acceptDynamic(_ p: Program) -> String? {
let lanes = 32
let loads = p.loadsPerHash
let mask: UInt32 = (1 << UInt32(acceptDatasetLog2)) - 1
let (d0, d1) = (p.seed[0], p.seed[1])
var andAcc = [UInt32](repeating: 0xffffffff, count: 8)
var orAcc = [UInt32](repeating: 0, count: 8)
var saturated: UInt32 = 0
var bitOnes = [UInt32](repeating: 0, count: 64)
var distinctSum: UInt64 = 0
var r = [UInt32](repeating: 0, count: 8 * lanes) // r[reg * 32 + lane]
var laneAddrs = [UInt32](repeating: 0, count: lanes * loads)
var sel = [UInt32](repeating: 0, count: lanes)
var idx = [UInt32](repeating: 0, count: lanes)
var tmp = [UInt32](repeating: 0, count: lanes)
for (unit, base) in acceptBaseNonces(p.seed).enumerated() {
for lane in 0..<lanes {
let nonce = base &+ UInt32(lane)
for i in 0..<8 {
var x = nonce ^ p.seed[i]
x &+= 0x9e3779b9 &* UInt32(i + 1)
x = splitmix32(x)
r[i * lanes + lane] = x ^ p.seed[(i + 1) & 7]
}
}
var nload = 0
for it in 0..<Program.iterations {
for lane in 0..<lanes { sel[lane] = r[lane] }
for (k, ins) in p.instrs.enumerated() {
let d = ins.dst * lanes, a = ins.a * lanes
switch ins.op {
case .add:
for lane in 0..<lanes {
let s = (sel[lane] >> UInt32(ins.bit)) & 1
r[d + lane] = r[d + lane] &+ r[a + lane] &+ (s != 0 ? ins.imm2 : ins.imm)
}
case .sub: for lane in 0..<lanes { r[d + lane] = r[d + lane] &- r[a + lane] }
case .mul: for lane in 0..<lanes { r[d + lane] = r[d + lane] &* r[a + lane] }
case .mulhi: for lane in 0..<lanes { r[d + lane] = mulhi32(r[d + lane], r[a + lane]) }
case .xor: for lane in 0..<lanes { r[d + lane] ^= r[a + lane] }
case .or: for lane in 0..<lanes { r[d + lane] |= r[a + lane] }
case .rotl: for lane in 0..<lanes { r[d + lane] = rotl32(r[d + lane], ins.rot) }
case .rotr: for lane in 0..<lanes { r[d + lane] = rotr32(r[d + lane], r[a + lane]) }
case .mad:
let b = ins.b * lanes
for lane in 0..<lanes { r[d + lane] = (r[a + lane] &* r[b + lane]) &+ r[d + lane] }
case .shfl:
for lane in 0..<lanes { tmp[lane] = r[a + lane] }
for lane in 0..<lanes { r[d + lane] ^= tmp[lane ^ ins.mask] }
case .load:
for lane in 0..<lanes { idx[lane] = r[a + lane] & mask }
var same = true
for lane in 1..<lanes where idx[lane] != idx[0] { same = false; break }
if same { return "(c) load at iteration \(it) instruction \(k) reads one address in all lanes of unit \(unit)" }
for lane in 0..<lanes {
r[d + lane] ^= datasetElem(idx[lane], d0, d1)
laneAddrs[lane * loads + nload] = idx[lane]
}
nload += 1
case .wload:
let b = (r[a] & mask) & ~31
for lane in 0..<lanes {
idx[lane] = b + UInt32(lane)
r[d + lane] ^= datasetElem(idx[lane], d0, d1)
laneAddrs[lane * loads + nload] = idx[lane]
}
nload += 1
}
}
}
for i in 0..<8 {
for lane in 0..<lanes {
let v = r[i * lanes + lane]
andAcc[i] &= v; orAcc[i] |= v
if v == 0 || v == 0xffffffff { saturated += 1 }
}
}
for lane in 0..<lanes {
let lo = r[lane] ^ rotl32(r[lanes + lane], 7) ^ rotl32(r[2 * lanes + lane], 14) ^ rotl32(r[3 * lanes + lane], 21)
let hi = r[4 * lanes + lane] ^ rotl32(r[5 * lanes + lane], 9) ^ rotl32(r[6 * lanes + lane], 18) ^ rotl32(r[7 * lanes + lane], 27)
let h = (UInt64(hi) << 32) | UInt64(lo)
for j in 0..<64 { bitOnes[j] += UInt32((h >> UInt64(j)) & 1) }
var sl = Array(laneAddrs[lane * loads..<(lane + 1) * loads])
sl.sort()
var distinct: UInt64 = 0
for k in 0..<loads where k == 0 || sl[k] != sl[k - 1] { distinct += 1 }
distinctSum += distinct
}
}
for i in 0..<8 {
let bits = (andAcc[i] | ~orAcc[i]).nonzeroBitCount
if bits != 0 { return "(c) r\(i) has \(bits) nonce-independent bits" }
}
if saturated >= acceptMaxSaturated { return "(c) \(saturated) of 16384 final register values saturated (limit 163)" }
let half = UInt32(acceptUnits * lanes / 2)
for j in 0..<64 {
let d = bitOnes[j] > half ? bitOnes[j] - half : half - bitOnes[j]
if d > acceptBiasTolerance { return "(c) output bit \(j) set in \(bitOnes[j]) of 2048 hashes" }
}
if distinctSum <= acceptMinDistinctSum { return "(c) distinct addresses \(distinctSum) over 2048 hashes (needs above 245760)" }
return nil
}
func acceptProgram(_ p: Program) -> String? { acceptStatic(p) ?? acceptDynamic(p) }
// The program of a seed under version 2: the first accepted candidate over attempts 0, 1, 2, ...
func generateProgramV2(seedString: String, bytes: [UInt8]) -> Program {
for attempt in 0..<maxAttempts {
let p = candidateProgram(seedString: seedString, words: attemptWords(bytes, attempt), attempt: attempt)
if acceptProgram(p) == nil { return p }
}
fatalError("seed \(seedString): \(maxAttempts) consecutive candidates rejected (consensus fault)")
}
// MARK: - MSL generation
func hex(_ v: UInt32) -> String { String(format: "0x%08xu", v) }
@ -2452,19 +2690,30 @@ func runTests(_ opts: Options) -> Never {
// MARK: - Serve mode (GPU worker for igneum-miner --worker), 3 October 2026
//
// Protocol, one line each. Only "found", "done" and "error" are parsed by the miner; every other line is logged.
// Protocol, one line each. Only "found", "done", "error", "prepared", "prepare-failed" and "ready" are parsed by the
// miner; every other line is logged.
// stdin: job <job_id> <header_prehash_hex 64> <target_hex 16> <nonce_start u64> <nonce_count u64> <epoch_seed_hex 64> <day_seed_hex>
// prepare <epoch_seed_hex 64> <day_seed_hex> [<pack_dir>] compile that program and build that day's dataset in the
// background while jobs on the current seeds keep running
// quit
// stdout: ready metal <device>
// stdout: ready metal <device> dataset-log2 <n> batch <n> prepare 1
// found <job_id> <nonce u64> <hash_hex 16> every nonce whose 64-bit hash is <= target (hash <= target64)
// done <job_id> <hashes> <ms> end of the job (wall ms, dispatch plus scan)
// error <job_id> <text>
// The program for an epoch seed is generateProgram(words: seedWordsBytes(epoch_seed)), compiled once and cached; the
// cache and 1 GiB dataset for a day seed come from seedWordsBytes(day_seed_bytes), built once and cached (two of each).
// prepared <epoch_seed_hex> <day_seed_hex> <ms> program <ms> dataset <ms> ... the pair is resident
// prepare-failed <epoch_seed_hex> <day_seed_hex> <text>
// The program for an epoch seed is generateProgramV2(bytes: epoch_seed) (version 2 with the acceptance rule, the same
// derivation as igneum-pow), compiled once and kept; the
// cache and 1 GiB dataset for a day seed come from seedWordsBytes(day_seed_bytes), built once and kept. At most two
// programs and two datasets are resident: the current job's pair and one more (the prepared pair, or the previous
// pair until the first job on the new one is done). A job whose pair is resident switches instantly; a job whose pair
// is not (no prepare, or a pair nobody predicted) compiles inline as before. The pack_dir of a prepare is ignored
// here (Metal compiles from the seed); ahead-of-time workers build their kernel from it.
// The init words of a dispatch are seedWordsBytes("igneum-block/" || prehash || nonce_hi_le32) and go to buffer 3 of
// igneum_hash_bound; the lane nonce is baseNonce + gid as in the bench kernel.
func emit(_ line: String) { print(line); fflush(stdout) }
let emitLock = NSLock()
func emit(_ line: String) { emitLock.lock(); print(line); fflush(stdout); emitLock.unlock() }
func unhex(_ s: String) -> [UInt8]? {
let chars = Array(s.utf8)
@ -2498,6 +2747,41 @@ final class ServeDataset {
init(dayHex: String, ctx: DatasetContext, buffer: MTLBuffer) { self.dayHex = dayHex; self.ctx = ctx; self.buffer = buffer }
}
// The resident programs and datasets, shared by the job loop (main thread) and the prepare queue (background).
final class ServeStore {
private let lock = NSLock()
private var programs = [ServeProgram]()
private var datasets = [ServeDataset]()
/// The pair of the last job (epoch seed hex, day seed hex); never evicted
var current: (String, String)? = nil
func program(_ seedHex: String) -> ServeProgram? { lock.lock(); defer { lock.unlock() }; return programs.first { $0.seedHex == seedHex } }
func dataset(_ dayHex: String) -> ServeDataset? { lock.lock(); defer { lock.unlock() }; return datasets.first { $0.dayHex == dayHex } }
func counts() -> (Int, Int) { lock.lock(); defer { lock.unlock() }; return (programs.count, datasets.count) }
/// Adds a program; with more than two resident, the oldest one that is not the current job's goes.
func add(_ p: ServeProgram) {
lock.lock(); defer { lock.unlock() }
if programs.contains(where: { $0.seedHex == p.seedHex }) { return }
programs.append(p)
while programs.count > 2, let i = programs.firstIndex(where: { $0.seedHex != current?.0 && $0.seedHex != p.seedHex }) { programs.remove(at: i) }
}
func add(_ d: ServeDataset) {
lock.lock(); defer { lock.unlock() }
if datasets.contains(where: { $0.dayHex == d.dayHex }) { return }
datasets.append(d)
while datasets.count > 2, let i = datasets.firstIndex(where: { $0.dayHex != current?.1 && $0.dayHex != d.dayHex }) { datasets.remove(at: i) }
}
/// After the first job on a new pair: drop everything but that pair (the old program and dataset are released).
func prune(to pair: (String, String)) -> (Int, Int) {
lock.lock(); defer { lock.unlock() }
let before = (programs.count, datasets.count)
programs.removeAll { $0.seedHex != pair.0 }
datasets.removeAll { $0.dayHex != pair.1 }
return (before.0 - programs.count, before.1 - datasets.count)
}
}
func compileBound(_ gpu: GPU, msl: String) throws -> CompiledHash {
let t0 = nowNs()
let lib = try gpu.device.makeLibrary(source: msl, options: MTLCompileOptions())
@ -2508,18 +2792,63 @@ func compileBound(_ gpu: GPU, msl: String) throws -> CompiledHash {
return CompiledHash(pipeline: pipe, libraryMs: ms(t0, t1), pipelineMs: ms(t1, t2))
}
// Builds the program for an epoch seed (hex) unless resident. Returns (program, compile ms) or throws.
func serveProgram(_ gpu: GPU, _ store: ServeStore, seedHex: String, seed: [UInt8], datasetLog2: Int) throws -> (ServeProgram, Double) {
if let p = store.program(seedHex) { return (p, 0) }
let t0 = nowNs()
let p = generateProgramV2(seedString: "epoch/" + seedHex, bytes: seed)
let msl = generateMSL(p, datasetLog2: datasetLog2, source: .stored, bound: true)
let c = try compileBound(gpu, msl: msl)
let sp = ServeProgram(seedHex: seedHex, program: p, compiled: c)
store.add(sp)
return (sp, ms(t0, nowNs()))
}
// Builds the cache and dataset for a day seed (hex) unless resident. Returns (dataset, build ms).
func serveDataset(_ gpu: GPU, _ store: ServeStore, dayHex: String, day: [UInt8], datasetLog2: Int) -> (ServeDataset, Double) {
if let d = store.dataset(dayHex) { return (d, 0) }
let t0 = nowNs()
let key = seedWordsBytes(day)
let ctx = DatasetContext(gpu: gpu, closedForm: false, dayString: "day/" + dayHex, key: key)
let buf = ctx.makeDataset(log2: datasetLog2)
let sd = ServeDataset(dayHex: dayHex, ctx: ctx, buffer: buf)
store.add(sd)
return (sd, ms(t0, nowNs()))
}
func runServe(_ opts: Options) -> Never {
let gpu = GPU()
let datasetLog2 = opts.datasetLog2
let batch = 1 << opts.batchLog2 // nonces per dispatch
var programs = [ServeProgram]()
var datasets = [ServeDataset]()
let store = ServeStore()
let prepareQueue = DispatchQueue(label: "igneum.prepare") // one prepare at a time, off the job loop
guard let outBuf = gpu.device.makeBuffer(length: batch * 8, options: .storageModeShared) else { emit("error 0 cannot allocate the output buffer"); exit(1) }
emit("ready metal \(gpu.device.name.replacingOccurrences(of: " ", with: "_")) dataset-log2 \(datasetLog2) batch \(batch)")
emit("ready metal \(gpu.device.name.replacingOccurrences(of: " ", with: "_")) dataset-log2 \(datasetLog2) batch \(batch) prepare \(opts.noPrepare ? 0 : 1)")
var lastPair: (String, String)? = nil
while let line = readLine(strippingNewline: true) {
let f = line.split(separator: " ").map(String.init)
if f.isEmpty { continue }
if f[0] == "quit" { break }
if f[0] == "prepare" {
if opts.noPrepare { emit("info ignored (started with --no-prepare): \(line)"); continue }
if f.count < 3 { emit("prepare-failed 0 0 malformed prepare line (need epoch_seed_hex and day_seed_hex)"); continue }
guard let epochSeed = unhex(f[1]), epochSeed.count == 32, let daySeed = unhex(f[2]) else {
emit("prepare-failed \(f[1]) \(f[2]) bad field (epoch_seed 64 hex, day_seed hex)"); continue
}
let (epochHex, dayHex) = (f[1], f[2])
let have = (store.program(epochHex) != nil, store.dataset(dayHex) != nil)
if have.0 && have.1 { emit("prepared \(epochHex) \(dayHex) 0 program 0 dataset 0 (already resident)"); continue }
prepareQueue.async {
let t0 = nowNs()
do {
let (sp, progMs) = try serveProgram(gpu, store, seedHex: epochHex, seed: epochSeed, datasetLog2: datasetLog2)
let (sd, dsMs) = serveDataset(gpu, store, dayHex: dayHex, day: daySeed, datasetLog2: datasetLog2)
let (np, nd) = store.counts()
emit("prepared \(epochHex) \(dayHex) \(fmt(ms(t0, nowNs()), 1)) program \(fmt(progMs, 1)) dataset \(fmt(dsMs, 1)) loads/hash \(sp.program.loadsPerHash) cache-fill \(fmt(sd.ctx.cacheFillGPUms, 1)) resident \(np) programs \(nd) datasets")
} catch { emit("prepare-failed \(epochHex) \(dayHex) Metal compile failed: \(error)") }
}
continue
}
if f[0] != "job" { emit("info ignored: \(line)"); continue }
if f.count < 8 { emit("error \(f.count > 1 ? f[1] : "0") malformed job line (need 7 fields after job)"); continue }
let jobId = f[1]
@ -2530,31 +2859,24 @@ func runServe(_ opts: Options) -> Never {
}
if nonceCount == 0 || nonceCount % 32 != 0 || (nonceStart & 31) != 0 { emit("error \(jobId) nonce_start must be 32-aligned and nonce_count a non-zero multiple of 32"); continue }
let t0 = nowNs()
// Program for the epoch seed
var prog = programs.first { $0.seedHex == f[6] }
if prog == nil {
let p = generateProgram(seedString: "epoch/" + f[6], words: seedWordsBytes(epochSeed))
let msl = generateMSL(p, datasetLog2: datasetLog2, source: .stored, bound: true)
do {
let c = try compileBound(gpu, msl: msl)
emit("info program epoch \(f[6].prefix(16)) loads/hash \(p.loadsPerHash) compiled in \(fmt(c.totalMs, 1)) ms")
let sp = ServeProgram(seedHex: f[6], program: p, compiled: c)
programs.append(sp); if programs.count > 2 { programs.removeFirst() }
prog = sp
} catch { emit("error \(jobId) Metal compile failed: \(error)"); continue }
let pair = (f[6], f[7])
let switched = lastPair == nil || lastPair! != pair
if switched { store.current = pair }
// Program and dataset for the pair: resident (prepared, or the current pair) or compiled inline now. A prepare
// of the same pair may be in flight on the queue; waiting for the queue makes this a join instead of a double build.
let program: ServeProgram
let dataset: ServeDataset
do {
if store.program(pair.0) == nil || store.dataset(pair.1) == nil { prepareQueue.sync {} }
let (sp, progMs) = try serveProgram(gpu, store, seedHex: pair.0, seed: epochSeed, datasetLog2: datasetLog2)
if progMs > 0 { emit("info program epoch \(pair.0.prefix(16)) loads/hash \(sp.program.loadsPerHash) compiled inline in \(fmt(progMs, 1)) ms (not prepared)") }
let (sd, dsMs) = serveDataset(gpu, store, dayHex: pair.1, day: daySeed, datasetLog2: datasetLog2)
if dsMs > 0 { emit("info dataset day \(pair.1) built inline in \(fmt(dsMs, 1)) ms (cache fill \(fmt(sd.ctx.cacheFillGPUms, 1)) ms, build \(fmt(sd.ctx.lastBuildGPUms, 1)) ms GPU; not prepared)") }
program = sp; dataset = sd
} catch { emit("error \(jobId) Metal compile failed: \(error)"); continue }
if switched, let prev = lastPair {
emit("info switched from epoch \(prev.0.prefix(16)) day \(prev.1) to epoch \(pair.0.prefix(16)) day \(pair.1) in \(fmt(ms(t0, nowNs()), 2)) ms (resident: \(store.counts().0) programs, \(store.counts().1) datasets)")
}
// Cache and dataset for the day seed
var ds = datasets.first { $0.dayHex == f[7] }
if ds == nil {
let key = seedWordsBytes(daySeed)
let ctx = DatasetContext(gpu: gpu, closedForm: false, dayString: "day/" + f[7], key: key)
let buf = ctx.makeDataset(log2: datasetLog2)
emit("info dataset day \(f[7]) key \(key.map { String(format: "%08x", $0) }.joined(separator: " ")) cache fill \(fmt(ctx.cacheFillGPUms, 1)) ms build \(fmt(ctx.lastBuildGPUms, 1)) ms GPU")
let sd = ServeDataset(dayHex: f[7], ctx: ctx, buffer: buf)
datasets.append(sd); if datasets.count > 2 { datasets.removeFirst() }
ds = sd
}
guard let program = prog, let dataset = ds else { continue }
// Mine: chunks of at most `batch` lane nonces that share one high word
var remaining = nonceCount
var hi = UInt32(truncatingIfNeeded: nonceStart >> 32)
@ -2592,6 +2914,12 @@ func runServe(_ opts: Options) -> Never {
}
if failed { continue }
emit("done \(jobId) \(hashes) \(fmt(ms(t0, nowNs()), 2))")
if switched, lastPair != nil {
// The first job on the new pair is done: the old pair goes (at most two of each were resident until now)
let (dp, dd) = store.prune(to: pair)
if dp + dd > 0 { emit("info dropped \(dp) program(s) and \(dd) dataset(s) of the previous pair") }
}
lastPair = pair
}
exit(0)
}

View file

@ -31,6 +31,7 @@
#else
#include <time.h>
#include <dlfcn.h>
#include <pthread.h>
#endif
#define IGNEUM_NO_CUDA
@ -192,6 +193,7 @@ typedef struct {
const char* kernelPath;
const char* extraOpts;
int serve; // --serve: GPU worker for igneum-miner --worker (jobs on stdin), 3 October 2026
int noPrepare; // --no-prepare: serve without the prepare command (ready line says "prepare 0"), to test the miner's fallback
int kernelGiven; // --kernel was passed
const char* vendor; // --vendor S: pick the first GPU whose vendor string contains S (default: first GPU of any vendor)
} Options;
@ -217,6 +219,7 @@ static void usage(void) {
" Apple's OpenCL runtime reports unusable event timestamps, so wall is the default on the Apple platform.\n"
" --vendor S choose the first GPU whose vendor string contains S (for example \"Advanced Micro Devices\"); fails if none\n"
" --serve GPU worker for igneum-miner --worker: reads \"job ...\" lines on stdin, prints found/done lines.\n"
" --no-prepare with --serve: no prepare support (the miner then falls back to exit 42 at a seed change).\n"
" Builds the pack's kernel_bound.cl (next to the compiled-in kernel.cl) unless --kernel says otherwise.\n", packMib(), IGNEUM_KERNEL_PATH);
}
@ -227,7 +230,7 @@ static Options parseArgs(int argc, char** argv) {
Options o;
int i;
o.datasetMib = 1024; o.batchLog2 = 24; o.batches = 5; o.groupWarps = 1; o.sweep = 0; o.device = -1;
o.exchange = 0; o.list = 0; o.timeWall = -1; o.kernelPath = IGNEUM_KERNEL_PATH; o.extraOpts = ""; o.serve = 0; o.kernelGiven = 0; o.vendor = NULL;
o.exchange = 0; o.list = 0; o.timeWall = -1; o.kernelPath = IGNEUM_KERNEL_PATH; o.extraOpts = ""; o.serve = 0; o.noPrepare = 0; o.kernelGiven = 0; o.vendor = NULL;
for (i = 1; i < argc; ++i) {
const char* a = argv[i];
int needs = (strcmp(a, "--dataset-mib") == 0 || strcmp(a, "--batch-log2") == 0 || strcmp(a, "--batches") == 0 ||
@ -241,6 +244,7 @@ static Options parseArgs(int argc, char** argv) {
else if (strcmp(a, "--device") == 0) o.device = atoi(argv[++i]);
else if (strcmp(a, "--kernel") == 0) { o.kernelPath = argv[++i]; o.kernelGiven = 1; }
else if (strcmp(a, "--serve") == 0) o.serve = 1;
else if (strcmp(a, "--no-prepare") == 0) o.noPrepare = 1;
else if (strcmp(a, "--vendor") == 0) { if (i + 1 >= argc) { usage(); exit(2); } o.vendor = argv[++i]; }
else if (strcmp(a, "--build-opts") == 0) o.extraOpts = argv[++i];
else if (strcmp(a, "--time") == 0) {
@ -848,9 +852,20 @@ static SizeResult runSize(Device* dv, const DeviceInfo* di, const Options* o, in
* found <job_id> <nonce u64> <hash_hex 16> every nonce whose 64-bit hash is <= target
* done <job_id> <hashes> <ms> end of the job (wall ms)
* error <job_id> <text>
* The pack's program.h is compiled in and kernel_bound.cl is built at runtime, so this worker serves exactly one
* epoch seed and one day seed: the pack's. A job for other seeds is answered with an error naming both; re-export the
* pack with `igneum-miner export-pack <node> <dir>` and rebuild. The init words of a dispatch are
* prepare <epoch_seed_hex 64> <day_seed_hex> <pack_dir> build <pack_dir>/kernel_bound.cl (the pack the miner wrote
* for those seeds) plus its cache and dataset in the background
* stdout: ready opencl <device> ... prepare 1
* prepared <epoch_seed_hex> <day_seed_hex> <ms> ... the pair is resident; a job on it switches instantly
* prepare-failed <epoch_seed_hex> <day_seed_hex> <text>
* The pack's program.h is compiled in and kernel_bound.cl is built at runtime, so at start this worker serves exactly
* one epoch seed and one day seed: the pack's. The next pair arrives through `prepare`: the miner writes the pack for the
* prepared seeds (igneum-miner --prepare-packs <dir>) and names its directory; a background thread builds that pack's
* kernel_bound.cl with the same build options, fills its cache and builds its dataset on a second queue while jobs on
* the current pair keep running (at most two pairs resident: the current one and the prepared one; the old pair is
* released after the first job on the new one). A job for seeds that are neither the current nor the prepared pair is
* answered with an error naming both. Without prepare, re-export the pack with `igneum-miner export-pack <node> <dir>`
* and rebuild. A prepared pair's cache is not cross-checked against a host fill (memhard.h is compiled in for the
* original day); the miner's CPU re-check of every found nonce covers it. The init words of a dispatch are
* seed_words_from_bytes("igneum-block/" || prehash || nonce_hi_le32), written to a small buffer that is the fifth
* argument of igneum_hash_bound; the lane nonce is baseNonce + gid as in the bench kernel. The exchange rule of
* WAVEFRONT.md applies unchanged (the bound kernel has the same body and the same IGNEUM_EXCHANGE build).
@ -881,6 +896,132 @@ static int unhexBuf(const char* s, uint8_t* out, size_t cap, size_t* len) {
return 1;
}
#if IGNEUM_DATASET_MODE == 1
/* One resident (program, cache, dataset) triple for a seed pair. The first one is the compiled-in pack on the main
* queue; prepared ones are built from a pack directory on their own queue. */
typedef struct {
char epochHex[65];
char dayHex[512];
uint32_t sw[8], kw[8]; /* seed words and key words, for job matching */
cl_program prog;
cl_kernel kHashBound, kCacheFill, kBuild;
cl_mem cache, ds;
double buildMs, cacheMs, datasetMs;
} ServePair;
static void releasePair(ServePair* p) {
if (!p) return;
if (p->ds) clReleaseMemObject(p->ds);
if (p->cache) clReleaseMemObject(p->cache);
if (p->kHashBound) clReleaseKernel(p->kHashBound);
if (p->kCacheFill) clReleaseKernel(p->kCacheFill);
if (p->kBuild) clReleaseKernel(p->kBuild);
if (p->prog) clReleaseProgram(p->prog);
free(p);
}
/* The prepare request and its result, handed between the main loop and the prepare thread. */
typedef struct {
Device* dv;
const DeviceInfo* di;
char epochHex[65];
char dayHex[512];
char packDir[1024];
uint32_t words;
char error[512];
ServePair* result; /* set by the thread on success */
volatile int done; /* 1 when the thread has finished (success or failure) */
double t0, doneAt;
} PrepareTask;
static void prepareFail(PrepareTask* t, const char* what, cl_int err) {
snprintf(t->error, sizeof(t->error), "%s (%s)", what, clErrName(err));
}
/* Builds the pair for a prepare request. Runs on its own thread with its own command queue. */
static void prepareRun(PrepareTask* t) {
cl_int err = 0;
size_t srcLen = 0;
char path[1200];
char* src;
ServePair* p = (ServePair*)calloc(1, sizeof(ServePair));
cl_command_queue q = NULL;
double tb;
cl_uint nSeg = IGNEUM_CACHE_SEGMENTS, nItems;
size_t local, bytes = (size_t)CACHE_WORDS_HOST * 4u;
strncpy(p->epochHex, t->epochHex, 64); p->epochHex[64] = 0;
strncpy(p->dayHex, t->dayHex, sizeof(p->dayHex) - 1);
snprintf(path, sizeof(path), "%s/kernel_bound.cl", t->packDir);
src = readFile(path, &srcLen);
if (!src) { snprintf(t->error, sizeof(t->error), "cannot read %s", path); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; }
tb = wallMs();
p->prog = clCreateProgramWithSource(t->dv->ctx, 1, (const char**)&src, &srcLen, &err);
free(src);
if (err != CL_SUCCESS) { prepareFail(t, "clCreateProgramWithSource", err); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; }
err = clBuildProgram(p->prog, 1, &t->di->device, t->dv->buildOptions, NULL, NULL);
if (err != CL_SUCCESS) {
size_t logLen = 0;
char* log;
clGetProgramBuildInfo(p->prog, t->di->device, CL_PROGRAM_BUILD_LOG, 0, NULL, &logLen);
log = (char*)calloc(logLen + 1, 1);
if (logLen) clGetProgramBuildInfo(p->prog, t->di->device, CL_PROGRAM_BUILD_LOG, logLen, log, NULL);
snprintf(t->error, sizeof(t->error), "clBuildProgram failed (%s): %.300s", clErrName(err), log);
free(log); releasePair(p); t->done = 1; return;
}
p->buildMs = wallMs() - tb;
p->kHashBound = clCreateKernel(p->prog, "igneum_hash_bound", &err);
if (err != CL_SUCCESS) { prepareFail(t, "clCreateKernel igneum_hash_bound (is this a kernel_bound.cl?)", err); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; }
p->kCacheFill = clCreateKernel(p->prog, "igneum_cache_fill", &err);
if (err != CL_SUCCESS) { prepareFail(t, "clCreateKernel igneum_cache_fill", err); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; }
p->kBuild = clCreateKernel(p->prog, "igneum_build", &err);
if (err != CL_SUCCESS) { prepareFail(t, "clCreateKernel igneum_build", err); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; }
q = clCreateCommandQueue(t->dv->ctx, t->di->device, 0, &err);
if (err != CL_SUCCESS) { prepareFail(t, "clCreateCommandQueue (prepare)", err); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; }
/* Cache: the same segment count as the compiled-in pack (the dataset schedule is a network constant) */
tb = wallMs();
p->cache = clCreateBuffer(t->dv->ctx, CL_MEM_READ_WRITE, bytes, NULL, &err);
if (err != CL_SUCCESS) { prepareFail(t, "clCreateBuffer cache", err); clReleaseCommandQueue(q); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; }
local = kernelMaxLocal(t->dv, p->kCacheFill, t->di, 256);
{
size_t g = ((nSeg + local - 1) / local) * local;
err = clSetKernelArg(p->kCacheFill, 0, sizeof(cl_mem), &p->cache);
if (err == CL_SUCCESS) err = clSetKernelArg(p->kCacheFill, 1, sizeof(cl_uint), &nSeg);
if (err == CL_SUCCESS) err = clEnqueueNDRangeKernel(q, p->kCacheFill, 1, NULL, &g, &local, 0, NULL, NULL);
if (err == CL_SUCCESS) err = clFinish(q);
}
if (err != CL_SUCCESS) { prepareFail(t, "cache fill", err); clReleaseCommandQueue(q); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; }
p->cacheMs = wallMs() - tb;
/* Dataset */
tb = wallMs();
nItems = t->words / 16u;
p->ds = clCreateBuffer(t->dv->ctx, CL_MEM_READ_WRITE, (size_t)t->words * 4u, NULL, &err);
if (err != CL_SUCCESS) { prepareFail(t, "clCreateBuffer dataset", err); clReleaseCommandQueue(q); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; }
local = kernelMaxLocal(t->dv, p->kBuild, t->di, 256);
{
size_t g = ((nItems + local - 1) / local) * local;
err = clSetKernelArg(p->kBuild, 0, sizeof(cl_mem), &p->ds);
if (err == CL_SUCCESS) err = clSetKernelArg(p->kBuild, 1, sizeof(cl_mem), &p->cache);
if (err == CL_SUCCESS) err = clSetKernelArg(p->kBuild, 2, sizeof(cl_uint), &nItems);
if (err == CL_SUCCESS) err = clEnqueueNDRangeKernel(q, p->kBuild, 1, NULL, &g, &local, 0, NULL, NULL);
if (err == CL_SUCCESS) err = clFinish(q);
}
if (err != CL_SUCCESS) { prepareFail(t, "dataset build", err); clReleaseCommandQueue(q); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; }
p->datasetMs = wallMs() - tb;
clReleaseCommandQueue(q);
t->result = p;
t->doneAt = wallMs();
t->done = 1;
}
#ifdef _WIN32
static DWORD WINAPI prepareThreadMain(LPVOID arg) { prepareRun((PrepareTask*)arg); return 0; }
static int startPrepareThread(PrepareTask* t) { HANDLE h = CreateThread(NULL, 0, prepareThreadMain, t, 0, NULL); if (!h) return 0; CloseHandle(h); return 1; }
#else
static void* prepareThreadMain(void* arg) { prepareRun((PrepareTask*)arg); return NULL; }
static int startPrepareThread(PrepareTask* t) { pthread_t th; if (pthread_create(&th, NULL, prepareThreadMain, t) != 0) return 0; pthread_detach(th); return 1; }
#endif
#endif
static int runServe(Device* dv, const DeviceInfo* di, const Options* o) {
#if IGNEUM_DATASET_MODE != 1
(void)dv; (void)di; (void)o;
@ -893,26 +1034,35 @@ static int runServe(Device* dv, const DeviceInfo* di, const Options* o) {
const uint32_t batch = 1u << (o->batchLog2 == 24 ? 22 : o->batchLog2); /* 2^22 nonces per dispatch by default */
size_t groupSize = 32 * (size_t)o->groupWarps;
cl_int err = 0;
cl_mem dDs, dOut, dInit;
cl_mem dOut, dInit;
cl_uint nItems = words / 16u;
uint64_t* hOut;
char devName[256];
char line[1024];
char line[2048];
size_t k;
ServePair* cur; /* the pair jobs run on */
ServePair* prepared = NULL; /* the pair the last prepare built, until a job switches to it */
ServePair* old = NULL; /* the previous pair, released after the first job on the new one */
PrepareTask* task = NULL; /* the prepare in flight */
if (!dv->kHashBound) { printf("error 0 the kernel source has no igneum_hash_bound (build from the pack's kernel_bound.cl, or pass --kernel)\n"); fflush(stdout); return 2; }
if (!setupCache(dv, di)) { printf("error 0 cache check failed (device cache differs from the host cache or the pack's FNV)\n"); fflush(stdout); return 1; }
dDs = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, (size_t)words * 4u, NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer dataset");
CL_CHECK(clSetKernelArg(dv->kBuild, 0, sizeof(cl_mem), &dDs));
CL_CHECK(clSetKernelArg(dv->kBuild, 1, sizeof(cl_mem), &gCache));
CL_CHECK(clSetKernelArg(dv->kBuild, 2, sizeof(cl_uint), &nItems));
clReleaseEvent(launch1D(dv, dv->kBuild, nItems, kernelMaxLocal(dv, dv->kBuild, di, 256)));
cur = (ServePair*)calloc(1, sizeof(ServePair));
memcpy(cur->sw, SEEDW, 32); memcpy(cur->kw, KEYW, 32);
cur->kHashBound = dv->kHashBound; cur->kCacheFill = dv->kCacheFill; cur->kBuild = dv->kBuild; cur->prog = dv->prog;
dv->kHashBound = dv->kCacheFill = dv->kBuild = NULL; dv->prog = NULL; /* owned by the pair now */
cur->cache = gCache; gCache = NULL;
cur->ds = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, (size_t)words * 4u, NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer dataset");
CL_CHECK(clSetKernelArg(cur->kBuild, 0, sizeof(cl_mem), &cur->ds));
CL_CHECK(clSetKernelArg(cur->kBuild, 1, sizeof(cl_mem), &cur->cache));
CL_CHECK(clSetKernelArg(cur->kBuild, 2, sizeof(cl_uint), &nItems));
clReleaseEvent(launch1D(dv, cur->kBuild, nItems, kernelMaxLocal(dv, cur->kBuild, di, 256)));
CL_CHECK(clFinish(dv->q));
dOut = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, (size_t)batch * sizeof(uint64_t), NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer out");
dInit = clCreateBuffer(dv->ctx, CL_MEM_READ_ONLY, 32, NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer init words");
hOut = (uint64_t*)malloc((size_t)batch * sizeof(uint64_t));
strncpy(devName, di->name, 255); devName[255] = 0;
for (k = 0; devName[k]; ++k) if (devName[k] == ' ') devName[k] = '_';
printf("ready opencl %s platform %s pack %s dataset-log2 %d batch %u exchange %d\n", devName, di->platformName, IGNEUM_SEED_STRING, IGNEUM_DATASET_LOG2, batch, dv->exchange);
printf("ready opencl %s platform %s pack %s dataset-log2 %d batch %u exchange %d prepare %d\n", devName, di->platformName, IGNEUM_SEED_STRING, IGNEUM_DATASET_LOG2, batch, dv->exchange, o->noPrepare ? 0 : 1);
fflush(stdout);
while (fgets(line, sizeof(line), stdin)) {
@ -928,11 +1078,42 @@ static int runServe(Device* dv, const DeviceInfo* di, const Options* o) {
uint64_t remaining, hashes = 0;
uint32_t hi, lo;
double t0;
int failed = 0;
int failed = 0, switched = 0;
line[strcspn(line, "\r\n")] = 0;
/* A finished prepare is reported here, between lines (the thread never prints) */
if (task && task->done) {
if (task->result) {
ServePair* p = task->result;
{
uint8_t eb[32], db[256]; size_t el = 0, dl = 0;
if (unhexBuf(p->epochHex, eb, 32, &el) && el == 32) seedWordsFromBytes(eb, 32, p->sw);
if (unhexBuf(p->dayHex, db, sizeof(db), &dl)) seedWordsFromBytes(db, dl, p->kw);
}
if (prepared) releasePair(prepared);
prepared = p;
printf("prepared %s %s %.1f build %.1f cache %.1f dataset %.1f resident 2 programs 2 datasets\n", p->epochHex, p->dayHex, task->doneAt - task->t0, p->buildMs, p->cacheMs, p->datasetMs);
} else {
printf("prepare-failed %s %s %s\n", task->epochHex, task->dayHex, task->error);
}
fflush(stdout);
free(task); task = NULL;
}
for (tok = strtok_r(line, " ", &save); tok && nf < 9; tok = strtok_r(NULL, " ", &save)) f[nf++] = tok;
if (nf == 0) continue;
if (strcmp(f[0], "quit") == 0) break;
if (strcmp(f[0], "prepare") == 0) {
if (o->noPrepare) { printf("info ignored (started with --no-prepare): prepare\n"); fflush(stdout); continue; }
if (nf < 4) { printf("prepare-failed %s %s this ahead-of-time worker needs a pack directory as the third field (igneum-miner --prepare-packs <dir>)\n", nf > 1 ? f[1] : "0", nf > 2 ? f[2] : "0"); fflush(stdout); continue; }
if (strlen(f[1]) != 64 || strlen(f[2]) >= 500) { printf("prepare-failed %s %s bad field (epoch_seed 64 hex, day_seed hex)\n", f[1], f[2]); fflush(stdout); continue; }
if (task) { printf("prepare-failed %s %s a prepare is still running\n", f[1], f[2]); fflush(stdout); continue; }
if (prepared && strcmp(prepared->epochHex, f[1]) == 0 && strcmp(prepared->dayHex, f[2]) == 0) { printf("prepared %s %s 0 (already resident)\n", f[1], f[2]); fflush(stdout); continue; }
task = (PrepareTask*)calloc(1, sizeof(PrepareTask));
task->dv = dv; task->di = di; task->words = words; task->t0 = wallMs();
strncpy(task->epochHex, f[1], 64); strncpy(task->dayHex, f[2], sizeof(task->dayHex) - 1); strncpy(task->packDir, f[3], sizeof(task->packDir) - 1);
if (!startPrepareThread(task)) { printf("prepare-failed %s %s cannot start the prepare thread\n", f[1], f[2]); fflush(stdout); free(task); task = NULL; continue; }
printf("info prepare started for epoch %.16s day %s from %s (builds in the background)\n", f[1], f[2], f[3]); fflush(stdout);
continue;
}
if (strcmp(f[0], "job") != 0) { printf("info ignored line\n"); fflush(stdout); continue; }
strncpy(jobId, nf > 1 ? f[1] : "0", 63); jobId[63] = 0;
if (nf < 8) { printf("error %s malformed job line (need 7 fields after job)\n", jobId); fflush(stdout); continue; }
@ -944,17 +1125,23 @@ static int runServe(Device* dv, const DeviceInfo* di, const Options* o) {
if (nonceCount == 0 || nonceCount % 32 != 0 || (nonceStart & 31) != 0) { printf("error %s nonce_start must be 32-aligned and nonce_count a non-zero multiple of 32\n", jobId); fflush(stdout); continue; }
seedWordsFromBytes(epochSeed, 32, sw);
seedWordsFromBytes(daySeed, dayLen, kw);
if (memcmp(sw, SEEDW, 32) != 0) {
printf("error %s epoch seed mismatch: this worker was built for pack \"%s\" (seed words %08x %08x ...), the job's epoch seed %.16s gives %08x %08x ...; run igneum-miner export-pack and rebuild\n",
jobId, IGNEUM_SEED_STRING, SEEDW[0], SEEDW[1], f[6], sw[0], sw[1]);
fflush(stdout); continue;
}
if (memcmp(kw, KEYW, 32) != 0) {
printf("error %s day seed mismatch: this worker's cache is for key %08x %08x ..., the job's day seed %s gives %08x %08x ...; run igneum-miner export-pack and rebuild\n",
jobId, KEYW[0], KEYW[1], f[7], kw[0], kw[1]);
fflush(stdout); continue;
}
t0 = wallMs();
if (memcmp(sw, cur->sw, 32) != 0 || memcmp(kw, cur->kw, 32) != 0) {
if (prepared && memcmp(sw, prepared->sw, 32) == 0 && memcmp(kw, prepared->kw, 32) == 0) {
/* The prepared pair: switch now, release the old one after this job */
if (old) releasePair(old);
old = cur; cur = prepared; prepared = NULL; switched = 1;
printf("info switched to the prepared pair epoch %.16s day %s in %.2f ms\n", cur->epochHex, cur->dayHex, wallMs() - t0); fflush(stdout);
} else if (memcmp(sw, cur->sw, 32) != 0) {
printf("error %s epoch seed mismatch: this worker holds %s%s (seed words %08x %08x ...)%s, the job's epoch seed %.16s gives %08x %08x ...; send prepare with a pack directory, or run igneum-miner export-pack and rebuild\n",
jobId, cur->epochHex[0] ? "prepared epoch " : "pack \"" IGNEUM_SEED_STRING "\"", cur->epochHex[0] ? cur->epochHex : "", cur->sw[0], cur->sw[1], prepared ? " plus one prepared pair" : "", f[6], sw[0], sw[1]);
fflush(stdout); continue;
} else {
printf("error %s day seed mismatch: this worker's cache is for key %08x %08x ..., the job's day seed %s gives %08x %08x ...; send prepare with a pack directory, or run igneum-miner export-pack and rebuild\n",
jobId, cur->kw[0], cur->kw[1], f[7], kw[0], kw[1]);
fflush(stdout); continue;
}
}
remaining = nonceCount;
hi = (uint32_t)(nonceStart >> 32); lo = (uint32_t)nonceStart;
while (remaining > 0) {
@ -973,13 +1160,15 @@ static int runServe(Device* dv, const DeviceInfo* di, const Options* o) {
seedWordsFromBytes(b, 49, iw);
CL_CHECK(clEnqueueWriteBuffer(dv->q, dInit, CL_TRUE, 0, 32, iw, 0, NULL, NULL));
baseNonce = lo;
CL_CHECK(clSetKernelArg(dv->kHashBound, 0, sizeof(cl_mem), &dDs));
CL_CHECK(clSetKernelArg(dv->kHashBound, 1, sizeof(cl_mem), &dOut));
CL_CHECK(clSetKernelArg(dv->kHashBound, 2, sizeof(cl_uint), &baseNonce));
CL_CHECK(clSetKernelArg(dv->kHashBound, 3, sizeof(cl_uint), &maskArg));
CL_CHECK(clSetKernelArg(dv->kHashBound, 4, sizeof(cl_mem), &dInit));
ev = launch1D(dv, dv->kHashBound, chunk, groupSize);
err = clFinish(dv->q);
CL_CHECK(clSetKernelArg(cur->kHashBound, 0, sizeof(cl_mem), &cur->ds));
CL_CHECK(clSetKernelArg(cur->kHashBound, 1, sizeof(cl_mem), &dOut));
CL_CHECK(clSetKernelArg(cur->kHashBound, 2, sizeof(cl_uint), &baseNonce));
CL_CHECK(clSetKernelArg(cur->kHashBound, 3, sizeof(cl_uint), &maskArg));
CL_CHECK(clSetKernelArg(cur->kHashBound, 4, sizeof(cl_mem), &dInit));
ev = launch1D(dv, cur->kHashBound, chunk, groupSize);
/* Wait on the dispatch event, not clFinish: the runtime can sleep the thread on an event, where clFinish
* on some drivers spins one core for the whole dispatch. */
err = clWaitForEvents(1, &ev);
clReleaseEvent(ev);
if (err != CL_SUCCESS) { printf("error %s dispatch failed: %s\n", jobId, clErrName(err)); fflush(stdout); failed = 1; break; }
CL_CHECK(clEnqueueReadBuffer(dv->q, dOut, CL_TRUE, 0, (size_t)chunk * sizeof(uint64_t), hOut, 0, NULL, NULL));
@ -994,12 +1183,15 @@ static int runServe(Device* dv, const DeviceInfo* di, const Options* o) {
}
if (failed) continue;
printf("done %s %llu %.2f\n", jobId, (unsigned long long)hashes, wallMs() - t0);
if (switched && old) { releasePair(old); old = NULL; printf("info dropped the previous pair (its program, cache and dataset)\n"); }
fflush(stdout);
}
free(hOut);
clReleaseMemObject(dInit);
clReleaseMemObject(dOut);
clReleaseMemObject(dDs);
if (old) releasePair(old);
if (prepared) releasePair(prepared);
releasePair(cur);
return 0;
#endif
}

File diff suppressed because one or more lines are too long

View file

@ -1,5 +1,5 @@
{
"updated": "2026-10-03",
"updated": "2026-10-04",
"stage": "phase-3",
"phases": [
{
@ -50,6 +50,54 @@
}
],
"log": [
{
"date": "2026-10-04",
"text": "Sim/economy: mining versus proving under stress, agent-based"
},
{
"date": "2026-10-04",
"text": "Difficulty rule under attack: pool hopping, pulsed rental, timestamp stretching, short-lane oscillation, epoch games, polluted window, block flood"
},
{
"date": "2026-10-04",
"text": "Difficulty rule: timestamp attack fixed , simulator regression, 3-node forger test"
},
{
"date": "2026-10-04",
"text": "Devnet-v4 integration: nine branches merged, 3-node test network on the merged node, Windows cross-build"
},
{
"date": "2026-10-03",
"text": "Per-identity hash rate \"decay\" on the RTX 5090: diagnosis and Metal reproduction"
},
{
"date": "2026-10-03",
"text": "Igneum-node devnet v2: sustained-mining finality rule v2 on a four-miner test network, and as a follower of the live devnet"
},
{
"date": "2026-10-03",
"text": "Execution layer devnet v3: revm over the selected chain, 3-node simnet, viem smoke test"
},
{
"date": "2026-10-03",
"text": "Weak-program census: 400,000 program runs through the CPU reference, the redundant-load finding, and the rules for M5 and M6"
},
{
"date": "2026-10-03",
"text": "Proving v0: first SP1 proof of an Igneum block, Apple M5 Max CPU, loaded machine"
},
{
"date": "2026-10-03",
"text": "Windows node package: igneumd cross-compiled for x86_64-pc-windows-gnu, two-peer sync test"
},
{
"date": "2026-10-03",
"text": "R3.26 / M15: PoW checked after the cheap checks, cache-build cap, attack before and after"
},
{
"date": "2026-10-03",
"text": "Difficulty controller: devnet record, simulator, Igneum dual-lane rule, 3-node CPU test network"
},
{
"date": "2026-10-03",
"text": "Devnet started four months ahead of plan. Finality, the EVM layer and the Igneum difficulty controller are in build."