Workers: compile-ahead hot swap in the CUDA, OpenCL and Metal hosts, generator v2 port in the Metal host; windows-miner one worker per card; site bench and journey regenerated
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
parent
afaa2ed97e
commit
c2120542dd
7 changed files with 981 additions and 80 deletions
|
|
@ -134,6 +134,7 @@ struct Options {
|
|||
bool sweep = false;
|
||||
int device = 0;
|
||||
bool serve = false; // --serve: GPU worker for igneum-miner --worker (jobs on stdin), 3 October 2026
|
||||
bool noPrepare = false; // --no-prepare: serve without the prepare command (ready line says "prepare 0"), to test the miner's fallback
|
||||
};
|
||||
|
||||
static int packMib() { return (int)(((1ull << IGNEUM_DATASET_LOG2) * 4ull) >> 20); }
|
||||
|
|
@ -148,7 +149,8 @@ static void usage() {
|
|||
" --block-warps W warps per thread block, 1..32 (default 1 = one warp per block, like the Metal run)\n"
|
||||
" --device D CUDA device index (default 0)\n"
|
||||
" --serve GPU worker for igneum-miner --worker: reads \"job ...\" lines on stdin, prints found/done lines\n"
|
||||
" (needs the pack's kernel_bound.cu compiled in: build.bat adds it when the pack has one)\n", packMib());
|
||||
" (needs the pack's kernel_bound.cu compiled in: build.bat adds it when the pack has one)\n"
|
||||
" --no-prepare with --serve: no prepare support (the miner then falls back to exit 42 at a seed change)\n", packMib());
|
||||
}
|
||||
|
||||
static bool isPow2(long long v) { return v > 0 && (v & (v - 1)) == 0; }
|
||||
|
|
@ -168,6 +170,7 @@ static Options parseArgs(int argc, char** argv) {
|
|||
else if (a == "--device") next(o.device);
|
||||
else if (a == "--sweep") o.sweep = true;
|
||||
else if (a == "--serve") o.serve = true;
|
||||
else if (a == "--no-prepare") o.noPrepare = true;
|
||||
else if (a == "-h" || a == "--help") { usage(); std::exit(0); }
|
||||
else { std::printf("unknown argument %s\n", argv[i]); usage(); std::exit(2); }
|
||||
}
|
||||
|
|
@ -401,8 +404,21 @@ static SizeResult runSize(const Options& o, int mib, uint64_t* dOut, uint32_t no
|
|||
// found <job_id> <nonce u64> <hash_hex 16> every nonce whose 64-bit hash is <= target
|
||||
// done <job_id> <hashes> <ms> end of the job (wall ms)
|
||||
// error <job_id> <text>
|
||||
// The program is compiled ahead of time from the pack (no NVRTC), so this worker serves exactly one epoch seed and
|
||||
// one day seed: the pack's. A job for other seeds is answered with an error naming both; re-export the pack with
|
||||
// prepare <epoch_seed_hex 64> <day_seed_hex> <pack_dir> build <pack_dir>/kernel.cu and kernel_bound.cu (the pack the
|
||||
// miner wrote for those seeds) to cubins with nvcc in the background,
|
||||
// then load them and build that pair's cache and dataset
|
||||
// prepared <epoch_seed_hex> <day_seed_hex> <ms> ... the pair is resident; a job on it switches instantly
|
||||
// prepare-failed <epoch_seed_hex> <day_seed_hex> <text>
|
||||
// The program is compiled ahead of time from the pack (no NVRTC), so at start this worker serves exactly one epoch
|
||||
// seed and one day seed: the pack's. The next pair arrives through `prepare`: the miner writes the pack for the
|
||||
// prepared seeds (igneum-miner --prepare-packs <dir>) and names its directory; a background thread runs nvcc on that
|
||||
// pack's kernel.cu (cache fill and build kernels, memhard.h for its day) and kernel_bound.cu (the bound hash kernel)
|
||||
// to two cubins for this device's architecture, and the main loop loads them through the driver API
|
||||
// (cudaGetDriverEntryPoint, so nothing new is linked), fills the cache and builds the dataset while jobs on the current
|
||||
// pair keep running (at most two pairs resident; the old one is released after the first job on the new one). nvcc
|
||||
// must be on PATH with a host compiler, as build.bat needs it; the ready line says "prepare 1" only when `nvcc --version`
|
||||
// answers. A prepared pair's cache is not cross-checked against the host fill (memhard.h is compiled in for the
|
||||
// original day); the miner's CPU re-check of every found nonce covers it. Without prepare, re-export the pack with
|
||||
// `igneum-miner export-pack <node> <dir>` and rebuild. The init words of a dispatch are
|
||||
// seed_words_from_bytes("igneum-block/" || prehash || nonce_hi_le32), passed by value to igneum_hash_bound
|
||||
// (kernel_bound.cu); the lane nonce is baseNonce + gid as in the bench kernel.
|
||||
|
|
@ -438,6 +454,162 @@ static bool unhexStr(const std::string& s, std::vector<uint8_t>& out) {
|
|||
return true;
|
||||
}
|
||||
|
||||
#if defined(IGNEUM_BOUND) && IGNEUM_DATASET_MODE == 1 && defined(__has_include)
|
||||
#if __has_include(<cuda.h>)
|
||||
#define IGNEUM_CUDA_PREPARE 1
|
||||
#include <cuda.h>
|
||||
#include <thread>
|
||||
#include <atomic>
|
||||
#include <fstream>
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#ifdef IGNEUM_CUDA_PREPARE
|
||||
// The few driver API entry points the hot swap needs, fetched through the runtime so the link line is unchanged.
|
||||
struct DriverApi {
|
||||
CUresult (*moduleLoad)(CUmodule*, const char*) = nullptr;
|
||||
CUresult (*moduleUnload)(CUmodule) = nullptr;
|
||||
CUresult (*moduleGetFunction)(CUfunction*, CUmodule, const char*) = nullptr;
|
||||
CUresult (*moduleGetFunctionCount)(unsigned int*, CUmodule) = nullptr; // CUDA 12.4 and newer
|
||||
CUresult (*moduleEnumerateFunctions)(CUfunction*, unsigned int, CUmodule) = nullptr;
|
||||
CUresult (*funcGetName)(const char**, CUfunction) = nullptr; // CUDA 12.3 and newer
|
||||
CUresult (*launchKernel)(CUfunction, unsigned, unsigned, unsigned, unsigned, unsigned, unsigned, unsigned, CUstream, void**, void**) = nullptr;
|
||||
CUresult (*getErrorString)(CUresult, const char**) = nullptr;
|
||||
bool ok = false;
|
||||
std::string why;
|
||||
template <typename F> bool get(const char* name, F& fn, bool required) {
|
||||
void* p = nullptr;
|
||||
cudaError_t e = cudaGetDriverEntryPoint(name, &p, cudaEnableDefault);
|
||||
if (e != cudaSuccess || !p) { if (required) { why = std::string("no driver entry point ") + name; } return false; }
|
||||
fn = reinterpret_cast<F>(p);
|
||||
return true;
|
||||
}
|
||||
void load() {
|
||||
ok = get("cuModuleLoad", moduleLoad, true) && get("cuModuleUnload", moduleUnload, true) && get("cuModuleGetFunction", moduleGetFunction, true) &&
|
||||
get("cuLaunchKernel", launchKernel, true) && get("cuGetErrorString", getErrorString, true);
|
||||
get("cuModuleGetFunctionCount", moduleGetFunctionCount, false);
|
||||
get("cuModuleEnumerateFunctions", moduleEnumerateFunctions, false);
|
||||
get("cuFuncGetName", funcGetName, false);
|
||||
}
|
||||
std::string err(CUresult r) { const char* s = nullptr; if (getErrorString) getErrorString(r, &s); return s ? s : "CUDA driver error"; }
|
||||
// A kernel by its plain name: the Itanium mangling nvcc gives device code first, then the enumeration (12.4+).
|
||||
bool find(CUmodule m, const char* plain, const char* mangled, CUfunction* out) {
|
||||
if (moduleGetFunction(out, m, mangled) == CUDA_SUCCESS) return true;
|
||||
if (moduleGetFunction(out, m, plain) == CUDA_SUCCESS) return true;
|
||||
if (!moduleGetFunctionCount || !moduleEnumerateFunctions || !funcGetName) return false;
|
||||
unsigned int n = 0;
|
||||
if (moduleGetFunctionCount(&n, m) != CUDA_SUCCESS || n == 0) return false;
|
||||
std::vector<CUfunction> fns(n);
|
||||
if (moduleEnumerateFunctions(fns.data(), n, m) != CUDA_SUCCESS) return false;
|
||||
for (CUfunction f : fns) {
|
||||
const char* name = nullptr;
|
||||
if (funcGetName(&name, f) == CUDA_SUCCESS && name && std::strstr(name, plain)) { *out = f; return true; }
|
||||
}
|
||||
return false;
|
||||
}
|
||||
};
|
||||
|
||||
// One resident pair: the compiled-in pack (runtime launchers, gCache) or a prepared pack (two cubins, driver launches).
|
||||
struct CudaPair {
|
||||
std::string epochHex, dayHex;
|
||||
uint32_t sw[8] = {0}, kw[8] = {0};
|
||||
bool builtIn = false;
|
||||
CUmodule modKernel = nullptr, modBound = nullptr;
|
||||
CUfunction fCacheFill = nullptr, fBuild = nullptr, fHashBound = nullptr;
|
||||
uint32_t* cache = nullptr;
|
||||
uint32_t* ds = nullptr;
|
||||
double nvccMs = 0, cacheMs = 0, dsMs = 0;
|
||||
};
|
||||
|
||||
static void releasePair(DriverApi& drv, CudaPair* p) {
|
||||
if (!p) return;
|
||||
if (p->ds) cudaFree(p->ds);
|
||||
if (p->cache) cudaFree(p->cache);
|
||||
if (p->modBound) drv.moduleUnload(p->modBound);
|
||||
if (p->modKernel) drv.moduleUnload(p->modKernel);
|
||||
delete p;
|
||||
}
|
||||
|
||||
// The nvcc step of a prepare, on its own thread. Only the compiler runs here; every CUDA call stays on the main thread.
|
||||
struct PrepareTask {
|
||||
std::string epochHex, dayHex, packDir, arch, error;
|
||||
std::atomic<bool> done{false};
|
||||
bool ok = false;
|
||||
double t0 = 0, nvccMs = 0;
|
||||
std::thread thread;
|
||||
};
|
||||
|
||||
static bool nvccAvailable() {
|
||||
#ifdef _WIN32
|
||||
return std::system("nvcc --version >NUL 2>&1") == 0;
|
||||
#else
|
||||
return std::system("nvcc --version >/dev/null 2>&1") == 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
static void prepareCompile(PrepareTask* t) {
|
||||
double c0 = wallMs();
|
||||
const char* files[2] = { "kernel", "kernel_bound" };
|
||||
for (const char* f : files) {
|
||||
std::string cmd = "nvcc -cubin -O3 -std=c++17 -arch=" + t->arch + " -allow-unsupported-compiler -I \"" + t->packDir + "\" -o \"" + t->packDir + "/" + f +
|
||||
".cubin\" \"" + t->packDir + "/" + f + ".cu\" > \"" + t->packDir + "/nvcc-" + f + ".log\" 2>&1";
|
||||
int rc = std::system(cmd.c_str());
|
||||
if (rc != 0) {
|
||||
std::string log;
|
||||
std::ifstream in(t->packDir + "/nvcc-" + f + ".log");
|
||||
std::string line;
|
||||
while (std::getline(in, line) && log.size() < 300) { log += line; log += " | "; }
|
||||
t->error = std::string("nvcc failed on ") + f + ".cu (exit " + std::to_string(rc) + "): " + log;
|
||||
t->done = true;
|
||||
return;
|
||||
}
|
||||
}
|
||||
t->nvccMs = wallMs() - c0;
|
||||
t->ok = true;
|
||||
t->done = true;
|
||||
}
|
||||
|
||||
// Loads the two cubins, builds the cache and dataset (main thread). Returns the pair or null with `error` set.
|
||||
static CudaPair* prepareLoad(DriverApi& drv, const PrepareTask& t, uint32_t words, std::string& error) {
|
||||
CudaPair* p = new CudaPair();
|
||||
p->epochHex = t.epochHex; p->dayHex = t.dayHex; p->nvccMs = t.nvccMs;
|
||||
{
|
||||
std::vector<uint8_t> eb, db;
|
||||
if (unhexStr(t.epochHex, eb) && eb.size() == 32) seedWordsFromBytes(eb.data(), 32, p->sw);
|
||||
if (unhexStr(t.dayHex, db)) seedWordsFromBytes(db.data(), db.size(), p->kw);
|
||||
}
|
||||
CUresult r = drv.moduleLoad(&p->modKernel, (t.packDir + "/kernel.cubin").c_str());
|
||||
if (r != CUDA_SUCCESS) { error = "cuModuleLoad kernel.cubin: " + drv.err(r); releasePair(drv, p); return nullptr; }
|
||||
r = drv.moduleLoad(&p->modBound, (t.packDir + "/kernel_bound.cubin").c_str());
|
||||
if (r != CUDA_SUCCESS) { error = "cuModuleLoad kernel_bound.cubin: " + drv.err(r); releasePair(drv, p); return nullptr; }
|
||||
if (!drv.find(p->modKernel, "igneum_cache_fill", "_Z17igneum_cache_fillPjj", &p->fCacheFill)) { error = "igneum_cache_fill not found in kernel.cubin"; releasePair(drv, p); return nullptr; }
|
||||
if (!drv.find(p->modKernel, "igneum_build", "_Z12igneum_buildPjPKjj", &p->fBuild)) { error = "igneum_build not found in kernel.cubin"; releasePair(drv, p); return nullptr; }
|
||||
if (!drv.find(p->modBound, "igneum_hash_bound", "_Z17igneum_hash_boundPKjPyjj15IgneumInitWords", &p->fHashBound)) { error = "igneum_hash_bound not found in kernel_bound.cubin"; releasePair(drv, p); return nullptr; }
|
||||
// Cache (the same segment count as the compiled-in pack: the dataset schedule is a network constant)
|
||||
double c0 = wallMs();
|
||||
size_t cacheBytes = (size_t)CACHE_WORDS_HOST * 4u;
|
||||
if (cudaMalloc((void**)&p->cache, cacheBytes) != cudaSuccess) { error = "cudaMalloc cache"; p->cache = nullptr; releasePair(drv, p); return nullptr; }
|
||||
{
|
||||
uint32_t nSeg = IGNEUM_CACHE_SEGMENTS, block = 256u, grid = (nSeg + block - 1u) / block;
|
||||
void* args[2] = { &p->cache, &nSeg };
|
||||
r = drv.launchKernel(p->fCacheFill, grid, 1, 1, block, 1, 1, 0, nullptr, args, nullptr);
|
||||
if (r != CUDA_SUCCESS || cudaDeviceSynchronize() != cudaSuccess) { error = "cache fill launch: " + drv.err(r); releasePair(drv, p); return nullptr; }
|
||||
}
|
||||
p->cacheMs = wallMs() - c0;
|
||||
// Dataset
|
||||
c0 = wallMs();
|
||||
if (cudaMalloc((void**)&p->ds, (size_t)words * 4u) != cudaSuccess) { error = "cudaMalloc dataset"; p->ds = nullptr; releasePair(drv, p); return nullptr; }
|
||||
{
|
||||
uint32_t nItems = words / 16u, block = 256u, grid = (nItems + block - 1u) / block;
|
||||
void* args[3] = { &p->ds, &p->cache, &nItems };
|
||||
r = drv.launchKernel(p->fBuild, grid, 1, 1, block, 1, 1, 0, nullptr, args, nullptr);
|
||||
if (r != CUDA_SUCCESS || cudaDeviceSynchronize() != cudaSuccess) { error = "dataset build launch: " + drv.err(r); releasePair(drv, p); return nullptr; }
|
||||
}
|
||||
p->dsMs = wallMs() - c0;
|
||||
return p;
|
||||
}
|
||||
#endif
|
||||
|
||||
static int runServe(const Options& o) {
|
||||
#if !defined(IGNEUM_BOUND) || IGNEUM_DATASET_MODE != 1
|
||||
(void)o;
|
||||
|
|
@ -465,7 +637,27 @@ static int runServe(const Options& o) {
|
|||
std::vector<uint64_t> hOut(batch);
|
||||
int regs = 0, bps = 0;
|
||||
igneum_hash_bound_info(®s, &bps, (uint32_t)o.blockWarps);
|
||||
std::printf("ready cuda %s pack %s dataset-log2 %d batch %u regs %d\n", devName.c_str(), IGNEUM_SEED_STRING, IGNEUM_DATASET_LOG2, batch, regs);
|
||||
// Prepare support: the driver entry points and nvcc on PATH
|
||||
int prepareOk = 0;
|
||||
#ifdef IGNEUM_CUDA_PREPARE
|
||||
DriverApi drv;
|
||||
drv.load();
|
||||
std::string arch = "sm_" + std::to_string(prop.major) + std::to_string(prop.minor);
|
||||
bool haveNvcc = !o.noPrepare && nvccAvailable();
|
||||
prepareOk = (!o.noPrepare && drv.ok && haveNvcc) ? 1 : 0;
|
||||
CudaPair* cur = new CudaPair();
|
||||
cur->builtIn = true; cur->cache = gCache; cur->ds = dDs;
|
||||
std::memcpy(cur->sw, SEEDW, 32); std::memcpy(cur->kw, KEYW, 32);
|
||||
CudaPair* prepared = nullptr;
|
||||
CudaPair* old = nullptr;
|
||||
PrepareTask* task = nullptr;
|
||||
#endif
|
||||
std::printf("ready cuda %s pack %s dataset-log2 %d batch %u regs %d prepare %d\n", devName.c_str(), IGNEUM_SEED_STRING, IGNEUM_DATASET_LOG2, batch, regs, prepareOk);
|
||||
#ifdef IGNEUM_CUDA_PREPARE
|
||||
if (!o.noPrepare && !prepareOk) std::printf("info prepare unavailable: %s\n", !drv.ok ? drv.why.c_str() : "nvcc is not on PATH (open the build prompt, or install the CUDA Toolkit)");
|
||||
#else
|
||||
std::printf("info prepare unavailable: this binary was built without cuda.h (CPU emulation or an old toolkit)\n");
|
||||
#endif
|
||||
std::fflush(stdout);
|
||||
|
||||
std::string line;
|
||||
|
|
@ -474,6 +666,41 @@ static int runServe(const Options& o) {
|
|||
std::vector<std::string> f;
|
||||
{ size_t i = 0; while (i < line.size()) { while (i < line.size() && line[i] == ' ') ++i; size_t j = i; while (j < line.size() && line[j] != ' ') ++j; if (j > i) f.push_back(line.substr(i, j - i)); i = j; } }
|
||||
if (f.empty()) continue;
|
||||
#ifdef IGNEUM_CUDA_PREPARE
|
||||
// A finished nvcc step is loaded here, between lines, on this thread
|
||||
if (task && task->done) {
|
||||
task->thread.join();
|
||||
if (task->ok) {
|
||||
std::string error;
|
||||
CudaPair* p = prepareLoad(drv, *task, words, error);
|
||||
if (p) {
|
||||
if (prepared) releasePair(drv, prepared);
|
||||
prepared = p;
|
||||
std::printf("prepared %s %s %.1f nvcc %.1f cache %.1f dataset %.1f resident 2 programs 2 datasets\n", p->epochHex.c_str(), p->dayHex.c_str(), wallMs() - task->t0, p->nvccMs, p->cacheMs, p->dsMs);
|
||||
} else {
|
||||
std::printf("prepare-failed %s %s %s\n", task->epochHex.c_str(), task->dayHex.c_str(), error.c_str());
|
||||
}
|
||||
} else {
|
||||
std::printf("prepare-failed %s %s %s\n", task->epochHex.c_str(), task->dayHex.c_str(), task->error.c_str());
|
||||
}
|
||||
std::fflush(stdout);
|
||||
delete task; task = nullptr;
|
||||
}
|
||||
if (f[0] == "prepare") {
|
||||
if (!prepareOk) { std::printf("info ignored (no prepare support): %s\n", line.c_str()); std::fflush(stdout); continue; }
|
||||
if (f.size() < 4) { std::printf("prepare-failed %s %s this ahead-of-time worker needs a pack directory as the third field (igneum-miner --prepare-packs <dir>)\n", f.size() > 1 ? f[1].c_str() : "0", f.size() > 2 ? f[2].c_str() : "0"); std::fflush(stdout); continue; }
|
||||
if (f[1].size() != 64) { std::printf("prepare-failed %s %s bad field (epoch_seed 64 hex, day_seed hex)\n", f[1].c_str(), f[2].c_str()); std::fflush(stdout); continue; }
|
||||
if (task) { std::printf("prepare-failed %s %s a prepare is still running\n", f[1].c_str(), f[2].c_str()); std::fflush(stdout); continue; }
|
||||
if (prepared && prepared->epochHex == f[1] && prepared->dayHex == f[2]) { std::printf("prepared %s %s 0 (already resident)\n", f[1].c_str(), f[2].c_str()); std::fflush(stdout); continue; }
|
||||
task = new PrepareTask();
|
||||
task->epochHex = f[1]; task->dayHex = f[2]; task->packDir = f[3]; task->arch = arch; task->t0 = wallMs();
|
||||
task->thread = std::thread(prepareCompile, task);
|
||||
std::printf("info prepare started for epoch %.16s day %s from %s (nvcc -arch=%s in the background)\n", f[1].c_str(), f[2].c_str(), f[3].c_str(), arch.c_str()); std::fflush(stdout);
|
||||
continue;
|
||||
}
|
||||
#else
|
||||
if (f[0] == "prepare") { std::printf("info ignored (no prepare support): %s\n", line.c_str()); std::fflush(stdout); continue; }
|
||||
#endif
|
||||
if (f[0] != "job") { std::printf("info ignored: %s\n", line.c_str()); std::fflush(stdout); continue; }
|
||||
std::string jobId = f.size() > 1 ? f[1] : "0";
|
||||
if (f.size() < 8) { std::printf("error %s malformed job line (need 7 fields after job)\n", jobId.c_str()); std::fflush(stdout); continue; }
|
||||
|
|
@ -488,6 +715,26 @@ static int runServe(const Options& o) {
|
|||
uint32_t sw[8], kw[8];
|
||||
seedWordsFromBytes(epochSeed.data(), epochSeed.size(), sw);
|
||||
seedWordsFromBytes(daySeed.data(), daySeed.size(), kw);
|
||||
double t0 = wallMs();
|
||||
#ifdef IGNEUM_CUDA_PREPARE
|
||||
bool switched = false;
|
||||
if (std::memcmp(sw, cur->sw, 32) != 0 || std::memcmp(kw, cur->kw, 32) != 0) {
|
||||
if (prepared && std::memcmp(sw, prepared->sw, 32) == 0 && std::memcmp(kw, prepared->kw, 32) == 0) {
|
||||
if (old) releasePair(drv, old);
|
||||
old = cur; cur = prepared; prepared = nullptr; switched = true;
|
||||
std::printf("info switched to the prepared pair epoch %.16s day %s in %.2f ms\n", cur->epochHex.c_str(), cur->dayHex.c_str(), wallMs() - t0); std::fflush(stdout);
|
||||
} else if (std::memcmp(sw, cur->sw, 32) != 0) {
|
||||
std::printf("error %s epoch seed mismatch: this worker holds %s%s (seed words %08x %08x ...)%s, the job's epoch seed %s gives %08x %08x ...; send prepare with a pack directory, or run igneum-miner export-pack and rebuild\n",
|
||||
jobId.c_str(), cur->builtIn ? "pack \"" IGNEUM_SEED_STRING "\"" : "prepared epoch ", cur->builtIn ? "" : cur->epochHex.c_str(), cur->sw[0], cur->sw[1], prepared ? " plus one prepared pair" : "", f[6].substr(0, 16).c_str(), sw[0], sw[1]);
|
||||
std::fflush(stdout); continue;
|
||||
} else {
|
||||
std::printf("error %s day seed mismatch: this worker's cache is for key %08x %08x ..., the job's day seed %s gives %08x %08x ...; send prepare with a pack directory, or run igneum-miner export-pack and rebuild\n",
|
||||
jobId.c_str(), cur->kw[0], cur->kw[1], f[7].c_str(), kw[0], kw[1]);
|
||||
std::fflush(stdout); continue;
|
||||
}
|
||||
}
|
||||
const uint32_t* jobDs = cur->ds;
|
||||
#else
|
||||
if (std::memcmp(sw, SEEDW, 32) != 0) {
|
||||
std::printf("error %s epoch seed mismatch: this worker was built for pack \"%s\" (seed words %08x %08x ...), the job's epoch seed %s gives %08x %08x ...; run igneum-miner export-pack and rebuild\n",
|
||||
jobId.c_str(), IGNEUM_SEED_STRING, SEEDW[0], SEEDW[1], f[6].substr(0, 16).c_str(), sw[0], sw[1]);
|
||||
|
|
@ -498,7 +745,8 @@ static int runServe(const Options& o) {
|
|||
jobId.c_str(), KEYW[0], KEYW[1], f[7].c_str(), kw[0], kw[1]);
|
||||
std::fflush(stdout); continue;
|
||||
}
|
||||
double t0 = wallMs();
|
||||
const uint32_t* jobDs = dDs;
|
||||
#endif
|
||||
uint64_t remaining = nonceCount, hashes = 0;
|
||||
uint32_t hi = (uint32_t)(nonceStart >> 32), lo = (uint32_t)nonceStart;
|
||||
bool failed = false;
|
||||
|
|
@ -515,7 +763,18 @@ static int runServe(const Options& o) {
|
|||
b[45] = (uint8_t)hi; b[46] = (uint8_t)(hi >> 8); b[47] = (uint8_t)(hi >> 16); b[48] = (uint8_t)(hi >> 24);
|
||||
seedWordsFromBytes(b, 49, iw.w);
|
||||
}
|
||||
cudaError_t e = igneum_launch_hash_bound(dDs, dOut, lo, mask, iw, chunk, (uint32_t)o.blockWarps);
|
||||
cudaError_t e = cudaSuccess;
|
||||
#ifdef IGNEUM_CUDA_PREPARE
|
||||
if (!cur->builtIn) {
|
||||
// A prepared pair: the same launch shape as igneum_launch_hash_bound, through the driver API
|
||||
uint32_t block = 32u * (uint32_t)o.blockWarps;
|
||||
uint32_t baseNonce = lo, maskArg = mask;
|
||||
void* args[5] = { (void*)&jobDs, (void*)&dOut, &baseNonce, &maskArg, &iw };
|
||||
CUresult r = drv.launchKernel(cur->fHashBound, chunk / block, 1, 1, block, 1, 1, 0, nullptr, args, nullptr);
|
||||
if (r != CUDA_SUCCESS) { std::printf("error %s dispatch failed: %s\n", jobId.c_str(), drv.err(r).c_str()); std::fflush(stdout); failed = true; break; }
|
||||
} else
|
||||
#endif
|
||||
e = igneum_launch_hash_bound(jobDs, dOut, lo, mask, iw, chunk, (uint32_t)o.blockWarps);
|
||||
if (e == cudaSuccess) e = cudaDeviceSynchronize();
|
||||
if (e == cudaSuccess) e = cudaMemcpy(hOut.data(), dOut, (size_t)chunk * sizeof(uint64_t), cudaMemcpyDeviceToHost);
|
||||
if (e != cudaSuccess) { std::printf("error %s dispatch failed: %s\n", jobId.c_str(), cudaGetErrorString(e)); std::fflush(stdout); failed = true; break; }
|
||||
|
|
@ -530,11 +789,21 @@ static int runServe(const Options& o) {
|
|||
}
|
||||
if (failed) continue;
|
||||
std::printf("done %s %llu %.2f\n", jobId.c_str(), (unsigned long long)hashes, wallMs() - t0);
|
||||
#ifdef IGNEUM_CUDA_PREPARE
|
||||
if (switched && old) { releasePair(drv, old); old = nullptr; std::printf("info dropped the previous pair (its program, cache and dataset)\n"); }
|
||||
#endif
|
||||
std::fflush(stdout);
|
||||
}
|
||||
cudaFree(dOut);
|
||||
#ifdef IGNEUM_CUDA_PREPARE
|
||||
if (task) { task->thread.join(); delete task; }
|
||||
if (old) releasePair(drv, old);
|
||||
if (prepared) releasePair(drv, prepared);
|
||||
releasePair(drv, cur); // frees gCache and dDs when the built-in pair is still current
|
||||
#else
|
||||
cudaFree(dDs);
|
||||
cudaFree(gCache);
|
||||
#endif
|
||||
return 0;
|
||||
#endif
|
||||
}
|
||||
|
|
@ -551,6 +820,12 @@ int main(int argc, char** argv) {
|
|||
if (count == 0) { std::printf("FAIL: no CUDA device\n"); return 2; }
|
||||
if (o.device < 0 || o.device >= count) { std::printf("FAIL: device %d out of range (%d devices)\n", o.device, count); return 2; }
|
||||
CUDA_CHECK(cudaSetDevice(o.device));
|
||||
// Blocking sync, set before the context exists: the host thread sleeps in cudaDeviceSynchronize instead of
|
||||
// spinning (one full core per worker process at the default spin schedule, measured on the RTX 5090 with
|
||||
// eight workers, 3 Oct 2026). The microseconds of wake-up latency are nothing against a 100 ms dispatch.
|
||||
#ifdef cudaDeviceScheduleBlockingSync
|
||||
CUDA_CHECK(cudaSetDeviceFlags(cudaDeviceScheduleBlockingSync));
|
||||
#endif
|
||||
if (o.serve) {
|
||||
if (o.batchLog2 == 24) { Options s2 = o; s2.batchLog2 = 22; return runServe(s2); } // 2^22 nonces per dispatch by default
|
||||
return runServe(o);
|
||||
|
|
|
|||
|
|
@ -5,7 +5,8 @@ rem one igneum-miner per worker against the Mac node. Logs land next to this fil
|
|||
rem The window shows a dashboard: hash rate, blocks found and one line per miner. Stop with Ctrl+C; the window stays open at the end.
|
||||
|
||||
rem ---- settings: edit these lines only ----------------------------------------------------------
|
||||
set "NODE_HOST=192.168.68.64"
|
||||
rem auto = a node on this PC (START-NODE.bat, 127.0.0.1) when one is running, else the Mac node at 192.168.68.64.
|
||||
set "NODE_HOST=auto"
|
||||
set "NODE_PORT=26610"
|
||||
rem Devnet payout address (igneumdev:...). Leave empty for the miner's fixed test address.
|
||||
set "PAYOUT_ADDRESS="
|
||||
|
|
|
|||
|
|
@ -70,6 +70,9 @@ $vcvars = if ($env:VCVARS) { $env:VCVARS } else { 'C:\Program Files\Microsoft Vi
|
|||
$vcvarsVer = if ($env:VCVARS_VER) { $env:VCVARS_VER } else { '14.30' }
|
||||
$minersPerVendor = 1
|
||||
if ($env:MINERS -and [int]$env:MINERS -ge 1) { $minersPerVendor = [int]$env:MINERS }
|
||||
# One worker process per card (3 Oct 2026): the MINERS identities run through ONE igneum-miner with --identities N and one
|
||||
# worker (one context, one cache and dataset, one host thread). ONE_WORKER_PER_CARD=0 restores a worker per identity.
|
||||
$oneWorkerPerCard = -not ($env:ONE_WORKER_PER_CARD -eq '0')
|
||||
# Nonces per job, per vendor (a multiple of 32). Empty = the miner's default of 16,777,216. A job is one template, so a
|
||||
# job should take well under a minute: a slow GPU with a long job mines a stale template and reports late.
|
||||
$nvidiaJobNonces = $env:NVIDIA_JOB_NONCES
|
||||
|
|
@ -321,8 +324,9 @@ foreach ($name in $plan) {
|
|||
$exe = Build-Worker $name $seeds
|
||||
if (-not $exe) { Log "$name skipped: no worker binary"; continue }
|
||||
$instances = @()
|
||||
for ($i = 1; $i -le $minersPerVendor; $i++) {
|
||||
$suffix = if ($minersPerVendor -gt 1) { "-$i" } else { '' }
|
||||
$processes = if ($oneWorkerPerCard) { 1 } else { $minersPerVendor }
|
||||
for ($i = 1; $i -le $processes; $i++) {
|
||||
$suffix = if ($processes -gt 1) { "-$i" } else { '' }
|
||||
$instances += @{ vendor = $name; index = $i; label = "$name-$machine$suffix"; file = "$name$suffix-$stamp"; log = $null; err = $null;
|
||||
runId = "$name-$machine$suffix-$stamp"; proc = $null; restarts = 0; starts = 0; startedAt = $null; restartAt = $null; exitCode = $null;
|
||||
logPos = [long]0; logRem = ''; errPos = [long]0; errRem = ''; accepted = 0; found = (New-Object System.Collections.ArrayList);
|
||||
|
|
@ -332,7 +336,8 @@ foreach ($name in $plan) {
|
|||
$vendors[$name] = @{ name = $name; card = (Get-CardName $name); exe = $exe; instances = $instances; rebuilds = 0; building = $false }
|
||||
}
|
||||
if ($vendors.Count -eq 0) { Log 'no worker could be built; see the launcher log'; exit 1 }
|
||||
Log ("identities per vendor: $minersPerVendor (each identity runs its own worker: about 1.3 GiB of GPU memory per instance)")
|
||||
if ($oneWorkerPerCard) { Log ("identities per vendor: $minersPerVendor through ONE worker process per card (labels <vendor>-<PC>-1..$minersPerVendor; about 1.3 GiB of GPU memory per card)") }
|
||||
else { Log ("identities per vendor: $minersPerVendor (each identity runs its own worker: about 1.3 GiB of GPU memory per instance)") }
|
||||
Log ("job nonces: nvidia " + $(if ($nvidiaJobNonces) { $nvidiaJobNonces } else { 'miner default (16,777,216)' }) + ", amd $amdJobNonces")
|
||||
|
||||
function Start-Miner($v, $inst) {
|
||||
|
|
@ -349,6 +354,7 @@ function Start-Miner($v, $inst) {
|
|||
# --exit-on-seed-change is the fallback only: a worker that answers "prepare 1" on its ready line is never exited at a
|
||||
# seed change; the miner writes the next pack under --prepare-packs and the worker builds it in the background.
|
||||
$margs = @('mine', $nodeUrl, '1', '100000000', $inst.label, '--worker', ('"' + $v.exe + '"'), '--status-secs', '30', '--exit-on-seed-change', '--prepare-packs', ('"' + $preparedPacks + '"'))
|
||||
if ($oneWorkerPerCard -and $minersPerVendor -gt 1) { $margs += @('--identities', $minersPerVendor) }
|
||||
if ($payout) { $margs += @('--address', $payout) } else { $margs += @('--payout-label', $inst.label) }
|
||||
if ($v.name -eq 'nvidia') {
|
||||
if ($nvidiaJobNonces) { $margs += @('--job-nonces', $nvidiaJobNonces) }
|
||||
|
|
|
|||
|
|
@ -18,6 +18,7 @@ struct Options {
|
|||
var dumpDir: String? = nil
|
||||
var exportPack: String? = nil // write a CUDA program pack for --seed into this directory and exit
|
||||
var serve = false // --serve: GPU worker for igneum-miner (jobs on stdin, results on stdout), 3 October 2026
|
||||
var noPrepare = false // --no-prepare: serve without the prepare command (ready line says "prepare 0"), to test the miner's fallback
|
||||
// Hardening tests (added 3 October 2026). Any of these runs instead of the bench.
|
||||
var fuzz: Int? = nil // --fuzz N: N random programs, GPU vs CPU on 4 random warps each
|
||||
var fuzzSeed = "igneum-fuzz-2026-10-03"
|
||||
|
|
@ -54,6 +55,7 @@ func parseArgs() -> Options {
|
|||
case "--dump": o.dumpDir = take()
|
||||
case "--export-pack": o.exportPack = take()
|
||||
case "--serve": o.serve = true
|
||||
case "--no-prepare": o.noPrepare = true
|
||||
case "--fuzz": o.fuzz = Int(take()) ?? 200
|
||||
case "--fuzz-seed": o.fuzzSeed = take()
|
||||
case "--edge": o.edge = true
|
||||
|
|
@ -356,6 +358,8 @@ struct Program {
|
|||
let seedString: String
|
||||
let seed: [UInt32]
|
||||
let instrs: [Instr]
|
||||
var generator = 1 // 2 for every current program (generateProgramV2); 1 for the retired lever generator
|
||||
var attempt: UInt32 = 0 // attempt index under the acceptance rule (0 = the bare seed)
|
||||
static let iterations = 8
|
||||
static let count = 64
|
||||
var loadsPerHash: Int { instrs.filter { $0.op == .load || $0.op == .wload }.count * Program.iterations }
|
||||
|
|
@ -397,10 +401,18 @@ struct GeneratorConfig {
|
|||
}
|
||||
var generatorConfig = GeneratorConfig()
|
||||
|
||||
func generateProgram(seedString: String) -> Program { generateProgram(seedString: seedString, words: seedWords(seedString)) }
|
||||
// The program of a seed string: generator version 2 with the acceptance rule (below), the same program the Rust crate
|
||||
// derives. The retired version 1 generator is used only when a lever (--load-weight, --wide-frac) is set, for the
|
||||
// MEMHARD.md section 2.4 measurements; those programs are not the lottery hash.
|
||||
func generateProgram(seedString: String) -> Program {
|
||||
if generatorConfig.loadWeight != 25 || generatorConfig.wideFrac != 0 {
|
||||
return generateProgramV1(seedString: seedString, words: seedWords(seedString))
|
||||
}
|
||||
return generateProgramV2(seedString: seedString, bytes: Array(seedString.utf8))
|
||||
}
|
||||
|
||||
// The generator from already-derived seed words (what the chain feeds: the epoch block hash on devnet v0, the VDF output later).
|
||||
func generateProgram(seedString: String, words sw: [UInt32]) -> Program {
|
||||
// Version 1 (retired 4 October 2026): op rolled per instruction against the 11-family table, load count free.
|
||||
func generateProgramV1(seedString: String, words sw: [UInt32]) -> Program {
|
||||
var rng = SplitMix64(s: (UInt64(sw[0]) | (UInt64(sw[1]) << 32)) ^ ((UInt64(sw[2]) | (UInt64(sw[3]) << 32)) &* 0x9E3779B97F4A7C15))
|
||||
var instrs = [Instr]()
|
||||
let weights = generatorConfig.weights
|
||||
|
|
@ -423,6 +435,232 @@ func generateProgram(seedString: String, words sw: [UInt32]) -> Program {
|
|||
return Program(seedString: seedString, seed: sw, instrs: instrs)
|
||||
}
|
||||
|
||||
// MARK: - Generator version 2 and the acceptance rule (4 October 2026)
|
||||
//
|
||||
// Draw for draw the Rust generator (igneum-pow/src/generator.rs, candidate_from_words) and acceptance rule
|
||||
// (igneum-pow/src/accept.rs), spec 01 sections 1.4.3 and 1.4.6. Exactly 16 load slots drawn first from instructions
|
||||
// 1..63; a load's source is drawn from the registers other than dst written by an earlier instruction and not read by
|
||||
// a load since; a candidate that fails the rule is replaced by attempt k + 1, seedWordsBytes(seed || k_le32).
|
||||
|
||||
let generatorVersion = 2
|
||||
let loadSlots = 16
|
||||
let maxAttempts: UInt32 = 32
|
||||
let nonloadWeights: [(Op, Int)] = [(.add, 12), (.xor, 10), (.mul, 8), (.mad, 8), (.shfl, 8),
|
||||
(.rotl, 7), (.sub, 6), (.mulhi, 6), (.rotr, 6), (.or, 4)]
|
||||
let acceptUnits = 64
|
||||
let acceptDatasetLog2 = 28
|
||||
let acceptMaxSaturated: UInt32 = 164
|
||||
let acceptBiasTolerance: UInt32 = 136
|
||||
let acceptMinDistinctSum: UInt64 = 245_760
|
||||
|
||||
func fnv1a64Bytes(_ bytes: [UInt8]) -> UInt64 {
|
||||
var h: UInt64 = 0xcbf29ce484222325
|
||||
for b in bytes { h ^= UInt64(b); h &*= 0x100000001b3 }
|
||||
return h
|
||||
}
|
||||
|
||||
func le32(_ v: UInt32) -> [UInt8] { [UInt8(v & 0xff), UInt8((v >> 8) & 0xff), UInt8((v >> 16) & 0xff), UInt8((v >> 24) & 0xff)] }
|
||||
|
||||
// The seed words of attempt k: seedWordsBytes(seed) for k = 0, seedWordsBytes(seed || k_le32) otherwise.
|
||||
func attemptWords(_ seedBytes: [UInt8], _ attempt: UInt32) -> [UInt32] {
|
||||
attempt == 0 ? seedWordsBytes(seedBytes) : seedWordsBytes(seedBytes + le32(attempt))
|
||||
}
|
||||
|
||||
// FNV-1a 64 over "igneum-program/" || generator_le32 || seed words LE || attempt_le32 (Program::program_id in Rust).
|
||||
func programId(_ p: Program) -> UInt64 {
|
||||
var b = Array("igneum-program/".utf8) + le32(UInt32(p.generator))
|
||||
for w in p.seed { b += le32(w) }
|
||||
b += le32(p.attempt)
|
||||
return fnv1a64Bytes(b)
|
||||
}
|
||||
|
||||
// One version 2 candidate from its seed words, before the acceptance rule.
|
||||
func candidateProgram(seedString: String, words sw: [UInt32], attempt: UInt32) -> Program {
|
||||
var rng = SplitMix64(s: (UInt64(sw[0]) | (UInt64(sw[1]) << 32)) ^ ((UInt64(sw[2]) | (UInt64(sw[3]) << 32)) &* 0x9E3779B97F4A7C15))
|
||||
var slots = Array(1..<Program.count)
|
||||
for i in 0..<loadSlots {
|
||||
let j = i + rng.below(Program.count - 1 - i)
|
||||
slots.swapAt(i, j)
|
||||
}
|
||||
var isLoad = [Bool](repeating: false, count: Program.count)
|
||||
for i in 0..<loadSlots { isLoad[slots[i]] = true }
|
||||
var fresh = [Bool](repeating: false, count: 8)
|
||||
var instrs = [Instr]()
|
||||
for k in 0..<Program.count {
|
||||
var roll = rng.below(75)
|
||||
var op = Op.add
|
||||
for (o, w) in nonloadWeights { if roll < w { op = o; break }; roll -= w }
|
||||
if isLoad[k] { op = .load }
|
||||
let dst = rng.below(8)
|
||||
var a: Int
|
||||
if op == .load {
|
||||
let eligible = (0..<8).filter { $0 != dst && fresh[$0] }
|
||||
if eligible.isEmpty {
|
||||
a = rng.below(7); if a >= dst { a += 1 }
|
||||
} else {
|
||||
a = eligible[rng.below(eligible.count)]
|
||||
}
|
||||
} else {
|
||||
a = rng.below(7); if a >= dst { a += 1 }
|
||||
}
|
||||
let b = rng.below(8)
|
||||
let imm = UInt32(truncatingIfNeeded: rng.next())
|
||||
let imm2 = UInt32(truncatingIfNeeded: rng.next())
|
||||
let rot = UInt32(1 + rng.below(31))
|
||||
let bit = rng.below(32)
|
||||
let mask = 1 << rng.below(5)
|
||||
if op == .load { fresh[a] = false }
|
||||
fresh[dst] = true
|
||||
instrs.append(Instr(op: op, dst: dst, a: a, b: b, imm: imm, imm2: imm2, rot: rot, bit: bit, mask: mask))
|
||||
}
|
||||
return Program(seedString: seedString, seed: sw, instrs: instrs, generator: generatorVersion, attempt: attempt)
|
||||
}
|
||||
|
||||
@inline(__always) func opInjects(_ op: Op) -> Bool {
|
||||
switch op { case .add, .sub, .xor, .mad, .shfl, .load, .wload: return true; default: return false }
|
||||
}
|
||||
|
||||
// Parts (a) and (b) of the rule. Returns nil when the program passes, else the reason.
|
||||
func acceptStatic(_ p: Program) -> String? {
|
||||
var pending = [Bool](repeating: false, count: 8)
|
||||
for _ in 0..<2 {
|
||||
for (k, ins) in p.instrs.enumerated() {
|
||||
let isLoad = ins.op == .load || ins.op == .wload
|
||||
if isLoad && pending[ins.a] { return "(a) load at instruction \(k) reads r\(ins.a), unwritten since the previous load from it" }
|
||||
pending[ins.dst] = false
|
||||
if isLoad { pending[ins.a] = true }
|
||||
}
|
||||
}
|
||||
var injected = [Bool](repeating: false, count: 8)
|
||||
for ins in p.instrs where opInjects(ins.op) { injected[ins.dst] = true }
|
||||
for r in 0..<8 where !injected[r] { return "(b) r\(r) has no add, sub, xor, mad, shfl or load write" }
|
||||
return nil
|
||||
}
|
||||
|
||||
// The 64 base nonces of the dynamic test: SplitMix64 seeded with FNV-1a 64("igneum-accept/" || seed words LE).
|
||||
func acceptBaseNonces(_ seed: [UInt32]) -> [UInt32] {
|
||||
var b = Array("igneum-accept/".utf8)
|
||||
for w in seed { b += le32(w) }
|
||||
var rng = SplitMix64(s: fnv1a64Bytes(b))
|
||||
return (0..<acceptUnits).map { _ in UInt32(truncatingIfNeeded: rng.next()) & ~31 }
|
||||
}
|
||||
|
||||
// Part (c): 64 units on the closed-form dataset keyed by the seed words, init words = seed words. Returns nil
|
||||
// when the program passes, else the reason. Mirrors run_unit and check_dynamic in igneum-pow/src/accept.rs.
|
||||
func acceptDynamic(_ p: Program) -> String? {
|
||||
let lanes = 32
|
||||
let loads = p.loadsPerHash
|
||||
let mask: UInt32 = (1 << UInt32(acceptDatasetLog2)) - 1
|
||||
let (d0, d1) = (p.seed[0], p.seed[1])
|
||||
var andAcc = [UInt32](repeating: 0xffffffff, count: 8)
|
||||
var orAcc = [UInt32](repeating: 0, count: 8)
|
||||
var saturated: UInt32 = 0
|
||||
var bitOnes = [UInt32](repeating: 0, count: 64)
|
||||
var distinctSum: UInt64 = 0
|
||||
var r = [UInt32](repeating: 0, count: 8 * lanes) // r[reg * 32 + lane]
|
||||
var laneAddrs = [UInt32](repeating: 0, count: lanes * loads)
|
||||
var sel = [UInt32](repeating: 0, count: lanes)
|
||||
var idx = [UInt32](repeating: 0, count: lanes)
|
||||
var tmp = [UInt32](repeating: 0, count: lanes)
|
||||
for (unit, base) in acceptBaseNonces(p.seed).enumerated() {
|
||||
for lane in 0..<lanes {
|
||||
let nonce = base &+ UInt32(lane)
|
||||
for i in 0..<8 {
|
||||
var x = nonce ^ p.seed[i]
|
||||
x &+= 0x9e3779b9 &* UInt32(i + 1)
|
||||
x = splitmix32(x)
|
||||
r[i * lanes + lane] = x ^ p.seed[(i + 1) & 7]
|
||||
}
|
||||
}
|
||||
var nload = 0
|
||||
for it in 0..<Program.iterations {
|
||||
for lane in 0..<lanes { sel[lane] = r[lane] }
|
||||
for (k, ins) in p.instrs.enumerated() {
|
||||
let d = ins.dst * lanes, a = ins.a * lanes
|
||||
switch ins.op {
|
||||
case .add:
|
||||
for lane in 0..<lanes {
|
||||
let s = (sel[lane] >> UInt32(ins.bit)) & 1
|
||||
r[d + lane] = r[d + lane] &+ r[a + lane] &+ (s != 0 ? ins.imm2 : ins.imm)
|
||||
}
|
||||
case .sub: for lane in 0..<lanes { r[d + lane] = r[d + lane] &- r[a + lane] }
|
||||
case .mul: for lane in 0..<lanes { r[d + lane] = r[d + lane] &* r[a + lane] }
|
||||
case .mulhi: for lane in 0..<lanes { r[d + lane] = mulhi32(r[d + lane], r[a + lane]) }
|
||||
case .xor: for lane in 0..<lanes { r[d + lane] ^= r[a + lane] }
|
||||
case .or: for lane in 0..<lanes { r[d + lane] |= r[a + lane] }
|
||||
case .rotl: for lane in 0..<lanes { r[d + lane] = rotl32(r[d + lane], ins.rot) }
|
||||
case .rotr: for lane in 0..<lanes { r[d + lane] = rotr32(r[d + lane], r[a + lane]) }
|
||||
case .mad:
|
||||
let b = ins.b * lanes
|
||||
for lane in 0..<lanes { r[d + lane] = (r[a + lane] &* r[b + lane]) &+ r[d + lane] }
|
||||
case .shfl:
|
||||
for lane in 0..<lanes { tmp[lane] = r[a + lane] }
|
||||
for lane in 0..<lanes { r[d + lane] ^= tmp[lane ^ ins.mask] }
|
||||
case .load:
|
||||
for lane in 0..<lanes { idx[lane] = r[a + lane] & mask }
|
||||
var same = true
|
||||
for lane in 1..<lanes where idx[lane] != idx[0] { same = false; break }
|
||||
if same { return "(c) load at iteration \(it) instruction \(k) reads one address in all lanes of unit \(unit)" }
|
||||
for lane in 0..<lanes {
|
||||
r[d + lane] ^= datasetElem(idx[lane], d0, d1)
|
||||
laneAddrs[lane * loads + nload] = idx[lane]
|
||||
}
|
||||
nload += 1
|
||||
case .wload:
|
||||
let b = (r[a] & mask) & ~31
|
||||
for lane in 0..<lanes {
|
||||
idx[lane] = b + UInt32(lane)
|
||||
r[d + lane] ^= datasetElem(idx[lane], d0, d1)
|
||||
laneAddrs[lane * loads + nload] = idx[lane]
|
||||
}
|
||||
nload += 1
|
||||
}
|
||||
}
|
||||
}
|
||||
for i in 0..<8 {
|
||||
for lane in 0..<lanes {
|
||||
let v = r[i * lanes + lane]
|
||||
andAcc[i] &= v; orAcc[i] |= v
|
||||
if v == 0 || v == 0xffffffff { saturated += 1 }
|
||||
}
|
||||
}
|
||||
for lane in 0..<lanes {
|
||||
let lo = r[lane] ^ rotl32(r[lanes + lane], 7) ^ rotl32(r[2 * lanes + lane], 14) ^ rotl32(r[3 * lanes + lane], 21)
|
||||
let hi = r[4 * lanes + lane] ^ rotl32(r[5 * lanes + lane], 9) ^ rotl32(r[6 * lanes + lane], 18) ^ rotl32(r[7 * lanes + lane], 27)
|
||||
let h = (UInt64(hi) << 32) | UInt64(lo)
|
||||
for j in 0..<64 { bitOnes[j] += UInt32((h >> UInt64(j)) & 1) }
|
||||
var sl = Array(laneAddrs[lane * loads..<(lane + 1) * loads])
|
||||
sl.sort()
|
||||
var distinct: UInt64 = 0
|
||||
for k in 0..<loads where k == 0 || sl[k] != sl[k - 1] { distinct += 1 }
|
||||
distinctSum += distinct
|
||||
}
|
||||
}
|
||||
for i in 0..<8 {
|
||||
let bits = (andAcc[i] | ~orAcc[i]).nonzeroBitCount
|
||||
if bits != 0 { return "(c) r\(i) has \(bits) nonce-independent bits" }
|
||||
}
|
||||
if saturated >= acceptMaxSaturated { return "(c) \(saturated) of 16384 final register values saturated (limit 163)" }
|
||||
let half = UInt32(acceptUnits * lanes / 2)
|
||||
for j in 0..<64 {
|
||||
let d = bitOnes[j] > half ? bitOnes[j] - half : half - bitOnes[j]
|
||||
if d > acceptBiasTolerance { return "(c) output bit \(j) set in \(bitOnes[j]) of 2048 hashes" }
|
||||
}
|
||||
if distinctSum <= acceptMinDistinctSum { return "(c) distinct addresses \(distinctSum) over 2048 hashes (needs above 245760)" }
|
||||
return nil
|
||||
}
|
||||
|
||||
func acceptProgram(_ p: Program) -> String? { acceptStatic(p) ?? acceptDynamic(p) }
|
||||
|
||||
// The program of a seed under version 2: the first accepted candidate over attempts 0, 1, 2, ...
|
||||
func generateProgramV2(seedString: String, bytes: [UInt8]) -> Program {
|
||||
for attempt in 0..<maxAttempts {
|
||||
let p = candidateProgram(seedString: seedString, words: attemptWords(bytes, attempt), attempt: attempt)
|
||||
if acceptProgram(p) == nil { return p }
|
||||
}
|
||||
fatalError("seed \(seedString): \(maxAttempts) consecutive candidates rejected (consensus fault)")
|
||||
}
|
||||
|
||||
// MARK: - MSL generation
|
||||
|
||||
func hex(_ v: UInt32) -> String { String(format: "0x%08xu", v) }
|
||||
|
|
@ -2452,19 +2690,30 @@ func runTests(_ opts: Options) -> Never {
|
|||
|
||||
// MARK: - Serve mode (GPU worker for igneum-miner --worker), 3 October 2026
|
||||
//
|
||||
// Protocol, one line each. Only "found", "done" and "error" are parsed by the miner; every other line is logged.
|
||||
// Protocol, one line each. Only "found", "done", "error", "prepared", "prepare-failed" and "ready" are parsed by the
|
||||
// miner; every other line is logged.
|
||||
// stdin: job <job_id> <header_prehash_hex 64> <target_hex 16> <nonce_start u64> <nonce_count u64> <epoch_seed_hex 64> <day_seed_hex>
|
||||
// prepare <epoch_seed_hex 64> <day_seed_hex> [<pack_dir>] compile that program and build that day's dataset in the
|
||||
// background while jobs on the current seeds keep running
|
||||
// quit
|
||||
// stdout: ready metal <device>
|
||||
// stdout: ready metal <device> dataset-log2 <n> batch <n> prepare 1
|
||||
// found <job_id> <nonce u64> <hash_hex 16> every nonce whose 64-bit hash is <= target (hash <= target64)
|
||||
// done <job_id> <hashes> <ms> end of the job (wall ms, dispatch plus scan)
|
||||
// error <job_id> <text>
|
||||
// The program for an epoch seed is generateProgram(words: seedWordsBytes(epoch_seed)), compiled once and cached; the
|
||||
// cache and 1 GiB dataset for a day seed come from seedWordsBytes(day_seed_bytes), built once and cached (two of each).
|
||||
// prepared <epoch_seed_hex> <day_seed_hex> <ms> program <ms> dataset <ms> ... the pair is resident
|
||||
// prepare-failed <epoch_seed_hex> <day_seed_hex> <text>
|
||||
// The program for an epoch seed is generateProgramV2(bytes: epoch_seed) (version 2 with the acceptance rule, the same
|
||||
// derivation as igneum-pow), compiled once and kept; the
|
||||
// cache and 1 GiB dataset for a day seed come from seedWordsBytes(day_seed_bytes), built once and kept. At most two
|
||||
// programs and two datasets are resident: the current job's pair and one more (the prepared pair, or the previous
|
||||
// pair until the first job on the new one is done). A job whose pair is resident switches instantly; a job whose pair
|
||||
// is not (no prepare, or a pair nobody predicted) compiles inline as before. The pack_dir of a prepare is ignored
|
||||
// here (Metal compiles from the seed); ahead-of-time workers build their kernel from it.
|
||||
// The init words of a dispatch are seedWordsBytes("igneum-block/" || prehash || nonce_hi_le32) and go to buffer 3 of
|
||||
// igneum_hash_bound; the lane nonce is baseNonce + gid as in the bench kernel.
|
||||
|
||||
func emit(_ line: String) { print(line); fflush(stdout) }
|
||||
let emitLock = NSLock()
|
||||
func emit(_ line: String) { emitLock.lock(); print(line); fflush(stdout); emitLock.unlock() }
|
||||
|
||||
func unhex(_ s: String) -> [UInt8]? {
|
||||
let chars = Array(s.utf8)
|
||||
|
|
@ -2498,6 +2747,41 @@ final class ServeDataset {
|
|||
init(dayHex: String, ctx: DatasetContext, buffer: MTLBuffer) { self.dayHex = dayHex; self.ctx = ctx; self.buffer = buffer }
|
||||
}
|
||||
|
||||
// The resident programs and datasets, shared by the job loop (main thread) and the prepare queue (background).
|
||||
final class ServeStore {
|
||||
private let lock = NSLock()
|
||||
private var programs = [ServeProgram]()
|
||||
private var datasets = [ServeDataset]()
|
||||
/// The pair of the last job (epoch seed hex, day seed hex); never evicted
|
||||
var current: (String, String)? = nil
|
||||
|
||||
func program(_ seedHex: String) -> ServeProgram? { lock.lock(); defer { lock.unlock() }; return programs.first { $0.seedHex == seedHex } }
|
||||
func dataset(_ dayHex: String) -> ServeDataset? { lock.lock(); defer { lock.unlock() }; return datasets.first { $0.dayHex == dayHex } }
|
||||
func counts() -> (Int, Int) { lock.lock(); defer { lock.unlock() }; return (programs.count, datasets.count) }
|
||||
|
||||
/// Adds a program; with more than two resident, the oldest one that is not the current job's goes.
|
||||
func add(_ p: ServeProgram) {
|
||||
lock.lock(); defer { lock.unlock() }
|
||||
if programs.contains(where: { $0.seedHex == p.seedHex }) { return }
|
||||
programs.append(p)
|
||||
while programs.count > 2, let i = programs.firstIndex(where: { $0.seedHex != current?.0 && $0.seedHex != p.seedHex }) { programs.remove(at: i) }
|
||||
}
|
||||
func add(_ d: ServeDataset) {
|
||||
lock.lock(); defer { lock.unlock() }
|
||||
if datasets.contains(where: { $0.dayHex == d.dayHex }) { return }
|
||||
datasets.append(d)
|
||||
while datasets.count > 2, let i = datasets.firstIndex(where: { $0.dayHex != current?.1 && $0.dayHex != d.dayHex }) { datasets.remove(at: i) }
|
||||
}
|
||||
/// After the first job on a new pair: drop everything but that pair (the old program and dataset are released).
|
||||
func prune(to pair: (String, String)) -> (Int, Int) {
|
||||
lock.lock(); defer { lock.unlock() }
|
||||
let before = (programs.count, datasets.count)
|
||||
programs.removeAll { $0.seedHex != pair.0 }
|
||||
datasets.removeAll { $0.dayHex != pair.1 }
|
||||
return (before.0 - programs.count, before.1 - datasets.count)
|
||||
}
|
||||
}
|
||||
|
||||
func compileBound(_ gpu: GPU, msl: String) throws -> CompiledHash {
|
||||
let t0 = nowNs()
|
||||
let lib = try gpu.device.makeLibrary(source: msl, options: MTLCompileOptions())
|
||||
|
|
@ -2508,18 +2792,63 @@ func compileBound(_ gpu: GPU, msl: String) throws -> CompiledHash {
|
|||
return CompiledHash(pipeline: pipe, libraryMs: ms(t0, t1), pipelineMs: ms(t1, t2))
|
||||
}
|
||||
|
||||
// Builds the program for an epoch seed (hex) unless resident. Returns (program, compile ms) or throws.
|
||||
func serveProgram(_ gpu: GPU, _ store: ServeStore, seedHex: String, seed: [UInt8], datasetLog2: Int) throws -> (ServeProgram, Double) {
|
||||
if let p = store.program(seedHex) { return (p, 0) }
|
||||
let t0 = nowNs()
|
||||
let p = generateProgramV2(seedString: "epoch/" + seedHex, bytes: seed)
|
||||
let msl = generateMSL(p, datasetLog2: datasetLog2, source: .stored, bound: true)
|
||||
let c = try compileBound(gpu, msl: msl)
|
||||
let sp = ServeProgram(seedHex: seedHex, program: p, compiled: c)
|
||||
store.add(sp)
|
||||
return (sp, ms(t0, nowNs()))
|
||||
}
|
||||
|
||||
// Builds the cache and dataset for a day seed (hex) unless resident. Returns (dataset, build ms).
|
||||
func serveDataset(_ gpu: GPU, _ store: ServeStore, dayHex: String, day: [UInt8], datasetLog2: Int) -> (ServeDataset, Double) {
|
||||
if let d = store.dataset(dayHex) { return (d, 0) }
|
||||
let t0 = nowNs()
|
||||
let key = seedWordsBytes(day)
|
||||
let ctx = DatasetContext(gpu: gpu, closedForm: false, dayString: "day/" + dayHex, key: key)
|
||||
let buf = ctx.makeDataset(log2: datasetLog2)
|
||||
let sd = ServeDataset(dayHex: dayHex, ctx: ctx, buffer: buf)
|
||||
store.add(sd)
|
||||
return (sd, ms(t0, nowNs()))
|
||||
}
|
||||
|
||||
func runServe(_ opts: Options) -> Never {
|
||||
let gpu = GPU()
|
||||
let datasetLog2 = opts.datasetLog2
|
||||
let batch = 1 << opts.batchLog2 // nonces per dispatch
|
||||
var programs = [ServeProgram]()
|
||||
var datasets = [ServeDataset]()
|
||||
let store = ServeStore()
|
||||
let prepareQueue = DispatchQueue(label: "igneum.prepare") // one prepare at a time, off the job loop
|
||||
guard let outBuf = gpu.device.makeBuffer(length: batch * 8, options: .storageModeShared) else { emit("error 0 cannot allocate the output buffer"); exit(1) }
|
||||
emit("ready metal \(gpu.device.name.replacingOccurrences(of: " ", with: "_")) dataset-log2 \(datasetLog2) batch \(batch)")
|
||||
emit("ready metal \(gpu.device.name.replacingOccurrences(of: " ", with: "_")) dataset-log2 \(datasetLog2) batch \(batch) prepare \(opts.noPrepare ? 0 : 1)")
|
||||
var lastPair: (String, String)? = nil
|
||||
while let line = readLine(strippingNewline: true) {
|
||||
let f = line.split(separator: " ").map(String.init)
|
||||
if f.isEmpty { continue }
|
||||
if f[0] == "quit" { break }
|
||||
if f[0] == "prepare" {
|
||||
if opts.noPrepare { emit("info ignored (started with --no-prepare): \(line)"); continue }
|
||||
if f.count < 3 { emit("prepare-failed 0 0 malformed prepare line (need epoch_seed_hex and day_seed_hex)"); continue }
|
||||
guard let epochSeed = unhex(f[1]), epochSeed.count == 32, let daySeed = unhex(f[2]) else {
|
||||
emit("prepare-failed \(f[1]) \(f[2]) bad field (epoch_seed 64 hex, day_seed hex)"); continue
|
||||
}
|
||||
let (epochHex, dayHex) = (f[1], f[2])
|
||||
let have = (store.program(epochHex) != nil, store.dataset(dayHex) != nil)
|
||||
if have.0 && have.1 { emit("prepared \(epochHex) \(dayHex) 0 program 0 dataset 0 (already resident)"); continue }
|
||||
prepareQueue.async {
|
||||
let t0 = nowNs()
|
||||
do {
|
||||
let (sp, progMs) = try serveProgram(gpu, store, seedHex: epochHex, seed: epochSeed, datasetLog2: datasetLog2)
|
||||
let (sd, dsMs) = serveDataset(gpu, store, dayHex: dayHex, day: daySeed, datasetLog2: datasetLog2)
|
||||
let (np, nd) = store.counts()
|
||||
emit("prepared \(epochHex) \(dayHex) \(fmt(ms(t0, nowNs()), 1)) program \(fmt(progMs, 1)) dataset \(fmt(dsMs, 1)) loads/hash \(sp.program.loadsPerHash) cache-fill \(fmt(sd.ctx.cacheFillGPUms, 1)) resident \(np) programs \(nd) datasets")
|
||||
} catch { emit("prepare-failed \(epochHex) \(dayHex) Metal compile failed: \(error)") }
|
||||
}
|
||||
continue
|
||||
}
|
||||
if f[0] != "job" { emit("info ignored: \(line)"); continue }
|
||||
if f.count < 8 { emit("error \(f.count > 1 ? f[1] : "0") malformed job line (need 7 fields after job)"); continue }
|
||||
let jobId = f[1]
|
||||
|
|
@ -2530,31 +2859,24 @@ func runServe(_ opts: Options) -> Never {
|
|||
}
|
||||
if nonceCount == 0 || nonceCount % 32 != 0 || (nonceStart & 31) != 0 { emit("error \(jobId) nonce_start must be 32-aligned and nonce_count a non-zero multiple of 32"); continue }
|
||||
let t0 = nowNs()
|
||||
// Program for the epoch seed
|
||||
var prog = programs.first { $0.seedHex == f[6] }
|
||||
if prog == nil {
|
||||
let p = generateProgram(seedString: "epoch/" + f[6], words: seedWordsBytes(epochSeed))
|
||||
let msl = generateMSL(p, datasetLog2: datasetLog2, source: .stored, bound: true)
|
||||
do {
|
||||
let c = try compileBound(gpu, msl: msl)
|
||||
emit("info program epoch \(f[6].prefix(16)) loads/hash \(p.loadsPerHash) compiled in \(fmt(c.totalMs, 1)) ms")
|
||||
let sp = ServeProgram(seedHex: f[6], program: p, compiled: c)
|
||||
programs.append(sp); if programs.count > 2 { programs.removeFirst() }
|
||||
prog = sp
|
||||
} catch { emit("error \(jobId) Metal compile failed: \(error)"); continue }
|
||||
let pair = (f[6], f[7])
|
||||
let switched = lastPair == nil || lastPair! != pair
|
||||
if switched { store.current = pair }
|
||||
// Program and dataset for the pair: resident (prepared, or the current pair) or compiled inline now. A prepare
|
||||
// of the same pair may be in flight on the queue; waiting for the queue makes this a join instead of a double build.
|
||||
let program: ServeProgram
|
||||
let dataset: ServeDataset
|
||||
do {
|
||||
if store.program(pair.0) == nil || store.dataset(pair.1) == nil { prepareQueue.sync {} }
|
||||
let (sp, progMs) = try serveProgram(gpu, store, seedHex: pair.0, seed: epochSeed, datasetLog2: datasetLog2)
|
||||
if progMs > 0 { emit("info program epoch \(pair.0.prefix(16)) loads/hash \(sp.program.loadsPerHash) compiled inline in \(fmt(progMs, 1)) ms (not prepared)") }
|
||||
let (sd, dsMs) = serveDataset(gpu, store, dayHex: pair.1, day: daySeed, datasetLog2: datasetLog2)
|
||||
if dsMs > 0 { emit("info dataset day \(pair.1) built inline in \(fmt(dsMs, 1)) ms (cache fill \(fmt(sd.ctx.cacheFillGPUms, 1)) ms, build \(fmt(sd.ctx.lastBuildGPUms, 1)) ms GPU; not prepared)") }
|
||||
program = sp; dataset = sd
|
||||
} catch { emit("error \(jobId) Metal compile failed: \(error)"); continue }
|
||||
if switched, let prev = lastPair {
|
||||
emit("info switched from epoch \(prev.0.prefix(16)) day \(prev.1) to epoch \(pair.0.prefix(16)) day \(pair.1) in \(fmt(ms(t0, nowNs()), 2)) ms (resident: \(store.counts().0) programs, \(store.counts().1) datasets)")
|
||||
}
|
||||
// Cache and dataset for the day seed
|
||||
var ds = datasets.first { $0.dayHex == f[7] }
|
||||
if ds == nil {
|
||||
let key = seedWordsBytes(daySeed)
|
||||
let ctx = DatasetContext(gpu: gpu, closedForm: false, dayString: "day/" + f[7], key: key)
|
||||
let buf = ctx.makeDataset(log2: datasetLog2)
|
||||
emit("info dataset day \(f[7]) key \(key.map { String(format: "%08x", $0) }.joined(separator: " ")) cache fill \(fmt(ctx.cacheFillGPUms, 1)) ms build \(fmt(ctx.lastBuildGPUms, 1)) ms GPU")
|
||||
let sd = ServeDataset(dayHex: f[7], ctx: ctx, buffer: buf)
|
||||
datasets.append(sd); if datasets.count > 2 { datasets.removeFirst() }
|
||||
ds = sd
|
||||
}
|
||||
guard let program = prog, let dataset = ds else { continue }
|
||||
// Mine: chunks of at most `batch` lane nonces that share one high word
|
||||
var remaining = nonceCount
|
||||
var hi = UInt32(truncatingIfNeeded: nonceStart >> 32)
|
||||
|
|
@ -2592,6 +2914,12 @@ func runServe(_ opts: Options) -> Never {
|
|||
}
|
||||
if failed { continue }
|
||||
emit("done \(jobId) \(hashes) \(fmt(ms(t0, nowNs()), 2))")
|
||||
if switched, lastPair != nil {
|
||||
// The first job on the new pair is done: the old pair goes (at most two of each were resident until now)
|
||||
let (dp, dd) = store.prune(to: pair)
|
||||
if dp + dd > 0 { emit("info dropped \(dp) program(s) and \(dd) dataset(s) of the previous pair") }
|
||||
}
|
||||
lastPair = pair
|
||||
}
|
||||
exit(0)
|
||||
}
|
||||
|
|
|
|||
|
|
@ -31,6 +31,7 @@
|
|||
#else
|
||||
#include <time.h>
|
||||
#include <dlfcn.h>
|
||||
#include <pthread.h>
|
||||
#endif
|
||||
|
||||
#define IGNEUM_NO_CUDA
|
||||
|
|
@ -192,6 +193,7 @@ typedef struct {
|
|||
const char* kernelPath;
|
||||
const char* extraOpts;
|
||||
int serve; // --serve: GPU worker for igneum-miner --worker (jobs on stdin), 3 October 2026
|
||||
int noPrepare; // --no-prepare: serve without the prepare command (ready line says "prepare 0"), to test the miner's fallback
|
||||
int kernelGiven; // --kernel was passed
|
||||
const char* vendor; // --vendor S: pick the first GPU whose vendor string contains S (default: first GPU of any vendor)
|
||||
} Options;
|
||||
|
|
@ -217,6 +219,7 @@ static void usage(void) {
|
|||
" Apple's OpenCL runtime reports unusable event timestamps, so wall is the default on the Apple platform.\n"
|
||||
" --vendor S choose the first GPU whose vendor string contains S (for example \"Advanced Micro Devices\"); fails if none\n"
|
||||
" --serve GPU worker for igneum-miner --worker: reads \"job ...\" lines on stdin, prints found/done lines.\n"
|
||||
" --no-prepare with --serve: no prepare support (the miner then falls back to exit 42 at a seed change).\n"
|
||||
" Builds the pack's kernel_bound.cl (next to the compiled-in kernel.cl) unless --kernel says otherwise.\n", packMib(), IGNEUM_KERNEL_PATH);
|
||||
}
|
||||
|
||||
|
|
@ -227,7 +230,7 @@ static Options parseArgs(int argc, char** argv) {
|
|||
Options o;
|
||||
int i;
|
||||
o.datasetMib = 1024; o.batchLog2 = 24; o.batches = 5; o.groupWarps = 1; o.sweep = 0; o.device = -1;
|
||||
o.exchange = 0; o.list = 0; o.timeWall = -1; o.kernelPath = IGNEUM_KERNEL_PATH; o.extraOpts = ""; o.serve = 0; o.kernelGiven = 0; o.vendor = NULL;
|
||||
o.exchange = 0; o.list = 0; o.timeWall = -1; o.kernelPath = IGNEUM_KERNEL_PATH; o.extraOpts = ""; o.serve = 0; o.noPrepare = 0; o.kernelGiven = 0; o.vendor = NULL;
|
||||
for (i = 1; i < argc; ++i) {
|
||||
const char* a = argv[i];
|
||||
int needs = (strcmp(a, "--dataset-mib") == 0 || strcmp(a, "--batch-log2") == 0 || strcmp(a, "--batches") == 0 ||
|
||||
|
|
@ -241,6 +244,7 @@ static Options parseArgs(int argc, char** argv) {
|
|||
else if (strcmp(a, "--device") == 0) o.device = atoi(argv[++i]);
|
||||
else if (strcmp(a, "--kernel") == 0) { o.kernelPath = argv[++i]; o.kernelGiven = 1; }
|
||||
else if (strcmp(a, "--serve") == 0) o.serve = 1;
|
||||
else if (strcmp(a, "--no-prepare") == 0) o.noPrepare = 1;
|
||||
else if (strcmp(a, "--vendor") == 0) { if (i + 1 >= argc) { usage(); exit(2); } o.vendor = argv[++i]; }
|
||||
else if (strcmp(a, "--build-opts") == 0) o.extraOpts = argv[++i];
|
||||
else if (strcmp(a, "--time") == 0) {
|
||||
|
|
@ -848,9 +852,20 @@ static SizeResult runSize(Device* dv, const DeviceInfo* di, const Options* o, in
|
|||
* found <job_id> <nonce u64> <hash_hex 16> every nonce whose 64-bit hash is <= target
|
||||
* done <job_id> <hashes> <ms> end of the job (wall ms)
|
||||
* error <job_id> <text>
|
||||
* The pack's program.h is compiled in and kernel_bound.cl is built at runtime, so this worker serves exactly one
|
||||
* epoch seed and one day seed: the pack's. A job for other seeds is answered with an error naming both; re-export the
|
||||
* pack with `igneum-miner export-pack <node> <dir>` and rebuild. The init words of a dispatch are
|
||||
* prepare <epoch_seed_hex 64> <day_seed_hex> <pack_dir> build <pack_dir>/kernel_bound.cl (the pack the miner wrote
|
||||
* for those seeds) plus its cache and dataset in the background
|
||||
* stdout: ready opencl <device> ... prepare 1
|
||||
* prepared <epoch_seed_hex> <day_seed_hex> <ms> ... the pair is resident; a job on it switches instantly
|
||||
* prepare-failed <epoch_seed_hex> <day_seed_hex> <text>
|
||||
* The pack's program.h is compiled in and kernel_bound.cl is built at runtime, so at start this worker serves exactly
|
||||
* one epoch seed and one day seed: the pack's. The next pair arrives through `prepare`: the miner writes the pack for the
|
||||
* prepared seeds (igneum-miner --prepare-packs <dir>) and names its directory; a background thread builds that pack's
|
||||
* kernel_bound.cl with the same build options, fills its cache and builds its dataset on a second queue while jobs on
|
||||
* the current pair keep running (at most two pairs resident: the current one and the prepared one; the old pair is
|
||||
* released after the first job on the new one). A job for seeds that are neither the current nor the prepared pair is
|
||||
* answered with an error naming both. Without prepare, re-export the pack with `igneum-miner export-pack <node> <dir>`
|
||||
* and rebuild. A prepared pair's cache is not cross-checked against a host fill (memhard.h is compiled in for the
|
||||
* original day); the miner's CPU re-check of every found nonce covers it. The init words of a dispatch are
|
||||
* seed_words_from_bytes("igneum-block/" || prehash || nonce_hi_le32), written to a small buffer that is the fifth
|
||||
* argument of igneum_hash_bound; the lane nonce is baseNonce + gid as in the bench kernel. The exchange rule of
|
||||
* WAVEFRONT.md applies unchanged (the bound kernel has the same body and the same IGNEUM_EXCHANGE build).
|
||||
|
|
@ -881,6 +896,132 @@ static int unhexBuf(const char* s, uint8_t* out, size_t cap, size_t* len) {
|
|||
return 1;
|
||||
}
|
||||
|
||||
#if IGNEUM_DATASET_MODE == 1
|
||||
/* One resident (program, cache, dataset) triple for a seed pair. The first one is the compiled-in pack on the main
|
||||
* queue; prepared ones are built from a pack directory on their own queue. */
|
||||
typedef struct {
|
||||
char epochHex[65];
|
||||
char dayHex[512];
|
||||
uint32_t sw[8], kw[8]; /* seed words and key words, for job matching */
|
||||
cl_program prog;
|
||||
cl_kernel kHashBound, kCacheFill, kBuild;
|
||||
cl_mem cache, ds;
|
||||
double buildMs, cacheMs, datasetMs;
|
||||
} ServePair;
|
||||
|
||||
static void releasePair(ServePair* p) {
|
||||
if (!p) return;
|
||||
if (p->ds) clReleaseMemObject(p->ds);
|
||||
if (p->cache) clReleaseMemObject(p->cache);
|
||||
if (p->kHashBound) clReleaseKernel(p->kHashBound);
|
||||
if (p->kCacheFill) clReleaseKernel(p->kCacheFill);
|
||||
if (p->kBuild) clReleaseKernel(p->kBuild);
|
||||
if (p->prog) clReleaseProgram(p->prog);
|
||||
free(p);
|
||||
}
|
||||
|
||||
/* The prepare request and its result, handed between the main loop and the prepare thread. */
|
||||
typedef struct {
|
||||
Device* dv;
|
||||
const DeviceInfo* di;
|
||||
char epochHex[65];
|
||||
char dayHex[512];
|
||||
char packDir[1024];
|
||||
uint32_t words;
|
||||
char error[512];
|
||||
ServePair* result; /* set by the thread on success */
|
||||
volatile int done; /* 1 when the thread has finished (success or failure) */
|
||||
double t0, doneAt;
|
||||
} PrepareTask;
|
||||
|
||||
static void prepareFail(PrepareTask* t, const char* what, cl_int err) {
|
||||
snprintf(t->error, sizeof(t->error), "%s (%s)", what, clErrName(err));
|
||||
}
|
||||
|
||||
/* Builds the pair for a prepare request. Runs on its own thread with its own command queue. */
|
||||
static void prepareRun(PrepareTask* t) {
|
||||
cl_int err = 0;
|
||||
size_t srcLen = 0;
|
||||
char path[1200];
|
||||
char* src;
|
||||
ServePair* p = (ServePair*)calloc(1, sizeof(ServePair));
|
||||
cl_command_queue q = NULL;
|
||||
double tb;
|
||||
cl_uint nSeg = IGNEUM_CACHE_SEGMENTS, nItems;
|
||||
size_t local, bytes = (size_t)CACHE_WORDS_HOST * 4u;
|
||||
strncpy(p->epochHex, t->epochHex, 64); p->epochHex[64] = 0;
|
||||
strncpy(p->dayHex, t->dayHex, sizeof(p->dayHex) - 1);
|
||||
snprintf(path, sizeof(path), "%s/kernel_bound.cl", t->packDir);
|
||||
src = readFile(path, &srcLen);
|
||||
if (!src) { snprintf(t->error, sizeof(t->error), "cannot read %s", path); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; }
|
||||
tb = wallMs();
|
||||
p->prog = clCreateProgramWithSource(t->dv->ctx, 1, (const char**)&src, &srcLen, &err);
|
||||
free(src);
|
||||
if (err != CL_SUCCESS) { prepareFail(t, "clCreateProgramWithSource", err); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; }
|
||||
err = clBuildProgram(p->prog, 1, &t->di->device, t->dv->buildOptions, NULL, NULL);
|
||||
if (err != CL_SUCCESS) {
|
||||
size_t logLen = 0;
|
||||
char* log;
|
||||
clGetProgramBuildInfo(p->prog, t->di->device, CL_PROGRAM_BUILD_LOG, 0, NULL, &logLen);
|
||||
log = (char*)calloc(logLen + 1, 1);
|
||||
if (logLen) clGetProgramBuildInfo(p->prog, t->di->device, CL_PROGRAM_BUILD_LOG, logLen, log, NULL);
|
||||
snprintf(t->error, sizeof(t->error), "clBuildProgram failed (%s): %.300s", clErrName(err), log);
|
||||
free(log); releasePair(p); t->done = 1; return;
|
||||
}
|
||||
p->buildMs = wallMs() - tb;
|
||||
p->kHashBound = clCreateKernel(p->prog, "igneum_hash_bound", &err);
|
||||
if (err != CL_SUCCESS) { prepareFail(t, "clCreateKernel igneum_hash_bound (is this a kernel_bound.cl?)", err); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; }
|
||||
p->kCacheFill = clCreateKernel(p->prog, "igneum_cache_fill", &err);
|
||||
if (err != CL_SUCCESS) { prepareFail(t, "clCreateKernel igneum_cache_fill", err); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; }
|
||||
p->kBuild = clCreateKernel(p->prog, "igneum_build", &err);
|
||||
if (err != CL_SUCCESS) { prepareFail(t, "clCreateKernel igneum_build", err); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; }
|
||||
q = clCreateCommandQueue(t->dv->ctx, t->di->device, 0, &err);
|
||||
if (err != CL_SUCCESS) { prepareFail(t, "clCreateCommandQueue (prepare)", err); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; }
|
||||
/* Cache: the same segment count as the compiled-in pack (the dataset schedule is a network constant) */
|
||||
tb = wallMs();
|
||||
p->cache = clCreateBuffer(t->dv->ctx, CL_MEM_READ_WRITE, bytes, NULL, &err);
|
||||
if (err != CL_SUCCESS) { prepareFail(t, "clCreateBuffer cache", err); clReleaseCommandQueue(q); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; }
|
||||
local = kernelMaxLocal(t->dv, p->kCacheFill, t->di, 256);
|
||||
{
|
||||
size_t g = ((nSeg + local - 1) / local) * local;
|
||||
err = clSetKernelArg(p->kCacheFill, 0, sizeof(cl_mem), &p->cache);
|
||||
if (err == CL_SUCCESS) err = clSetKernelArg(p->kCacheFill, 1, sizeof(cl_uint), &nSeg);
|
||||
if (err == CL_SUCCESS) err = clEnqueueNDRangeKernel(q, p->kCacheFill, 1, NULL, &g, &local, 0, NULL, NULL);
|
||||
if (err == CL_SUCCESS) err = clFinish(q);
|
||||
}
|
||||
if (err != CL_SUCCESS) { prepareFail(t, "cache fill", err); clReleaseCommandQueue(q); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; }
|
||||
p->cacheMs = wallMs() - tb;
|
||||
/* Dataset */
|
||||
tb = wallMs();
|
||||
nItems = t->words / 16u;
|
||||
p->ds = clCreateBuffer(t->dv->ctx, CL_MEM_READ_WRITE, (size_t)t->words * 4u, NULL, &err);
|
||||
if (err != CL_SUCCESS) { prepareFail(t, "clCreateBuffer dataset", err); clReleaseCommandQueue(q); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; }
|
||||
local = kernelMaxLocal(t->dv, p->kBuild, t->di, 256);
|
||||
{
|
||||
size_t g = ((nItems + local - 1) / local) * local;
|
||||
err = clSetKernelArg(p->kBuild, 0, sizeof(cl_mem), &p->ds);
|
||||
if (err == CL_SUCCESS) err = clSetKernelArg(p->kBuild, 1, sizeof(cl_mem), &p->cache);
|
||||
if (err == CL_SUCCESS) err = clSetKernelArg(p->kBuild, 2, sizeof(cl_uint), &nItems);
|
||||
if (err == CL_SUCCESS) err = clEnqueueNDRangeKernel(q, p->kBuild, 1, NULL, &g, &local, 0, NULL, NULL);
|
||||
if (err == CL_SUCCESS) err = clFinish(q);
|
||||
}
|
||||
if (err != CL_SUCCESS) { prepareFail(t, "dataset build", err); clReleaseCommandQueue(q); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; }
|
||||
p->datasetMs = wallMs() - tb;
|
||||
clReleaseCommandQueue(q);
|
||||
t->result = p;
|
||||
t->doneAt = wallMs();
|
||||
t->done = 1;
|
||||
}
|
||||
|
||||
#ifdef _WIN32
|
||||
static DWORD WINAPI prepareThreadMain(LPVOID arg) { prepareRun((PrepareTask*)arg); return 0; }
|
||||
static int startPrepareThread(PrepareTask* t) { HANDLE h = CreateThread(NULL, 0, prepareThreadMain, t, 0, NULL); if (!h) return 0; CloseHandle(h); return 1; }
|
||||
#else
|
||||
static void* prepareThreadMain(void* arg) { prepareRun((PrepareTask*)arg); return NULL; }
|
||||
static int startPrepareThread(PrepareTask* t) { pthread_t th; if (pthread_create(&th, NULL, prepareThreadMain, t) != 0) return 0; pthread_detach(th); return 1; }
|
||||
#endif
|
||||
#endif
|
||||
|
||||
static int runServe(Device* dv, const DeviceInfo* di, const Options* o) {
|
||||
#if IGNEUM_DATASET_MODE != 1
|
||||
(void)dv; (void)di; (void)o;
|
||||
|
|
@ -893,26 +1034,35 @@ static int runServe(Device* dv, const DeviceInfo* di, const Options* o) {
|
|||
const uint32_t batch = 1u << (o->batchLog2 == 24 ? 22 : o->batchLog2); /* 2^22 nonces per dispatch by default */
|
||||
size_t groupSize = 32 * (size_t)o->groupWarps;
|
||||
cl_int err = 0;
|
||||
cl_mem dDs, dOut, dInit;
|
||||
cl_mem dOut, dInit;
|
||||
cl_uint nItems = words / 16u;
|
||||
uint64_t* hOut;
|
||||
char devName[256];
|
||||
char line[1024];
|
||||
char line[2048];
|
||||
size_t k;
|
||||
ServePair* cur; /* the pair jobs run on */
|
||||
ServePair* prepared = NULL; /* the pair the last prepare built, until a job switches to it */
|
||||
ServePair* old = NULL; /* the previous pair, released after the first job on the new one */
|
||||
PrepareTask* task = NULL; /* the prepare in flight */
|
||||
if (!dv->kHashBound) { printf("error 0 the kernel source has no igneum_hash_bound (build from the pack's kernel_bound.cl, or pass --kernel)\n"); fflush(stdout); return 2; }
|
||||
if (!setupCache(dv, di)) { printf("error 0 cache check failed (device cache differs from the host cache or the pack's FNV)\n"); fflush(stdout); return 1; }
|
||||
dDs = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, (size_t)words * 4u, NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer dataset");
|
||||
CL_CHECK(clSetKernelArg(dv->kBuild, 0, sizeof(cl_mem), &dDs));
|
||||
CL_CHECK(clSetKernelArg(dv->kBuild, 1, sizeof(cl_mem), &gCache));
|
||||
CL_CHECK(clSetKernelArg(dv->kBuild, 2, sizeof(cl_uint), &nItems));
|
||||
clReleaseEvent(launch1D(dv, dv->kBuild, nItems, kernelMaxLocal(dv, dv->kBuild, di, 256)));
|
||||
cur = (ServePair*)calloc(1, sizeof(ServePair));
|
||||
memcpy(cur->sw, SEEDW, 32); memcpy(cur->kw, KEYW, 32);
|
||||
cur->kHashBound = dv->kHashBound; cur->kCacheFill = dv->kCacheFill; cur->kBuild = dv->kBuild; cur->prog = dv->prog;
|
||||
dv->kHashBound = dv->kCacheFill = dv->kBuild = NULL; dv->prog = NULL; /* owned by the pair now */
|
||||
cur->cache = gCache; gCache = NULL;
|
||||
cur->ds = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, (size_t)words * 4u, NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer dataset");
|
||||
CL_CHECK(clSetKernelArg(cur->kBuild, 0, sizeof(cl_mem), &cur->ds));
|
||||
CL_CHECK(clSetKernelArg(cur->kBuild, 1, sizeof(cl_mem), &cur->cache));
|
||||
CL_CHECK(clSetKernelArg(cur->kBuild, 2, sizeof(cl_uint), &nItems));
|
||||
clReleaseEvent(launch1D(dv, cur->kBuild, nItems, kernelMaxLocal(dv, cur->kBuild, di, 256)));
|
||||
CL_CHECK(clFinish(dv->q));
|
||||
dOut = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, (size_t)batch * sizeof(uint64_t), NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer out");
|
||||
dInit = clCreateBuffer(dv->ctx, CL_MEM_READ_ONLY, 32, NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer init words");
|
||||
hOut = (uint64_t*)malloc((size_t)batch * sizeof(uint64_t));
|
||||
strncpy(devName, di->name, 255); devName[255] = 0;
|
||||
for (k = 0; devName[k]; ++k) if (devName[k] == ' ') devName[k] = '_';
|
||||
printf("ready opencl %s platform %s pack %s dataset-log2 %d batch %u exchange %d\n", devName, di->platformName, IGNEUM_SEED_STRING, IGNEUM_DATASET_LOG2, batch, dv->exchange);
|
||||
printf("ready opencl %s platform %s pack %s dataset-log2 %d batch %u exchange %d prepare %d\n", devName, di->platformName, IGNEUM_SEED_STRING, IGNEUM_DATASET_LOG2, batch, dv->exchange, o->noPrepare ? 0 : 1);
|
||||
fflush(stdout);
|
||||
|
||||
while (fgets(line, sizeof(line), stdin)) {
|
||||
|
|
@ -928,11 +1078,42 @@ static int runServe(Device* dv, const DeviceInfo* di, const Options* o) {
|
|||
uint64_t remaining, hashes = 0;
|
||||
uint32_t hi, lo;
|
||||
double t0;
|
||||
int failed = 0;
|
||||
int failed = 0, switched = 0;
|
||||
line[strcspn(line, "\r\n")] = 0;
|
||||
/* A finished prepare is reported here, between lines (the thread never prints) */
|
||||
if (task && task->done) {
|
||||
if (task->result) {
|
||||
ServePair* p = task->result;
|
||||
{
|
||||
uint8_t eb[32], db[256]; size_t el = 0, dl = 0;
|
||||
if (unhexBuf(p->epochHex, eb, 32, &el) && el == 32) seedWordsFromBytes(eb, 32, p->sw);
|
||||
if (unhexBuf(p->dayHex, db, sizeof(db), &dl)) seedWordsFromBytes(db, dl, p->kw);
|
||||
}
|
||||
if (prepared) releasePair(prepared);
|
||||
prepared = p;
|
||||
printf("prepared %s %s %.1f build %.1f cache %.1f dataset %.1f resident 2 programs 2 datasets\n", p->epochHex, p->dayHex, task->doneAt - task->t0, p->buildMs, p->cacheMs, p->datasetMs);
|
||||
} else {
|
||||
printf("prepare-failed %s %s %s\n", task->epochHex, task->dayHex, task->error);
|
||||
}
|
||||
fflush(stdout);
|
||||
free(task); task = NULL;
|
||||
}
|
||||
for (tok = strtok_r(line, " ", &save); tok && nf < 9; tok = strtok_r(NULL, " ", &save)) f[nf++] = tok;
|
||||
if (nf == 0) continue;
|
||||
if (strcmp(f[0], "quit") == 0) break;
|
||||
if (strcmp(f[0], "prepare") == 0) {
|
||||
if (o->noPrepare) { printf("info ignored (started with --no-prepare): prepare\n"); fflush(stdout); continue; }
|
||||
if (nf < 4) { printf("prepare-failed %s %s this ahead-of-time worker needs a pack directory as the third field (igneum-miner --prepare-packs <dir>)\n", nf > 1 ? f[1] : "0", nf > 2 ? f[2] : "0"); fflush(stdout); continue; }
|
||||
if (strlen(f[1]) != 64 || strlen(f[2]) >= 500) { printf("prepare-failed %s %s bad field (epoch_seed 64 hex, day_seed hex)\n", f[1], f[2]); fflush(stdout); continue; }
|
||||
if (task) { printf("prepare-failed %s %s a prepare is still running\n", f[1], f[2]); fflush(stdout); continue; }
|
||||
if (prepared && strcmp(prepared->epochHex, f[1]) == 0 && strcmp(prepared->dayHex, f[2]) == 0) { printf("prepared %s %s 0 (already resident)\n", f[1], f[2]); fflush(stdout); continue; }
|
||||
task = (PrepareTask*)calloc(1, sizeof(PrepareTask));
|
||||
task->dv = dv; task->di = di; task->words = words; task->t0 = wallMs();
|
||||
strncpy(task->epochHex, f[1], 64); strncpy(task->dayHex, f[2], sizeof(task->dayHex) - 1); strncpy(task->packDir, f[3], sizeof(task->packDir) - 1);
|
||||
if (!startPrepareThread(task)) { printf("prepare-failed %s %s cannot start the prepare thread\n", f[1], f[2]); fflush(stdout); free(task); task = NULL; continue; }
|
||||
printf("info prepare started for epoch %.16s day %s from %s (builds in the background)\n", f[1], f[2], f[3]); fflush(stdout);
|
||||
continue;
|
||||
}
|
||||
if (strcmp(f[0], "job") != 0) { printf("info ignored line\n"); fflush(stdout); continue; }
|
||||
strncpy(jobId, nf > 1 ? f[1] : "0", 63); jobId[63] = 0;
|
||||
if (nf < 8) { printf("error %s malformed job line (need 7 fields after job)\n", jobId); fflush(stdout); continue; }
|
||||
|
|
@ -944,17 +1125,23 @@ static int runServe(Device* dv, const DeviceInfo* di, const Options* o) {
|
|||
if (nonceCount == 0 || nonceCount % 32 != 0 || (nonceStart & 31) != 0) { printf("error %s nonce_start must be 32-aligned and nonce_count a non-zero multiple of 32\n", jobId); fflush(stdout); continue; }
|
||||
seedWordsFromBytes(epochSeed, 32, sw);
|
||||
seedWordsFromBytes(daySeed, dayLen, kw);
|
||||
if (memcmp(sw, SEEDW, 32) != 0) {
|
||||
printf("error %s epoch seed mismatch: this worker was built for pack \"%s\" (seed words %08x %08x ...), the job's epoch seed %.16s gives %08x %08x ...; run igneum-miner export-pack and rebuild\n",
|
||||
jobId, IGNEUM_SEED_STRING, SEEDW[0], SEEDW[1], f[6], sw[0], sw[1]);
|
||||
fflush(stdout); continue;
|
||||
}
|
||||
if (memcmp(kw, KEYW, 32) != 0) {
|
||||
printf("error %s day seed mismatch: this worker's cache is for key %08x %08x ..., the job's day seed %s gives %08x %08x ...; run igneum-miner export-pack and rebuild\n",
|
||||
jobId, KEYW[0], KEYW[1], f[7], kw[0], kw[1]);
|
||||
fflush(stdout); continue;
|
||||
}
|
||||
t0 = wallMs();
|
||||
if (memcmp(sw, cur->sw, 32) != 0 || memcmp(kw, cur->kw, 32) != 0) {
|
||||
if (prepared && memcmp(sw, prepared->sw, 32) == 0 && memcmp(kw, prepared->kw, 32) == 0) {
|
||||
/* The prepared pair: switch now, release the old one after this job */
|
||||
if (old) releasePair(old);
|
||||
old = cur; cur = prepared; prepared = NULL; switched = 1;
|
||||
printf("info switched to the prepared pair epoch %.16s day %s in %.2f ms\n", cur->epochHex, cur->dayHex, wallMs() - t0); fflush(stdout);
|
||||
} else if (memcmp(sw, cur->sw, 32) != 0) {
|
||||
printf("error %s epoch seed mismatch: this worker holds %s%s (seed words %08x %08x ...)%s, the job's epoch seed %.16s gives %08x %08x ...; send prepare with a pack directory, or run igneum-miner export-pack and rebuild\n",
|
||||
jobId, cur->epochHex[0] ? "prepared epoch " : "pack \"" IGNEUM_SEED_STRING "\"", cur->epochHex[0] ? cur->epochHex : "", cur->sw[0], cur->sw[1], prepared ? " plus one prepared pair" : "", f[6], sw[0], sw[1]);
|
||||
fflush(stdout); continue;
|
||||
} else {
|
||||
printf("error %s day seed mismatch: this worker's cache is for key %08x %08x ..., the job's day seed %s gives %08x %08x ...; send prepare with a pack directory, or run igneum-miner export-pack and rebuild\n",
|
||||
jobId, cur->kw[0], cur->kw[1], f[7], kw[0], kw[1]);
|
||||
fflush(stdout); continue;
|
||||
}
|
||||
}
|
||||
remaining = nonceCount;
|
||||
hi = (uint32_t)(nonceStart >> 32); lo = (uint32_t)nonceStart;
|
||||
while (remaining > 0) {
|
||||
|
|
@ -973,13 +1160,15 @@ static int runServe(Device* dv, const DeviceInfo* di, const Options* o) {
|
|||
seedWordsFromBytes(b, 49, iw);
|
||||
CL_CHECK(clEnqueueWriteBuffer(dv->q, dInit, CL_TRUE, 0, 32, iw, 0, NULL, NULL));
|
||||
baseNonce = lo;
|
||||
CL_CHECK(clSetKernelArg(dv->kHashBound, 0, sizeof(cl_mem), &dDs));
|
||||
CL_CHECK(clSetKernelArg(dv->kHashBound, 1, sizeof(cl_mem), &dOut));
|
||||
CL_CHECK(clSetKernelArg(dv->kHashBound, 2, sizeof(cl_uint), &baseNonce));
|
||||
CL_CHECK(clSetKernelArg(dv->kHashBound, 3, sizeof(cl_uint), &maskArg));
|
||||
CL_CHECK(clSetKernelArg(dv->kHashBound, 4, sizeof(cl_mem), &dInit));
|
||||
ev = launch1D(dv, dv->kHashBound, chunk, groupSize);
|
||||
err = clFinish(dv->q);
|
||||
CL_CHECK(clSetKernelArg(cur->kHashBound, 0, sizeof(cl_mem), &cur->ds));
|
||||
CL_CHECK(clSetKernelArg(cur->kHashBound, 1, sizeof(cl_mem), &dOut));
|
||||
CL_CHECK(clSetKernelArg(cur->kHashBound, 2, sizeof(cl_uint), &baseNonce));
|
||||
CL_CHECK(clSetKernelArg(cur->kHashBound, 3, sizeof(cl_uint), &maskArg));
|
||||
CL_CHECK(clSetKernelArg(cur->kHashBound, 4, sizeof(cl_mem), &dInit));
|
||||
ev = launch1D(dv, cur->kHashBound, chunk, groupSize);
|
||||
/* Wait on the dispatch event, not clFinish: the runtime can sleep the thread on an event, where clFinish
|
||||
* on some drivers spins one core for the whole dispatch. */
|
||||
err = clWaitForEvents(1, &ev);
|
||||
clReleaseEvent(ev);
|
||||
if (err != CL_SUCCESS) { printf("error %s dispatch failed: %s\n", jobId, clErrName(err)); fflush(stdout); failed = 1; break; }
|
||||
CL_CHECK(clEnqueueReadBuffer(dv->q, dOut, CL_TRUE, 0, (size_t)chunk * sizeof(uint64_t), hOut, 0, NULL, NULL));
|
||||
|
|
@ -994,12 +1183,15 @@ static int runServe(Device* dv, const DeviceInfo* di, const Options* o) {
|
|||
}
|
||||
if (failed) continue;
|
||||
printf("done %s %llu %.2f\n", jobId, (unsigned long long)hashes, wallMs() - t0);
|
||||
if (switched && old) { releasePair(old); old = NULL; printf("info dropped the previous pair (its program, cache and dataset)\n"); }
|
||||
fflush(stdout);
|
||||
}
|
||||
free(hOut);
|
||||
clReleaseMemObject(dInit);
|
||||
clReleaseMemObject(dOut);
|
||||
clReleaseMemObject(dDs);
|
||||
if (old) releasePair(old);
|
||||
if (prepared) releasePair(prepared);
|
||||
releasePair(cur);
|
||||
return 0;
|
||||
#endif
|
||||
}
|
||||
|
|
|
|||
File diff suppressed because one or more lines are too long
|
|
@ -1,5 +1,5 @@
|
|||
{
|
||||
"updated": "2026-10-03",
|
||||
"updated": "2026-10-04",
|
||||
"stage": "phase-3",
|
||||
"phases": [
|
||||
{
|
||||
|
|
@ -50,6 +50,54 @@
|
|||
}
|
||||
],
|
||||
"log": [
|
||||
{
|
||||
"date": "2026-10-04",
|
||||
"text": "Sim/economy: mining versus proving under stress, agent-based"
|
||||
},
|
||||
{
|
||||
"date": "2026-10-04",
|
||||
"text": "Difficulty rule under attack: pool hopping, pulsed rental, timestamp stretching, short-lane oscillation, epoch games, polluted window, block flood"
|
||||
},
|
||||
{
|
||||
"date": "2026-10-04",
|
||||
"text": "Difficulty rule: timestamp attack fixed , simulator regression, 3-node forger test"
|
||||
},
|
||||
{
|
||||
"date": "2026-10-04",
|
||||
"text": "Devnet-v4 integration: nine branches merged, 3-node test network on the merged node, Windows cross-build"
|
||||
},
|
||||
{
|
||||
"date": "2026-10-03",
|
||||
"text": "Per-identity hash rate \"decay\" on the RTX 5090: diagnosis and Metal reproduction"
|
||||
},
|
||||
{
|
||||
"date": "2026-10-03",
|
||||
"text": "Igneum-node devnet v2: sustained-mining finality rule v2 on a four-miner test network, and as a follower of the live devnet"
|
||||
},
|
||||
{
|
||||
"date": "2026-10-03",
|
||||
"text": "Execution layer devnet v3: revm over the selected chain, 3-node simnet, viem smoke test"
|
||||
},
|
||||
{
|
||||
"date": "2026-10-03",
|
||||
"text": "Weak-program census: 400,000 program runs through the CPU reference, the redundant-load finding, and the rules for M5 and M6"
|
||||
},
|
||||
{
|
||||
"date": "2026-10-03",
|
||||
"text": "Proving v0: first SP1 proof of an Igneum block, Apple M5 Max CPU, loaded machine"
|
||||
},
|
||||
{
|
||||
"date": "2026-10-03",
|
||||
"text": "Windows node package: igneumd cross-compiled for x86_64-pc-windows-gnu, two-peer sync test"
|
||||
},
|
||||
{
|
||||
"date": "2026-10-03",
|
||||
"text": "R3.26 / M15: PoW checked after the cheap checks, cache-build cap, attack before and after"
|
||||
},
|
||||
{
|
||||
"date": "2026-10-03",
|
||||
"text": "Difficulty controller: devnet record, simulator, Igneum dual-lane rule, 3-node CPU test network"
|
||||
},
|
||||
{
|
||||
"date": "2026-10-03",
|
||||
"text": "Devnet started four months ahead of plan. Finality, the EVM layer and the Igneum difficulty controller are in build."
|
||||
|
|
|
|||
Loading…
Reference in a new issue