igneum/proto-cuda/nvrtc/emu/emu_backend.cpp
igneum-labs 83c414a03b OpenCL worker fault guard, miner-side fault state in the launcher, NVRTC annotation rule in the emulation
After PC 2's gfx1036 (prebuilt-generic path) completed 2,000 jobs a second with no hash from 600 s on: every OpenCL
call in host.c's job path is now fatal on error (exit 3, the miner restarts the worker), the dispatch event must read
CL_COMPLETE, a chunk 20x faster per nonce than the running mean or an output buffer unchanged since the previous
dispatch is a fault, and a stats line every 200 jobs carries the live event and buffer counts (a leak over 16 events
or 12 buffers is fatal too). IGNEUM_FAULT_TEST=N exercises the detectors on a healthy device (verified on Apple
OpenCL: the stale-output guard fires on the chunk after the injected fault). The launcher shows 'worker fault' and
'restarting' for the card on the miner's WORKER FAULT line and drops the last rate. The emulation's NVRTC stand-in
now applies NVRTC's execution-space rule (program.h(46) igneum_launch_* declarations are host code unless
-default-device or -DIGNEUM_NO_CUDA), which is what the RTX 5090 reported; test.sh checks the rejection.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
2026-10-04 10:42:36 +00:00

250 lines
17 KiB
C++

// emu_backend.cpp: the CUDA driver API and NVRTC replaced by host functions, for checking igneum-worker-cuda on a
// machine without an NVIDIA GPU (the Mac). Not part of the deliverable. 4 October 2026.
//
// What it proves: the worker's protocol, job walk, init words, pack reading, self-test, prepare thread and swap,
// with the pack's real kernel text running on host threads through proto-cuda/emu (the same shim the host.cu
// emulation uses, 32 threads per warp, a barrier inside __shfl_xor_sync). What it cannot prove: that NVRTC accepts
// the text and that the driver runs it; that is the RTX 5090 run (windows-app/TEST.md).
//
// The NVRTC stand-in records every program the worker creates and checks it against the pack on disk: the source
// must be the pack's kernel file up to the host launch wrappers, byte for byte, the dropped tail must hold host
// wrappers only, and program.h and memhard.h must be the pack's files byte for byte. Two packs may be compiled in
// (IGNEUM_EMU_PACK and IGNEUM_EMU_PACK2, the directories), so a prepare and the swap run with real second-pack code.
#include <cstdint>
#include <cstdio>
#include <cstdlib>
#include <cstring>
#include <string>
#include <vector>
#include <list>
#include <map>
#include <regex>
#include <sstream>
#include "../cuda_api.h" // the real cuda.h / nvrtc.h types
#include "cuda_runtime.h" // the shim (proto-cuda/emu): emu_launch and the kernel symbols' world
namespace emu_pack_a {
struct IgneumInitWords { uint32_t w[8]; }; // the same definition as the pack's kernel_bound.cu, inside its namespace
void igneum_cache_fill(uint32_t* cache, uint32_t nSegments);
void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);
void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw);
}
#ifdef IGNEUM_EMU_TWO_PACKS
namespace emu_pack_b {
struct IgneumInitWords { uint32_t w[8]; };
void igneum_cache_fill(uint32_t* cache, uint32_t nSegments);
void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);
void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw);
}
#endif
static std::string readAll(const std::string& path) {
FILE* f = std::fopen(path.c_str(), "rb");
if (!f) return std::string();
std::string s;
char buf[65536];
size_t n;
while ((n = std::fread(buf, 1, sizeof(buf), f)) > 0) s.append(buf, n);
std::fclose(f);
return s;
}
// ---------------------------------------------------------------------------------------------
// NVRTC stand-in
struct EmuProg {
std::string src, name, log;
std::map<std::string, std::string> headers;
std::list<std::string> exprs;
int pack = 0; // 1 = IGNEUM_EMU_PACK, 2 = IGNEUM_EMU_PACK2, 0 = unknown
bool checked = false, ok = false, compiled = false;
};
static const char* packDir(int which) {
const char* e = std::getenv(which == 1 ? "IGNEUM_EMU_PACK" : "IGNEUM_EMU_PACK2");
return e ? e : "";
}
// NVRTC's rule, which bit on the RTX 5090 on 4 October 2026 ("A function without execution space annotations
// (__host__/__device__/__global__) is considered a host function, and host functions are not allowed in JIT mode",
// program.h line 46, the igneum_launch_* declarations): every function declaration or definition in the source and in
// the pack headers must carry an annotation (IGNEUM_HD expands to one under __CUDACC__, which NVRTC defines), unless
// the compile passes -default-device (then unannotated functions are device code) or the text is inside an
// #ifndef IGNEUM_NO_CUDA block and -DIGNEUM_NO_CUDA was passed. Returns the first offender as "file(line): text" or "".
// IGNEUM_EMU_NO_DEFAULT_DEVICE=1 makes the stand-in ignore -default-device, to test this check.
static std::string firstUnannotated(const std::string& name, const std::string& text, bool noCuda) {
std::istringstream in(text);
std::string line;
int ln = 0, skip = 0, depth = 0;
std::vector<bool> skipping;
static const std::regex fn(R"(^\s*(static\s+|inline\s+|extern\s+)*[A-Za-z_][A-Za-z_0-9:<>]*\s*\**\s+\**\s*[A-Za-z_][A-Za-z_0-9]*\s*\()");
while (std::getline(in, line)) {
++ln;
std::string t = line;
if (t.rfind("#ifndef IGNEUM_NO_CUDA", 0) == 0) { skipping.push_back(noCuda); if (noCuda) ++skip; ++depth; continue; }
if (t.rfind("#if", 0) == 0) { skipping.push_back(false); ++depth; continue; }
if (t.rfind("#endif", 0) == 0) { if (!skipping.empty()) { if (skipping.back()) --skip; skipping.pop_back(); --depth; } continue; }
if (skip > 0) continue;
if (t.empty() || t[0] == '#' || t.find("//") == 0 || t.find("/*") == 0 || t.find(" *") == 0) continue;
if (t.find("__device__") != std::string::npos || t.find("__global__") != std::string::npos || t.find("__host__") != std::string::npos || t.find("IGNEUM_HD") != std::string::npos) continue;
if (t.find("typedef") != std::string::npos || t.find("struct ") == 0 || t.find("return") != std::string::npos || t.find("if (") != std::string::npos || t.find("for (") != std::string::npos || t.find("while (") != std::string::npos) continue;
if (std::regex_search(t, fn)) return name + "(" + std::to_string(ln) + "): " + t;
}
return "";
}
// The source check. Fills p->pack, p->ok, p->log.
static void checkSource(EmuProg* p, bool defaultDevice, bool noCuda) {
p->checked = true;
if (std::getenv("IGNEUM_EMU_NO_DEFAULT_DEVICE")) defaultDevice = false;
if (!defaultDevice) {
std::string bad = firstUnannotated(p->name, p->src, noCuda);
if (bad.empty() && p->headers.count("program.h")) bad = firstUnannotated("program.h", p->headers["program.h"], noCuda);
if (bad.empty() && p->headers.count("memhard.h")) bad = firstUnannotated("memhard.h", p->headers["memhard.h"], noCuda);
if (!bad.empty()) { p->log = "emu-nvrtc: " + bad + ": A function without execution space annotations (__host__/__device__/__global__) is considered a host function, and host functions are not allowed in JIT mode (pass -default-device or -DIGNEUM_NO_CUDA)"; p->ok = false; return; }
}
for (int which = 1; which <= 2; ++which) {
std::string dir = packDir(which);
if (dir.empty()) continue;
std::string file = readAll(dir + "/" + p->name);
if (file.empty() || file.size() < p->src.size() || file.compare(0, p->src.size(), p->src) != 0) continue;
std::string tail = file.substr(p->src.size());
bool tailOk = tail.empty() || tail.rfind("// Host-side launch wrappers", 0) == 0 || tail.rfind("cudaError_t ", 0) == 0;
tailOk = tailOk && tail.find("__global__") == std::string::npos && tail.find("__device__") == std::string::npos;
if (!tailOk) { p->log = "emu-nvrtc: the dropped tail of " + p->name + " is not the host launch wrappers"; p->ok = false; p->pack = which; return; }
std::string ph = readAll(dir + "/program.h"), mh = readAll(dir + "/memhard.h");
if (!p->headers.count("program.h") || p->headers["program.h"] != ph) { p->log = "emu-nvrtc: program.h handed to NVRTC differs from the pack's"; p->ok = false; p->pack = which; return; }
if (!p->headers.count("memhard.h") || p->headers["memhard.h"] != mh) { p->log = "emu-nvrtc: memhard.h handed to NVRTC differs from the pack's"; p->ok = false; p->pack = which; return; }
if (!p->headers.count("cuda_runtime.h") || !p->headers.count("cstdint")) { p->log = "emu-nvrtc: the stub headers cuda_runtime.h and cstdint were not handed over"; p->ok = false; p->pack = which; return; }
p->pack = which; p->ok = true;
std::printf("info emu-nvrtc: %s source check PASS for pack %d: %zu bytes handed over = the pack file's first %zu of %zu bytes; dropped tail = host launch wrappers only (%zu bytes); program.h (%zu bytes) and memhard.h (%zu bytes) byte-identical\n",
p->name.c_str(), which, p->src.size(), p->src.size(), file.size(), tail.size(), ph.size(), mh.size());
std::fflush(stdout);
return;
}
p->log = "emu-nvrtc: the source of " + p->name + " is not a prefix of any pack's file (IGNEUM_EMU_PACK / IGNEUM_EMU_PACK2)";
p->ok = false;
}
static nvrtcResult e_version(int* major, int* minor) { *major = 12; *minor = 8; return NVRTC_SUCCESS; }
static int ARCHS[] = { 50, 52, 53, 60, 61, 62, 70, 72, 75, 80, 86, 87, 89, 90, 100, 101, 120 };
static nvrtcResult e_numArchs(int* n) { *n = (int)(sizeof(ARCHS) / sizeof(ARCHS[0])); return NVRTC_SUCCESS; }
static nvrtcResult e_archs(int* out) { for (size_t i = 0; i < sizeof(ARCHS) / sizeof(ARCHS[0]); ++i) out[i] = ARCHS[i]; return NVRTC_SUCCESS; }
static nvrtcResult e_create(nvrtcProgram* prog, const char* src, const char* name, int numHeaders, const char* const* headers, const char* const* includeNames) {
EmuProg* p = new EmuProg();
p->src = src ? src : ""; p->name = name ? name : "";
for (int i = 0; i < numHeaders; ++i) p->headers[includeNames[i]] = headers[i];
*prog = (nvrtcProgram)p;
return NVRTC_SUCCESS;
}
static nvrtcResult e_destroy(nvrtcProgram* prog) { delete (EmuProg*)*prog; *prog = nullptr; return NVRTC_SUCCESS; }
static nvrtcResult e_compile(nvrtcProgram prog, int numOptions, const char* const* options) {
EmuProg* p = (EmuProg*)prog;
bool arch = false, std17 = false, defaultDevice = false, noCuda = false;
for (int i = 0; i < numOptions; ++i) { std::string o = options[i]; if (o.rfind("--gpu-architecture=", 0) == 0) arch = true; if (o == "--std=c++17") std17 = true; if (o == "-default-device" || o == "--device-as-default-execution-space") defaultDevice = true; if (o == "-DIGNEUM_NO_CUDA" || o == "-DIGNEUM_NO_CUDA=1") noCuda = true; }
if (!arch || !std17) { p->log = "emu-nvrtc: expected --gpu-architecture=... and --std=c++17"; return NVRTC_ERROR_COMPILATION; }
checkSource(p, defaultDevice, noCuda);
if (!p->ok) return NVRTC_ERROR_COMPILATION;
p->compiled = true;
return NVRTC_SUCCESS;
}
static nvrtcResult e_logSize(nvrtcProgram prog, size_t* n) { *n = ((EmuProg*)prog)->log.size() + 1; return NVRTC_SUCCESS; }
static nvrtcResult e_log(nvrtcProgram prog, char* out) { std::strcpy(out, ((EmuProg*)prog)->log.c_str()); return NVRTC_SUCCESS; }
static std::string image(EmuProg* p) { return std::string("EMU-IMAGE:") + (p->pack == 2 ? "B" : "A"); }
static nvrtcResult e_imgSize(nvrtcProgram prog, size_t* n) { *n = image((EmuProg*)prog).size() + 1; return NVRTC_SUCCESS; }
static nvrtcResult e_img(nvrtcProgram prog, char* out) { std::strcpy(out, image((EmuProg*)prog).c_str()); return NVRTC_SUCCESS; }
static nvrtcResult e_addName(nvrtcProgram prog, const char* e) { ((EmuProg*)prog)->exprs.push_back(e); return NVRTC_SUCCESS; }
static nvrtcResult e_lowered(nvrtcProgram prog, const char* e, const char** out) {
EmuProg* p = (EmuProg*)prog;
for (const std::string& s : p->exprs) if (s == e) { *out = s.c_str(); return NVRTC_SUCCESS; }
return NVRTC_ERROR_NAME_EXPRESSION_NOT_VALID;
}
static const char* e_errstr(nvrtcResult r) { return r == NVRTC_SUCCESS ? "NVRTC_SUCCESS" : r == NVRTC_ERROR_COMPILATION ? "NVRTC_ERROR_COMPILATION" : "NVRTC_ERROR (emulated)"; }
void emu_fill_nvrtc(Rtc& r) {
r.version = e_version; r.getNumSupportedArchs = e_numArchs; r.getSupportedArchs = e_archs;
r.createProgram = e_create; r.destroyProgram = e_destroy; r.compileProgram = e_compile;
r.getProgramLogSize = e_logSize; r.getProgramLog = e_log;
r.getPTXSize = e_imgSize; r.getPTX = e_img; r.getCUBINSize = e_imgSize; r.getCUBIN = e_img;
r.addNameExpression = e_addName; r.getLoweredName = e_lowered; r.getErrorString = e_errstr;
}
// ---------------------------------------------------------------------------------------------
// Driver API stand-in
static CUresult d_init(unsigned) { return CUDA_SUCCESS; }
static CUresult d_driverVersion(int* v) { *v = 12080; return CUDA_SUCCESS; }
static CUresult d_count(int* c) { *c = 1; return CUDA_SUCCESS; }
static CUresult d_get(CUdevice* d, int i) { *d = i; return CUDA_SUCCESS; }
static CUresult d_name(char* out, int len, CUdevice) { std::snprintf(out, (size_t)len, "CPU emulation shim (not a GPU)"); return CUDA_SUCCESS; }
static CUresult d_attr(int* v, CUdevice_attribute a, CUdevice) {
switch (a) {
case CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MAJOR: *v = 12; break;
case CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MINOR: *v = 0; break;
case CU_DEVICE_ATTRIBUTE_WARP_SIZE: *v = 32; break;
default: *v = 0; break;
}
return CUDA_SUCCESS;
}
static CUresult d_totalMem(size_t* b, CUdevice) { *b = 16ull << 30; return CUDA_SUCCESS; }
static CUresult d_ctxFlags(CUdevice, unsigned) { return CUDA_SUCCESS; }
static int gCtx = 0;
static CUresult d_ctxRetain(CUcontext* c, CUdevice) { *c = (CUcontext)&gCtx; return CUDA_SUCCESS; }
static CUresult d_ctxRelease(CUdevice) { return CUDA_SUCCESS; }
static CUresult d_ctxSet(CUcontext) { return CUDA_SUCCESS; }
static CUresult d_ctxSync() { return CUDA_SUCCESS; }
static CUresult d_memInfo(size_t* f, size_t* t) { *f = 8ull << 30; *t = 16ull << 30; return CUDA_SUCCESS; }
static CUresult d_alloc(CUdeviceptr* p, size_t n) { void* m = std::malloc(n); if (!m) return CUDA_ERROR_OUT_OF_MEMORY; *p = (CUdeviceptr)(uintptr_t)m; return CUDA_SUCCESS; }
static CUresult d_free(CUdeviceptr p) { std::free((void*)(uintptr_t)p); return CUDA_SUCCESS; }
static CUresult d_dtoh(void* dst, CUdeviceptr src, size_t n) { std::memcpy(dst, (const void*)(uintptr_t)src, n); return CUDA_SUCCESS; }
static CUresult d_modLoad(CUmodule* m, const void* img) {
const char* s = (const char*)img;
if (std::strncmp(s, "EMU-IMAGE:", 10) != 0) return CUDA_ERROR_INVALID_IMAGE;
*m = (CUmodule)(uintptr_t)(s[10] == 'B' ? 2 : 1);
return CUDA_SUCCESS;
}
static CUresult d_modUnload(CUmodule) { return CUDA_SUCCESS; }
static CUresult d_getFn(CUfunction* f, CUmodule m, const char* name) {
int pack = (int)(uintptr_t)m, idx;
if (std::strcmp(name, "igneum_cache_fill") == 0) idx = 1;
else if (std::strcmp(name, "igneum_build") == 0) idx = 2;
else if (std::strcmp(name, "igneum_hash_bound") == 0) idx = 3;
else return CUDA_ERROR_NOT_FOUND;
#ifndef IGNEUM_EMU_TWO_PACKS
if (pack == 2) return CUDA_ERROR_NOT_FOUND;
#endif
*f = (CUfunction)(uintptr_t)((pack - 1) * 3 + idx);
return CUDA_SUCCESS;
}
template <class T> static T arg(void** params, int i) { return *(T*)params[i]; }
template <class T> static T* dptr(void** params, int i) { return (T*)(uintptr_t)(*(CUdeviceptr*)params[i]); }
static CUresult d_launch(CUfunction f, unsigned gx, unsigned, unsigned, unsigned bx, unsigned, unsigned, unsigned, CUstream, void** params, void**) {
switch ((int)(uintptr_t)f) {
case 1: emu_launch(emu_pack_a::igneum_cache_fill, gx, bx, dptr<uint32_t>(params, 0), arg<uint32_t>(params, 1)); return CUDA_SUCCESS;
case 2: emu_launch(emu_pack_a::igneum_build, gx, bx, dptr<uint32_t>(params, 0), dptr<const uint32_t>(params, 1), arg<uint32_t>(params, 2)); return CUDA_SUCCESS;
case 3: emu_launch(emu_pack_a::igneum_hash_bound, gx, bx, dptr<const uint32_t>(params, 0), dptr<uint64_t>(params, 1), arg<uint32_t>(params, 2), arg<uint32_t>(params, 3), arg<emu_pack_a::IgneumInitWords>(params, 4)); return CUDA_SUCCESS;
#ifdef IGNEUM_EMU_TWO_PACKS
case 4: emu_launch(emu_pack_b::igneum_cache_fill, gx, bx, dptr<uint32_t>(params, 0), arg<uint32_t>(params, 1)); return CUDA_SUCCESS;
case 5: emu_launch(emu_pack_b::igneum_build, gx, bx, dptr<uint32_t>(params, 0), dptr<const uint32_t>(params, 1), arg<uint32_t>(params, 2)); return CUDA_SUCCESS;
case 6: emu_launch(emu_pack_b::igneum_hash_bound, gx, bx, dptr<const uint32_t>(params, 0), dptr<uint64_t>(params, 1), arg<uint32_t>(params, 2), arg<uint32_t>(params, 3), arg<emu_pack_b::IgneumInitWords>(params, 4)); return CUDA_SUCCESS;
#endif
default: return CUDA_ERROR_INVALID_HANDLE;
}
}
static CUresult d_streamCreate(CUstream* s, unsigned) { *s = (CUstream)&gCtx; return CUDA_SUCCESS; }
static CUresult d_streamSync(CUstream) { return CUDA_SUCCESS; }
static CUresult d_streamDestroy(CUstream) { return CUDA_SUCCESS; }
static CUresult d_funcAttr(int* v, CUfunction_attribute, CUfunction) { *v = 0; return CUDA_SUCCESS; }
static CUresult d_occ(int* n, CUfunction, int, size_t) { *n = 0; return CUDA_SUCCESS; }
static CUresult d_errStr(CUresult r, const char** s) { *s = r == CUDA_SUCCESS ? "no error" : "emulated driver error"; return CUDA_SUCCESS; }
static CUresult d_errName(CUresult r, const char** s) { *s = r == CUDA_SUCCESS ? "CUDA_SUCCESS" : "CUDA_ERROR (emulated)"; return CUDA_SUCCESS; }
void emu_fill_driver(Drv& d) {
d.init = d_init; d.driverGetVersion = d_driverVersion; d.deviceGetCount = d_count; d.deviceGet = d_get; d.deviceGetName = d_name;
d.deviceGetAttribute = d_attr; d.deviceTotalMem = d_totalMem; d.primaryCtxSetFlags = d_ctxFlags; d.primaryCtxRetain = d_ctxRetain;
d.primaryCtxRelease = d_ctxRelease; d.ctxSetCurrent = d_ctxSet; d.ctxSynchronize = d_ctxSync; d.memGetInfo = d_memInfo;
d.memAlloc = d_alloc; d.memFree = d_free; d.memcpyDtoH = d_dtoh; d.moduleLoadData = d_modLoad; d.moduleUnload = d_modUnload;
d.moduleGetFunction = d_getFn; d.launchKernel = d_launch; d.streamCreate = d_streamCreate; d.streamSynchronize = d_streamSync;
d.streamDestroy = d_streamDestroy; d.funcGetAttribute = d_funcAttr; d.occupancy = d_occ; d.getErrorString = d_errStr; d.getErrorName = d_errName;
}