// emu_backend.cpp: the CUDA driver API and NVRTC as host functions for igneum-inline-bench on the Mac (no NVIDIA // GPU). The same shape as proto-cuda/nvrtc/emu/emu_backend.cpp: the bench's own code path is unchanged, the two // library tables are filled with stand-ins, and the five kernels are the real texts compiled by clang and run on // host threads through proto-cuda/emu (32 threads per warp, a barrier inside __shfl_xor_sync). What it proves: the // three bit-exact checks of inline_bench.cpp on the real kernel texts. What it cannot prove: that NVRTC accepts the // texts and that the driver runs them; that is the RTX 5090 run. Not part of any deliverable. // // check-mac.sh compiles the pack's kernel.cu and kernel_bound.cu into namespace emu_pack and the generated // kernel_inline.cu into namespace emu_inline (launch syntax rewritten by sed, as emu/emu.sh does), then links them // with this file and inline_bench.cpp built with -DIGNEUM_EMU. #include #include #include #include #include #include #include #include "../nvrtc/cuda_api.h" #include "cuda_runtime.h" // the shim (proto-cuda/emu) namespace emu_pack { struct IgneumInitWords { uint32_t w[8]; }; void igneum_cache_fill(uint32_t* cache, uint32_t nSegments); void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems); void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw); } namespace emu_inline { struct IgneumInitWords { uint32_t w[8]; }; void igneum_build_inline(uint32_t* ds, const uint32_t* cache, uint32_t nItems, uint32_t lineMask); void igneum_hash_inline(const uint32_t* cache, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw, uint32_t lineMask); } // ---- NVRTC stand-in: records the source and its headers, hands back a tag the module loader maps to the compiled-in kernels struct EmuProg { std::string src, name, log; std::map headers; std::vector exprs; int which = 0; }; static nvrtcResult e_version(int* major, int* minor) { *major = 12; *minor = 8; return NVRTC_SUCCESS; } static int ARCHS[] = { 75, 80, 86, 89, 90, 100, 120 }; static nvrtcResult e_numArchs(int* n) { *n = (int)(sizeof(ARCHS) / sizeof(ARCHS[0])); return NVRTC_SUCCESS; } static nvrtcResult e_archs(int* out) { for (size_t i = 0; i < sizeof(ARCHS) / sizeof(ARCHS[0]); ++i) out[i] = ARCHS[i]; return NVRTC_SUCCESS; } static nvrtcResult e_create(nvrtcProgram* prog, const char* src, const char* name, int numHeaders, const char* const* headers, const char* const* includeNames) { EmuProg* p = new EmuProg(); p->src = src ? src : ""; p->name = name ? name : ""; for (int i = 0; i < numHeaders; ++i) p->headers[includeNames[i]] = headers[i]; *prog = (nvrtcProgram)p; return NVRTC_SUCCESS; } static nvrtcResult e_destroy(nvrtcProgram* prog) { delete (EmuProg*)*prog; *prog = nullptr; return NVRTC_SUCCESS; } static nvrtcResult e_compile(nvrtcProgram prog, int numOptions, const char* const* options) { EmuProg* p = (EmuProg*)prog; bool arch = false, std17 = false; for (int i = 0; i < numOptions; ++i) { std::string o = options[i]; if (o.rfind("--gpu-architecture=", 0) == 0) arch = true; if (o == "--std=c++17") std17 = true; } if (!arch || !std17) { p->log = "emu-nvrtc: expected --gpu-architecture=... and --std=c++17"; return NVRTC_ERROR_COMPILATION; } if (p->name == "kernel.cu") p->which = 1; else if (p->name == "kernel_bound.cu") p->which = 2; else if (p->name == "kernel_inline.cu") p->which = 3; else { p->log = "emu-nvrtc: unknown program " + p->name; return NVRTC_ERROR_COMPILATION; } // the texts handed over must be the ones check-mac.sh compiled: the same bytes are read from the same files, so // the sizes recorded here are what the bench prints; the hashes are printed by the bench itself std::printf("info emu-nvrtc: %s %zu bytes, headers %zu, compiled in as unit %d\n", p->name.c_str(), p->src.size(), p->headers.size(), p->which); return NVRTC_SUCCESS; } static nvrtcResult e_logSize(nvrtcProgram prog, size_t* n) { *n = ((EmuProg*)prog)->log.size() + 1; return NVRTC_SUCCESS; } static nvrtcResult e_log(nvrtcProgram prog, char* out) { std::strcpy(out, ((EmuProg*)prog)->log.c_str()); return NVRTC_SUCCESS; } static std::string image(EmuProg* p) { return "EMU-IMAGE:" + std::to_string(p->which); } static nvrtcResult e_imgSize(nvrtcProgram prog, size_t* n) { *n = image((EmuProg*)prog).size() + 1; return NVRTC_SUCCESS; } static nvrtcResult e_img(nvrtcProgram prog, char* out) { std::strcpy(out, image((EmuProg*)prog).c_str()); return NVRTC_SUCCESS; } static nvrtcResult e_addName(nvrtcProgram prog, const char* e) { ((EmuProg*)prog)->exprs.push_back(e); return NVRTC_SUCCESS; } static nvrtcResult e_lowered(nvrtcProgram prog, const char* e, const char** out) { EmuProg* p = (EmuProg*)prog; for (const std::string& s : p->exprs) if (s == e) { *out = s.c_str(); return NVRTC_SUCCESS; } return NVRTC_ERROR_NAME_EXPRESSION_NOT_VALID; } static const char* e_errstr(nvrtcResult r) { return r == NVRTC_SUCCESS ? "NVRTC_SUCCESS" : r == NVRTC_ERROR_COMPILATION ? "NVRTC_ERROR_COMPILATION" : "NVRTC_ERROR (emulated)"; } void emu_fill_nvrtc(Rtc& r) { r.version = e_version; r.getNumSupportedArchs = e_numArchs; r.getSupportedArchs = e_archs; r.createProgram = e_create; r.destroyProgram = e_destroy; r.compileProgram = e_compile; r.getProgramLogSize = e_logSize; r.getProgramLog = e_log; r.getPTXSize = e_imgSize; r.getPTX = e_img; r.getCUBINSize = e_imgSize; r.getCUBIN = e_img; r.addNameExpression = e_addName; r.getLoweredName = e_lowered; r.getErrorString = e_errstr; } // ---- Driver API stand-in static CUresult d_init(unsigned) { return CUDA_SUCCESS; } static CUresult d_driverVersion(int* v) { *v = 12080; return CUDA_SUCCESS; } static CUresult d_count(int* c) { *c = 1; return CUDA_SUCCESS; } static CUresult d_get(CUdevice* d, int i) { *d = i; return CUDA_SUCCESS; } static CUresult d_name(char* out, int len, CUdevice) { std::snprintf(out, (size_t)len, "CPU emulation shim (not a GPU)"); return CUDA_SUCCESS; } static CUresult d_attr(int* v, CUdevice_attribute a, CUdevice) { switch (a) { case CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MAJOR: *v = 12; break; case CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MINOR: *v = 0; break; case CU_DEVICE_ATTRIBUTE_WARP_SIZE: *v = 32; break; default: *v = 0; break; } return CUDA_SUCCESS; } static CUresult d_totalMem(size_t* b, CUdevice) { *b = 16ull << 30; return CUDA_SUCCESS; } static CUresult d_ctxFlags(CUdevice, unsigned) { return CUDA_SUCCESS; } static int gCtx = 0; static CUresult d_ctxRetain(CUcontext* c, CUdevice) { *c = (CUcontext)&gCtx; return CUDA_SUCCESS; } static CUresult d_ctxRelease(CUdevice) { return CUDA_SUCCESS; } static CUresult d_ctxSet(CUcontext) { return CUDA_SUCCESS; } static CUresult d_ctxSync() { return CUDA_SUCCESS; } static CUresult d_memInfo(size_t* f, size_t* t) { *f = 8ull << 30; *t = 16ull << 30; return CUDA_SUCCESS; } static CUresult d_alloc(CUdeviceptr* p, size_t n) { void* m = std::malloc(n); if (!m) return CUDA_ERROR_OUT_OF_MEMORY; *p = (CUdeviceptr)(uintptr_t)m; return CUDA_SUCCESS; } static CUresult d_free(CUdeviceptr p) { std::free((void*)(uintptr_t)p); return CUDA_SUCCESS; } static CUresult d_dtoh(void* dst, CUdeviceptr src, size_t n) { std::memcpy(dst, (const void*)(uintptr_t)src, n); return CUDA_SUCCESS; } static CUresult d_modLoad(CUmodule* m, const void* img) { const char* s = (const char*)img; if (std::strncmp(s, "EMU-IMAGE:", 10) != 0) return CUDA_ERROR_INVALID_IMAGE; *m = (CUmodule)(uintptr_t)(s[10] - '0'); return CUDA_SUCCESS; } static CUresult d_modUnload(CUmodule) { return CUDA_SUCCESS; } static CUresult d_getFn(CUfunction* f, CUmodule m, const char* name) { int unit = (int)(uintptr_t)m, idx = 0; if (unit == 1 && std::strcmp(name, "igneum_cache_fill") == 0) idx = 1; else if (unit == 1 && std::strcmp(name, "igneum_build") == 0) idx = 2; else if (unit == 2 && std::strcmp(name, "igneum_hash_bound") == 0) idx = 3; else if (unit == 3 && std::strcmp(name, "igneum_build_inline") == 0) idx = 4; else if (unit == 3 && std::strcmp(name, "igneum_hash_inline") == 0) idx = 5; else return CUDA_ERROR_NOT_FOUND; *f = (CUfunction)(uintptr_t)idx; return CUDA_SUCCESS; } template static T arg(void** params, int i) { return *(T*)params[i]; } template static T* dptr(void** params, int i) { return (T*)(uintptr_t)(*(CUdeviceptr*)params[i]); } static CUresult d_launch(CUfunction f, unsigned gx, unsigned, unsigned, unsigned bx, unsigned, unsigned, unsigned, CUstream, void** params, void**) { switch ((int)(uintptr_t)f) { case 1: emu_launch(emu_pack::igneum_cache_fill, gx, bx, dptr(params, 0), arg(params, 1)); return CUDA_SUCCESS; case 2: emu_launch(emu_pack::igneum_build, gx, bx, dptr(params, 0), dptr(params, 1), arg(params, 2)); return CUDA_SUCCESS; case 3: emu_launch(emu_pack::igneum_hash_bound, gx, bx, dptr(params, 0), dptr(params, 1), arg(params, 2), arg(params, 3), arg(params, 4)); return CUDA_SUCCESS; case 4: emu_launch(emu_inline::igneum_build_inline, gx, bx, dptr(params, 0), dptr(params, 1), arg(params, 2), arg(params, 3)); return CUDA_SUCCESS; case 5: emu_launch(emu_inline::igneum_hash_inline, gx, bx, dptr(params, 0), dptr(params, 1), arg(params, 2), arg(params, 3), arg(params, 4), arg(params, 5)); return CUDA_SUCCESS; default: return CUDA_ERROR_INVALID_HANDLE; } } static CUresult d_streamCreate(CUstream* s, unsigned) { *s = (CUstream)&gCtx; return CUDA_SUCCESS; } static CUresult d_streamSync(CUstream) { return CUDA_SUCCESS; } static CUresult d_streamDestroy(CUstream) { return CUDA_SUCCESS; } static CUresult d_funcAttr(int* v, CUfunction_attribute, CUfunction) { *v = 0; return CUDA_SUCCESS; } static CUresult d_occ(int* n, CUfunction, int, size_t) { *n = 0; return CUDA_SUCCESS; } static CUresult d_errStr(CUresult r, const char** s) { *s = r == CUDA_SUCCESS ? "no error" : "emulated driver error"; return CUDA_SUCCESS; } static CUresult d_errName(CUresult r, const char** s) { *s = r == CUDA_SUCCESS ? "CUDA_SUCCESS" : "CUDA_ERROR (emulated)"; return CUDA_SUCCESS; } void emu_fill_driver(Drv& d) { d.init = d_init; d.driverGetVersion = d_driverVersion; d.deviceGetCount = d_count; d.deviceGet = d_get; d.deviceGetName = d_name; d.deviceGetAttribute = d_attr; d.deviceTotalMem = d_totalMem; d.primaryCtxSetFlags = d_ctxFlags; d.primaryCtxRetain = d_ctxRetain; d.primaryCtxRelease = d_ctxRelease; d.ctxSetCurrent = d_ctxSet; d.ctxSynchronize = d_ctxSync; d.memGetInfo = d_memInfo; d.memAlloc = d_alloc; d.memFree = d_free; d.memcpyDtoH = d_dtoh; d.moduleLoadData = d_modLoad; d.moduleUnload = d_modUnload; d.moduleGetFunction = d_getFn; d.launchKernel = d_launch; d.streamCreate = d_streamCreate; d.streamSynchronize = d_streamSync; d.streamDestroy = d_streamDestroy; d.funcGetAttribute = d_funcAttr; d.occupancy = d_occ; d.getErrorString = d_errStr; d.getErrorName = d_errName; }