igneum/proto-cuda/inline-bench/emu_backend.cpp

150 lines
11 KiB
C++

// emu_backend.cpp: the CUDA driver API and NVRTC as host functions for igneum-inline-bench on the Mac (no NVIDIA
// GPU). The same shape as proto-cuda/nvrtc/emu/emu_backend.cpp: the bench's own code path is unchanged, the two
// library tables are filled with stand-ins, and the five kernels are the real texts compiled by clang and run on
// host threads through proto-cuda/emu (32 threads per warp, a barrier inside __shfl_xor_sync). What it proves: the
// three bit-exact checks of inline_bench.cpp on the real kernel texts. What it cannot prove: that NVRTC accepts the
// texts and that the driver runs them; that is the RTX 5090 run. Not part of any deliverable.
//
// check-mac.sh compiles the pack's kernel.cu and kernel_bound.cu into namespace emu_pack and the generated
// kernel_inline.cu into namespace emu_inline (launch syntax rewritten by sed, as emu/emu.sh does), then links them
// with this file and inline_bench.cpp built with -DIGNEUM_EMU.
#include <cstdint>
#include <cstdio>
#include <cstdlib>
#include <cstring>
#include <string>
#include <vector>
#include <map>
#include "../nvrtc/cuda_api.h"
#include "cuda_runtime.h" // the shim (proto-cuda/emu)
namespace emu_pack {
struct IgneumInitWords { uint32_t w[8]; };
void igneum_cache_fill(uint32_t* cache, uint32_t nSegments);
void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);
void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw);
}
namespace emu_inline {
struct IgneumInitWords { uint32_t w[8]; };
void igneum_build_inline(uint32_t* ds, const uint32_t* cache, uint32_t nItems, uint32_t lineMask);
void igneum_hash_inline(const uint32_t* cache, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw, uint32_t lineMask);
}
// ---- NVRTC stand-in: records the source and its headers, hands back a tag the module loader maps to the compiled-in kernels
struct EmuProg { std::string src, name, log; std::map<std::string, std::string> headers; std::vector<std::string> exprs; int which = 0; };
static nvrtcResult e_version(int* major, int* minor) { *major = 12; *minor = 8; return NVRTC_SUCCESS; }
static int ARCHS[] = { 75, 80, 86, 89, 90, 100, 120 };
static nvrtcResult e_numArchs(int* n) { *n = (int)(sizeof(ARCHS) / sizeof(ARCHS[0])); return NVRTC_SUCCESS; }
static nvrtcResult e_archs(int* out) { for (size_t i = 0; i < sizeof(ARCHS) / sizeof(ARCHS[0]); ++i) out[i] = ARCHS[i]; return NVRTC_SUCCESS; }
static nvrtcResult e_create(nvrtcProgram* prog, const char* src, const char* name, int numHeaders, const char* const* headers, const char* const* includeNames) {
EmuProg* p = new EmuProg(); p->src = src ? src : ""; p->name = name ? name : "";
for (int i = 0; i < numHeaders; ++i) p->headers[includeNames[i]] = headers[i];
*prog = (nvrtcProgram)p; return NVRTC_SUCCESS;
}
static nvrtcResult e_destroy(nvrtcProgram* prog) { delete (EmuProg*)*prog; *prog = nullptr; return NVRTC_SUCCESS; }
static nvrtcResult e_compile(nvrtcProgram prog, int numOptions, const char* const* options) {
EmuProg* p = (EmuProg*)prog;
bool arch = false, std17 = false;
for (int i = 0; i < numOptions; ++i) { std::string o = options[i]; if (o.rfind("--gpu-architecture=", 0) == 0) arch = true; if (o == "--std=c++17") std17 = true; }
if (!arch || !std17) { p->log = "emu-nvrtc: expected --gpu-architecture=... and --std=c++17"; return NVRTC_ERROR_COMPILATION; }
if (p->name == "kernel.cu") p->which = 1; else if (p->name == "kernel_bound.cu") p->which = 2; else if (p->name == "kernel_inline.cu") p->which = 3;
else { p->log = "emu-nvrtc: unknown program " + p->name; return NVRTC_ERROR_COMPILATION; }
// the texts handed over must be the ones check-mac.sh compiled: the same bytes are read from the same files, so
// the sizes recorded here are what the bench prints; the hashes are printed by the bench itself
std::printf("info emu-nvrtc: %s %zu bytes, headers %zu, compiled in as unit %d\n", p->name.c_str(), p->src.size(), p->headers.size(), p->which);
return NVRTC_SUCCESS;
}
static nvrtcResult e_logSize(nvrtcProgram prog, size_t* n) { *n = ((EmuProg*)prog)->log.size() + 1; return NVRTC_SUCCESS; }
static nvrtcResult e_log(nvrtcProgram prog, char* out) { std::strcpy(out, ((EmuProg*)prog)->log.c_str()); return NVRTC_SUCCESS; }
static std::string image(EmuProg* p) { return "EMU-IMAGE:" + std::to_string(p->which); }
static nvrtcResult e_imgSize(nvrtcProgram prog, size_t* n) { *n = image((EmuProg*)prog).size() + 1; return NVRTC_SUCCESS; }
static nvrtcResult e_img(nvrtcProgram prog, char* out) { std::strcpy(out, image((EmuProg*)prog).c_str()); return NVRTC_SUCCESS; }
static nvrtcResult e_addName(nvrtcProgram prog, const char* e) { ((EmuProg*)prog)->exprs.push_back(e); return NVRTC_SUCCESS; }
static nvrtcResult e_lowered(nvrtcProgram prog, const char* e, const char** out) {
EmuProg* p = (EmuProg*)prog;
for (const std::string& s : p->exprs) if (s == e) { *out = s.c_str(); return NVRTC_SUCCESS; }
return NVRTC_ERROR_NAME_EXPRESSION_NOT_VALID;
}
static const char* e_errstr(nvrtcResult r) { return r == NVRTC_SUCCESS ? "NVRTC_SUCCESS" : r == NVRTC_ERROR_COMPILATION ? "NVRTC_ERROR_COMPILATION" : "NVRTC_ERROR (emulated)"; }
void emu_fill_nvrtc(Rtc& r) {
r.version = e_version; r.getNumSupportedArchs = e_numArchs; r.getSupportedArchs = e_archs;
r.createProgram = e_create; r.destroyProgram = e_destroy; r.compileProgram = e_compile;
r.getProgramLogSize = e_logSize; r.getProgramLog = e_log;
r.getPTXSize = e_imgSize; r.getPTX = e_img; r.getCUBINSize = e_imgSize; r.getCUBIN = e_img;
r.addNameExpression = e_addName; r.getLoweredName = e_lowered; r.getErrorString = e_errstr;
}
// ---- Driver API stand-in
static CUresult d_init(unsigned) { return CUDA_SUCCESS; }
static CUresult d_driverVersion(int* v) { *v = 12080; return CUDA_SUCCESS; }
static CUresult d_count(int* c) { *c = 1; return CUDA_SUCCESS; }
static CUresult d_get(CUdevice* d, int i) { *d = i; return CUDA_SUCCESS; }
static CUresult d_name(char* out, int len, CUdevice) { std::snprintf(out, (size_t)len, "CPU emulation shim (not a GPU)"); return CUDA_SUCCESS; }
static CUresult d_attr(int* v, CUdevice_attribute a, CUdevice) {
switch (a) {
case CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MAJOR: *v = 12; break;
case CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MINOR: *v = 0; break;
case CU_DEVICE_ATTRIBUTE_WARP_SIZE: *v = 32; break;
default: *v = 0; break;
}
return CUDA_SUCCESS;
}
static CUresult d_totalMem(size_t* b, CUdevice) { *b = 16ull << 30; return CUDA_SUCCESS; }
static CUresult d_ctxFlags(CUdevice, unsigned) { return CUDA_SUCCESS; }
static int gCtx = 0;
static CUresult d_ctxRetain(CUcontext* c, CUdevice) { *c = (CUcontext)&gCtx; return CUDA_SUCCESS; }
static CUresult d_ctxRelease(CUdevice) { return CUDA_SUCCESS; }
static CUresult d_ctxSet(CUcontext) { return CUDA_SUCCESS; }
static CUresult d_ctxSync() { return CUDA_SUCCESS; }
static CUresult d_memInfo(size_t* f, size_t* t) { *f = 8ull << 30; *t = 16ull << 30; return CUDA_SUCCESS; }
static CUresult d_alloc(CUdeviceptr* p, size_t n) { void* m = std::malloc(n); if (!m) return CUDA_ERROR_OUT_OF_MEMORY; *p = (CUdeviceptr)(uintptr_t)m; return CUDA_SUCCESS; }
static CUresult d_free(CUdeviceptr p) { std::free((void*)(uintptr_t)p); return CUDA_SUCCESS; }
static CUresult d_dtoh(void* dst, CUdeviceptr src, size_t n) { std::memcpy(dst, (const void*)(uintptr_t)src, n); return CUDA_SUCCESS; }
static CUresult d_modLoad(CUmodule* m, const void* img) {
const char* s = (const char*)img;
if (std::strncmp(s, "EMU-IMAGE:", 10) != 0) return CUDA_ERROR_INVALID_IMAGE;
*m = (CUmodule)(uintptr_t)(s[10] - '0');
return CUDA_SUCCESS;
}
static CUresult d_modUnload(CUmodule) { return CUDA_SUCCESS; }
static CUresult d_getFn(CUfunction* f, CUmodule m, const char* name) {
int unit = (int)(uintptr_t)m, idx = 0;
if (unit == 1 && std::strcmp(name, "igneum_cache_fill") == 0) idx = 1;
else if (unit == 1 && std::strcmp(name, "igneum_build") == 0) idx = 2;
else if (unit == 2 && std::strcmp(name, "igneum_hash_bound") == 0) idx = 3;
else if (unit == 3 && std::strcmp(name, "igneum_build_inline") == 0) idx = 4;
else if (unit == 3 && std::strcmp(name, "igneum_hash_inline") == 0) idx = 5;
else return CUDA_ERROR_NOT_FOUND;
*f = (CUfunction)(uintptr_t)idx;
return CUDA_SUCCESS;
}
template <class T> static T arg(void** params, int i) { return *(T*)params[i]; }
template <class T> static T* dptr(void** params, int i) { return (T*)(uintptr_t)(*(CUdeviceptr*)params[i]); }
static CUresult d_launch(CUfunction f, unsigned gx, unsigned, unsigned, unsigned bx, unsigned, unsigned, unsigned, CUstream, void** params, void**) {
switch ((int)(uintptr_t)f) {
case 1: emu_launch(emu_pack::igneum_cache_fill, gx, bx, dptr<uint32_t>(params, 0), arg<uint32_t>(params, 1)); return CUDA_SUCCESS;
case 2: emu_launch(emu_pack::igneum_build, gx, bx, dptr<uint32_t>(params, 0), dptr<const uint32_t>(params, 1), arg<uint32_t>(params, 2)); return CUDA_SUCCESS;
case 3: emu_launch(emu_pack::igneum_hash_bound, gx, bx, dptr<const uint32_t>(params, 0), dptr<uint64_t>(params, 1), arg<uint32_t>(params, 2), arg<uint32_t>(params, 3), arg<emu_pack::IgneumInitWords>(params, 4)); return CUDA_SUCCESS;
case 4: emu_launch(emu_inline::igneum_build_inline, gx, bx, dptr<uint32_t>(params, 0), dptr<const uint32_t>(params, 1), arg<uint32_t>(params, 2), arg<uint32_t>(params, 3)); return CUDA_SUCCESS;
case 5: emu_launch(emu_inline::igneum_hash_inline, gx, bx, dptr<const uint32_t>(params, 0), dptr<uint64_t>(params, 1), arg<uint32_t>(params, 2), arg<uint32_t>(params, 3), arg<emu_inline::IgneumInitWords>(params, 4), arg<uint32_t>(params, 5)); return CUDA_SUCCESS;
default: return CUDA_ERROR_INVALID_HANDLE;
}
}
static CUresult d_streamCreate(CUstream* s, unsigned) { *s = (CUstream)&gCtx; return CUDA_SUCCESS; }
static CUresult d_streamSync(CUstream) { return CUDA_SUCCESS; }
static CUresult d_streamDestroy(CUstream) { return CUDA_SUCCESS; }
static CUresult d_funcAttr(int* v, CUfunction_attribute, CUfunction) { *v = 0; return CUDA_SUCCESS; }
static CUresult d_occ(int* n, CUfunction, int, size_t) { *n = 0; return CUDA_SUCCESS; }
static CUresult d_errStr(CUresult r, const char** s) { *s = r == CUDA_SUCCESS ? "no error" : "emulated driver error"; return CUDA_SUCCESS; }
static CUresult d_errName(CUresult r, const char** s) { *s = r == CUDA_SUCCESS ? "CUDA_SUCCESS" : "CUDA_ERROR (emulated)"; return CUDA_SUCCESS; }
void emu_fill_driver(Drv& d) {
d.init = d_init; d.driverGetVersion = d_driverVersion; d.deviceGetCount = d_count; d.deviceGet = d_get; d.deviceGetName = d_name;
d.deviceGetAttribute = d_attr; d.deviceTotalMem = d_totalMem; d.primaryCtxSetFlags = d_ctxFlags; d.primaryCtxRetain = d_ctxRetain;
d.primaryCtxRelease = d_ctxRelease; d.ctxSetCurrent = d_ctxSet; d.ctxSynchronize = d_ctxSync; d.memGetInfo = d_memInfo;
d.memAlloc = d_alloc; d.memFree = d_free; d.memcpyDtoH = d_dtoh; d.moduleLoadData = d_modLoad; d.moduleUnload = d_modUnload;
d.moduleGetFunction = d_getFn; d.launchKernel = d_launch; d.streamCreate = d_streamCreate; d.streamSynchronize = d_streamSync;
d.streamDestroy = d_streamDestroy; d.funcGetAttribute = d_funcAttr; d.occupancy = d_occ; d.getErrorString = d_errStr; d.getErrorName = d_errName;
}