igneum/infra/gpu-bench/nvrtc-time.cu
igneum-labs d8dc8c6ba3 infra: cloud devnet (20 nodes), rented GPU bench and seed node scripts; plans for both
infra/cloud-devnet: hcloud (doctl variant) create, builder-VM provision from a git-archive source tarball,
systemd units for igneumd --devnet-suffix with a sparse --addpeer mesh and a CPU trickle miner per node,
stdlib wRPC client, experiments (latency, partition, hop, collect, observer hookup), README with the command
sequence and the Hetzner API prices of 3 Oct 2026.
infra/gpu-bench: RunPod image recipes (CUDA 12.8, ROCm), bundle, run.sh (vectors gate, 10-min raw, sweep,
inline shortcut ratio, nvcc/NVRTC/OpenCL recompile timings, results row, intake upload), bench-log template.
infra/seed-nodes: create-seed (persistent IPv4, firewall), provision on the VM, health check, addPeer from the
Mac over grpcurl, seeds.txt; igneum-seed-1 created at 188.245.5.161 (Hetzner cx23, fsn1).
docs/plans/cloud-devnet.md and docs/plans/seed-nodes.md.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
2026-10-03 22:01:50 +00:00

82 lines
4.7 KiB
Text

// Times an in-process NVRTC compile of a pack's kernel (the device part of kernel.cu with program.h and memhard.h
// as in-memory headers), N times, then loads the cubin through the driver API to prove it is usable.
// This is the "hourly JIT in the miner process" figure of ledger M11 and M17; the worker today runs nvcc out of
// process (proto-cuda/host.cu, prepareCompile), which run.sh times separately.
// nvcc -O2 -std=c++17 -o nvrtc-time nvrtc-time.cu -lnvrtc -lcuda
// ./nvrtc-time <pack-dir> [iterations=3]
// Prints one line per iteration (compile ms, cubin bytes, load ms) and "median <ms>".
#include <nvrtc.h>
#include <cuda.h>
#include <algorithm>
#include <chrono>
#include <cstdio>
#include <cstdlib>
#include <fstream>
#include <sstream>
#include <string>
#include <vector>
static std::string readFile(const std::string& p) {
std::ifstream f(p); if (!f) { std::fprintf(stderr, "cannot read %s\n", p.c_str()); std::exit(2); }
std::stringstream ss; ss << f.rdbuf(); return ss.str();
}
// Drop the includes NVRTC cannot see and the host launch wrappers; the typedefs replace <cstdint>.
static std::string deviceOnly(const std::string& src, bool cutHost) {
std::istringstream in(src); std::string line, out = "typedef unsigned int uint32_t;\ntypedef unsigned long long uint64_t;\ntypedef unsigned char uint8_t;\ntypedef unsigned long size_t;\n";
while (std::getline(in, line)) {
if (cutHost && line.rfind("cudaError_t igneum_launch", 0) == 0) break;
if (line.find("#include <cuda_runtime.h>") != std::string::npos || line.find("#include <cstdint>") != std::string::npos ||
line.find("#include <stdint.h>") != std::string::npos || line.find("#include <stddef.h>") != std::string::npos) { out += "\n"; continue; }
out += line + "\n";
}
return out;
}
static double ms() { return std::chrono::duration<double, std::milli>(std::chrono::steady_clock::now().time_since_epoch()).count(); }
int main(int argc, char** argv) {
if (argc < 2) { std::fprintf(stderr, "usage: nvrtc-time <pack-dir> [iterations]\n"); return 2; }
std::string pack = argv[1]; int iters = argc > 2 ? std::atoi(argv[2]) : 3;
std::string kernel = deviceOnly(readFile(pack + "/kernel.cu"), true);
std::string program = deviceOnly(readFile(pack + "/program.h"), false);
std::string memhard = deviceOnly(readFile(pack + "/memhard.h"), false);
const char* headers[2] = { program.c_str(), memhard.c_str() };
const char* names[2] = { "program.h", "memhard.h" };
if (cuInit(0) != CUDA_SUCCESS) { std::fprintf(stderr, "cuInit failed\n"); return 2; }
CUdevice dev; cuDeviceGet(&dev, 0); int major = 0, minor = 0;
cuDeviceGetAttribute(&major, CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MAJOR, dev);
cuDeviceGetAttribute(&minor, CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MINOR, dev);
CUcontext ctx; cuCtxCreate(&ctx, 0, dev);
char arch[64]; std::snprintf(arch, sizeof arch, "--gpu-architecture=sm_%d%d", major, minor);
const char* opts[] = { arch, "-std=c++17", "-O3", "-default-device" };
int nvrtcMajor = 0, nvrtcMinor = 0; nvrtcVersion(&nvrtcMajor, &nvrtcMinor);
std::printf("nvrtc %d.%d, device sm_%d%d, kernel source %zu bytes\n", nvrtcMajor, nvrtcMinor, major, minor, kernel.size());
std::vector<double> times;
for (int i = 0; i < iters; ++i) {
double t0 = ms();
nvrtcProgram prog;
if (nvrtcCreateProgram(&prog, kernel.c_str(), "kernel.cu", 2, headers, names) != NVRTC_SUCCESS) { std::fprintf(stderr, "nvrtcCreateProgram failed\n"); return 1; }
nvrtcResult r = nvrtcCompileProgram(prog, 4, opts);
double t1 = ms();
size_t logSize = 0; nvrtcGetProgramLogSize(prog, &logSize);
if (r != NVRTC_SUCCESS) {
std::string log(logSize, '\0'); nvrtcGetProgramLog(prog, &log[0]);
std::fprintf(stderr, "nvrtc compile failed:\n%s\n", log.c_str()); return 1;
}
size_t cubinSize = 0; nvrtcGetCUBINSize(prog, &cubinSize);
std::vector<char> cubin(cubinSize); nvrtcGetCUBIN(prog, cubin.data());
nvrtcDestroyProgram(&prog);
double t2 = ms();
CUmodule mod; CUfunction fn;
if (cuModuleLoadData(&mod, cubin.data()) != CUDA_SUCCESS || cuModuleGetFunction(&fn, mod, "_Z11igneum_hashPKjPyjj") != CUDA_SUCCESS) {
std::fprintf(stderr, "cubin load or igneum_hash lookup failed (name mangling differs?)\n");
} else { cuModuleUnload(mod); }
double t3 = ms();
std::printf("iteration %d: compile %.1f ms, cubin %zu bytes, load %.1f ms\n", i + 1, t1 - t0, cubinSize, t3 - t2);
times.push_back(t1 - t0);
}
std::sort(times.begin(), times.end());
std::printf("median %.1f ms (nvrtc compile of the device kernel, %d iterations)\n", times[times.size() / 2], iters);
return 0;
}