infra/cloud-devnet: hcloud (doctl variant) create, builder-VM provision from a git-archive source tarball, systemd units for igneumd --devnet-suffix with a sparse --addpeer mesh and a CPU trickle miner per node, stdlib wRPC client, experiments (latency, partition, hop, collect, observer hookup), README with the command sequence and the Hetzner API prices of 3 Oct 2026. infra/gpu-bench: RunPod image recipes (CUDA 12.8, ROCm), bundle, run.sh (vectors gate, 10-min raw, sweep, inline shortcut ratio, nvcc/NVRTC/OpenCL recompile timings, results row, intake upload), bench-log template. infra/seed-nodes: create-seed (persistent IPv4, firewall), provision on the VM, health check, addPeer from the Mac over grpcurl, seeds.txt; igneum-seed-1 created at 188.245.5.161 (Hetzner cx23, fsn1). docs/plans/cloud-devnet.md and docs/plans/seed-nodes.md. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
82 lines
4.7 KiB
Text
82 lines
4.7 KiB
Text
// Times an in-process NVRTC compile of a pack's kernel (the device part of kernel.cu with program.h and memhard.h
|
|
// as in-memory headers), N times, then loads the cubin through the driver API to prove it is usable.
|
|
// This is the "hourly JIT in the miner process" figure of ledger M11 and M17; the worker today runs nvcc out of
|
|
// process (proto-cuda/host.cu, prepareCompile), which run.sh times separately.
|
|
// nvcc -O2 -std=c++17 -o nvrtc-time nvrtc-time.cu -lnvrtc -lcuda
|
|
// ./nvrtc-time <pack-dir> [iterations=3]
|
|
// Prints one line per iteration (compile ms, cubin bytes, load ms) and "median <ms>".
|
|
#include <nvrtc.h>
|
|
#include <cuda.h>
|
|
#include <algorithm>
|
|
#include <chrono>
|
|
#include <cstdio>
|
|
#include <cstdlib>
|
|
#include <fstream>
|
|
#include <sstream>
|
|
#include <string>
|
|
#include <vector>
|
|
|
|
static std::string readFile(const std::string& p) {
|
|
std::ifstream f(p); if (!f) { std::fprintf(stderr, "cannot read %s\n", p.c_str()); std::exit(2); }
|
|
std::stringstream ss; ss << f.rdbuf(); return ss.str();
|
|
}
|
|
// Drop the includes NVRTC cannot see and the host launch wrappers; the typedefs replace <cstdint>.
|
|
static std::string deviceOnly(const std::string& src, bool cutHost) {
|
|
std::istringstream in(src); std::string line, out = "typedef unsigned int uint32_t;\ntypedef unsigned long long uint64_t;\ntypedef unsigned char uint8_t;\ntypedef unsigned long size_t;\n";
|
|
while (std::getline(in, line)) {
|
|
if (cutHost && line.rfind("cudaError_t igneum_launch", 0) == 0) break;
|
|
if (line.find("#include <cuda_runtime.h>") != std::string::npos || line.find("#include <cstdint>") != std::string::npos ||
|
|
line.find("#include <stdint.h>") != std::string::npos || line.find("#include <stddef.h>") != std::string::npos) { out += "\n"; continue; }
|
|
out += line + "\n";
|
|
}
|
|
return out;
|
|
}
|
|
static double ms() { return std::chrono::duration<double, std::milli>(std::chrono::steady_clock::now().time_since_epoch()).count(); }
|
|
|
|
int main(int argc, char** argv) {
|
|
if (argc < 2) { std::fprintf(stderr, "usage: nvrtc-time <pack-dir> [iterations]\n"); return 2; }
|
|
std::string pack = argv[1]; int iters = argc > 2 ? std::atoi(argv[2]) : 3;
|
|
std::string kernel = deviceOnly(readFile(pack + "/kernel.cu"), true);
|
|
std::string program = deviceOnly(readFile(pack + "/program.h"), false);
|
|
std::string memhard = deviceOnly(readFile(pack + "/memhard.h"), false);
|
|
const char* headers[2] = { program.c_str(), memhard.c_str() };
|
|
const char* names[2] = { "program.h", "memhard.h" };
|
|
|
|
if (cuInit(0) != CUDA_SUCCESS) { std::fprintf(stderr, "cuInit failed\n"); return 2; }
|
|
CUdevice dev; cuDeviceGet(&dev, 0); int major = 0, minor = 0;
|
|
cuDeviceGetAttribute(&major, CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MAJOR, dev);
|
|
cuDeviceGetAttribute(&minor, CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MINOR, dev);
|
|
CUcontext ctx; cuCtxCreate(&ctx, 0, dev);
|
|
char arch[64]; std::snprintf(arch, sizeof arch, "--gpu-architecture=sm_%d%d", major, minor);
|
|
const char* opts[] = { arch, "-std=c++17", "-O3", "-default-device" };
|
|
int nvrtcMajor = 0, nvrtcMinor = 0; nvrtcVersion(&nvrtcMajor, &nvrtcMinor);
|
|
std::printf("nvrtc %d.%d, device sm_%d%d, kernel source %zu bytes\n", nvrtcMajor, nvrtcMinor, major, minor, kernel.size());
|
|
|
|
std::vector<double> times;
|
|
for (int i = 0; i < iters; ++i) {
|
|
double t0 = ms();
|
|
nvrtcProgram prog;
|
|
if (nvrtcCreateProgram(&prog, kernel.c_str(), "kernel.cu", 2, headers, names) != NVRTC_SUCCESS) { std::fprintf(stderr, "nvrtcCreateProgram failed\n"); return 1; }
|
|
nvrtcResult r = nvrtcCompileProgram(prog, 4, opts);
|
|
double t1 = ms();
|
|
size_t logSize = 0; nvrtcGetProgramLogSize(prog, &logSize);
|
|
if (r != NVRTC_SUCCESS) {
|
|
std::string log(logSize, '\0'); nvrtcGetProgramLog(prog, &log[0]);
|
|
std::fprintf(stderr, "nvrtc compile failed:\n%s\n", log.c_str()); return 1;
|
|
}
|
|
size_t cubinSize = 0; nvrtcGetCUBINSize(prog, &cubinSize);
|
|
std::vector<char> cubin(cubinSize); nvrtcGetCUBIN(prog, cubin.data());
|
|
nvrtcDestroyProgram(&prog);
|
|
double t2 = ms();
|
|
CUmodule mod; CUfunction fn;
|
|
if (cuModuleLoadData(&mod, cubin.data()) != CUDA_SUCCESS || cuModuleGetFunction(&fn, mod, "_Z11igneum_hashPKjPyjj") != CUDA_SUCCESS) {
|
|
std::fprintf(stderr, "cubin load or igneum_hash lookup failed (name mangling differs?)\n");
|
|
} else { cuModuleUnload(mod); }
|
|
double t3 = ms();
|
|
std::printf("iteration %d: compile %.1f ms, cubin %zu bytes, load %.1f ms\n", i + 1, t1 - t0, cubinSize, t3 - t2);
|
|
times.push_back(t1 - t0);
|
|
}
|
|
std::sort(times.begin(), times.end());
|
|
std::printf("median %.1f ms (nvrtc compile of the device kernel, %d iterations)\n", times[times.size() / 2], iters);
|
|
return 0;
|
|
}
|