605 lines
41 KiB
C++
605 lines
41 KiB
C++
// igneum-inline-bench: ledger M16 (the 256 MiB cache on a die) and E17 (the draw line per setting).
|
|
//
|
|
// BENCHMARK ONLY, beside the shipped worker (proto-cuda/nvrtc/worker.cpp), never inside it. No pool, no network,
|
|
// no wallet. It measures, on one NVIDIA card and one pack, the honest kernel against the recompute attacker's:
|
|
//
|
|
// honest the pack's bound hash kernel reading the 1 GiB dataset (what every miner runs)
|
|
// inline256 the same program with every dataset read replaced by its derivation from the 256 MiB cache
|
|
// (8 dependent cache-line reads and 9 mixer applications per item, spec 01 section 1.8); the
|
|
// cache sits in VRAM, so this is the Mac's MEMHARD.md 2.2 number on NVIDIA
|
|
// inline64 the same derivation with a 64 MiB cache-line mask: the first quarter of the cache is all the kernel
|
|
// touches, which fits inside the RTX 5090's 96 MiB L2. That is the on-die SRAM emulation the ledger
|
|
// entry names: the attacker's rate with the cache in SRAM-class memory and the GPU's own integer
|
|
// engine doing the mixer work. A 64 MiB cache is NOT the construction; it is the attacker's device
|
|
// modelled on the honest card, so its outputs equal no pack vector and are checked another way
|
|
// inline32 the same at 32 MiB (a second point inside the L2)
|
|
//
|
|
// Three bit-exact checks run before anything is timed, and the run is void if one fails:
|
|
// 1. the pack's own self-test (cache head, last line, FNV; dataset head, last word, samples; the 96 vector lanes
|
|
// through the honest kernel)
|
|
// 2. the inline kernel at the pack's 256 MiB mask on the 96 vector lanes == the pack's vectors (this proves the
|
|
// text substitution of gen.py and the derivation: the dataset is never read and the hashes are the same)
|
|
// 3. at the 64 MiB mask (and 32 MiB): a dataset of --check-mib MiB built by igneum_build_inline at that mask,
|
|
// read by the honest kernel, against igneum_hash_inline at the same mask on the same nonces: the stored and
|
|
// the recomputed path of one construction agree on every lane (the vector warps and a batch of 2^13 nonces)
|
|
//
|
|
// Timing: each setting runs launches of 2^batch-log2 nonces until --seconds have passed (the first launch warms
|
|
// up and is not counted; the worker's race times the same way), rate = hashes / wall seconds. A sampler thread runs
|
|
// the --smi command every --smi-every seconds during each timed window and prints each line with the setting's
|
|
// name (E17: nvidia-smi power.draw, clocks.sm, temperature.gpu, utilization.gpu). Every line that matters starts
|
|
// with RESULT so a run job's report carries it.
|
|
//
|
|
// The kernel texts: the pack's kernel.cu (cache fill, dataset build), kernel_bound.cu (the honest hash) and
|
|
// kernel_inline.cu + memhard_inline.h, which gen.py derives from the pack by text substitution. All four are handed
|
|
// to NVRTC at run time with program.h and memhard.h, exactly as the shipped worker does; the exe prints the sha256
|
|
// of every text it compiled.
|
|
//
|
|
// igneum-inline-bench --pack <dir> [--seconds 20] [--settings honest,inline256,inline64,inline32] [--batch-log2 22]
|
|
// [--block-warps 1] [--check-mib 64] [--check] [--device 0] [--arch auto]
|
|
// [--smi "nvidia-smi --query-gpu=power.draw,clocks.sm,temperature.gpu,utilization.gpu --format=csv,noheader"]
|
|
// [--smi-every 2]
|
|
//
|
|
// The Mac check: built with -DIGNEUM_EMU against emu_backend.cpp (the driver API and NVRTC as host functions, the
|
|
// kernel texts compiled by clang and run on host threads through proto-cuda/emu), `--check` runs the three checks
|
|
// with no GPU. Rates from the emulation are not numbers.
|
|
#include <cstdint>
|
|
#include <cstdarg>
|
|
#include <cstdio>
|
|
#include <cstdlib>
|
|
#include <cstring>
|
|
#include <chrono>
|
|
#include <string>
|
|
#include <vector>
|
|
#include <thread>
|
|
#include <atomic>
|
|
#include <mutex>
|
|
#include <algorithm>
|
|
|
|
#include "../nvrtc/cuda_api.h"
|
|
#include "../nvrtc/packfile.h"
|
|
|
|
#ifdef _WIN32
|
|
#include <windows.h>
|
|
#define POPEN _popen
|
|
#define PCLOSE _pclose
|
|
#else
|
|
#include <dlfcn.h>
|
|
#define POPEN popen
|
|
#define PCLOSE pclose
|
|
#endif
|
|
|
|
static const char* BENCH_VERSION = "1.0 (6 October 2026, ledger close round 2)";
|
|
|
|
static double wallMs() {
|
|
using namespace std::chrono;
|
|
return duration<double, std::milli>(steady_clock::now().time_since_epoch()).count();
|
|
}
|
|
static std::string fmt(const char* f, ...) {
|
|
char b[4096];
|
|
va_list ap; va_start(ap, f); std::vsnprintf(b, sizeof(b), f, ap); va_end(ap);
|
|
return b;
|
|
}
|
|
static void line(const std::string& s) { std::fputs(s.c_str(), stdout); std::fputc('\n', stdout); std::fflush(stdout); }
|
|
static void result(const std::string& s) { line("RESULT " + s); }
|
|
static std::string readText(const std::string& path, bool& ok) {
|
|
std::FILE* f = std::fopen(path.c_str(), "rb");
|
|
if (!f) { ok = false; return ""; }
|
|
std::string s; char buf[65536]; size_t n;
|
|
while ((n = std::fread(buf, 1, sizeof(buf), f)) > 0) s.append(buf, n);
|
|
std::fclose(f); ok = true; return s;
|
|
}
|
|
static std::string sha256Hex(const std::string& s) { char h[65]; pf_sha256_hex((const uint8_t*)s.data(), s.size(), h); return h; }
|
|
|
|
// ---------------------------------------------------------------------------------------------
|
|
// The two libraries (the same shape as the worker: run-time loading, no import library)
|
|
|
|
#ifndef IGNEUM_EMU
|
|
static void* libOpen(const std::string& name) {
|
|
#ifdef _WIN32
|
|
return (void*)LoadLibraryA(name.c_str());
|
|
#else
|
|
return dlopen(name.c_str(), RTLD_NOW);
|
|
#endif
|
|
}
|
|
static void* libSym(void* lib, const char* name) {
|
|
#ifdef _WIN32
|
|
return (void*)GetProcAddress((HMODULE)lib, name);
|
|
#else
|
|
return dlsym(lib, name);
|
|
#endif
|
|
}
|
|
#endif
|
|
#define LOAD_SYM(table, field, name) do { table.field = (decltype(table.field))libSym(lib, name); if (!table.field) { missing += std::string(missing.empty() ? "" : ", ") + name; } } while (0)
|
|
|
|
static bool loadDriver(Drv& d, std::string& err, std::string& libName) {
|
|
#ifdef IGNEUM_EMU
|
|
emu_fill_driver(d); libName = "emulation (host threads, no GPU)"; (void)err; return true;
|
|
#else
|
|
#ifdef _WIN32
|
|
const char* names[] = { "nvcuda.dll" };
|
|
#else
|
|
const char* names[] = { "libcuda.so.1", "libcuda.so" };
|
|
#endif
|
|
void* lib = nullptr;
|
|
for (const char* n : names) { lib = libOpen(n); if (lib) { libName = n; break; } }
|
|
if (!lib) { err = "the CUDA driver library is not installed"; return false; }
|
|
std::string missing;
|
|
LOAD_SYM(d, init, "cuInit"); LOAD_SYM(d, driverGetVersion, "cuDriverGetVersion"); LOAD_SYM(d, deviceGetCount, "cuDeviceGetCount");
|
|
LOAD_SYM(d, deviceGet, "cuDeviceGet"); LOAD_SYM(d, deviceGetName, "cuDeviceGetName"); LOAD_SYM(d, deviceGetAttribute, "cuDeviceGetAttribute");
|
|
LOAD_SYM(d, deviceTotalMem, "cuDeviceTotalMem_v2"); LOAD_SYM(d, primaryCtxSetFlags, "cuDevicePrimaryCtxSetFlags_v2");
|
|
LOAD_SYM(d, primaryCtxRetain, "cuDevicePrimaryCtxRetain"); LOAD_SYM(d, primaryCtxRelease, "cuDevicePrimaryCtxRelease_v2");
|
|
LOAD_SYM(d, ctxSetCurrent, "cuCtxSetCurrent"); LOAD_SYM(d, ctxSynchronize, "cuCtxSynchronize"); LOAD_SYM(d, memGetInfo, "cuMemGetInfo_v2");
|
|
LOAD_SYM(d, memAlloc, "cuMemAlloc_v2"); LOAD_SYM(d, memFree, "cuMemFree_v2"); LOAD_SYM(d, memcpyDtoH, "cuMemcpyDtoH_v2");
|
|
LOAD_SYM(d, moduleLoadData, "cuModuleLoadData"); LOAD_SYM(d, moduleUnload, "cuModuleUnload"); LOAD_SYM(d, moduleGetFunction, "cuModuleGetFunction");
|
|
LOAD_SYM(d, launchKernel, "cuLaunchKernel"); LOAD_SYM(d, streamCreate, "cuStreamCreate"); LOAD_SYM(d, streamSynchronize, "cuStreamSynchronize");
|
|
LOAD_SYM(d, streamDestroy, "cuStreamDestroy_v2"); LOAD_SYM(d, funcGetAttribute, "cuFuncGetAttribute");
|
|
LOAD_SYM(d, occupancy, "cuOccupancyMaxActiveBlocksPerMultiprocessor"); LOAD_SYM(d, getErrorString, "cuGetErrorString"); LOAD_SYM(d, getErrorName, "cuGetErrorName");
|
|
if (!missing.empty()) { err = "the driver library lacks " + missing; return false; }
|
|
return true;
|
|
#endif
|
|
}
|
|
|
|
#ifdef _WIN32
|
|
static std::string exeDir() {
|
|
char buf[MAX_PATH];
|
|
DWORD n = GetModuleFileNameA(nullptr, buf, MAX_PATH);
|
|
std::string p(buf, n);
|
|
size_t i = p.find_last_of("\\/");
|
|
return i == std::string::npos ? "." : p.substr(0, i);
|
|
}
|
|
#endif
|
|
|
|
static bool loadNvrtc(Rtc& r, std::string& err, std::string& libName) {
|
|
#ifdef IGNEUM_EMU
|
|
emu_fill_nvrtc(r); libName = "emulation (the texts are compiled by clang, see emu_backend.cpp)"; (void)err; return true;
|
|
#else
|
|
void* lib = nullptr;
|
|
#ifdef _WIN32
|
|
std::vector<std::string> names;
|
|
if (const char* o = std::getenv("IGNEUM_NVRTC_DLL")) names.push_back(o);
|
|
std::string dir = exeDir();
|
|
WIN32_FIND_DATAA fd;
|
|
HANDLE h = FindFirstFileA((dir + "\\nvrtc64_*_0.dll").c_str(), &fd);
|
|
if (h != INVALID_HANDLE_VALUE) { do { std::string n = fd.cFileName; if (n.find(".alt.") == std::string::npos) names.push_back(dir + "\\" + n); } while (FindNextFileA(h, &fd)); FindClose(h); }
|
|
names.push_back("nvrtc64_120_0.dll"); names.push_back("nvrtc64_130_0.dll");
|
|
for (const std::string& n : names) { lib = libOpen(n); if (lib) { libName = n; break; } }
|
|
if (!lib) { err = "nvrtc64_120_0.dll (and nvrtc-builtins64_128.dll) must sit next to the exe (copy them from the installed Igneum Miner)"; return false; }
|
|
#else
|
|
const char* names[] = { "libnvrtc.so.12", "libnvrtc.so" };
|
|
for (const char* n : names) { lib = libOpen(n); if (lib) { libName = n; break; } }
|
|
if (!lib) { err = "libnvrtc.so.12 not found"; return false; }
|
|
#endif
|
|
std::string missing;
|
|
LOAD_SYM(r, version, "nvrtcVersion"); LOAD_SYM(r, createProgram, "nvrtcCreateProgram"); LOAD_SYM(r, destroyProgram, "nvrtcDestroyProgram");
|
|
LOAD_SYM(r, compileProgram, "nvrtcCompileProgram"); LOAD_SYM(r, getProgramLogSize, "nvrtcGetProgramLogSize"); LOAD_SYM(r, getProgramLog, "nvrtcGetProgramLog");
|
|
LOAD_SYM(r, getPTXSize, "nvrtcGetPTXSize"); LOAD_SYM(r, getPTX, "nvrtcGetPTX"); LOAD_SYM(r, getCUBINSize, "nvrtcGetCUBINSize"); LOAD_SYM(r, getCUBIN, "nvrtcGetCUBIN");
|
|
LOAD_SYM(r, addNameExpression, "nvrtcAddNameExpression"); LOAD_SYM(r, getLoweredName, "nvrtcGetLoweredName"); LOAD_SYM(r, getErrorString, "nvrtcGetErrorString");
|
|
if (!missing.empty()) { err = "the NVRTC library lacks " + missing; return false; }
|
|
r.getNumSupportedArchs = (decltype(r.getNumSupportedArchs))libSym(lib, "nvrtcGetNumSupportedArchs");
|
|
r.getSupportedArchs = (decltype(r.getSupportedArchs))libSym(lib, "nvrtcGetSupportedArchs");
|
|
return true;
|
|
#endif
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------------------------
|
|
// Device and NVRTC (the worker's recipe: the device's own SASS when NVRTC knows it, else PTX for the driver)
|
|
|
|
struct Ctx {
|
|
Drv drv; Rtc rtc;
|
|
CUdevice dev = 0; CUcontext ctx = nullptr;
|
|
std::string name; int major = 0, minor = 0, sms = 0, driverVersion = 0, rtcMajor = 0, rtcMinor = 0, l2Bytes = 0, smClockKhz = 0;
|
|
std::string archOpt; bool ptx = false; std::string why;
|
|
std::string err(CUresult r) { const char* s = nullptr; if (drv.getErrorString) drv.getErrorString(r, &s); return s ? s : "CUDA driver error"; }
|
|
};
|
|
#define DRV_CHECK(c, call, what) do { CUresult r_ = (call); if (r_ != CUDA_SUCCESS) { err = std::string(what) + ": " + (c).err(r_); return false; } } while (0)
|
|
|
|
static bool openDevice(Ctx& c, int device, const std::string& archArg, std::string& err) {
|
|
DRV_CHECK(c, c.drv.init(0), "cuInit");
|
|
int count = 0;
|
|
DRV_CHECK(c, c.drv.deviceGetCount(&count), "cuDeviceGetCount");
|
|
if (count == 0) { err = "no CUDA device"; return false; }
|
|
if (device < 0 || device >= count) { err = fmt("device %d out of range (%d devices)", device, count); return false; }
|
|
DRV_CHECK(c, c.drv.deviceGet(&c.dev, device), "cuDeviceGet");
|
|
char name[256] = {0};
|
|
DRV_CHECK(c, c.drv.deviceGetName(name, 255, c.dev), "cuDeviceGetName");
|
|
c.name = name;
|
|
for (char& ch : c.name) if (ch == ' ') ch = '_';
|
|
DRV_CHECK(c, c.drv.deviceGetAttribute(&c.major, CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MAJOR, c.dev), "cc major");
|
|
DRV_CHECK(c, c.drv.deviceGetAttribute(&c.minor, CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MINOR, c.dev), "cc minor");
|
|
DRV_CHECK(c, c.drv.deviceGetAttribute(&c.sms, CU_DEVICE_ATTRIBUTE_MULTIPROCESSOR_COUNT, c.dev), "sm count");
|
|
c.drv.deviceGetAttribute(&c.l2Bytes, CU_DEVICE_ATTRIBUTE_L2_CACHE_SIZE, c.dev);
|
|
c.drv.deviceGetAttribute(&c.smClockKhz, CU_DEVICE_ATTRIBUTE_CLOCK_RATE, c.dev);
|
|
c.drv.driverGetVersion(&c.driverVersion);
|
|
c.drv.primaryCtxSetFlags(c.dev, CU_CTX_SCHED_BLOCKING_SYNC);
|
|
DRV_CHECK(c, c.drv.primaryCtxRetain(&c.ctx, c.dev), "cuDevicePrimaryCtxRetain");
|
|
DRV_CHECK(c, c.drv.ctxSetCurrent(c.ctx), "cuCtxSetCurrent");
|
|
c.rtc.version(&c.rtcMajor, &c.rtcMinor);
|
|
std::vector<int> archs;
|
|
if (c.rtc.getNumSupportedArchs && c.rtc.getSupportedArchs) {
|
|
int n = 0;
|
|
if (c.rtc.getNumSupportedArchs(&n) == NVRTC_SUCCESS && n > 0 && n < 256) { archs.assign((size_t)n, 0); if (c.rtc.getSupportedArchs(archs.data()) != NVRTC_SUCCESS) archs.clear(); }
|
|
}
|
|
int cc = c.major * 10 + c.minor;
|
|
if (archArg != "auto" && !archArg.empty()) { c.archOpt = archArg; c.ptx = archArg.rfind("compute_", 0) == 0; c.why = "--arch"; }
|
|
else if (archs.empty()) { c.archOpt = fmt("sm_%d", cc); c.why = "the device's architecture (NVRTC did not list its targets)"; }
|
|
else {
|
|
bool known = false; int best = 0;
|
|
for (int a : archs) { if (a == cc) known = true; if (a <= cc && a > best) best = a; }
|
|
if (known) { c.archOpt = fmt("sm_%d", cc); c.why = "the device's architecture, listed by NVRTC"; }
|
|
else if (best > 0) { c.archOpt = fmt("compute_%d", best); c.ptx = true; c.why = fmt("NVRTC does not know sm_%d; PTX for compute_%d", cc, best); }
|
|
else { c.archOpt = fmt("compute_%d", archs.front()); c.ptx = true; c.why = "PTX for NVRTC's oldest target"; }
|
|
}
|
|
return true;
|
|
}
|
|
|
|
static const char* STUB_CUDA_RUNTIME =
|
|
"#pragma once\n#ifndef __CUDACC_RTC__\n#error \"this stub is for NVRTC only\"\n#endif\n"
|
|
"#ifdef __SIZE_TYPE__\ntypedef __SIZE_TYPE__ size_t;\n#elif defined(__LP64__) || defined(_LP64)\ntypedef unsigned long size_t;\n#else\ntypedef unsigned long long size_t;\n#endif\n";
|
|
static const char* STUB_CSTDINT =
|
|
"#pragma once\ntypedef signed char int8_t; typedef unsigned char uint8_t; typedef short int16_t; typedef unsigned short uint16_t;\n"
|
|
"typedef int int32_t; typedef unsigned int uint32_t;\n"
|
|
"#if defined(__LP64__) || defined(_LP64)\ntypedef long int64_t; typedef unsigned long uint64_t;\n#else\ntypedef long long int64_t; typedef unsigned long long uint64_t;\n#endif\n";
|
|
|
|
struct Compiled { std::vector<char> image; std::vector<std::string> lowered; double ms = 0; std::string log; };
|
|
|
|
static bool rtcCompile(Ctx& c, const std::string& src, const char* name, const std::vector<std::pair<std::string, std::string>>& hdrs,
|
|
const std::vector<std::string>& nameExprs, Compiled& out, std::string& err) {
|
|
double t0 = wallMs();
|
|
std::vector<const char*> headers = { STUB_CUDA_RUNTIME, STUB_CSTDINT }, names = { "cuda_runtime.h", "cstdint" };
|
|
for (const auto& h : hdrs) { names.push_back(h.first.c_str()); headers.push_back(h.second.c_str()); }
|
|
nvrtcProgram prog = nullptr;
|
|
nvrtcResult r = c.rtc.createProgram(&prog, src.c_str(), name, (int)headers.size(), headers.data(), names.data());
|
|
if (r != NVRTC_SUCCESS) { err = std::string("nvrtcCreateProgram: ") + c.rtc.getErrorString(r); return false; }
|
|
for (const std::string& e : nameExprs) {
|
|
r = c.rtc.addNameExpression(prog, e.c_str());
|
|
if (r != NVRTC_SUCCESS) { err = "nvrtcAddNameExpression " + e + ": " + c.rtc.getErrorString(r); c.rtc.destroyProgram(&prog); return false; }
|
|
}
|
|
std::string archOpt = "--gpu-architecture=" + c.archOpt;
|
|
const char* opts[] = { archOpt.c_str(), "--std=c++17", "-default-device" };
|
|
r = c.rtc.compileProgram(prog, 3, opts);
|
|
{
|
|
size_t logSize = 0;
|
|
if (c.rtc.getProgramLogSize(prog, &logSize) == NVRTC_SUCCESS && logSize > 1) { std::vector<char> log(logSize); c.rtc.getProgramLog(prog, log.data()); out.log.assign(log.data(), logSize - 1); }
|
|
}
|
|
if (r != NVRTC_SUCCESS) {
|
|
std::string one;
|
|
for (char ch : out.log) { if (ch == '\n' || ch == '\r') { if (one.size() && one.back() != '|') one += " | "; } else one += ch; if (one.size() > 900) break; }
|
|
err = std::string("nvrtcCompileProgram ") + name + " for " + c.archOpt + ": " + c.rtc.getErrorString(r) + ": " + one;
|
|
c.rtc.destroyProgram(&prog); return false;
|
|
}
|
|
for (const std::string& e : nameExprs) {
|
|
const char* lowered = nullptr;
|
|
r = c.rtc.getLoweredName(prog, e.c_str(), &lowered);
|
|
if (r != NVRTC_SUCCESS || !lowered) { err = "nvrtcGetLoweredName " + e + ": " + c.rtc.getErrorString(r); c.rtc.destroyProgram(&prog); return false; }
|
|
out.lowered.push_back(lowered);
|
|
}
|
|
size_t n = 0;
|
|
if (c.ptx) { r = c.rtc.getPTXSize(prog, &n); if (r == NVRTC_SUCCESS) { out.image.resize(n); r = c.rtc.getPTX(prog, out.image.data()); } }
|
|
else { r = c.rtc.getCUBINSize(prog, &n); if (r == NVRTC_SUCCESS) { out.image.resize(n); r = c.rtc.getCUBIN(prog, out.image.data()); } }
|
|
c.rtc.destroyProgram(&prog);
|
|
if (r != NVRTC_SUCCESS || n == 0) { err = std::string(c.ptx ? "nvrtcGetPTX" : "nvrtcGetCUBIN") + ": " + c.rtc.getErrorString(r); return false; }
|
|
out.ms = wallMs() - t0;
|
|
return true;
|
|
}
|
|
|
|
static bool deviceOnly(const std::string& text, std::string& out, std::string& err) {
|
|
size_t cut = text.find("\n// Host-side launch wrappers");
|
|
if (cut == std::string::npos) cut = text.find("\ncudaError_t ");
|
|
if (cut == std::string::npos) { out = text; return true; }
|
|
std::string tail = text.substr(cut + 1);
|
|
if (tail.find("__global__") != std::string::npos || tail.find("__device__") != std::string::npos) { err = "device code after the host launch wrappers"; return false; }
|
|
out = text.substr(0, cut + 1);
|
|
return true;
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------------------------
|
|
// Options
|
|
|
|
struct Options {
|
|
std::string pack, settings = "honest,inline256,inline64,inline32", arch = "auto";
|
|
std::string smi = "nvidia-smi --query-gpu=power.draw,clocks.sm,temperature.gpu,utilization.gpu --format=csv,noheader";
|
|
double seconds = 20; int batchLog2 = 22, blockWarps = 1, checkMib = 64, device = 0, smiEvery = 2; bool checkOnly = false;
|
|
};
|
|
static void usage() {
|
|
line("igneum-inline-bench --pack <dir> [--seconds 20] [--settings honest,inline256,inline64,inline32] [--batch-log2 22] [--block-warps 1]");
|
|
line(" [--check-mib 64] [--check] [--device 0] [--arch auto] [--smi \"<command>\"] [--smi-every 2]");
|
|
}
|
|
static Options parseArgs(int argc, char** argv) {
|
|
Options o;
|
|
for (int i = 1; i < argc; ++i) {
|
|
std::string a = argv[i];
|
|
auto next = [&]() -> std::string { if (i + 1 >= argc) { usage(); std::exit(2); } return argv[++i]; };
|
|
if (a == "--pack") o.pack = next();
|
|
else if (a == "--seconds") o.seconds = std::atof(next().c_str());
|
|
else if (a == "--settings") o.settings = next();
|
|
else if (a == "--batch-log2") o.batchLog2 = std::atoi(next().c_str());
|
|
else if (a == "--block-warps") o.blockWarps = std::atoi(next().c_str());
|
|
else if (a == "--check-mib") o.checkMib = std::atoi(next().c_str());
|
|
else if (a == "--check") o.checkOnly = true;
|
|
else if (a == "--device") o.device = std::atoi(next().c_str());
|
|
else if (a == "--arch") o.arch = next();
|
|
else if (a == "--smi") o.smi = next();
|
|
else if (a == "--smi-every") o.smiEvery = std::atoi(next().c_str());
|
|
else if (a == "-h" || a == "--help") { usage(); std::exit(0); }
|
|
else { line("unknown argument " + a); usage(); std::exit(2); }
|
|
}
|
|
if (o.pack.empty()) { usage(); std::exit(2); }
|
|
if (o.batchLog2 < 10 || o.batchLog2 > 26 || o.blockWarps < 1 || o.blockWarps > 32) { line("--batch-log2 10..26, --block-warps 1..32"); std::exit(2); }
|
|
if (o.checkMib < 1 || o.checkMib > 1024 || (o.checkMib & (o.checkMib - 1))) { line("--check-mib must be a power of two between 1 and 1024"); std::exit(2); }
|
|
return o;
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------------------------
|
|
// The sampler (E17): one line of the --smi command every --smi-every seconds while a window is timed
|
|
|
|
struct Sampler {
|
|
std::string cmd; int everyS = 2; std::atomic<bool> stop{false}; std::thread th; std::mutex mu; std::vector<std::string> lines;
|
|
void start(const std::string& label) {
|
|
if (cmd.empty()) return;
|
|
stop = false;
|
|
th = std::thread([this, label]() {
|
|
while (!stop) {
|
|
std::string out; std::FILE* p = POPEN(cmd.c_str(), "r");
|
|
if (p) { char b[512]; while (std::fgets(b, sizeof(b), p)) out += b; PCLOSE(p); }
|
|
while (!out.empty() && (out.back() == '\n' || out.back() == '\r')) out.pop_back();
|
|
if (out.empty()) out = "(no output)";
|
|
for (char& ch : out) if (ch == '\n' || ch == '\r') ch = ' ';
|
|
{ std::lock_guard<std::mutex> g(mu); lines.push_back(out); }
|
|
result("smi " + label + " " + out);
|
|
for (int i = 0; i < everyS * 10 && !stop; ++i) std::this_thread::sleep_for(std::chrono::milliseconds(100));
|
|
}
|
|
});
|
|
}
|
|
std::vector<std::string> finish() { if (!th.joinable()) return {}; stop = true; th.join(); std::lock_guard<std::mutex> g(mu); std::vector<std::string> v = lines; lines.clear(); return v; }
|
|
};
|
|
// the median of the first comma field of each sample, read as a number ("123.45 W" -> 123.45); -1 when none parses
|
|
static double medianFirstField(const std::vector<std::string>& v) {
|
|
std::vector<double> xs;
|
|
for (const std::string& s : v) { double x = 0; if (std::sscanf(s.c_str(), "%lf", &x) == 1) xs.push_back(x); }
|
|
if (xs.empty()) return -1;
|
|
std::sort(xs.begin(), xs.end());
|
|
return xs[xs.size() / 2];
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------------------------
|
|
|
|
struct IgneumInitWordsArg { uint32_t w[8]; };
|
|
|
|
struct Bench {
|
|
Ctx& c; const PfPack& pk; CUfunction fCacheFill = nullptr, fBuild = nullptr, fHashBound = nullptr, fBuildInline = nullptr, fHashInline = nullptr;
|
|
CUdeviceptr cache = 0, ds = 0, dOut = 0;
|
|
uint32_t cacheWords = 0, dsWords = 0, lineMaskFull = 0, blockWarps = 1;
|
|
Bench(Ctx& c_, const PfPack& pk_) : c(c_), pk(pk_) {}
|
|
|
|
bool launchHonest(CUdeviceptr dsBuf, uint32_t dsMask, CUdeviceptr out, uint32_t base, uint32_t nonces, std::string& err) {
|
|
uint32_t block = 32u * blockWarps;
|
|
IgneumInitWordsArg a; std::memcpy(a.w, pk.seedw, 32);
|
|
void* args[5] = { &dsBuf, &out, &base, &dsMask, &a };
|
|
DRV_CHECK(c, c.drv.launchKernel(fHashBound, nonces / block, 1, 1, block, 1, 1, 0, nullptr, args, nullptr), "cuLaunchKernel igneum_hash_bound");
|
|
return true;
|
|
}
|
|
bool launchInline(uint32_t lineMask, uint32_t dsMask, CUdeviceptr out, uint32_t base, uint32_t nonces, std::string& err) {
|
|
uint32_t block = 32u * blockWarps;
|
|
IgneumInitWordsArg a; std::memcpy(a.w, pk.seedw, 32);
|
|
void* args[6] = { &cache, &out, &base, &dsMask, &a, &lineMask };
|
|
DRV_CHECK(c, c.drv.launchKernel(fHashInline, nonces / block, 1, 1, block, 1, 1, 0, nullptr, args, nullptr), "cuLaunchKernel igneum_hash_inline");
|
|
return true;
|
|
}
|
|
bool sync(std::string& err) { DRV_CHECK(c, c.drv.streamSynchronize(nullptr), "cuStreamSynchronize"); return true; }
|
|
bool readOut(CUdeviceptr out, std::vector<uint64_t>& v, size_t n, std::string& err) { v.resize(n); DRV_CHECK(c, c.drv.memcpyDtoH(v.data(), out, n * 8), "cuMemcpyDtoH"); return true; }
|
|
};
|
|
|
|
static int compareLanes(const std::vector<uint64_t>& a, const std::vector<uint64_t>& b, size_t n, size_t& firstBad) {
|
|
int bad = 0; firstBad = n;
|
|
for (size_t i = 0; i < n; ++i) if (a[i] != b[i]) { if (bad == 0) firstBad = i; ++bad; }
|
|
return bad;
|
|
}
|
|
|
|
int main(int argc, char** argv) {
|
|
Options o = parseArgs(argc, argv);
|
|
Ctx c;
|
|
std::string err, drvLib, rtcLib;
|
|
if (!loadDriver(c.drv, err, drvLib)) { result("error " + err); return 2; }
|
|
if (!loadNvrtc(c.rtc, err, rtcLib)) { result("error " + err); return 2; }
|
|
if (!openDevice(c, o.device, o.arch, err)) { result("error " + err); return 2; }
|
|
result(fmt("bench igneum-inline-bench %s: device %d %s (sm_%d%d, %d SMs, L2 %d MiB, SM clock %d MHz), driver %d.%d from %s, NVRTC %d.%d from %s, target %s (%s)",
|
|
BENCH_VERSION, o.device, c.name.c_str(), c.major, c.minor, c.sms, c.l2Bytes >> 20, c.smClockKhz / 1000, c.driverVersion / 1000, (c.driverVersion % 100) / 10,
|
|
drvLib.c_str(), c.rtcMajor, c.rtcMinor, rtcLib.c_str(), c.archOpt.c_str(), c.why.c_str()));
|
|
|
|
PfPack pk; char perr[512];
|
|
if (!pf_load(o.pack.c_str(), &pk, perr, sizeof(perr))) { result(std::string("error pack ") + o.pack + ": " + perr); return 2; }
|
|
if (pk.datasetMode != 1) { result("error the pack is not memory-hard (IGNEUM_DATASET_MODE 1); the inline kernel has no meaning for a closed-form dataset"); return 2; }
|
|
if (!pk.haveVectors) { result("error the pack carries no vectors.h; the bit-exact checks need it"); return 2; }
|
|
bool ok1, ok2, ok3, ok4, ok5, ok6;
|
|
std::string kernelCu = readText(o.pack + "/kernel.cu", ok1), boundCu = readText(o.pack + "/kernel_bound.cu", ok2);
|
|
std::string programH = readText(o.pack + "/program.h", ok3), memhardH = readText(o.pack + "/memhard.h", ok4);
|
|
std::string inlineCu = readText(o.pack + "/kernel_inline.cu", ok5), memhardInlineH = readText(o.pack + "/memhard_inline.h", ok6);
|
|
if (!ok1 || !ok2 || !ok3 || !ok4) { result("error the pack lacks kernel.cu, kernel_bound.cu, program.h or memhard.h"); return 2; }
|
|
if (!ok5 || !ok6) { result("error the pack lacks kernel_inline.cu or memhard_inline.h: run inline-bench/gen.py <pack> <pack> first"); return 2; }
|
|
std::string kernelDev, boundDev;
|
|
if (!deviceOnly(kernelCu, kernelDev, err) || !deviceOnly(boundCu, boundDev, err)) { result("error " + err); return 2; }
|
|
{
|
|
size_t n = 0, p = 0;
|
|
while ((p = inlineCu.find("mhi_word(cache,", p)) != std::string::npos) { ++n; p += 10; }
|
|
if (n == 0 || inlineCu.find("igneum_hash_inline") == std::string::npos || inlineCu.find("igneum_build_inline") == std::string::npos) { result("error kernel_inline.cu is not gen.py's output"); return 2; }
|
|
result(fmt("texts pack %s seed %s: kernel.cu %s kernel_bound.cu %s program.h %s memhard.h %s kernel_inline.cu %s (%zu inline loads) memhard_inline.h %s",
|
|
o.pack.c_str(), pk.seedString, sha256Hex(kernelCu).substr(0, 16).c_str(), sha256Hex(boundCu).substr(0, 16).c_str(), sha256Hex(programH).substr(0, 16).c_str(),
|
|
sha256Hex(memhardH).substr(0, 16).c_str(), sha256Hex(inlineCu).substr(0, 16).c_str(), n, sha256Hex(memhardInlineH).substr(0, 16).c_str()));
|
|
}
|
|
|
|
// Compile the three programs
|
|
std::vector<std::pair<std::string, std::string>> hdrs = { { "program.h", programH }, { "memhard.h", memhardH }, { "memhard_inline.h", memhardInlineH } };
|
|
Compiled ck, cb, ci;
|
|
if (!rtcCompile(c, kernelDev, "kernel.cu", hdrs, { "igneum_cache_fill", "igneum_build" }, ck, err)) { result("error " + err); return 2; }
|
|
if (!rtcCompile(c, boundDev, "kernel_bound.cu", hdrs, { "igneum_hash_bound" }, cb, err)) { result("error " + err); return 2; }
|
|
if (!rtcCompile(c, inlineCu, "kernel_inline.cu", hdrs, { "igneum_build_inline", "igneum_hash_inline" }, ci, err)) { result("error " + err); return 2; }
|
|
CUmodule mk = nullptr, mb = nullptr, mi = nullptr;
|
|
Bench b(c, pk);
|
|
b.blockWarps = (uint32_t)o.blockWarps;
|
|
{
|
|
CUresult r;
|
|
if ((r = c.drv.moduleLoadData(&mk, ck.image.data())) != CUDA_SUCCESS) { result("error cuModuleLoadData kernel.cu: " + c.err(r)); return 2; }
|
|
if ((r = c.drv.moduleLoadData(&mb, cb.image.data())) != CUDA_SUCCESS) { result("error cuModuleLoadData kernel_bound.cu: " + c.err(r)); return 2; }
|
|
if ((r = c.drv.moduleLoadData(&mi, ci.image.data())) != CUDA_SUCCESS) { result("error cuModuleLoadData kernel_inline.cu: " + c.err(r)); return 2; }
|
|
if (c.drv.moduleGetFunction(&b.fCacheFill, mk, ck.lowered[0].c_str()) != CUDA_SUCCESS || c.drv.moduleGetFunction(&b.fBuild, mk, ck.lowered[1].c_str()) != CUDA_SUCCESS) { result("error kernel.cu functions missing"); return 2; }
|
|
if (c.drv.moduleGetFunction(&b.fHashBound, mb, cb.lowered[0].c_str()) != CUDA_SUCCESS) { result("error igneum_hash_bound missing"); return 2; }
|
|
if (c.drv.moduleGetFunction(&b.fBuildInline, mi, ci.lowered[0].c_str()) != CUDA_SUCCESS || c.drv.moduleGetFunction(&b.fHashInline, mi, ci.lowered[1].c_str()) != CUDA_SUCCESS) { result("error kernel_inline.cu functions missing"); return 2; }
|
|
}
|
|
int regsH = 0, regsI = 0, occH = 0, occI = 0;
|
|
c.drv.funcGetAttribute(®sH, CU_FUNC_ATTRIBUTE_NUM_REGS, b.fHashBound); c.drv.funcGetAttribute(®sI, CU_FUNC_ATTRIBUTE_NUM_REGS, b.fHashInline);
|
|
c.drv.occupancy(&occH, b.fHashBound, 32 * o.blockWarps, 0); c.drv.occupancy(&occI, b.fHashInline, 32 * o.blockWarps, 0);
|
|
result(fmt("compile %s: kernel.cu %.0f ms, kernel_bound.cu %.0f ms, kernel_inline.cu %.0f ms; registers honest %d inline %d; resident blocks/SM at %d warp(s)/block honest %d inline %d",
|
|
c.archOpt.c_str(), ck.ms, cb.ms, ci.ms, regsH, regsI, o.blockWarps, occH, occI));
|
|
|
|
// Memory: the cache, the honest dataset, the check dataset, the outputs
|
|
b.cacheWords = 1u << pk.cacheLog2Words; b.dsWords = 1u << pk.datasetLog2; b.lineMaskFull = (b.cacheWords / 16u) - 1u;
|
|
uint32_t nonces = 1u << o.batchLog2, block = 32u * (uint32_t)o.blockWarps;
|
|
if (nonces % block) { result("error the batch is not a multiple of the block"); return 2; }
|
|
uint32_t checkWords = (uint32_t)(((uint64_t)o.checkMib << 20) / 4u);
|
|
size_t cacheBytes = (size_t)b.cacheWords * 4u, dsBytes = (size_t)b.dsWords * 4u, checkBytes = (size_t)checkWords * 4u;
|
|
{
|
|
size_t freeB = 0, totalB = 0;
|
|
if (c.drv.memGetInfo(&freeB, &totalB) == CUDA_SUCCESS) result(fmt("memory %llu MiB free of %llu; this run needs %llu MiB (cache %llu + dataset %llu + check dataset %llu + outputs)",
|
|
(unsigned long long)(freeB >> 20), (unsigned long long)(totalB >> 20), (unsigned long long)((cacheBytes + dsBytes + checkBytes) >> 20) + 64, (unsigned long long)(cacheBytes >> 20), (unsigned long long)(dsBytes >> 20), (unsigned long long)(checkBytes >> 20)));
|
|
CUresult r;
|
|
if ((r = c.drv.memAlloc(&b.cache, cacheBytes)) != CUDA_SUCCESS) { result("error cuMemAlloc cache: " + c.err(r)); return 2; }
|
|
if ((r = c.drv.memAlloc(&b.ds, dsBytes)) != CUDA_SUCCESS) { result("error cuMemAlloc dataset: " + c.err(r)); return 2; }
|
|
if ((r = c.drv.memAlloc(&b.dOut, (size_t)nonces * 8u)) != CUDA_SUCCESS) { result("error cuMemAlloc out: " + c.err(r)); return 2; }
|
|
}
|
|
// Cache fill, dataset build
|
|
double t0 = wallMs();
|
|
{
|
|
uint32_t nSeg = pk.cacheSegments, blk = 256u, grid = (nSeg + blk - 1u) / blk;
|
|
void* args[2] = { &b.cache, &nSeg };
|
|
CUresult r = c.drv.launchKernel(b.fCacheFill, grid, 1, 1, blk, 1, 1, 0, nullptr, args, nullptr);
|
|
if (r == CUDA_SUCCESS) r = c.drv.streamSynchronize(nullptr);
|
|
if (r != CUDA_SUCCESS) { result("error cache fill: " + c.err(r)); return 2; }
|
|
}
|
|
double cacheMs = wallMs() - t0; t0 = wallMs();
|
|
{
|
|
uint32_t nItems = b.dsWords / 16u, blk = 256u, grid = (nItems + blk - 1u) / blk;
|
|
void* args[3] = { &b.ds, &b.cache, &nItems };
|
|
CUresult r = c.drv.launchKernel(b.fBuild, grid, 1, 1, blk, 1, 1, 0, nullptr, args, nullptr);
|
|
if (r == CUDA_SUCCESS) r = c.drv.streamSynchronize(nullptr);
|
|
if (r != CUDA_SUCCESS) { result("error dataset build: " + c.err(r)); return 2; }
|
|
}
|
|
double dsMs = wallMs() - t0;
|
|
result(fmt("fill cache %u MiB in %.1f ms (%u segments); dataset %u MiB built in %.1f ms (%.0f M items/s)", (unsigned)(cacheBytes >> 20), cacheMs, pk.cacheSegments, (unsigned)(dsBytes >> 20), dsMs, (double)(b.dsWords / 16u) / 1e3 / dsMs));
|
|
|
|
// Check 1: the pack's self-test through the honest kernel
|
|
uint32_t dsMask = b.dsWords - 1u;
|
|
std::vector<uint64_t> vecH((size_t)pk.vecWarps * 32u), vecI((size_t)pk.vecWarps * 32u), tmp;
|
|
bool allPass = true;
|
|
{
|
|
std::vector<uint32_t> whole(b.cacheWords);
|
|
if (c.drv.memcpyDtoH(whole.data(), b.cache, cacheBytes) != CUDA_SUCCESS) { result("error cuMemcpyDtoH cache"); return 2; }
|
|
uint32_t cacheHead[16], cacheLast[16], dsHead[16], dsLast = 0;
|
|
std::memcpy(cacheHead, whole.data(), 64); std::memcpy(cacheLast, whole.data() + b.cacheWords - 16u, 64);
|
|
uint64_t fnv = pf_fnv1a64(whole.data(), cacheBytes);
|
|
whole.clear(); whole.shrink_to_fit();
|
|
c.drv.memcpyDtoH(dsHead, b.ds, 64);
|
|
if (pk.dsLastIndex < b.dsWords) c.drv.memcpyDtoH(&dsLast, b.ds + (CUdeviceptr)pk.dsLastIndex * 4u, 4);
|
|
std::vector<uint32_t> samples((size_t)(pk.nSamples > 0 ? pk.nSamples : 1), 0u);
|
|
for (int i = 0; i < pk.nSamples; ++i) if (pk.sampleIdx[i] < b.dsWords) c.drv.memcpyDtoH(&samples[(size_t)i], b.ds + (CUdeviceptr)pk.sampleIdx[i] * 4u, 4);
|
|
for (int w = 0; w < pk.vecWarps; ++w) {
|
|
if (!b.launchHonest(b.ds, dsMask, b.dOut, pk.vecBase[w], block, err) || !b.sync(err) || !b.readOut(b.dOut, tmp, 32, err)) { result("error check 1: " + err); return 2; }
|
|
std::memcpy(&vecH[(size_t)w * 32u], tmp.data(), 256);
|
|
}
|
|
char ln[1024];
|
|
int pass = pf_selftest(&pk, cacheHead, cacheLast, fnv, dsHead, dsLast, samples.data(), vecH.data(), ln, sizeof(ln));
|
|
result(std::string("check 1 honest kernel, the pack's self-test: ") + ln);
|
|
allPass = allPass && pass != 0;
|
|
}
|
|
// Check 2: inline at the pack's mask == the pack's vectors
|
|
{
|
|
for (int w = 0; w < pk.vecWarps; ++w) {
|
|
if (!b.launchInline(b.lineMaskFull, dsMask, b.dOut, pk.vecBase[w], block, err) || !b.sync(err) || !b.readOut(b.dOut, tmp, 32, err)) { result("error check 2: " + err); return 2; }
|
|
std::memcpy(&vecI[(size_t)w * 32u], tmp.data(), 256);
|
|
}
|
|
std::vector<uint64_t> want((size_t)pk.vecWarps * 32u);
|
|
for (int w = 0; w < pk.vecWarps; ++w) for (int l = 0; l < 32; ++l) want[(size_t)w * 32u + l] = pk.vecOut[w][l];
|
|
size_t firstBad = 0; int bad = compareLanes(vecI, want, want.size(), firstBad);
|
|
if (bad == 0) result(fmt("check 2 inline kernel at the %u MiB mask (0x%08x) on the %d vector lanes: PASS, every lane equals the pack's vectors (the dataset was never read)", (unsigned)(cacheBytes >> 20), b.lineMaskFull, pk.vecWarps * 32));
|
|
else { result(fmt("check 2 inline kernel at the pack's mask: FAIL, %d of %d lanes differ, first at warp %zu lane %zu: device %016llx expected %016llx", bad, pk.vecWarps * 32, firstBad / 32, firstBad % 32, (unsigned long long)vecI[firstBad], (unsigned long long)want[firstBad])); allPass = false; }
|
|
}
|
|
// Check 3: at each smaller mask, the stored twin (igneum_build_inline at the mask, read by the honest kernel)
|
|
// against the recomputed path (igneum_hash_inline at the mask), on a --check-mib dataset
|
|
struct Setting { std::string name; uint32_t lineMask; bool honest; };
|
|
std::vector<Setting> settings;
|
|
{
|
|
std::string s = o.settings; size_t p = 0;
|
|
while (p <= s.size()) {
|
|
size_t q = s.find(',', p); std::string t = s.substr(p, q == std::string::npos ? std::string::npos : q - p);
|
|
if (t == "honest") settings.push_back({ t, 0, true });
|
|
else if (t.rfind("inline", 0) == 0) { int mib = std::atoi(t.c_str() + 6); if (mib < 1 || mib > (int)(cacheBytes >> 20) || (mib & (mib - 1))) { result("error setting " + t + ": the MiB must be a power of two up to the cache size"); return 2; } settings.push_back({ t, (uint32_t)(((uint64_t)mib << 20) / 64u) - 1u, false }); }
|
|
else if (!t.empty()) { result("error unknown setting " + t); return 2; }
|
|
if (q == std::string::npos) break;
|
|
p = q + 1;
|
|
}
|
|
}
|
|
{
|
|
CUdeviceptr dsCheck = 0; CUresult r;
|
|
if ((r = c.drv.memAlloc(&dsCheck, checkBytes)) != CUDA_SUCCESS) { result("error cuMemAlloc check dataset: " + c.err(r)); return 2; }
|
|
uint32_t checkMask = checkWords - 1u, nCheck = 1u << 13;
|
|
if (nCheck < block) nCheck = block;
|
|
for (const Setting& st : settings) {
|
|
if (st.honest || st.lineMask == b.lineMaskFull) continue;
|
|
uint32_t nItems = checkWords / 16u, blk = 256u, grid = (nItems + blk - 1u) / blk, lm = st.lineMask;
|
|
void* args[4] = { &dsCheck, &b.cache, &nItems, &lm };
|
|
r = c.drv.launchKernel(b.fBuildInline, grid, 1, 1, blk, 1, 1, 0, nullptr, args, nullptr);
|
|
if (r == CUDA_SUCCESS) r = c.drv.streamSynchronize(nullptr);
|
|
if (r != CUDA_SUCCESS) { result("error check 3 build at " + st.name + ": " + c.err(r)); return 2; }
|
|
std::vector<uint64_t> hs, is; int bad = 0; size_t firstBad = 0, lanes = 0;
|
|
for (int w = 0; w < pk.vecWarps; ++w) {
|
|
if (!b.launchHonest(dsCheck, checkMask, b.dOut, pk.vecBase[w], block, err) || !b.sync(err) || !b.readOut(b.dOut, hs, block, err)) { result("error check 3: " + err); return 2; }
|
|
if (!b.launchInline(st.lineMask, checkMask, b.dOut, pk.vecBase[w], block, err) || !b.sync(err) || !b.readOut(b.dOut, is, block, err)) { result("error check 3: " + err); return 2; }
|
|
size_t fb; int bd = compareLanes(hs, is, block, fb); if (bd && bad == 0) firstBad = lanes + fb; bad += bd; lanes += block;
|
|
}
|
|
if (!b.launchHonest(dsCheck, checkMask, b.dOut, 0x20000000u, nCheck, err) || !b.sync(err) || !b.readOut(b.dOut, hs, nCheck, err)) { result("error check 3: " + err); return 2; }
|
|
if (!b.launchInline(st.lineMask, checkMask, b.dOut, 0x20000000u, nCheck, err) || !b.sync(err) || !b.readOut(b.dOut, is, nCheck, err)) { result("error check 3: " + err); return 2; }
|
|
{ size_t fb; int bd = compareLanes(hs, is, nCheck, fb); if (bd && bad == 0) firstBad = lanes + fb; bad += bd; lanes += nCheck; }
|
|
// the two paths must also differ from the pack's vectors (a smaller cache is a different construction)
|
|
size_t fbv; int diffFromPack = compareLanes(hs, vecH, (size_t)std::min<uint32_t>(32u, block), fbv);
|
|
if (bad == 0) result(fmt("check 3 %s (mask 0x%08x, %d MiB check dataset): PASS, stored twin == recomputed on all %zu lanes%s", st.name.c_str(), st.lineMask, o.checkMib, lanes, diffFromPack ? "; differs from the pack's vectors, as a smaller cache must" : "; WARNING equals the pack's vectors"));
|
|
else { result(fmt("check 3 %s: FAIL, %d of %zu lanes differ, first lane %zu: stored %016llx recomputed %016llx", st.name.c_str(), bad, lanes, firstBad, (unsigned long long)(firstBad < hs.size() ? hs[firstBad] : 0), (unsigned long long)(firstBad < is.size() ? is[firstBad] : 0))); allPass = false; }
|
|
}
|
|
c.drv.memFree(dsCheck);
|
|
}
|
|
result(std::string("checks ") + (allPass ? "PASS" : "FAIL") + ": every number below " + (allPass ? "stands" : "is VOID"));
|
|
if (o.checkOnly || !allPass) { c.drv.memFree(b.dOut); c.drv.memFree(b.ds); c.drv.memFree(b.cache); return allPass ? 0 : 1; }
|
|
|
|
// Timing
|
|
#ifdef IGNEUM_EMU
|
|
result("note: emulation build, the rates below are host-thread rates and are not numbers");
|
|
#endif
|
|
Sampler smi; smi.cmd = o.smi; smi.everyS = o.smiEvery;
|
|
double honestMhs = 0;
|
|
std::vector<std::string> table;
|
|
for (const Setting& st : settings) {
|
|
smi.start(st.name);
|
|
double tStart = 0; uint64_t hashes = 0; int launches = 0; bool ok = true;
|
|
while (true) {
|
|
uint32_t base = 0x40000000u + (uint32_t)launches * nonces;
|
|
bool l = st.honest ? b.launchHonest(b.ds, dsMask, b.dOut, base, nonces, err) : b.launchInline(st.lineMask, dsMask, b.dOut, base, nonces, err);
|
|
if (!l || !b.sync(err)) { result("error timing " + st.name + ": " + err); ok = false; break; }
|
|
double now = wallMs();
|
|
if (launches == 0) tStart = now; else hashes += nonces;
|
|
++launches;
|
|
if (launches >= 2 && now - tStart >= o.seconds * 1000.0) break;
|
|
}
|
|
std::vector<std::string> samples = smi.finish();
|
|
if (!ok) break;
|
|
double secs = (wallMs() - tStart) / 1000.0, mhs = (double)hashes / secs / 1e6;
|
|
if (st.honest) honestMhs = mhs;
|
|
double power = medianFirstField(samples);
|
|
std::string ratio = honestMhs > 0 ? fmt("%.3f", mhs / honestMhs) : "n/a";
|
|
std::string row = fmt("setting=%s lineMask=0x%08x cacheTouched=%u MiB blockWarps=%d batch=2^%d launches=%d hashes=%llu seconds=%.2f mhs=%.3f ratio_vs_honest=%s smi_samples=%zu smi_first_field_median=%s",
|
|
st.name.c_str(), st.lineMask, st.honest ? (unsigned)(dsBytes >> 20) : (unsigned)(((uint64_t)(st.lineMask + 1u) * 64u) >> 20), o.blockWarps, o.batchLog2, launches - 1, (unsigned long long)hashes, secs, mhs, ratio.c_str(), samples.size(), power < 0 ? "none" : fmt("%.1f", power).c_str());
|
|
result(row); table.push_back(row);
|
|
}
|
|
result("summary " + std::to_string(table.size()) + " settings timed; honest " + fmt("%.3f", honestMhs) + " Mhash/s; the inline ratios are the attacker's rate over the honest rate on this card");
|
|
c.drv.memFree(b.dOut); c.drv.memFree(b.ds); c.drv.memFree(b.cache);
|
|
c.drv.moduleUnload(mi); c.drv.moduleUnload(mb); c.drv.moduleUnload(mk);
|
|
c.drv.primaryCtxRelease(c.dev);
|
|
return 0;
|
|
}
|