483 lines
23 KiB
Text
483 lines
23 KiB
Text
// igneum-bench-cuda: host program for Igneum's random-program proof-of-work test harness on NVIDIA GPUs.
|
|
//
|
|
// TEST HARNESS ONLY. No pool, no network, no wallet, no mining protocol. It fills the dataset on the GPU,
|
|
// checks the GPU against vectors produced on the Mac (proto-metal), and times the kernel.
|
|
//
|
|
// C++17 plus the CUDA runtime API, nothing else. The kernel is compiled ahead of time by nvcc from
|
|
// packs/<seed>/kernel.cu, which was generated by proto-metal/igneum-bench --export-pack.
|
|
//
|
|
// Build (Linux, from proto-cuda/):
|
|
// nvcc -O3 -std=c++17 -arch=sm_120 -I packs/igneum-genesis -o igneum-bench-cuda-igneum-genesis host.cu packs/igneum-genesis/kernel.cu
|
|
// Windows and the -arch=native fallback are in README.md.
|
|
|
|
#include <cuda_runtime.h>
|
|
#include <cstdint>
|
|
#include <cstdio>
|
|
#include <cstdlib>
|
|
#include <cstring>
|
|
#include <chrono>
|
|
#include <string>
|
|
#include <vector>
|
|
|
|
#include "program.h"
|
|
#include "vectors.h"
|
|
|
|
// Packs written before 3 October 2026 have no dataset mode: they are closed-form (mode 0).
|
|
#ifndef IGNEUM_DATASET_MODE
|
|
#define IGNEUM_DATASET_MODE 0
|
|
#endif
|
|
#if IGNEUM_DATASET_MODE == 1
|
|
#include "memhard.h" // the memory-hard core, compiled here for the host reference (see proto-metal/MEMHARD.md)
|
|
#endif
|
|
|
|
#define CUDA_CHECK(call) do { cudaError_t err_ = (call); if (err_ != cudaSuccess) { \
|
|
std::fprintf(stderr, "CUDA error: %s (%d)\n at %s:%d\n in %s\n", cudaGetErrorString(err_), (int)err_, __FILE__, __LINE__, #call); \
|
|
std::exit(2); } } while (0)
|
|
|
|
static const uint32_t SEEDW[8] = IGNEUM_SEEDW_INIT;
|
|
|
|
#if IGNEUM_DATASET_MODE == 0
|
|
// Same closed form as ds_elem in kernel.cu and datasetElem in proto-metal/main.swift.
|
|
static uint32_t host_ds_elem(uint32_t i, uint32_t d0, uint32_t d1) {
|
|
uint32_t x = i ^ d0;
|
|
x *= 0x9E3779B1u; x ^= x >> 15;
|
|
x += d1;
|
|
x *= 0x85EBCA77u; x ^= x >> 13;
|
|
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
|
return x;
|
|
}
|
|
#endif
|
|
|
|
static double wallMs();
|
|
|
|
#if IGNEUM_DATASET_MODE == 1
|
|
static uint64_t fnv1a64(const void* p, size_t n) {
|
|
const uint8_t* b = (const uint8_t*)p;
|
|
uint64_t h = 0xcbf29ce484222325ull;
|
|
for (size_t i = 0; i < n; ++i) { h ^= b[i]; h *= 0x100000001b3ull; }
|
|
return h;
|
|
}
|
|
|
|
// Memory-hard mode: the 256 MiB cache on the device (filled by the pack's kernel) and on the host (filled by the
|
|
// same mh_cache_segment text on one thread). Both are built once per process in setupCache(), compared word for
|
|
// word, and checked against the head, last line and FNV-1a 64 the Mac recorded in vectors.h.
|
|
static const uint32_t CACHE_WORDS_HOST = 1u << IGNEUM_CACHE_LOG2_WORDS;
|
|
static uint32_t* gCache = nullptr; // device
|
|
static std::vector<uint32_t> hCache; // host
|
|
static double gCacheFillFirstMs = 0, gCacheFillSecondMs = 0, gCacheHostMs = 0;
|
|
static bool gCachePass = false;
|
|
|
|
static bool setupCache() {
|
|
size_t bytes = (size_t)CACHE_WORDS_HOST * 4u;
|
|
CUDA_CHECK(cudaMalloc((void**)&gCache, bytes));
|
|
cudaEvent_t e0, e1;
|
|
CUDA_CHECK(cudaEventCreate(&e0));
|
|
CUDA_CHECK(cudaEventCreate(&e1));
|
|
for (int pass = 0; pass < 2; ++pass) {
|
|
CUDA_CHECK(cudaEventRecord(e0));
|
|
CUDA_CHECK(igneum_launch_cache_fill(gCache, IGNEUM_CACHE_SEGMENTS));
|
|
CUDA_CHECK(cudaEventRecord(e1));
|
|
CUDA_CHECK(cudaEventSynchronize(e1));
|
|
float msf = 0.f;
|
|
CUDA_CHECK(cudaEventElapsedTime(&msf, e0, e1));
|
|
if (pass == 0) gCacheFillFirstMs = msf; else gCacheFillSecondMs = msf;
|
|
}
|
|
CUDA_CHECK(cudaEventDestroy(e0));
|
|
CUDA_CHECK(cudaEventDestroy(e1));
|
|
std::printf("cache fill (GPU): %.2f ms first, %.2f ms second (%u chains x %u ChaCha blocks, %u MiB)\n",
|
|
gCacheFillFirstMs, gCacheFillSecondMs, (unsigned)IGNEUM_CACHE_SEGMENTS,
|
|
1u << IGNEUM_CACHE_SEGMENT_LOG2_LINES, (unsigned)(bytes >> 20));
|
|
|
|
hCache.assign(CACHE_WORDS_HOST, 0u);
|
|
double h0 = wallMs();
|
|
for (uint32_t seg = 0; seg < IGNEUM_CACHE_SEGMENTS; ++seg) mh_cache_segment(hCache.data(), seg);
|
|
gCacheHostMs = wallMs() - h0;
|
|
std::printf("cache fill (host, one thread): %.1f ms\n", gCacheHostMs);
|
|
|
|
std::vector<uint32_t> dev(CACHE_WORDS_HOST);
|
|
CUDA_CHECK(cudaMemcpy(dev.data(), gCache, bytes, cudaMemcpyDeviceToHost));
|
|
bool same = std::memcmp(dev.data(), hCache.data(), bytes) == 0;
|
|
uint64_t fnv = fnv1a64(hCache.data(), bytes);
|
|
bool fnvOk = (fnv == IGNEUM_CACHE_FNV64);
|
|
bool headOk = std::memcmp(hCache.data(), IGNEUM_CACHE_HEAD, 64) == 0;
|
|
bool lastOk = std::memcmp(hCache.data() + CACHE_WORDS_HOST - 16u, IGNEUM_CACHE_LAST, 64) == 0;
|
|
if (!same) {
|
|
for (uint32_t i = 0; i < CACHE_WORDS_HOST; ++i) if (dev[i] != hCache[i]) {
|
|
std::printf(" cache[%u]: gpu 0x%08x host 0x%08x (first difference)\n", i, dev[i], hCache[i]); break;
|
|
}
|
|
}
|
|
gCachePass = same && fnvOk && headOk && lastOk;
|
|
std::printf("cache check: %s (GPU == host all %u words %s, host FNV-1a 64 %016llx vs Mac %016llx %s, head 16 vs Mac %s, last line vs Mac %s)\n",
|
|
gCachePass ? "PASS" : "FAIL", CACHE_WORDS_HOST, same ? "PASS" : "FAIL",
|
|
(unsigned long long)fnv, (unsigned long long)IGNEUM_CACHE_FNV64, fnvOk ? "PASS" : "FAIL",
|
|
headOk ? "PASS" : "FAIL", lastOk ? "PASS" : "FAIL");
|
|
return gCachePass;
|
|
}
|
|
|
|
// dataset[w] derived on the host from the host cache, exactly as proto-metal's verifier does it.
|
|
static uint32_t host_ds_word(uint32_t w) {
|
|
uint32_t s[16];
|
|
mh_item(hCache.data(), w >> 4u, s);
|
|
return s[w & 15u];
|
|
}
|
|
#endif
|
|
|
|
// ---------------------------------------------------------------------------------------------
|
|
// Options
|
|
|
|
struct Options {
|
|
int datasetMib = 1024;
|
|
int batchLog2 = 24;
|
|
int batches = 5;
|
|
int blockWarps = 1;
|
|
bool sweep = false;
|
|
int device = 0;
|
|
};
|
|
|
|
static int packMib() { return (int)(((1ull << IGNEUM_DATASET_LOG2) * 4ull) >> 20); }
|
|
|
|
static void usage() {
|
|
std::printf(
|
|
"igneum-bench-cuda [--dataset-mib N] [--sweep] [--batch-log2 24] [--batches 5] [--block-warps 1] [--device 0]\n"
|
|
" --dataset-mib N dataset size in MiB, power of two (default 1024; vectors are only checked at %d MiB)\n"
|
|
" --sweep run 4, 64, 256, 512 and 1024 MiB in sequence (same sweep as the Mac)\n"
|
|
" --batch-log2 B nonces per batch = 2^B (default 24)\n"
|
|
" --batches N timed batches after one warm-up batch (default 5)\n"
|
|
" --block-warps W warps per thread block, 1..32 (default 1 = one warp per block, like the Metal run)\n"
|
|
" --device D CUDA device index (default 0)\n", packMib());
|
|
}
|
|
|
|
static bool isPow2(long long v) { return v > 0 && (v & (v - 1)) == 0; }
|
|
|
|
static Options parseArgs(int argc, char** argv) {
|
|
Options o;
|
|
for (int i = 1; i < argc; ++i) {
|
|
std::string a = argv[i];
|
|
auto next = [&](int& dst) {
|
|
if (i + 1 >= argc) { usage(); std::exit(2); }
|
|
dst = std::atoi(argv[++i]);
|
|
};
|
|
if (a == "--dataset-mib") next(o.datasetMib);
|
|
else if (a == "--batch-log2") next(o.batchLog2);
|
|
else if (a == "--batches") next(o.batches);
|
|
else if (a == "--block-warps") next(o.blockWarps);
|
|
else if (a == "--device") next(o.device);
|
|
else if (a == "--sweep") o.sweep = true;
|
|
else if (a == "-h" || a == "--help") { usage(); std::exit(0); }
|
|
else { std::printf("unknown argument %s\n", argv[i]); usage(); std::exit(2); }
|
|
}
|
|
if (!isPow2(o.datasetMib) || o.datasetMib < 1 || o.datasetMib > 16384) {
|
|
std::printf("--dataset-mib must be a power of two between 1 and 16384\n"); std::exit(2);
|
|
}
|
|
if (o.batchLog2 < 10 || o.batchLog2 > 28) { std::printf("--batch-log2 must be between 10 and 28\n"); std::exit(2); }
|
|
if (o.batches < 1) { std::printf("--batches must be at least 1\n"); std::exit(2); }
|
|
if (o.blockWarps < 1 || o.blockWarps > 32) { std::printf("--block-warps must be between 1 and 32\n"); std::exit(2); }
|
|
return o;
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------------------------
|
|
// Helpers
|
|
|
|
static double wallMs() {
|
|
using namespace std::chrono;
|
|
return duration<double, std::milli>(steady_clock::now().time_since_epoch()).count();
|
|
}
|
|
|
|
static int log2u32(uint32_t v) { int n = 0; while (v > 1u) { v >>= 1; ++n; } return n; }
|
|
|
|
static bool compareWarp(const uint64_t* got, const uint64_t* want, uint32_t base, const char* how) {
|
|
int bad = 0, first = -1;
|
|
for (int l = 0; l < 32; ++l) if (got[l] != want[l]) { if (first < 0) first = l; ++bad; }
|
|
if (bad == 0) {
|
|
std::printf("verify warp base %u (nonces %u..%u) %s: PASS\n", base, base, base + 31u, how);
|
|
} else {
|
|
std::printf("verify warp base %u (nonces %u..%u) %s: FAIL %d of 32 lanes differ, first lane %d: gpu=%016llx expected=%016llx\n",
|
|
base, base, base + 31u, how, bad, first,
|
|
(unsigned long long)got[first], (unsigned long long)want[first]);
|
|
}
|
|
return bad == 0;
|
|
}
|
|
|
|
struct SizeResult {
|
|
int mib = 0;
|
|
uint32_t words = 0;
|
|
double fillFirstMs = 0, fillSecondMs = 0;
|
|
bool dsPass = false;
|
|
bool vecChecked = false, vecPass = false;
|
|
double gpuMs = 0, wallMsTimed = 0;
|
|
double hashesPerSec = 0, gbps = 0;
|
|
};
|
|
|
|
// ---------------------------------------------------------------------------------------------
|
|
// One dataset size: fill, self-test, vectors, bench
|
|
|
|
static SizeResult runSize(const Options& o, int mib, uint64_t* dOut, uint32_t nonces) {
|
|
SizeResult r;
|
|
r.mib = mib;
|
|
uint64_t bytes = (uint64_t)mib << 20;
|
|
r.words = (uint32_t)(bytes / 4ull);
|
|
uint32_t mask = r.words - 1u;
|
|
bool atPackSize = (r.words == (1u << IGNEUM_DATASET_LOG2));
|
|
std::printf("\n=== dataset %d MiB (2^%d words, mask 0x%08x)%s ===\n", mib, log2u32(r.words), mask,
|
|
atPackSize ? "" : " [not the pack size: vectors skipped, dataset head and random points still checked]");
|
|
|
|
size_t freeB = 0, totalB = 0;
|
|
CUDA_CHECK(cudaMemGetInfo(&freeB, &totalB));
|
|
if ((uint64_t)freeB < bytes + (64ull << 20)) {
|
|
std::printf("FAIL: %llu MiB free on the device, need %d MiB for the dataset\n", (unsigned long long)(freeB >> 20), mib);
|
|
std::exit(2);
|
|
}
|
|
uint32_t* dDs = nullptr;
|
|
CUDA_CHECK(cudaMalloc((void**)&dDs, (size_t)bytes));
|
|
|
|
cudaEvent_t e0, e1;
|
|
CUDA_CHECK(cudaEventCreate(&e0));
|
|
CUDA_CHECK(cudaEventCreate(&e1));
|
|
|
|
// Fill (or build) twice: the Mac showed a first-touch cost on the first fill of a process.
|
|
for (int pass = 0; pass < 2; ++pass) {
|
|
CUDA_CHECK(cudaEventRecord(e0));
|
|
#if IGNEUM_DATASET_MODE == 1
|
|
CUDA_CHECK(igneum_launch_build(dDs, gCache, r.words / 16u));
|
|
#else
|
|
CUDA_CHECK(igneum_launch_fill(dDs, r.words, IGNEUM_DAY0, IGNEUM_DAY1));
|
|
#endif
|
|
CUDA_CHECK(cudaEventRecord(e1));
|
|
CUDA_CHECK(cudaEventSynchronize(e1));
|
|
float msf = 0.f;
|
|
CUDA_CHECK(cudaEventElapsedTime(&msf, e0, e1));
|
|
if (pass == 0) r.fillFirstMs = msf; else r.fillSecondMs = msf;
|
|
}
|
|
#if IGNEUM_DATASET_MODE == 1
|
|
std::printf("dataset build (memory-hard, from the cache): %.2f ms first, %.2f ms second -> %.1f M items/s, %.2f G cache-line reads/s (second, GPU time)\n",
|
|
r.fillFirstMs, r.fillSecondMs, (double)(r.words / 16u) / 1e6 / (r.fillSecondMs / 1000.0),
|
|
(double)(r.words / 16u) * (double)IGNEUM_ITEM_ROUNDS / 1e9 / (r.fillSecondMs / 1000.0));
|
|
#else
|
|
std::printf("dataset fill: %.2f ms first, %.2f ms second -> %.0f GB/s write (second, GPU time)\n",
|
|
r.fillFirstMs, r.fillSecondMs, (double)bytes / 1e9 / (r.fillSecondMs / 1000.0));
|
|
#endif
|
|
|
|
// Dataset self-test: head 16 (any size), element [MASK] (pack size only), 64 pseudo-random points vs the host
|
|
// formula (closed form) or the host derivation from the host cache (memory-hard), and the Mac's 64 sampled words.
|
|
{
|
|
uint32_t head[16];
|
|
CUDA_CHECK(cudaMemcpy(head, dDs, sizeof(head), cudaMemcpyDeviceToHost));
|
|
int badHead = 0;
|
|
for (int i = 0; i < 16; ++i) {
|
|
if (head[i] != IGNEUM_DS_HEAD[i]) {
|
|
if (badHead == 0) std::printf(" dataset[%d] = 0x%08x, expected 0x%08x\n", i, head[i], IGNEUM_DS_HEAD[i]);
|
|
++badHead;
|
|
}
|
|
}
|
|
bool lastOk = true;
|
|
const char* lastText = "skipped";
|
|
if (atPackSize) {
|
|
uint32_t last = 0;
|
|
CUDA_CHECK(cudaMemcpy(&last, dDs + IGNEUM_DS_LAST_INDEX, sizeof(last), cudaMemcpyDeviceToHost));
|
|
lastOk = (last == IGNEUM_DS_LAST);
|
|
lastText = lastOk ? "PASS" : "FAIL";
|
|
if (!lastOk) std::printf(" dataset[%u] = 0x%08x, expected 0x%08x\n", IGNEUM_DS_LAST_INDEX, last, IGNEUM_DS_LAST);
|
|
}
|
|
int badRnd = 0;
|
|
uint64_t s = 0x9E3779B97F4A7C15ull ^ (uint64_t)r.words;
|
|
for (int k = 0; k < 64; ++k) {
|
|
s += 0x9E3779B97F4A7C15ull;
|
|
uint64_t z = s;
|
|
z = (z ^ (z >> 30)) * 0xBF58476D1CE4E5B9ull;
|
|
z = (z ^ (z >> 27)) * 0x94D049BB133111EBull;
|
|
z ^= z >> 31;
|
|
uint32_t idx = (uint32_t)z & mask;
|
|
uint32_t v = 0;
|
|
CUDA_CHECK(cudaMemcpy(&v, dDs + idx, sizeof(v), cudaMemcpyDeviceToHost));
|
|
#if IGNEUM_DATASET_MODE == 1
|
|
uint32_t want = host_ds_word(idx);
|
|
#else
|
|
uint32_t want = host_ds_elem(idx, IGNEUM_DAY0, IGNEUM_DAY1);
|
|
#endif
|
|
if (v != want) {
|
|
if (badRnd == 0) std::printf(" dataset[%u] = 0x%08x, host %s 0x%08x\n", idx, v, IGNEUM_DATASET_MODE == 1 ? "derivation" : "formula", want);
|
|
++badRnd;
|
|
}
|
|
}
|
|
// The Mac's sampled words: every sample whose index lies inside this dataset size (items are the same
|
|
// at every size, the smaller dataset is a prefix of the larger one).
|
|
int badSample = 0, nSample = 0;
|
|
#ifdef IGNEUM_DS_SAMPLES
|
|
for (int k = 0; k < IGNEUM_DS_SAMPLES; ++k) {
|
|
if (IGNEUM_DS_SAMPLE_INDEX[k] > mask) continue;
|
|
++nSample;
|
|
uint32_t v = 0;
|
|
CUDA_CHECK(cudaMemcpy(&v, dDs + IGNEUM_DS_SAMPLE_INDEX[k], sizeof(v), cudaMemcpyDeviceToHost));
|
|
if (v != IGNEUM_DS_SAMPLE_VALUE[k]) {
|
|
if (badSample == 0) std::printf(" dataset[%u] = 0x%08x, Mac 0x%08x\n", IGNEUM_DS_SAMPLE_INDEX[k], v, IGNEUM_DS_SAMPLE_VALUE[k]);
|
|
++badSample;
|
|
}
|
|
}
|
|
#endif
|
|
r.dsPass = (badHead == 0 && lastOk && badRnd == 0 && badSample == 0);
|
|
std::printf("dataset self-test: %s (head 16 vs Mac %s, element [MASK] vs Mac %s, 64 random points vs host %s %s, %d Mac samples %s)\n",
|
|
r.dsPass ? "PASS" : "FAIL", badHead == 0 ? "PASS" : "FAIL", lastText,
|
|
IGNEUM_DATASET_MODE == 1 ? "derivation" : "formula", badRnd == 0 ? "PASS" : "FAIL",
|
|
nSample, nSample == 0 ? "none in pack" : (badSample == 0 ? "PASS" : "FAIL"));
|
|
}
|
|
|
|
// Vectors, standalone: one 32-thread block per base nonce, exactly like the Mac cross-check.
|
|
uint64_t got[32];
|
|
if (atPackSize) {
|
|
r.vecChecked = true;
|
|
r.vecPass = true;
|
|
for (int w = 0; w < IGNEUM_VEC_WARPS; ++w) {
|
|
CUDA_CHECK(igneum_launch_hash(dDs, dOut, IGNEUM_VEC_BASE[w], mask, 32u, 1u));
|
|
CUDA_CHECK(cudaDeviceSynchronize());
|
|
CUDA_CHECK(cudaMemcpy(got, dOut, sizeof(got), cudaMemcpyDeviceToHost));
|
|
bool ok = compareWarp(got, IGNEUM_VEC_OUT[w], IGNEUM_VEC_BASE[w], "standalone, 1 warp/block");
|
|
r.vecPass = r.vecPass && ok;
|
|
}
|
|
} else {
|
|
std::printf("vectors: skipped (the pack's vectors are for %d MiB)\n", packMib());
|
|
}
|
|
|
|
// Warm-up batch at base nonce 0. With the default 2^24 nonces it contains all three vector warps,
|
|
// so the bench configuration itself (blockDim = 32 x block-warps) is also checked bit for bit.
|
|
double w0 = wallMs();
|
|
CUDA_CHECK(igneum_launch_hash(dDs, dOut, 0u, mask, nonces, (uint32_t)o.blockWarps));
|
|
CUDA_CHECK(cudaDeviceSynchronize());
|
|
double w1 = wallMs();
|
|
std::printf("warm-up batch: %u hashes in %.2f ms wall\n", nonces, w1 - w0);
|
|
if (atPackSize) {
|
|
char how[64];
|
|
std::snprintf(how, sizeof(how), "in batch, %d warp(s)/block", o.blockWarps);
|
|
for (int w = 0; w < IGNEUM_VEC_WARPS; ++w) {
|
|
if ((uint64_t)IGNEUM_VEC_BASE[w] + 32ull > (uint64_t)nonces) {
|
|
std::printf("verify warp base %u in batch: skipped (batch has %u nonces)\n", IGNEUM_VEC_BASE[w], nonces);
|
|
continue;
|
|
}
|
|
CUDA_CHECK(cudaMemcpy(got, dOut + IGNEUM_VEC_BASE[w], sizeof(got), cudaMemcpyDeviceToHost));
|
|
bool ok = compareWarp(got, IGNEUM_VEC_OUT[w], IGNEUM_VEC_BASE[w], how);
|
|
r.vecPass = r.vecPass && ok;
|
|
}
|
|
}
|
|
|
|
// Timed batches. Base nonces (b * nonces) mod 2^32, as in the Metal run.
|
|
CUDA_CHECK(cudaEventRecord(e0));
|
|
double t0 = wallMs();
|
|
for (int b = 1; b <= o.batches; ++b) {
|
|
uint32_t base = (uint32_t)((uint64_t)b * (uint64_t)nonces);
|
|
CUDA_CHECK(igneum_launch_hash(dDs, dOut, base, mask, nonces, (uint32_t)o.blockWarps));
|
|
}
|
|
CUDA_CHECK(cudaEventRecord(e1));
|
|
CUDA_CHECK(cudaEventSynchronize(e1));
|
|
double t1 = wallMs();
|
|
float gpuMsF = 0.f;
|
|
CUDA_CHECK(cudaEventElapsedTime(&gpuMsF, e0, e1));
|
|
double total = (double)nonces * (double)o.batches;
|
|
r.gpuMs = gpuMsF;
|
|
r.wallMsTimed = t1 - t0;
|
|
r.hashesPerSec = total / (r.gpuMs / 1000.0);
|
|
r.gbps = r.hashesPerSec * (double)IGNEUM_LOADS_PER_HASH * 4.0 / 1e9;
|
|
std::printf("timed: %d batches x %u hashes = %.0f hashes\n", o.batches, nonces, total);
|
|
std::printf(" GPU %.2f ms -> %.3f Mhash/s (%.0f hashes/s), %.2f GB/s useful (loads x 4 B)\n",
|
|
r.gpuMs, r.hashesPerSec / 1e6, r.hashesPerSec, r.gbps);
|
|
std::printf(" wall %.2f ms -> %.3f Mhash/s\n", r.wallMsTimed, total / (r.wallMsTimed / 1000.0) / 1e6);
|
|
|
|
CUDA_CHECK(cudaEventDestroy(e0));
|
|
CUDA_CHECK(cudaEventDestroy(e1));
|
|
CUDA_CHECK(cudaFree(dDs));
|
|
return r;
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------------------------
|
|
// Main
|
|
|
|
int main(int argc, char** argv) {
|
|
Options o = parseArgs(argc, argv);
|
|
std::printf("igneum-bench-cuda pack \"%s\" (test harness: no pool, no network, no wallet)\n", IGNEUM_SEED_STRING);
|
|
|
|
int count = 0;
|
|
CUDA_CHECK(cudaGetDeviceCount(&count));
|
|
if (count == 0) { std::printf("FAIL: no CUDA device\n"); return 2; }
|
|
if (o.device < 0 || o.device >= count) { std::printf("FAIL: device %d out of range (%d devices)\n", o.device, count); return 2; }
|
|
CUDA_CHECK(cudaSetDevice(o.device));
|
|
|
|
cudaDeviceProp prop;
|
|
std::memset(&prop, 0, sizeof(prop));
|
|
CUDA_CHECK(cudaGetDeviceProperties(&prop, o.device));
|
|
int drv = 0, rt = 0;
|
|
CUDA_CHECK(cudaDriverGetVersion(&drv));
|
|
CUDA_CHECK(cudaRuntimeGetVersion(&rt));
|
|
int warp = 0, clk = 0, memclk = 0, bus = 0, l2 = 0, thrSM = 0;
|
|
cudaDeviceGetAttribute(&warp, cudaDevAttrWarpSize, o.device);
|
|
cudaDeviceGetAttribute(&clk, cudaDevAttrClockRate, o.device);
|
|
cudaDeviceGetAttribute(&memclk, cudaDevAttrMemoryClockRate, o.device);
|
|
cudaDeviceGetAttribute(&bus, cudaDevAttrGlobalMemoryBusWidth, o.device);
|
|
cudaDeviceGetAttribute(&l2, cudaDevAttrL2CacheSize, o.device);
|
|
cudaDeviceGetAttribute(&thrSM, cudaDevAttrMaxThreadsPerMultiProcessor, o.device);
|
|
cudaGetLastError(); // attribute queries are informational; clear any error they left
|
|
|
|
std::printf("GPU: %s (%d SMs, compute capability %d.%d, %.0f MiB global memory)\n",
|
|
prop.name, prop.multiProcessorCount, prop.major, prop.minor, (double)prop.totalGlobalMem / 1048576.0);
|
|
std::printf(" SM clock %d MHz, memory clock %d MHz, bus %d bits, L2 %d MiB, max %d threads/SM, warp size %d\n",
|
|
clk / 1000, memclk / 1000, bus, l2 / 1048576, thrSM, warp);
|
|
std::printf("CUDA: driver %d.%d, runtime %d.%d\n", drv / 1000, (drv % 100) / 10, rt / 1000, (rt % 100) / 10);
|
|
if (warp != 32) {
|
|
std::printf("WARNING: warp size is %d, not 32. The 32-lane shuffle model does not hold on this device.\n", warp);
|
|
}
|
|
|
|
int regs = 0, blocksPerSM = 0;
|
|
CUDA_CHECK(igneum_hash_info(®s, &blocksPerSM, (uint32_t)o.blockWarps));
|
|
std::printf("kernel: %d registers/thread, %d resident blocks/SM at %d warp(s)/block = %d resident warps/SM\n",
|
|
regs, blocksPerSM, o.blockWarps, blocksPerSM * o.blockWarps);
|
|
std::printf("program: %d instructions x %d iterations, loads/hash %d, op mix %s\n",
|
|
IGNEUM_INSTR_COUNT, IGNEUM_ITERATIONS, IGNEUM_LOADS_PER_HASH, IGNEUM_OP_MIX);
|
|
std::printf("seed words: %08x %08x %08x %08x %08x %08x %08x %08x\n",
|
|
SEEDW[0], SEEDW[1], SEEDW[2], SEEDW[3], SEEDW[4], SEEDW[5], SEEDW[6], SEEDW[7]);
|
|
std::printf("day \"%s\" (d0 0x%08x, d1 0x%08x), pack dataset 2^%d words = %d MiB\n",
|
|
IGNEUM_DAY_STRING, IGNEUM_DAY0, IGNEUM_DAY1, IGNEUM_DATASET_LOG2, packMib());
|
|
#if IGNEUM_DATASET_MODE == 1
|
|
std::printf("dataset construction: memory-hard (256 MiB ChaCha cache, %d dependent cache reads per 64-byte item; proto-metal/MEMHARD.md)\n", IGNEUM_ITEM_ROUNDS);
|
|
bool cachePass = setupCache();
|
|
#else
|
|
std::printf("dataset construction: closed-form ds_elem (the original prototype dataset, not memory-hard)\n");
|
|
bool cachePass = true;
|
|
#endif
|
|
|
|
uint32_t nonces = 1u << o.batchLog2;
|
|
if (nonces % (32u * (uint32_t)o.blockWarps) != 0u) {
|
|
std::printf("FAIL: 2^%d nonces is not a multiple of %d threads per block\n", o.batchLog2, 32 * o.blockWarps);
|
|
return 2;
|
|
}
|
|
uint64_t* dOut = nullptr;
|
|
CUDA_CHECK(cudaMalloc((void**)&dOut, (size_t)nonces * sizeof(uint64_t)));
|
|
|
|
std::vector<int> sizes;
|
|
if (o.sweep) { sizes.push_back(4); sizes.push_back(64); sizes.push_back(256); sizes.push_back(512); sizes.push_back(1024); }
|
|
else sizes.push_back(o.datasetMib);
|
|
|
|
std::vector<SizeResult> results;
|
|
for (size_t i = 0; i < sizes.size(); ++i) results.push_back(runSize(o, sizes[i], dOut, nonces));
|
|
CUDA_CHECK(cudaFree(dOut));
|
|
|
|
std::printf("\n=== summary (%s, pack %s, batch 2^%d x %d, %d warp(s)/block, GPU-event time) ===\n",
|
|
prop.name, IGNEUM_SEED_STRING, o.batchLog2, o.batches, o.blockWarps);
|
|
std::printf("| dataset MiB | %s ms (second) | Mhash/s | GB/s useful | random loads/s (G) | loads/hash | dataset self-test | vectors |\n",
|
|
IGNEUM_DATASET_MODE == 1 ? "build" : "fill");
|
|
std::printf("|---|---|---|---|---|---|---|---|\n");
|
|
bool overall = cachePass, anyVec = false;
|
|
for (size_t i = 0; i < results.size(); ++i) {
|
|
const SizeResult& r = results[i];
|
|
overall = overall && r.dsPass && (!r.vecChecked || r.vecPass);
|
|
anyVec = anyVec || r.vecChecked;
|
|
std::printf("| %d | %.2f | %.3f | %.2f | %.2f | %d | %s | %s |\n",
|
|
r.mib, r.fillSecondMs, r.hashesPerSec / 1e6, r.gbps,
|
|
r.hashesPerSec * (double)IGNEUM_LOADS_PER_HASH / 1e9, IGNEUM_LOADS_PER_HASH,
|
|
r.dsPass ? "PASS" : "FAIL",
|
|
r.vecChecked ? (r.vecPass ? "PASS (3 warps, standalone and in batch)" : "FAIL") : "skipped (not pack size)");
|
|
}
|
|
#if IGNEUM_DATASET_MODE == 1
|
|
std::printf("cache: GPU fill %.2f ms (second), host fill %.1f ms one thread, cache check %s\n", gCacheFillSecondMs, gCacheHostMs, cachePass ? "PASS" : "FAIL");
|
|
CUDA_CHECK(cudaFree(gCache));
|
|
#endif
|
|
if (!anyVec) std::printf("NOTE: no vectors were checked. Run at %d MiB (the default) to verify against the Mac.\n", packMib());
|
|
std::printf("OVERALL: %s\n", overall ? "PASS" : "FAIL");
|
|
return overall ? 0 : 1;
|
|
}
|