igneum/proto-newpow/mma-shadow/verify_ref.c
igneum-labs a664af6fc9 Horizon: lane 8 (new-proof-of-work) lands: three schemes, two prototypes measured on rented 4090s
docs/analysis/horizon/new-pow.md sections 0 to 9: scheme A (mining is proving) never, on bytes,
the verifier and sampleability; scheme B (the tensor-shaped integer shadow) prototyped as
proto-newpow/mma-shadow and measured, never as class content on the energy reading, with the R8
two-output correction; scheme C (proof of stored state, sd1: the daily dataset derived from the
execution state) prototyped as proto-newpow/state-dataset, measured on the GPU and the box's
CPU, and put forward as the class v5 candidate with its spec items and the Devnet 2 gate. The
lane's standing rule: a shadow lever only works through joules the honest card is forced to
spend, so shadow work goes where the GPU is least efficient per op. Chip rows in
sim/horizon/new-pow/chip_rows.py by the chip-model-v3 method. Rented box addresses replaced by
placeholders in the READMEs and the run script.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
2026-10-06 20:18:39 +00:00

152 lines
7.3 KiB
C

// verify_ref.c (proto-newpow/mma-shadow): plain C CPU reference for the mx8+mm8xR prototype.
// Reads a bench --dump file (lines: base lane value_hex, 32 lanes per warp), recomputes every warp with a
// register-major 32-lane interpreter of the pack's program (ref_program.inc, generated from kernel.cu), lazy
// dataset words through memhard.h's mh_word over a host-filled 256 MiB cache, the mm8 block in the spec layout,
// then the fold. Prints "N of 1024 lanes equal" and the first mismatch, then times the verifier per 32-lane unit.
// Build: gcc -O2 -o verify_ref verify_ref.c Run: taskset -c 2 ./verify_ref dump.txt <R> [--time]
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <time.h>
#define IGNEUM_NO_CUDA
#include "program.h"
#include "vectors.h"
#include "memhard.h"
#include "mm8_block.h"
static const uint16_t MM8_TABLE[IGNEUM_MM8_R_MAX] = IGNEUM_MM8_TABLE_INIT;
static uint32_t splitmix32(uint32_t x) { x ^= x >> 16; x *= 0x7feb352du; x ^= x >> 15; x *= 0x846ca68bu; x ^= x >> 16; return x; }
static uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
static uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
static uint32_t umulhi(uint32_t a, uint32_t b) { return (uint32_t)(((uint64_t)a * (uint64_t)b) >> 32); }
static double nowMs(void) { struct timespec ts; clock_gettime(CLOCK_MONOTONIC, &ts); return ts.tv_sec * 1e3 + ts.tv_nsec / 1e6; }
#define L for (int l = 0; l < 32; ++l)
// One mm8 step on the 32-lane register file (spec layout, identical to family-probe.cu's mm8 warp_ref):
// r[c] += C[l >> 2][2 * (l & 3)] (d0), r[c2] += C[l >> 2][2 * (l & 3) + 1] (d1).
static void mm8_step(uint32_t R[8][32], int a, int b, int c, int c2) {
uint32_t A[32], B[32];
memcpy(A, R[a], sizeof A);
memcpy(B, R[b], sizeof B);
L {
int row = l >> 2, col0 = 2 * (l & 3);
uint32_t acc0 = 0u, acc1 = 0u;
for (int k = 0; k < 16; ++k) {
uint32_t av = (A[row * 4 + k / 4] >> (8 * (k % 4))) & 0xffu;
acc0 += av * ((B[col0 * 4 + k / 4] >> (8 * (k % 4))) & 0xffu);
acc1 += av * ((B[(col0 + 1) * 4 + k / 4] >> (8 * (k % 4))) & 0xffu);
}
R[c][l] += acc0;
R[c2][l] += acc1;
}
}
static void hash_unit(const uint32_t* cache, uint32_t base, int Rsteps, uint64_t out[32]) {
static const uint32_t SEEDW[8] = IGNEUM_SEEDW_INIT;
uint32_t R[8][32], T[32], SEL[32];
const uint32_t mask = IGNEUM_MASK;
L {
uint32_t nonce = base + (uint32_t)l;
for (int i = 0; i < 8; ++i) {
uint32_t x = nonce ^ SEEDW[i]; x += 0x9e3779b9u * (uint32_t)(i + 1); x = splitmix32(x);
R[i][l] = x ^ SEEDW[(i + 1) & 7];
}
}
for (int it = 0; it < 8; ++it) {
L { SEL[l] = R[0][l]; }
#include "ref_program.inc"
for (int k = 0; k < Rsteps; ++k) {
uint16_t v = MM8_TABLE[k];
mm8_step(R, v & 7, (v >> 3) & 7, (v >> 6) & 7, (v >> 9) & 7);
}
}
L {
uint32_t lo = R[0][l] ^ rotl_imm(R[1][l], 7u) ^ rotl_imm(R[2][l], 14u) ^ rotl_imm(R[3][l], 21u);
uint32_t hi = R[4][l] ^ rotl_imm(R[5][l], 9u) ^ rotl_imm(R[6][l], 18u) ^ rotl_imm(R[7][l], 27u);
out[l] = ((uint64_t)hi << 32) | (uint64_t)lo;
}
}
static uint64_t fnv1a64(const void* p, size_t n) {
const uint8_t* b = (const uint8_t*)p; uint64_t h = 0xcbf29ce484222325ull;
for (size_t i = 0; i < n; ++i) { h ^= b[i]; h *= 0x100000001b3ull; }
return h;
}
int main(int argc, char** argv) {
if (argc < 3) { printf("usage: verify_ref <dump file> <R> [--time] [--self-test]\n"); return 2; }
const char* path = argv[1];
int Rsteps = atoi(argv[2]);
int doTime = 0, selfTest = 0;
for (int i = 3; i < argc; ++i) { if (!strcmp(argv[i], "--time")) doTime = 1; if (!strcmp(argv[i], "--self-test")) selfTest = 1; }
if (Rsteps < 0 || Rsteps > IGNEUM_MM8_R_MAX) { printf("R out of range\n"); return 2; }
size_t words = (size_t)1u << IGNEUM_CACHE_LOG2_WORDS;
uint32_t* cache = (uint32_t*)malloc(words * 4u);
if (!cache) { printf("no memory for the cache\n"); return 2; }
double c0 = nowMs();
for (uint32_t seg = 0; seg < IGNEUM_CACHE_SEGMENTS; ++seg) mh_cache_segment(cache, seg);
double c1 = nowMs();
uint64_t fnv = fnv1a64(cache, words * 4u);
printf("cache: host fill %.1f ms, FNV-1a 64 %016llx vs Mac %016llx %s\n", c1 - c0, (unsigned long long)fnv,
(unsigned long long)IGNEUM_CACHE_FNV64, fnv == IGNEUM_CACHE_FNV64 ? "PASS" : "FAIL");
if (fnv != IGNEUM_CACHE_FNV64) return 1;
if (selfTest) { // R = 0 interpreter against the pack's vectors
int ok = 1;
for (int w = 0; w < IGNEUM_VEC_WARPS; ++w) {
uint64_t out[32]; hash_unit(cache, IGNEUM_VEC_BASE[w], 0, out);
int bad = 0; L { if (out[l] != IGNEUM_VEC_OUT[w][l]) ++bad; }
printf("self-test R=0 vector base %u: %s (%d lanes differ)\n", IGNEUM_VEC_BASE[w], bad ? "FAIL" : "PASS", bad);
ok = ok && !bad;
}
if (!ok) return 1;
}
// Read the dump: 32 consecutive lines per warp, same base
FILE* f = fopen(path, "r");
if (!f) { printf("cannot open %s\n", path); return 2; }
uint32_t* bases = NULL; uint64_t* vals = NULL; size_t n = 0, cap = 0;
unsigned base; int lane; unsigned long long v;
while (fscanf(f, "%u %d %llx", &base, &lane, &v) == 3) {
if (n == cap) { cap = cap ? cap * 2 : 1024; bases = realloc(bases, cap * 4); vals = realloc(vals, cap * 8); }
bases[n] = base; vals[n] = v; ++n;
}
fclose(f);
if (n % 32 != 0) { printf("dump has %zu lines, not a multiple of 32\n", n); return 2; }
size_t units = n / 32;
printf("dump: %zu lines, %zu units, R = %d (%d mm8 per hash)\n", n, units, Rsteps, 8 * Rsteps);
size_t equal = 0; int firstPrinted = 0;
double v0 = nowMs();
for (size_t u = 0; u < units; ++u) {
uint64_t out[32]; hash_unit(cache, bases[u * 32], Rsteps, out);
L {
if (out[l] == vals[u * 32 + l]) ++equal;
else if (!firstPrinted) { firstPrinted = 1; printf("first mismatch: base %u lane %d: cpu %016llx gpu %016llx\n", bases[u * 32], l, (unsigned long long)out[l], (unsigned long long)vals[u * 32 + l]); }
}
}
double v1 = nowMs();
printf("%zu of %zu lanes equal (%s) [%.3f ms per unit during the check]\n", equal, n, equal == n ? "PASS" : "FAIL", (v1 - v0) / units);
if (doTime) {
int Rs[2] = { 0, Rsteps }; double ms[2] = { 0, 0 };
for (int t = 0; t < 2; ++t) {
uint64_t out[32]; volatile uint64_t sink = 0;
hash_unit(cache, bases[0], Rs[t], out); // warm
double t0 = nowMs();
for (size_t u = 0; u < units; ++u) { hash_unit(cache, bases[u * 32], Rs[t], out); sink ^= out[0]; }
double t1 = nowMs();
ms[t] = (t1 - t0) / units;
(void)sink;
}
printf("verifier: %.3f ms per unit at R=0, %.3f ms per unit at R=%d, mm8 block delta %.3f ms per unit (%.2f us per mm8 step), averaged over %zu units, one core\n",
ms[0], ms[1], Rsteps, ms[1] - ms[0], Rsteps ? (ms[1] - ms[0]) * 1000.0 / (8.0 * Rsteps) : 0.0, units);
printf("VERIFY R=%d equal=%zu of=%zu ms_r0=%.3f ms_r=%.3f delta=%.3f\n", Rsteps, equal, n, ms[0], ms[1], ms[1] - ms[0]);
}
free(cache); free(bases); free(vals);
return equal == n ? 0 : 1;
}