igneum-pow 0.2.0: generator v2 draws exactly 16 load slots from instructions 1..63, a load's source from the registers written earlier and not read by a load since, the other 48 ops from the ten non-load weights; accept.rs is spec 01 section 1.4.6 (static: no stale load source, every register injected; dynamic: 64 units on the seed-keyed closed-form dataset, no constant bit, no lane-constant site, under 164 saturated, bias within 136 of 1024, distinct addresses above 245,760); a rejected candidate is replaced by the next attempt of the seed (seed || k_le32), 32 a consensus fault. Packs carry the generator version, attempt and program id. Version 1 kept as generate_v1 for the census. Packs: igneum-genesis, igneum-hourly, igneum-genesis-mh regenerated by igneum-pow export; new igneum-devnet-v4-epoch0 (devnet genesis hash, day bytes 20730). Checks: Rust 39 of 39 tests; Metal natively via the Swift port (export cross-check 3 of 3 warps, identical programs and vectors on five seeds incl. three with attempt 1, fuzz 2,000 of 2,000); CUDA emu 4 of 4 packs; OpenCL emu 2 packs x 2 configurations; Apple OpenCL 4 of 4 packs at 27.9 Mhash/s. Census 20,000: 5.225 percent rejected, accepted distinct mean 127.887. Spec 01 0.2 (1.4.2, 1.4.3, 1.4.6, 1.11, 1.15, 1.16, 1.17), igneum-pow README, the CUDA, OpenCL and Metal test notes, bench-log entry, ledger M5 and M6 Fixed. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
108 lines
5.7 KiB
C++
108 lines
5.7 KiB
C++
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000". Do not edit by hand.
|
|
// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against.
|
|
// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference).
|
|
// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C.
|
|
#pragma once
|
|
#ifdef __cplusplus
|
|
#include <cstdint>
|
|
#else
|
|
#include <stdint.h>
|
|
#endif
|
|
#if defined(__CUDACC__)
|
|
#define IGNEUM_HD __host__ __device__ __forceinline__
|
|
#elif defined(_MSC_VER) && !defined(__cplusplus)
|
|
#define IGNEUM_HD static __inline
|
|
#else
|
|
#define IGNEUM_HD static inline
|
|
#endif
|
|
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
|
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
|
|
#define MH_CACHE_LINE_MASK 0x003fffffu
|
|
#define MH_SEGMENT_LINES 64u
|
|
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
|
IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
|
|
|
// y = ChaCha12 core(x) + x
|
|
IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) {
|
|
for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i];
|
|
for (uint32_t r = 0u; r < 6u; ++r) {
|
|
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
|
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
|
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
|
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
|
}
|
|
for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i];
|
|
}
|
|
|
|
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
|
IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) {
|
|
uint32_t prev[16]; uint32_t x[16]; uint32_t y[16];
|
|
for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u;
|
|
for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
|
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
|
x[4] = 0xceed56d7u ^ prev[4];
|
|
x[5] = 0x9ba270d2u ^ prev[5];
|
|
x[6] = 0x82caab2du ^ prev[6];
|
|
x[7] = 0x81ebce0eu ^ prev[7];
|
|
x[8] = 0x12b6ecf1u ^ prev[8];
|
|
x[9] = 0xd0f3fd7cu ^ prev[9];
|
|
x[10] = 0xd872eefeu ^ prev[10];
|
|
x[11] = 0xc158c7bdu ^ prev[11];
|
|
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
|
mh_chacha_block(x, y);
|
|
uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
|
for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
|
}
|
|
}
|
|
|
|
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
|
IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) {
|
|
s[0] = (s[0] ^ (0xc6892460u + rk)) * 0xf351d601u;
|
|
s[1] = (s[1] ^ (0x25b7228au + rk)) * 0xa3bb398fu;
|
|
s[2] = (s[2] ^ (0xcd515004u + rk)) * 0xb5a09e35u;
|
|
s[3] = (s[3] ^ (0x2846527au + rk)) * 0x7509c9c1u;
|
|
s[4] = (s[4] ^ (0xa6324241u + rk)) * 0x6bbf31e9u;
|
|
s[5] = (s[5] ^ (0x36e3ec53u + rk)) * 0xfc849a79u;
|
|
s[6] = (s[6] ^ (0x82961bacu + rk)) * 0xded91851u;
|
|
s[7] = (s[7] ^ (0x0f97ba7du + rk)) * 0x8d9113d1u;
|
|
s[8] = (s[8] ^ (0xb6f921a9u + rk)) * 0x0ff15225u;
|
|
s[9] = (s[9] ^ (0x3ada24e5u + rk)) * 0x3a5bdd41u;
|
|
s[10] = (s[10] ^ (0xde20ab91u + rk)) * 0xab533435u;
|
|
s[11] = (s[11] ^ (0x5378eeb2u + rk)) * 0xe1c55ad5u;
|
|
s[12] = (s[12] ^ (0x7d161662u + rk)) * 0xe6d3bd0du;
|
|
s[13] = (s[13] ^ (0x89353cc1u + rk)) * 0x9d9ffbbdu;
|
|
s[14] = (s[14] ^ (0xb1aa03a2u + rk)) * 0xbb2a3cf3u;
|
|
s[15] = (s[15] ^ (0x788acae6u + rk)) * 0x50a7c08du;
|
|
MH_QR(s[0], s[4], s[8], s[12], 17u, 12u, 20u, 23u) MH_QR(s[1], s[5], s[9], s[13], 17u, 12u, 20u, 23u)
|
|
MH_QR(s[2], s[6], s[10], s[14], 17u, 12u, 20u, 23u) MH_QR(s[3], s[7], s[11], s[15], 17u, 12u, 20u, 23u)
|
|
MH_QR(s[0], s[5], s[10], s[15], 7u, 3u, 27u, 16u) MH_QR(s[1], s[6], s[11], s[12], 7u, 3u, 27u, 16u)
|
|
MH_QR(s[2], s[7], s[8], s[13], 7u, 3u, 27u, 16u) MH_QR(s[3], s[4], s[9], s[14], 7u, 3u, 27u, 16u)
|
|
}
|
|
|
|
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
|
|
IGNEUM_HD void mh_item(const uint32_t* cache, uint32_t t, uint32_t* s) {
|
|
s[0] = 0xceed56d7u;
|
|
s[1] = 0x9ba270d2u;
|
|
s[2] = 0x82caab2du;
|
|
s[3] = 0x81ebce0eu;
|
|
s[4] = 0x12b6ecf1u;
|
|
s[5] = 0xd0f3fd7cu;
|
|
s[6] = 0xd872eefeu;
|
|
s[7] = 0xc158c7bdu;
|
|
s[8] = t * 0xf351d601u + 0xc6892460u;
|
|
s[9] = t * 0xa3bb398fu + 0x25b7228au;
|
|
s[10] = t * 0xb5a09e35u + 0xcd515004u;
|
|
s[11] = t * 0x7509c9c1u + 0x2846527au;
|
|
s[12] = t * 0x6bbf31e9u + 0xa6324241u;
|
|
s[13] = t * 0xfc849a79u + 0x36e3ec53u;
|
|
s[14] = t * 0xded91851u + 0x82961bacu;
|
|
s[15] = t * 0x8d9113d1u + 0x0f97ba7du;
|
|
for (uint32_t r = 0u; r < 8u; ++r) {
|
|
mh_mixer(s, 0x9E3779B9u * (r + 1u));
|
|
const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
|
for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
|
}
|
|
mh_mixer(s, 0x9E3779B9u * 9u);
|
|
}
|
|
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
|
IGNEUM_HD uint32_t mh_word(const uint32_t* cache, uint32_t w) { uint32_t s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|