diff --git a/proto-cuda/nvrtc/emu/packfile-test.c b/proto-cuda/nvrtc/emu/packfile-test.c index 0bcc3d3b9..f962386df 100644 --- a/proto-cuda/nvrtc/emu/packfile-test.c +++ b/proto-cuda/nvrtc/emu/packfile-test.c @@ -110,6 +110,37 @@ int main(int argc, char** argv) { CHECK(pf_load(dir, &pk, err, sizeof(err)) == 1 && pk.attempt == 0, "no attempt line reads as attempt 0 and the bare words load"); CHECK(strcmp(pk.programClass, "v2") == 0 && pk.eraHex[0] == 0, "a generator 2 pack without a class line is class v2 with no era"); } + + // 4. research class ds55 (8 October 2026): a pack whose program.h carries IGNEUM_DATASET_WORDS (a dataset that is not a + // power of two, every load through the multiply-shift of spec 01 section 1.13.3) is sized by that count, never by + // 1 << IGNEUM_DATASET_LOG2 (the floor); a pack without the line keeps the power-of-two count, so every existing pack + // loads as before. Known-failed first: the old reader sized this pack at 2^30 words (a quarter of the items built, + // the sampled words above 2^30 never read), and the self-test failed on the 5.5 GiB pack. + { + char text[2400], sw[200], kw[200]; + words_hex(att1, sw); words_hex(keyw, kw); + snprintf(text, sizeof(text), + "#define IGNEUM_SEED_BYTES_HEX \"%s\"\n#define IGNEUM_DAY_BYTES_HEX \"%s\"\n#define IGNEUM_GENERATOR 2\n#define IGNEUM_PROGRAM_ATTEMPT 1\n" + "#define IGNEUM_DATASET_LOG2 30\n#define IGNEUM_DATASET_WORDS 1476395008u\n#define IGNEUM_DATASET_ITEMS 92274688u\n#define IGNEUM_DATASET_BYTES 5905580032ull\n" + "#define IGNEUM_DATASET_MULSHIFT 1\n#define IGNEUM_MASK 0x57ffffffu\n#define IGNEUM_DATASET_MODE 1\n" + "#define IGNEUM_SEEDW_INIT { %s }\n#define IGNEUM_KEY_INIT { %s }\n#define IGNEUM_CACHE_LOG2_WORDS 26\n#define IGNEUM_CACHE_SEGMENTS 4096u\n", + EPOCH_34, DAY_20731, sw, kw); + write_file(dir, "program.h", text); + write_file(dir, "seeds.txt", "epoch_seed_hex " EPOCH_34 "\nday_seed_hex " DAY_20731 "\n"); + err[0] = 0; + CHECK(pf_load(dir, &pk, err, sizeof(err)) == 1 && pk.datasetLog2 == 30 && pk.datasetWords == 1476395008u && pk.datasetItems == 92274688u && pk.datasetMulshift == 1, + "known-failed: a pack with IGNEUM_DATASET_WORDS is sized by the word count (1476395008 words, 92274688 items, multiply-shift), not by 1 << 30"); + if (err[0]) printf(" (%s)\n", err); + CHECK(pk.datasetWords != (1u << pk.datasetLog2) && pk.datasetWords / 16u == pk.datasetItems, "the count is not the floor's power of two and the items are words / 16"); + write_program_h(dir, EPOCH_34, DAY_20731, 1, att1, keyw); + err[0] = 0; + CHECK(pf_load(dir, &pk, err, sizeof(err)) == 1 && pk.datasetLog2 == 28 && pk.datasetWords == (1u << 28) && pk.datasetItems == (1u << 24) && pk.datasetMulshift == 0, + "known-good: a pack without the line keeps 1 << IGNEUM_DATASET_LOG2 (2^28 words, 2^24 items, the mask path)"); + if (checked_in) { + err[0] = 0; + CHECK(pf_load(checked_in, &pk, err, sizeof(err)) == 1 && pk.datasetWords == (1u << pk.datasetLog2) && pk.datasetMulshift == 0, "known-good: the checked-in pack is sized by its power of two"); + } + } // 4. Program classes (Counter ASIC 2.0, 5 October 2026, spec 01 section 1.4.5): a generator this worker does not // run is refused; a generator 3 pack is class v3 and carries its era seed; a class line that contradicts the // generator is refused; a pack with no generator line at all (generator 1, the retired lever generator) is refused. diff --git a/proto-cuda/nvrtc/packfile.h b/proto-cuda/nvrtc/packfile.h index 932be43b7..57e332d0d 100644 --- a/proto-cuda/nvrtc/packfile.h +++ b/proto-cuda/nvrtc/packfile.h @@ -23,6 +23,10 @@ typedef struct { // program.h uint32_t datasetLog2, cacheLog2Words, cacheSegments, datasetMode, generator; + /* Research class ds55 (8 October 2026): IGNEUM_DATASET_WORDS when program.h carries it (a dataset that is not a power of + * two; every load address is (src * words) >> 32, IGNEUM_DATASET_MULSHIFT 1), else 1 << datasetLog2, so every existing + * pack is sized as before. A host allocates datasetWords words and builds datasetItems (= words / 16) items. */ + uint32_t datasetWords, datasetItems, datasetMulshift; uint32_t attempt; // IGNEUM_PROGRAM_ATTEMPT: seedw are the words of this attempt of the epoch seed (0 = bare seed) uint32_t seedw[8], keyw[8]; char seedString[600]; @@ -297,6 +301,13 @@ static int pf_load(const char* dir, PfPack* pk, char* err, size_t cap) { if (!prog) { char m[480]; snprintf(m, sizeof(m), "cannot read %.400s/program.h", dir); return pf_fail(err, cap, m); } if (!pf_define_u32(prog, "IGNEUM_DATASET_LOG2", &pk->datasetLog2)) { free(prog); return pf_fail(err, cap, "program.h has no IGNEUM_DATASET_LOG2"); } if (!pf_define_u32(prog, "IGNEUM_DATASET_MODE", &pk->datasetMode)) pk->datasetMode = 0; + if (pk->datasetLog2 > 31) { free(prog); return pf_fail(err, cap, "program.h IGNEUM_DATASET_LOG2 above 31: the word index is 32-bit"); } + if (!pf_define_u32(prog, "IGNEUM_DATASET_WORDS", &pk->datasetWords)) pk->datasetWords = 1u << pk->datasetLog2; + if (pk->datasetWords < (1u << 16) || (pk->datasetWords & 0xffffu) != 0) { free(prog); return pf_fail(err, cap, "program.h IGNEUM_DATASET_WORDS is not a multiple of 65,536 words"); } + if (!pf_define_u32(prog, "IGNEUM_DATASET_ITEMS", &pk->datasetItems)) pk->datasetItems = pk->datasetWords / 16u; + if (pk->datasetItems != pk->datasetWords / 16u) { free(prog); return pf_fail(err, cap, "program.h IGNEUM_DATASET_ITEMS is not IGNEUM_DATASET_WORDS / 16"); } + if (!pf_define_u32(prog, "IGNEUM_DATASET_MULSHIFT", &pk->datasetMulshift)) pk->datasetMulshift = 0; + if (pk->datasetMulshift != ((pk->datasetWords & (pk->datasetWords - 1u)) != 0u)) { free(prog); return pf_fail(err, cap, "program.h IGNEUM_DATASET_MULSHIFT does not match the word count (1 exactly when the count is not a power of two)"); } if (pk->datasetMode != 1) { free(prog); return pf_fail(err, cap, "the pack is not memory-hard (IGNEUM_DATASET_MODE 1); the one-click workers serve memory-hard packs only"); } if (!pf_define_u32(prog, "IGNEUM_CACHE_LOG2_WORDS", &pk->cacheLog2Words)) { free(prog); return pf_fail(err, cap, "program.h has no IGNEUM_CACHE_LOG2_WORDS"); } if (!pf_define_u32(prog, "IGNEUM_CACHE_SEGMENTS", &pk->cacheSegments)) { free(prog); return pf_fail(err, cap, "program.h has no IGNEUM_CACHE_SEGMENTS"); } diff --git a/proto-cuda/nvrtc/worker.cpp b/proto-cuda/nvrtc/worker.cpp index f373a81a2..af6976d74 100644 --- a/proto-cuda/nvrtc/worker.cpp +++ b/proto-cuda/nvrtc/worker.cpp @@ -415,6 +415,7 @@ struct Pair { std::string dir, epochHex, dayHex, seedString; uint32_t sw[8] = {0}, kw[8] = {0}; uint32_t datasetLog2 = 0, words = 0, cacheWords = 0, cacheSegments = 0; + uint32_t items = 0, mulshift = 0; // research class ds55: words / 16 items; 1 when the pack's loads are (src * words) >> 32 CUmodule modKernel = nullptr, modBound = nullptr; CUfunction fCacheFill = nullptr, fBuild = nullptr, fHashBound = nullptr; CUdeviceptr cache = 0, ds = 0; @@ -821,7 +822,10 @@ static Pair* buildPair(Ctx& c, const std::string& dir, CUstream s, std::string& Pair* p = new Pair(); p->dir = dir; p->epochHex = pk.epochHex; p->dayHex = pk.dayHex; p->seedString = pk.seedString; std::memcpy(p->sw, pk.seedw, 32); std::memcpy(p->kw, pk.keyw, 32); - p->datasetLog2 = pk.datasetLog2; p->words = 1u << pk.datasetLog2; p->cacheWords = 1u << pk.cacheLog2Words; p->cacheSegments = pk.cacheSegments; + // the dataset's size is the pack's word count (IGNEUM_DATASET_WORDS when present, else 1 << IGNEUM_DATASET_LOG2: packfile.h), + // never the power of two of the floor: a 5.5 GiB research pack holds 1,476,395,008 words and 92,274,688 items + p->datasetLog2 = pk.datasetLog2; p->words = pk.datasetWords; p->items = pk.datasetItems; p->mulshift = pk.datasetMulshift; + p->cacheWords = 1u << pk.cacheLog2Words; p->cacheSegments = pk.cacheSegments; // Compile Compiled ck, cb; std::vector kernelNames = { "igneum_cache_fill", "igneum_build" }; @@ -899,7 +903,7 @@ static Pair* buildPair(Ctx& c, const std::string& dir, CUstream s, std::string& { CUresult r = c.drv.memAlloc(&p->ds, dsBytes); if (r != CUDA_SUCCESS) { err = "cuMemAlloc dataset: " + c.err(r); p->ds = 0; std::free(hLeaves); releasePair(c, p); return nullptr; } - uint32_t nItems = p->words / 16u, block = 256u, grid = (nItems + block - 1u) / block; + uint32_t nItems = p->items, block = 256u, grid = (nItems + block - 1u) / block; if (hLeaves) { // class v5: igneum_build(ds, cache, leaves, nLeaves, nItems), the leaf buffer freed once the build has run CUdeviceptr dLeaves = 0; @@ -985,8 +989,9 @@ static Pair* buildPair(Ctx& c, const std::string& dir, CUstream s, std::string& } static std::string pairSummary(const Pair* p) { - return fmt("nvrtc %.0f cache %.0f dataset %.0f hot %.0f check %.0f race %.0f ms variant %s class %s%s; %s", p->compileMs, p->cacheMs, p->dsMs, p->hotMs, p->checkMs, p->raceMs, p->variant.c_str(), p->programClass.c_str(), - p->stateLeaves ? fmt(" (state leaves %u, uploaded for the build and freed)", p->stateLeaves).c_str() : "", p->check.c_str()); + return fmt("nvrtc %.0f cache %.0f dataset %.0f hot %.0f check %.0f race %.0f ms variant %s class %s%s%s; %s", p->compileMs, p->cacheMs, p->dsMs, p->hotMs, p->checkMs, p->raceMs, p->variant.c_str(), p->programClass.c_str(), + p->stateLeaves ? fmt(" (state leaves %u, uploaded for the build and freed)", p->stateLeaves).c_str() : "", + p->mulshift ? fmt(" (dataset %u words, %u items, not a power of two: loads are (src * words) >> 32)", p->words, p->items).c_str() : "", p->check.c_str()); } // --------------------------------------------------------------------------------------------- @@ -1523,8 +1528,8 @@ int main(int argc, char** argv) { } if (o.check) { std::printf("check PASS %s in %.0f ms: %s\n", o.pack.c_str(), wallMs() - t0, pairSummary(cur).c_str()); - std::printf(" epoch %s day %s, dataset 2^%u words, cache 2^%u words in %u segments, %d registers, %d blocks/SM at %d warp(s)/block, target %s\n", - cur->epochHex.c_str(), cur->dayHex.c_str(), cur->datasetLog2, (unsigned)__builtin_ctz(cur->cacheWords), cur->cacheSegments, cur->regs, cur->blocksPerSM, c.blockWarps, c.archOpt.c_str()); + std::printf(" epoch %s day %s, dataset %s words, cache 2^%u words in %u segments, %d registers, %d blocks/SM at %d warp(s)/block, target %s\n", + cur->epochHex.c_str(), cur->dayHex.c_str(), (cur->mulshift ? fmt("%u (%u items, multiply-shift)", cur->words, cur->items) : fmt("2^%u", cur->datasetLog2)).c_str(), (unsigned)__builtin_ctz(cur->cacheWords), cur->cacheSegments, cur->regs, cur->blocksPerSM, c.blockWarps, c.archOpt.c_str()); releasePair(c, cur); return 0; } diff --git a/proto-opencl/host.c b/proto-opencl/host.c index 563e8d81b..eedc8259e 100644 --- a/proto-opencl/host.c +++ b/proto-opencl/host.c @@ -235,7 +235,19 @@ typedef struct { int warps; // --warps N: persistent warps for a variant-5 pack (IGNEUM_PERSISTENT_WARPS); 0 = 2048 } Options; -static int packMib(void) { return (int)(((1ull << IGNEUM_DATASET_LOG2) * 4ull) >> 20); } +/* The compiled-in pack's word count: IGNEUM_DATASET_WORDS when program.h carries it (research class ds55, 8 October 2026: a + * dataset that is not a power of two, every load (src * words) >> 32), else 1 << IGNEUM_DATASET_LOG2. */ +#ifdef IGNEUM_DATASET_WORDS +#define IGNEUM_PACK_WORDS ((uint32_t)IGNEUM_DATASET_WORDS) +#else +#define IGNEUM_PACK_WORDS (1u << IGNEUM_DATASET_LOG2) +#endif +#ifndef IGNEUM_DATASET_MULSHIFT +#define IGNEUM_DATASET_MULSHIFT 0 +#endif +/* The pack's range reduction of a 32-bit source value to a word index, for the host's own sample points. */ +static uint32_t packIndex(uint32_t src, uint32_t words, int mulshift) { return mulshift ? (uint32_t)(((uint64_t)src * (uint64_t)words) >> 32) : (src & (words - 1u)); } +static int packMib(void) { return (int)(((uint64_t)IGNEUM_PACK_WORDS * 4ull) >> 20); } static void usage(void) { printf( @@ -843,9 +855,13 @@ static SizeResult runSize(Device* dv, const DeviceInfo* di, const Options* o, in r.mib = mib; r.words = (uint32_t)(bytes / 4ull); mask = r.words - 1u; - atPackSize = (r.words == (1u << IGNEUM_DATASET_LOG2)); - printf("\n=== dataset %d MiB (2^%d words, mask 0x%08x)%s ===\n", mib, log2u32(r.words), mask, - atPackSize ? "" : " [not the pack size: vectors skipped, dataset head and random points still checked]"); + atPackSize = (r.words == IGNEUM_PACK_WORDS); + if (IGNEUM_DATASET_MULSHIFT && !atPackSize) { printf("FAIL: the pack's dataset is %u words (not a power of two, multiply-shift loads): run it at the pack size only\n", (unsigned)IGNEUM_PACK_WORDS); exit(2); } + if (IGNEUM_DATASET_MULSHIFT) + printf("\n=== dataset %d MiB (%u words, %u items, not a power of two: loads are (src * words) >> 32) ===\n", mib, (unsigned)r.words, (unsigned)(r.words / 16u)); + else + printf("\n=== dataset %d MiB (2^%d words, mask 0x%08x)%s ===\n", mib, log2u32(r.words), mask, + atPackSize ? "" : " [not the pack size: vectors skipped, dataset head and random points still checked]"); if ((uint64_t)di->maxAlloc < bytes) { printf("FAIL: CL_DEVICE_MAX_MEM_ALLOC_SIZE is %llu MiB, the dataset needs %d MiB in one buffer\n", (unsigned long long)(di->maxAlloc >> 20), mib); exit(2); @@ -921,7 +937,7 @@ static SizeResult runSize(Device* dv, const DeviceInfo* di, const Options* o, in z = (z ^ (z >> 30)) * 0xBF58476D1CE4E5B9ull; z = (z ^ (z >> 27)) * 0x94D049BB133111EBull; z ^= z >> 31; - idx = (uint32_t)z & mask; + idx = packIndex((uint32_t)z, r.words, IGNEUM_DATASET_MULSHIFT); readWords(dv, dDs, idx, 1, &v); #if IGNEUM_DATASET_MODE == 1 want = host_ds_word(idx); @@ -1110,7 +1126,7 @@ static cl_int setScratchArgs(cl_kernel k, cl_uint firstArg, cl_uint units) { gSalt += units; return e; } -static uint32_t gServeWords = 1u << IGNEUM_DATASET_LOG2; +static uint32_t gServeWords = IGNEUM_PACK_WORDS; #if IGNEUM_DATASET_MODE == 1 static uint32_t gServeCacheWords = 1u << IGNEUM_CACHE_LOG2_WORDS; static uint32_t gServeSegments = IGNEUM_CACHE_SEGMENTS; @@ -2232,7 +2248,7 @@ int main(int argc, char** argv) { if (n > 1 && (o.packDir[n - 1] == '/' || o.packDir[n - 1] == '\\')) ((char*)o.packDir)[n - 1] = 0; if (!pf_load(o.packDir, &gPack, perr, sizeof(perr))) { printf("error 0 pack %s: %s\n", o.packDir, perr); fflush(stdout); return 2; } gGeneric = 1; - gServeWords = 1u << gPack.datasetLog2; + gServeWords = gPack.datasetWords; /* the pack's own count: not a power of two under research class ds55 */ #if IGNEUM_DATASET_MODE == 1 gServeCacheWords = 1u << gPack.cacheLog2Words; gServeSegments = gPack.cacheSegments;