Exporter writes kernel.cl next to kernel.cu (same instruction list; memory-hard core emitted in a third, OpenCL C dialect with the same literals as memhard.h). Pack headers are now C99-safe so a plain C host can include them. proto-opencl/host.c: C99 + OpenCL 1.2 API, device list, runtime build, cache fill and FNV check, dataset build and self-test, 3 vector warps standalone and in batch, bench and sweep as host.cu, whole-batch fingerprint. The 32-lane exchange is sub_group_shuffle_xor only when the queried sub-group size for a 32-item work-group is exactly 32; otherwise a local-memory exchange with one barrier per exchange, so wave64 hardware cannot change the hash (WAVEFRONT.md). build.sh (macOS, Linux), build.bat (MSVC), README with the exact AMD-rig commands. Proven without AMD silicon: Apple OpenCL 1.2 on the M5 Max 96/96 on all three packs (45.0 Mhash/s at 1 GiB, Apple number, not AMD); pocl 7.2 CPU device 96/96 on both exchange paths including the real sub_group_shuffle_xor text; CPU emulator 7 configurations incl. 64-wide sub-groups, identical fingerprint f99fb375b3abeaf5 everywhere. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
36 lines
1.6 KiB
C++
36 lines
1.6 KiB
C++
// Generated by proto-metal/igneum-bench --export-pack for seed "igneum-genesis". Do not edit by hand.
|
|
// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.
|
|
// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros.
|
|
#pragma once
|
|
#ifdef __cplusplus
|
|
#include <cstdint>
|
|
#else
|
|
#include <stdint.h>
|
|
#endif
|
|
#ifndef IGNEUM_NO_CUDA
|
|
#include <cuda_runtime.h>
|
|
#endif
|
|
|
|
#define IGNEUM_SEED_STRING "igneum-genesis"
|
|
#define IGNEUM_DAY_STRING "2026-10-03"
|
|
#define IGNEUM_DAY0 0x3067619fu
|
|
#define IGNEUM_DAY1 0x3c269176u
|
|
#define IGNEUM_DATASET_LOG2 28
|
|
#define IGNEUM_MASK 0x0fffffffu
|
|
#define IGNEUM_LANES 32
|
|
#define IGNEUM_ITERATIONS 8
|
|
#define IGNEUM_INSTR_COUNT 64
|
|
#define IGNEUM_LOADS_PER_HASH 104
|
|
#define IGNEUM_WIDE_LOADS_PER_HASH 0
|
|
#define IGNEUM_OP_MIX "load=13 xor=13 sub=7 shfl=6 add=5 mulhi=5 mad=4 rotr=4 mul=3 rotl=3 or=1"
|
|
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
|
|
#define IGNEUM_DATASET_MODE 0
|
|
|
|
#define IGNEUM_SEEDW_INIT { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }
|
|
#ifndef IGNEUM_NO_CUDA
|
|
// Defined in kernel.cu. Both launch on the default stream and return cudaGetLastError().
|
|
cudaError_t igneum_launch_fill(uint32_t* ds, uint32_t nWords, uint32_t d0, uint32_t d1);
|
|
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
|
uint32_t nonces, uint32_t blockWarps);
|
|
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
|
|
#endif
|