#!/usr/bin/env bash # Build the inline-dataset shortcut variant of a CUDA pack: the measurement of proto-metal/MEMHARD.md section 2.2 # (ledger M16) on NVIDIA. usage: make-inline.sh # Two text edits on kernel.cu, nothing else: # 1. every dataset load `ds[expr]` in igneum_hash becomes `mh_word(ds, expr)`: the word is recomputed from the # cache through memhard.h's mh_item (8 dependent cache reads and 9 mixer applications per word) instead of read # 2. igneum_build copies the 256 MiB cache into the head of the dataset buffer instead of building items, so the # `ds` pointer the hash kernel receives points at the cache (mh_word indexes words below 2^IGNEUM_CACHE_LOG2_WORDS) # The dataset self-test of host.cu then FAILS by construction (it reads ds words expecting items) and the vectors # PASS (mh_word gives the true words). The rate line is the number; run.sh checks the vector lines and ignores OVERALL. set -euo pipefail src="$1"; out="$2" mkdir -p "$out" cp "$src"/*.h "$out/" perl -0pe ' s/ds\[([^\]]*)\]/mh_word(ds, $1)/g; s/__global__ void igneum_build\(uint32_t\* ds, const uint32_t\* cache, uint32_t nItems\) \{.*?\n\}\n/__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {\n \/\/ INLINE VARIANT (infra\/gpu-bench\/make-inline.sh): copy the cache into the dataset buffer; igneum_hash recomputes words from it\n uint32_t t = blockIdx.x * blockDim.x + threadIdx.x;\n if (t < nItems && t < (1u << (IGNEUM_CACHE_LOG2_WORDS - 4u))) { for (uint32_t i = 0u; i < 16u; ++i) ds[(size_t)t * 16u + i] = cache[(size_t)t * 16u + i]; }\n}\n/s; ' "$src/kernel.cu" > "$out/kernel.cu" n=$(grep -c 'mh_word(ds,' "$out/kernel.cu" || true) grep -q 'INLINE VARIANT' "$out/kernel.cu" || { echo "make-inline: igneum_build not rewritten (pack layout changed?)"; exit 1; } [ "$n" -gt 0 ] || { echo "make-inline: no dataset loads rewritten"; exit 1; } echo "make-inline: $n loads rewritten to mh_word, build kernel replaced -> $out/kernel.cu"