110 lines
6.5 KiB
Metal
110 lines
6.5 KiB
Metal
#include <metal_stdlib>
|
|
using namespace metal;
|
|
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
|
// Item: 8 rounds of 8 x seed-parameterised mixer + one 64-byte cache read, then 8 x final mixer (class v3, mixer multiplier 8,
|
|
// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.
|
|
#define MH_CACHE_LINE_MASK 0x003fffffu
|
|
#define MH_SEGMENT_LINES 64u
|
|
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
|
inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
|
|
|
// y = ChaCha12 core(x) + x
|
|
inline void mh_chacha_block(const thread uint* x, thread uint* y) {
|
|
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
|
for (uint r = 0u; r < 6u; ++r) {
|
|
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
|
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
|
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
|
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
|
}
|
|
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
|
}
|
|
|
|
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
|
inline void mh_cache_segment(device uint* cache, uint seg) {
|
|
uint prev[16]; uint x[16]; uint y[16];
|
|
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
|
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
|
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
|
x[4] = 0xceed56d7u ^ prev[4];
|
|
x[5] = 0x9ba270d2u ^ prev[5];
|
|
x[6] = 0x82caab2du ^ prev[6];
|
|
x[7] = 0x81ebce0eu ^ prev[7];
|
|
x[8] = 0x12b6ecf1u ^ prev[8];
|
|
x[9] = 0xd0f3fd7cu ^ prev[9];
|
|
x[10] = 0xd872eefeu ^ prev[10];
|
|
x[11] = 0xc158c7bdu ^ prev[11];
|
|
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
|
mh_chacha_block(x, y);
|
|
device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
|
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
|
}
|
|
}
|
|
|
|
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
|
inline void mh_mixer(thread uint* s, uint rk) {
|
|
s[0] = (s[0] ^ (0xc6892460u + rk)) * 0xf351d601u;
|
|
s[1] = (s[1] ^ (0x25b7228au + rk)) * 0xa3bb398fu;
|
|
s[2] = (s[2] ^ (0xcd515004u + rk)) * 0xb5a09e35u;
|
|
s[3] = (s[3] ^ (0x2846527au + rk)) * 0x7509c9c1u;
|
|
s[4] = (s[4] ^ (0xa6324241u + rk)) * 0x6bbf31e9u;
|
|
s[5] = (s[5] ^ (0x36e3ec53u + rk)) * 0xfc849a79u;
|
|
s[6] = (s[6] ^ (0x82961bacu + rk)) * 0xded91851u;
|
|
s[7] = (s[7] ^ (0x0f97ba7du + rk)) * 0x8d9113d1u;
|
|
s[8] = (s[8] ^ (0xb6f921a9u + rk)) * 0x0ff15225u;
|
|
s[9] = (s[9] ^ (0x3ada24e5u + rk)) * 0x3a5bdd41u;
|
|
s[10] = (s[10] ^ (0xde20ab91u + rk)) * 0xab533435u;
|
|
s[11] = (s[11] ^ (0x5378eeb2u + rk)) * 0xe1c55ad5u;
|
|
s[12] = (s[12] ^ (0x7d161662u + rk)) * 0xe6d3bd0du;
|
|
s[13] = (s[13] ^ (0x89353cc1u + rk)) * 0x9d9ffbbdu;
|
|
s[14] = (s[14] ^ (0xb1aa03a2u + rk)) * 0xbb2a3cf3u;
|
|
s[15] = (s[15] ^ (0x788acae6u + rk)) * 0x50a7c08du;
|
|
MH_QR(s[0], s[4], s[8], s[12], 17u, 12u, 20u, 23u) MH_QR(s[1], s[5], s[9], s[13], 17u, 12u, 20u, 23u)
|
|
MH_QR(s[2], s[6], s[10], s[14], 17u, 12u, 20u, 23u) MH_QR(s[3], s[7], s[11], s[15], 17u, 12u, 20u, 23u)
|
|
MH_QR(s[0], s[5], s[10], s[15], 7u, 3u, 27u, 16u) MH_QR(s[1], s[6], s[11], s[12], 7u, 3u, 27u, 16u)
|
|
MH_QR(s[2], s[7], s[8], s[13], 7u, 3u, 27u, 16u) MH_QR(s[3], s[4], s[9], s[14], 7u, 3u, 27u, 16u)
|
|
}
|
|
|
|
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of 8 x mixer + cache line s[0] & mask; 8 x final mixer.
|
|
inline void mh_item(device const uint* cache, uint t, thread uint* s) {
|
|
s[0] = 0xceed56d7u;
|
|
s[1] = 0x9ba270d2u;
|
|
s[2] = 0x82caab2du;
|
|
s[3] = 0x81ebce0eu;
|
|
s[4] = 0x12b6ecf1u;
|
|
s[5] = 0xd0f3fd7cu;
|
|
s[6] = 0xd872eefeu;
|
|
s[7] = 0xc158c7bdu;
|
|
s[8] = t * 0xf351d601u + 0xc6892460u;
|
|
s[9] = t * 0xa3bb398fu + 0x25b7228au;
|
|
s[10] = t * 0xb5a09e35u + 0xcd515004u;
|
|
s[11] = t * 0x7509c9c1u + 0x2846527au;
|
|
s[12] = t * 0x6bbf31e9u + 0xa6324241u;
|
|
s[13] = t * 0xfc849a79u + 0x36e3ec53u;
|
|
s[14] = t * 0xded91851u + 0x82961bacu;
|
|
s[15] = t * 0x8d9113d1u + 0x0f97ba7du;
|
|
for (uint r = 0u; r < 8u; ++r) {
|
|
for (uint j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 8u + j + 1u));
|
|
device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
|
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
|
}
|
|
for (uint j = 0u; j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (64u + j + 1u));
|
|
}
|
|
// Era layout (docs/plans/era-layout.md 1.2): dataset word w holds word mh_j(w) of item mh_t(w); j's bits sit at positions 0 2 10 15 of w.
|
|
inline uint mh_j(uint w) { return ((w >> 0u) & 1u) | (((w >> 2u) & 1u) << 1) | (((w >> 10u) & 1u) << 2) | (((w >> 15u) & 1u) << 3); }
|
|
inline uint mh_t(uint w) { w = (w & 0x00007fffu) | ((w >> 16u) << 15u); w = (w & 0x000003ffu) | ((w >> 11u) << 10u); w = (w & 0x00000003u) | ((w >> 3u) << 2u); w = (w & 0x00000000u) | ((w >> 1u) << 0u); return w; }
|
|
inline uint mh_addr(uint t, uint j) { uint w = t; w = ((w >> 0u) << 1u) | (w & 0x00000000u) | (((j >> 0u) & 1u) << 0u); w = ((w >> 2u) << 3u) | (w & 0x00000003u) | (((j >> 1u) & 1u) << 2u); w = ((w >> 10u) << 11u) | (w & 0x000003ffu) | (((j >> 2u) & 1u) << 10u); w = ((w >> 15u) << 16u) | (w & 0x00007fffu) | (((j >> 3u) & 1u) << 15u); return w; }
|
|
// dataset[w] without the dataset: derive item mh_t(w) and take word mh_j(w).
|
|
inline uint mh_word(device const uint* cache, uint w) { uint s[16]; mh_item(cache, mh_t(w), s); return s[mh_j(w)]; }
|
|
|
|
// One thread per segment (2^16 threads).
|
|
kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) {
|
|
mh_cache_segment(cache, gid);
|
|
}
|
|
// One thread per 64-byte item (dataset words / 16 threads).
|
|
kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],
|
|
uint gid [[thread_position_in_grid]]) {
|
|
uint s[16];
|
|
mh_item(cache, gid, s);
|
|
for (uint i = 0u; i < 16u; ++i) dataset[mh_addr(gid, i)] = s[i];
|
|
}
|