mixer x4: the two pinned class v3 packs (proto-cuda/packs-ca2-mixer/mx4-genesis and mx4-devnet-epoch0, generator 3 on V3_CLASS = mx4, the devnet pack with era 0's stand-in), tests/packs.rs runs every pinned check over them plus v3_packs_are_the_v2_seeds_under_mixer_x4; igneum-pow --program-class v2|v3 and --era-hex through one epoch_of helper (export, bench, hash, hash-bound); fresh exports of igneum-genesis-mh and igneum-devnet-v4-epoch0 diff clean against the checked-in packs
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
parent
e08909f138
commit
f74a087443
26 changed files with 3481 additions and 32 deletions
|
|
@ -12,10 +12,10 @@
|
|||
//! on every command selects the load class (default v2, the lottery hash). Nothing in a v2 run changes.
|
||||
|
||||
use igneum_pow::emit::export_pack;
|
||||
use igneum_pow::generator::LoadClass;
|
||||
use igneum_pow::generator::{LoadClass, ProgramClass};
|
||||
use igneum_pow::memhard::{Cache, Shape};
|
||||
use igneum_pow::seed::day_key;
|
||||
use igneum_pow::verify::{DatasetMode, Epoch, DEFAULT_DATASET_LOG2};
|
||||
use igneum_pow::verify::{DatasetMode, DatasetSource, Epoch, DEFAULT_DATASET_LOG2};
|
||||
use std::time::Instant;
|
||||
|
||||
struct Args {
|
||||
|
|
@ -33,6 +33,10 @@ struct Args {
|
|||
class: LoadClass,
|
||||
/// Days since genesis for the cache growth rule of a class with `growth` (0: the genesis cache).
|
||||
days: u64,
|
||||
/// The program class (Counter ASIC 2.0 seam): v2 (default) or v3, which draws from V3_CLASS with generator 3.
|
||||
program_class: Option<ProgramClass>,
|
||||
/// The era seed bytes a class v3 chain program records (`--era-hex`).
|
||||
era_hex: Option<String>,
|
||||
}
|
||||
|
||||
fn usage() -> ! {
|
||||
|
|
@ -45,7 +49,8 @@ fn usage() -> ! {
|
|||
\x20 accept every candidate of the seed (or --epoch-hex) with its acceptance verdict\n\
|
||||
\x20 show the accepted program, one instruction per line\n\
|
||||
\x20 --class C load class: v2 (default), mx4 (class v3: mixer x4, cache growth), w4, w16, w64, w64x4, p4,p16,p64[xN], <class>m<mult>[g]\n\
|
||||
\x20 --days N days since genesis for the cache growth rule of a class with it (default 0: the 2^26-word cache)"
|
||||
\x20 --days N days since genesis for the cache growth rule of a class with it (default 0: the 2^26-word cache)\n\
|
||||
\x20 --program-class v2|v3 the program class of the seam (v3 = generator 3 on V3_CLASS, the chain's own derivation; --era-hex records the era seed)"
|
||||
);
|
||||
std::process::exit(2)
|
||||
}
|
||||
|
|
@ -65,6 +70,8 @@ fn parse() -> Args {
|
|||
day_hex: None,
|
||||
class: LoadClass::V2,
|
||||
days: 0,
|
||||
program_class: None,
|
||||
era_hex: None,
|
||||
};
|
||||
let mut it = std::env::args().skip(1);
|
||||
a.cmd = it.next().unwrap_or_else(|| usage());
|
||||
|
|
@ -83,6 +90,8 @@ fn parse() -> Args {
|
|||
"--day-hex" => a.day_hex = Some(val()),
|
||||
"--class" => a.class = LoadClass::parse(&val()).unwrap_or_else(|| usage()),
|
||||
"--days" => a.days = val().parse().unwrap_or_else(|_| usage()),
|
||||
"--program-class" => a.program_class = Some(ProgramClass::parse(&val()).unwrap_or_else(|| usage())),
|
||||
"--era-hex" => a.era_hex = Some(val()),
|
||||
_ => usage(),
|
||||
}
|
||||
}
|
||||
|
|
@ -98,21 +107,14 @@ fn main() {
|
|||
"accept" => accept(&a),
|
||||
"show" => show(&a),
|
||||
"hash" => {
|
||||
let e = Epoch::new_class_day(&a.seed, &a.day, mode, a.dataset_log2, a.class, a.days);
|
||||
let (e, _) = epoch_of(&a, mode);
|
||||
println!("{:016x}", e.hash(a.nonce as u32));
|
||||
}
|
||||
"hash-bound" => {
|
||||
let bytes = igneum_pow::bind::unhex(&a.prehash).unwrap_or_else(|| usage());
|
||||
let prehash: [u8; 32] = bytes.as_slice().try_into().unwrap_or_else(|_| usage());
|
||||
// --epoch-hex / --day-hex: the chain's byte seeds (Epoch::from_seed_bytes), as the worker protocol carries them
|
||||
let e = match (&a.epoch_hex, &a.day_hex) {
|
||||
(Some(eh), Some(dh)) => {
|
||||
let eb = igneum_pow::bind::unhex(eh).unwrap_or_else(|| usage());
|
||||
let db = igneum_pow::bind::unhex(dh).unwrap_or_else(|| usage());
|
||||
Epoch::from_seed_bytes_class(&eb, &db, "cli", a.class)
|
||||
}
|
||||
_ => Epoch::new_class_day(&a.seed, &a.day, mode, a.dataset_log2, a.class, a.days),
|
||||
};
|
||||
let (e, _) = epoch_of(&a, mode);
|
||||
let init = igneum_pow::bind::block_init_words(&prehash, a.nonce);
|
||||
println!("init words {}", init.iter().map(|w| format!("{w:08x}")).collect::<Vec<_>>().join(" "));
|
||||
println!("{:016x}", e.hash_bound(&prehash, a.nonce));
|
||||
|
|
@ -121,6 +123,43 @@ fn main() {
|
|||
}
|
||||
}
|
||||
|
||||
/// The epoch every command works on, and the day label for packs. `--epoch-hex`/`--day-hex` give the chain's byte
|
||||
/// seeds (the day label then names the day bytes); else the string seed and day. `--program-class v3` draws the
|
||||
/// program through the seam (generator 3 on `V3_CLASS`, the era bytes of `--era-hex` recorded) and sizes the
|
||||
/// dataset for `--days` through `Epoch::chain_dataset_day`; `--class` is ignored under a program class (the class
|
||||
/// names the load class). Closed-form mode is only for string seeds under the default class.
|
||||
fn epoch_of(a: &Args, mode: DatasetMode) -> (Epoch, String) {
|
||||
let era = a.era_hex.as_ref().map(|h| igneum_pow::bind::unhex(h).unwrap_or_else(|| usage()));
|
||||
match (&a.epoch_hex, &a.day_hex) {
|
||||
(Some(eh), Some(dh)) => {
|
||||
let eb = igneum_pow::bind::unhex(eh).unwrap_or_else(|| usage());
|
||||
let db = igneum_pow::bind::unhex(dh).unwrap_or_else(|| usage());
|
||||
let label = format!("igneum-epoch/{eh}/day/{dh}");
|
||||
let e = match a.program_class {
|
||||
Some(pc) => Epoch {
|
||||
program: Epoch::chain_program(&eb, era.as_deref(), pc, &label),
|
||||
dataset: Epoch::chain_dataset_day(&db, pc, a.days, a.dataset_log2),
|
||||
},
|
||||
None => Epoch::from_seed_bytes_day(&eb, &db, &label, a.class, a.days, a.dataset_log2),
|
||||
};
|
||||
(e, format!("bytes:{dh}"))
|
||||
}
|
||||
_ => {
|
||||
let e = match a.program_class {
|
||||
Some(pc) => {
|
||||
let program = igneum_pow::generator::generate_from_seed_bytes_program_class(&a.seed, a.seed.as_bytes(), pc, era.as_deref());
|
||||
let lc = pc.load_class();
|
||||
let shape = Shape::for_class_day(&lc, a.days);
|
||||
let log2 = if lc.growth { igneum_pow::memhard::dataset_log2_words(a.dataset_log2, a.days) } else { a.dataset_log2 };
|
||||
Epoch { program, dataset: DatasetSource::new_shape(&a.day, mode, log2, shape) }
|
||||
}
|
||||
None => Epoch::new_class_day(&a.seed, &a.day, mode, a.dataset_log2, a.class, a.days),
|
||||
};
|
||||
(e, a.day.clone())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn bench(a: &Args, mode: DatasetMode) {
|
||||
println!(
|
||||
"igneum-pow bench: seed \"{}\", day \"{}\", dataset 2^{} words ({})",
|
||||
|
|
@ -129,7 +168,7 @@ fn bench(a: &Args, mode: DatasetMode) {
|
|||
a.dataset_log2,
|
||||
mode.name()
|
||||
);
|
||||
let shape = Shape::for_class_day(&a.class, a.days);
|
||||
let shape = Shape::for_class_day(&a.program_class.map(|pc| pc.load_class()).unwrap_or(a.class), a.days);
|
||||
if mode == DatasetMode::MemoryHard {
|
||||
// Time the cache fill on its own first (one core), then build the epoch (which fills it again).
|
||||
let t0 = Instant::now();
|
||||
|
|
@ -145,7 +184,7 @@ fn bench(a: &Args, mode: DatasetMode) {
|
|||
drop(c);
|
||||
}
|
||||
let t0 = Instant::now();
|
||||
let e = Epoch::new_class_day(&a.seed, &a.day, mode, a.dataset_log2, a.class, a.days);
|
||||
let (e, _) = epoch_of(a, mode);
|
||||
let build_ms = t0.elapsed().as_secs_f64() * 1e3;
|
||||
println!(
|
||||
"program: class {}, {} loads/hash, {} bytes/hash, widths (1,4,16 words) {:?}, {} items/warp, mixer x{} ({} mixers/item), cache 2^{} words, op mix {}; epoch built in {build_ms:.1} ms",
|
||||
|
|
@ -184,14 +223,7 @@ fn export(a: &Args, mode: DatasetMode) {
|
|||
let out = a.out.clone().unwrap_or_else(|| usage());
|
||||
let t0 = Instant::now();
|
||||
// --epoch-hex / --day-hex: the chain's byte seeds; the day label then names the day bytes
|
||||
let (e, day_label) = match (&a.epoch_hex, &a.day_hex) {
|
||||
(Some(eh), Some(dh)) => {
|
||||
let eb = igneum_pow::bind::unhex(eh).unwrap_or_else(|| usage());
|
||||
let db = igneum_pow::bind::unhex(dh).unwrap_or_else(|| usage());
|
||||
(Epoch::from_seed_bytes_class(&eb, &db, &format!("igneum-epoch/{eh}/day/{dh}"), a.class), format!("bytes:{dh}"))
|
||||
}
|
||||
_ => (Epoch::new_class_day(&a.seed, &a.day, mode, a.dataset_log2, a.class, a.days), a.day.clone()),
|
||||
};
|
||||
let (e, day_label) = epoch_of(a, mode);
|
||||
let build_ms = t0.elapsed().as_secs_f64() * 1e3;
|
||||
println!("igneum-pow export {out}");
|
||||
println!(
|
||||
|
|
|
|||
|
|
@ -5,28 +5,43 @@
|
|||
//!
|
||||
//! Packs: igneum-genesis-mh and igneum-devnet-v4-epoch0 (memory-hard; the latter from the devnet genesis hash as
|
||||
//! the epoch seed and the day bytes of 2026-10-04), igneum-genesis and igneum-hourly (closed-form dataset,
|
||||
//! interpreter regression only).
|
||||
//! interpreter regression only); and, under `proto-cuda/packs-ca2-mixer/`, the class v3 packs mx4-genesis and
|
||||
//! mx4-devnet-epoch0 (Counter ASIC 2.0, 5 October 2026: generator 3 on `V3_CLASS` = mixer x4 with the cache growth
|
||||
//! rule, the same seeds and days as the two memory-hard v2 packs, so the v2 program and cache carry over and only
|
||||
//! the dataset words and the hashes change).
|
||||
|
||||
use igneum_pow::accept;
|
||||
use igneum_pow::emit::{
|
||||
cuda_kernel, cuda_kernel_bound, cuda_memhard_header, export_pack, metal_memhard, metal_program,
|
||||
metal_program_bound, opencl_kernel, opencl_kernel_bound, program_header, program_json, LoadSource,
|
||||
};
|
||||
use igneum_pow::generator::{generate_from_seed_bytes, Op, GENERATOR_VERSION, LOAD_SLOTS};
|
||||
use igneum_pow::memhard::CACHE_WORDS;
|
||||
use igneum_pow::generator::{generate_from_seed_bytes, generate_from_seed_bytes_program_class, Op, ProgramClass, GENERATOR_VERSION, GENERATOR_VERSION_V3, LOAD_SLOTS, V3_CLASS};
|
||||
use igneum_pow::memhard::{Shape, CACHE_WORDS};
|
||||
use igneum_pow::verify::{DatasetMode, DatasetSource, Epoch};
|
||||
use serde_json::Value;
|
||||
use std::path::PathBuf;
|
||||
use std::sync::OnceLock;
|
||||
|
||||
const PACKS: [&str; 4] = ["igneum-genesis-mh", "igneum-devnet-v4-epoch0", "igneum-genesis", "igneum-hourly"];
|
||||
const PACKS: [&str; 6] =
|
||||
["igneum-genesis-mh", "igneum-devnet-v4-epoch0", "igneum-genesis", "igneum-hourly", "mx4-genesis", "mx4-devnet-epoch0"];
|
||||
/// The class v3 packs (the last two of `PACKS`).
|
||||
const PACKS_V3: [&str; 2] = ["mx4-genesis", "mx4-devnet-epoch0"];
|
||||
|
||||
fn packs_dir() -> PathBuf {
|
||||
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs")
|
||||
}
|
||||
|
||||
/// The directory a pack lives in: the class v3 packs under packs-ca2-mixer, the rest under packs.
|
||||
fn pack_dir(pack: &str) -> PathBuf {
|
||||
if pack.starts_with("mx4-") {
|
||||
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca2-mixer").join(pack)
|
||||
} else {
|
||||
packs_dir().join(pack)
|
||||
}
|
||||
}
|
||||
|
||||
fn read(pack: &str, file: &str) -> String {
|
||||
let p = packs_dir().join(pack).join(file);
|
||||
let p = pack_dir(pack).join(file);
|
||||
std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display()))
|
||||
}
|
||||
|
||||
|
|
@ -62,9 +77,18 @@ fn epoch(pack: &str) -> &'static Epoch {
|
|||
_ => DatasetMode::ClosedForm,
|
||||
};
|
||||
let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32;
|
||||
let program = generate_from_seed_bytes(seed, &seed_bytes);
|
||||
// a class v3 pack: generator 3 on V3_CLASS through the seam, the era bytes it records, the dataset
|
||||
// in the class's shape on day 0 (the growth rule's genesis cache: every pinned pack is a day-0 size)
|
||||
let program = match j["generator"].as_u64().unwrap() as u32 {
|
||||
GENERATOR_VERSION_V3 => {
|
||||
let era = j.get("era_seed_bytes").map(unhex);
|
||||
generate_from_seed_bytes_program_class(seed, &seed_bytes, ProgramClass::V3, era.as_deref())
|
||||
}
|
||||
_ => generate_from_seed_bytes(seed, &seed_bytes),
|
||||
};
|
||||
let shape = Shape::for_class(&program.class);
|
||||
let mut dataset =
|
||||
DatasetSource::from_key(igneum_pow::seed::seed_words_from_bytes(&day_bytes), mode, log2);
|
||||
DatasetSource::from_key_shape(igneum_pow::seed::seed_words_from_bytes(&day_bytes), mode, log2, shape);
|
||||
dataset.key_bytes = day_bytes;
|
||||
(p.to_string(), Epoch { program, dataset })
|
||||
})
|
||||
|
|
@ -81,7 +105,8 @@ fn check_program_json(pack: &str) {
|
|||
let j = json(pack, "program.json");
|
||||
let p = &epoch(pack).program;
|
||||
assert_eq!(j["format"].as_str().unwrap(), "igneum-program-pack-3");
|
||||
assert_eq!(j["generator"].as_u64().unwrap() as u32, GENERATOR_VERSION, "{pack}: generator version");
|
||||
assert_eq!(j["generator"].as_u64().unwrap() as u32, p.generator, "{pack}: generator version");
|
||||
assert_eq!(p.generator, if PACKS_V3.contains(&pack) { GENERATOR_VERSION_V3 } else { GENERATOR_VERSION });
|
||||
assert_eq!(j["attempt"].as_u64().unwrap() as u32, p.attempt, "{pack}: attempt");
|
||||
assert_eq!(hex64(&j["program_id"]), p.program_id(), "{pack}: program id");
|
||||
let sw: Vec<u32> = j["seed_words"].as_array().unwrap().iter().map(hex32).collect();
|
||||
|
|
@ -129,7 +154,7 @@ fn genesis_program_shape() {
|
|||
|
||||
#[test]
|
||||
fn mixer_params_match_pack() {
|
||||
for pack in ["igneum-genesis-mh", "igneum-devnet-v4-epoch0"] {
|
||||
for pack in ["igneum-genesis-mh", "igneum-devnet-v4-epoch0", "mx4-genesis", "mx4-devnet-epoch0"] {
|
||||
let j = json(pack, "program.json");
|
||||
let mp = &epoch(pack).dataset.memhard().unwrap().params;
|
||||
let key: Vec<u32> = j["dataset"]["key"].as_array().unwrap().iter().map(hex32).collect();
|
||||
|
|
@ -287,6 +312,64 @@ fn emitted_sources_match_all_packs() {
|
|||
}
|
||||
|
||||
/// The whole pack as `export` writes it: vectors.json and vectors.h match, and the file list is the full set.
|
||||
/// The class v3 packs (Counter ASIC 2.0, `docs/plans/mixer-x4.md`): generator 3 on V3_CLASS = mx4; the program of
|
||||
/// each is the v2 program of the same seed instruction for instruction (v2 loads take no width roll); the cache is
|
||||
/// the v2 cache (day 0 of the growth rule: 2^26 words, the same FNV-1a 64); the dataset words differ from v2's;
|
||||
/// program.json, program.h and the emitted memhard core say so; the id carries generator 3.
|
||||
#[test]
|
||||
fn v3_packs_are_the_v2_seeds_under_mixer_x4() {
|
||||
assert_eq!(V3_CLASS.name(), "mx4");
|
||||
assert_eq!(V3_CLASS.mixer_mult, 4);
|
||||
assert!(V3_CLASS.growth);
|
||||
for (v3, v2) in [("mx4-genesis", "igneum-genesis-mh"), ("mx4-devnet-epoch0", "igneum-devnet-v4-epoch0")] {
|
||||
let e3 = epoch(v3);
|
||||
let e2 = epoch(v2);
|
||||
let j = json(v3, "program.json");
|
||||
assert_eq!(j["program_class"].as_str().unwrap(), "v3");
|
||||
assert_eq!(j["load_class"].as_str().unwrap(), "mx4");
|
||||
assert_eq!(j["mixer_mult"].as_u64().unwrap(), 4);
|
||||
assert_eq!(j["cache_growth"].as_bool().unwrap(), true);
|
||||
assert_eq!(j["dataset"]["mixer_mult"].as_u64().unwrap(), 4);
|
||||
assert_eq!(j["dataset"]["cache"]["log2_words"].as_u64().unwrap(), 26);
|
||||
assert_eq!(e3.program.generator, GENERATOR_VERSION_V3);
|
||||
assert_eq!(e3.program.class, V3_CLASS);
|
||||
assert_eq!(e3.program.instrs, e2.program.instrs, "{v3}: the v2 program under the v3 construction");
|
||||
assert_eq!(e3.program.seed, e2.program.seed);
|
||||
assert_eq!(e3.program.attempt, e2.program.attempt);
|
||||
assert_ne!(e3.program.program_id(), e2.program.program_id());
|
||||
assert_eq!(e3.program.program_id(), igneum_pow::generator::program_id(GENERATOR_VERSION_V3, &e3.program.seed, e3.program.attempt));
|
||||
let m3 = e3.dataset.memhard().unwrap();
|
||||
let m2 = e2.dataset.memhard().unwrap();
|
||||
assert_eq!(m3.shape(), Shape { mixer_mult: 4, cache_log2_words: 26 });
|
||||
assert_eq!(m3.cache.fnv1a64(), m2.cache.fnv1a64(), "{v3}: the same cache as v2 on day 0");
|
||||
assert_eq!(m3.params.rot, m2.params.rot);
|
||||
assert_eq!(e3.dataset.log2_words, 28);
|
||||
assert_ne!(e3.dataset.word(0), e2.dataset.word(0), "{v3}: the dataset words differ");
|
||||
assert_ne!(e3.hash_warp(0), e2.hash_warp(0));
|
||||
let h = read(v3, "program.h");
|
||||
assert!(h.contains("#define IGNEUM_GENERATOR 3\n"));
|
||||
assert!(h.contains("#define IGNEUM_PROGRAM_CLASS \"v3\"\n"));
|
||||
assert!(h.contains("#define IGNEUM_MIXER_MULT 4"));
|
||||
assert!(h.contains("#define IGNEUM_CACHE_GROWTH 1"));
|
||||
assert!(h.contains("#define IGNEUM_CACHE_LOG2_WORDS 26\n"));
|
||||
assert!(h.contains("#define IGNEUM_LOAD_CLASS \"mx4\"\n"));
|
||||
for file in ["memhard.h", "memhard.metal", "kernel.cl"] {
|
||||
let text = read(v3, file);
|
||||
assert_eq!(text.matches("j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 4u + j + 1u))").count(), 1, "{v3}/{file}");
|
||||
assert_eq!(text.matches("j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (32u + j + 1u))").count(), 1, "{v3}/{file}");
|
||||
}
|
||||
for file in ["memhard.h", "memhard.metal", "kernel.cl"] {
|
||||
let text = read(v2, file);
|
||||
assert_eq!(text.matches("j < 4u").count(), 0, "{v2}/{file}: the v2 text has no multiplier loop");
|
||||
}
|
||||
}
|
||||
// the devnet v3 pack records era 0's stand-in, the devnet genesis hash
|
||||
let j = json("mx4-devnet-epoch0", "program.json");
|
||||
assert_eq!(unhex(&j["era_seed_bytes"]), unhex(&j["seed_bytes"]));
|
||||
assert!(read("mx4-devnet-epoch0", "program.h").contains("#define IGNEUM_ERA_SEED_HEX \"edc4fa844da9dc98"));
|
||||
assert!(json("mx4-genesis", "program.json").get("era_seed_bytes").is_none());
|
||||
}
|
||||
|
||||
fn check_export(pack: &str) {
|
||||
let e = epoch(pack);
|
||||
let v = json(pack, "vectors.json");
|
||||
|
|
@ -311,7 +394,7 @@ fn check_export(pack: &str) {
|
|||
expected.extend(["memhard.h", "memhard.metal"]);
|
||||
}
|
||||
assert_eq!(out.files.iter().map(|(n, _)| n.as_str()).collect::<Vec<_>>(), expected);
|
||||
let mut on_disk: Vec<String> = std::fs::read_dir(packs_dir().join(pack))
|
||||
let mut on_disk: Vec<String> = std::fs::read_dir(pack_dir(pack))
|
||||
.unwrap()
|
||||
.map(|d| d.unwrap().file_name().to_string_lossy().to_string())
|
||||
.filter(|n| !n.starts_with('.'))
|
||||
|
|
|
|||
279
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/kernel.cl
Normal file
279
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/kernel.cl
Normal file
|
|
@ -0,0 +1,279 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000". Do not edit by hand.
|
||||
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
|
||||
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
|
||||
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
|
||||
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
|
||||
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
|
||||
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
|
||||
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
|
||||
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
|
||||
#ifndef IGNEUM_GROUP
|
||||
#define IGNEUM_GROUP 32
|
||||
#endif
|
||||
#ifndef IGNEUM_EXCHANGE
|
||||
#define IGNEUM_EXCHANGE 0
|
||||
#endif
|
||||
#ifdef __OPENCL_VERSION__
|
||||
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
|
||||
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#ifdef cl_khr_subgroups
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
|
||||
#endif
|
||||
#ifdef cl_khr_subgroup_shuffle
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
|
||||
#endif
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
|
||||
#endif
|
||||
#else
|
||||
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
|
||||
#include "emu_opencl.h"
|
||||
#endif
|
||||
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#else
|
||||
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
|
||||
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
|
||||
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
|
||||
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
|
||||
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
|
||||
#endif
|
||||
|
||||
static inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
|
||||
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
|
||||
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
|
||||
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
|
||||
static inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of 4 x seed-parameterised mixer + one 64-byte cache read, then 4 x final mixer (class v3, mixer multiplier 4,
|
||||
// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
static inline void mh_chacha_block(const uint* x, uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
static inline void mh_cache_segment(__global uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0xceed56d7u ^ prev[4];
|
||||
x[5] = 0x9ba270d2u ^ prev[5];
|
||||
x[6] = 0x82caab2du ^ prev[6];
|
||||
x[7] = 0x81ebce0eu ^ prev[7];
|
||||
x[8] = 0x12b6ecf1u ^ prev[8];
|
||||
x[9] = 0xd0f3fd7cu ^ prev[9];
|
||||
x[10] = 0xd872eefeu ^ prev[10];
|
||||
x[11] = 0xc158c7bdu ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
static inline void mh_mixer(uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xc6892460u + rk)) * 0xf351d601u;
|
||||
s[1] = (s[1] ^ (0x25b7228au + rk)) * 0xa3bb398fu;
|
||||
s[2] = (s[2] ^ (0xcd515004u + rk)) * 0xb5a09e35u;
|
||||
s[3] = (s[3] ^ (0x2846527au + rk)) * 0x7509c9c1u;
|
||||
s[4] = (s[4] ^ (0xa6324241u + rk)) * 0x6bbf31e9u;
|
||||
s[5] = (s[5] ^ (0x36e3ec53u + rk)) * 0xfc849a79u;
|
||||
s[6] = (s[6] ^ (0x82961bacu + rk)) * 0xded91851u;
|
||||
s[7] = (s[7] ^ (0x0f97ba7du + rk)) * 0x8d9113d1u;
|
||||
s[8] = (s[8] ^ (0xb6f921a9u + rk)) * 0x0ff15225u;
|
||||
s[9] = (s[9] ^ (0x3ada24e5u + rk)) * 0x3a5bdd41u;
|
||||
s[10] = (s[10] ^ (0xde20ab91u + rk)) * 0xab533435u;
|
||||
s[11] = (s[11] ^ (0x5378eeb2u + rk)) * 0xe1c55ad5u;
|
||||
s[12] = (s[12] ^ (0x7d161662u + rk)) * 0xe6d3bd0du;
|
||||
s[13] = (s[13] ^ (0x89353cc1u + rk)) * 0x9d9ffbbdu;
|
||||
s[14] = (s[14] ^ (0xb1aa03a2u + rk)) * 0xbb2a3cf3u;
|
||||
s[15] = (s[15] ^ (0x788acae6u + rk)) * 0x50a7c08du;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 17u, 12u, 20u, 23u) MH_QR(s[1], s[5], s[9], s[13], 17u, 12u, 20u, 23u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 17u, 12u, 20u, 23u) MH_QR(s[3], s[7], s[11], s[15], 17u, 12u, 20u, 23u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 7u, 3u, 27u, 16u) MH_QR(s[1], s[6], s[11], s[12], 7u, 3u, 27u, 16u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 7u, 3u, 27u, 16u) MH_QR(s[3], s[4], s[9], s[14], 7u, 3u, 27u, 16u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of 4 x mixer + cache line s[0] & mask; 4 x final mixer.
|
||||
static inline void mh_item(__global const uint* cache, uint t, uint* s) {
|
||||
s[0] = 0xceed56d7u;
|
||||
s[1] = 0x9ba270d2u;
|
||||
s[2] = 0x82caab2du;
|
||||
s[3] = 0x81ebce0eu;
|
||||
s[4] = 0x12b6ecf1u;
|
||||
s[5] = 0xd0f3fd7cu;
|
||||
s[6] = 0xd872eefeu;
|
||||
s[7] = 0xc158c7bdu;
|
||||
s[8] = t * 0xf351d601u + 0xc6892460u;
|
||||
s[9] = t * 0xa3bb398fu + 0x25b7228au;
|
||||
s[10] = t * 0xb5a09e35u + 0xcd515004u;
|
||||
s[11] = t * 0x7509c9c1u + 0x2846527au;
|
||||
s[12] = t * 0x6bbf31e9u + 0xa6324241u;
|
||||
s[13] = t * 0xfc849a79u + 0x36e3ec53u;
|
||||
s[14] = t * 0xded91851u + 0x82961bacu;
|
||||
s[15] = t * 0x8d9113d1u + 0x0f97ba7du;
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
for (uint j = 0u; j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 4u + j + 1u));
|
||||
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
for (uint j = 0u; j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (32u + j + 1u));
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
|
||||
// The same constants as memhard.h in this pack (one emitter, three dialects).
|
||||
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
|
||||
uint seg = (uint)get_global_id(0);
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
|
||||
uint t = (uint)get_global_id(0);
|
||||
if (t < nItems) {
|
||||
uint s[16];
|
||||
mh_item(cache, t, s);
|
||||
__global uint* d = ds + ((ulong)t * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask) {
|
||||
uint gid = (uint)get_global_id(0);
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ 0x667d0fbdu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x7b8e5963u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint x = nonce ^ 0x7b8e5963u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0x31c67e5eu; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint x = nonce ^ 0x31c67e5eu; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4529ddc6u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint x = nonce ^ 0x4529ddc6u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xef19d6d8u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint x = nonce ^ 0xef19d6d8u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xaccf6211u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint x = nonce ^ 0xaccf6211u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0xda0aed32u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint x = nonce ^ 0xda0aed32u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0xabc6df31u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint x = nonce ^ 0xabc6df31u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x667d0fbdu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r4 = r4 + r5 + ((((sel >> 13u) & 1u) != 0u) ? 0x5810667au : 0xea86e152u); // 0 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 4u); r2 = r2 ^ t_; } // 1 shfl
|
||||
r3 = r3 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0x642e66dbu : 0x2cccb6cau); // 2 add
|
||||
r0 = rotl_imm(r0, 19u); // 3 rotl
|
||||
r7 = rotr_var(r7, r6); // 4 rotr
|
||||
r7 = r7 + r4 + ((((sel >> 21u) & 1u) != 0u) ? 0xc1535555u : 0xee02465fu); // 5 add
|
||||
r1 = mul_hi(r1, r7); // 6 mulhi
|
||||
r4 = r4 ^ ds[r2 & mask]; // 7 load
|
||||
r7 = r7 ^ ds[r4 & mask]; // 8 load
|
||||
r0 = r0 ^ ds[r3 & mask]; // 9 load
|
||||
r5 = r5 ^ ds[r1 & mask]; // 10 load
|
||||
r1 = r1 ^ ds[r5 & mask]; // 11 load
|
||||
r3 = mul_hi(r3, r5); // 12 mulhi
|
||||
r1 = r1 ^ ds[r3 & mask]; // 13 load
|
||||
r0 = r0 - r3; // 14 sub
|
||||
r5 = r1 * r3 + r5; // 15 mad
|
||||
r6 = mul_hi(r6, r1); // 16 mulhi
|
||||
r5 = r5 + r2 + ((((sel >> 28u) & 1u) != 0u) ? 0x8b965b57u : 0x697b3d00u); // 17 add
|
||||
r0 = mul_hi(r0, r6); // 18 mulhi
|
||||
r5 = rotr_var(r5, r3); // 19 rotr
|
||||
r5 = mul_hi(r5, r2); // 20 mulhi
|
||||
r1 = r1 + r0 + ((((sel >> 1u) & 1u) != 0u) ? 0x6d7e8d05u : 0xebcf247au); // 21 add
|
||||
r7 = r7 + r5 + ((((sel >> 12u) & 1u) != 0u) ? 0xb9e3577eu : 0xf66e7017u); // 22 add
|
||||
r1 = mul_hi(r1, r5); // 23 mulhi
|
||||
r2 = r2 - r5; // 24 sub
|
||||
r7 = r7 + r4 + ((((sel >> 2u) & 1u) != 0u) ? 0x699ef1bbu : 0x08ffa6c7u); // 25 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 26 shfl
|
||||
r7 = r7 + r1 + ((((sel >> 14u) & 1u) != 0u) ? 0xb4ead2fbu : 0xe60fea84u); // 27 add
|
||||
r3 = r3 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x8f30d21du : 0x65c76dabu); // 28 add
|
||||
r2 = r2 ^ ds[r1 & mask]; // 29 load
|
||||
r5 = r5 ^ ds[r7 & mask]; // 30 load
|
||||
r2 = r2 ^ ds[r5 & mask]; // 31 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 4u); r1 = r1 ^ t_; } // 32 shfl
|
||||
r4 = r5 * r7 + r4; // 33 mad
|
||||
r4 = r4 + r2 + ((((sel >> 21u) & 1u) != 0u) ? 0xc7ce690cu : 0x0480debeu); // 34 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 8u); r3 = r3 ^ t_; } // 35 shfl
|
||||
r7 = r7 + r1 + ((((sel >> 2u) & 1u) != 0u) ? 0xe10c2c95u : 0xc53b542eu); // 36 add
|
||||
r5 = r5 ^ r7; // 37 xor
|
||||
r2 = r2 | r1; // 38 or
|
||||
r1 = mul_hi(r1, r0); // 39 mulhi
|
||||
r6 = rotl_imm(r6, 19u); // 40 rotl
|
||||
r4 = mul_hi(r4, r6); // 41 mulhi
|
||||
r6 = r6 - r0; // 42 sub
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 43 shfl
|
||||
r4 = r4 ^ ds[r2 & mask]; // 44 load
|
||||
r1 = r1 ^ r3; // 45 xor
|
||||
r7 = r7 ^ ds[r0 & mask]; // 46 load
|
||||
r3 = r3 ^ ds[r1 & mask]; // 47 load
|
||||
r5 = r5 * r3; // 48 mul
|
||||
r1 = r1 - r5; // 49 sub
|
||||
r2 = rotl_imm(r2, 8u); // 50 rotl
|
||||
r1 = r1 + r5 + ((((sel >> 23u) & 1u) != 0u) ? 0x77b9bd43u : 0xa900fec4u); // 51 add
|
||||
r4 = r4 ^ ds[r7 & mask]; // 52 load
|
||||
r2 = r2 - r7; // 53 sub
|
||||
r4 = r4 ^ r0; // 54 xor
|
||||
r1 = r1 + r6 + ((((sel >> 14u) & 1u) != 0u) ? 0x83e825bfu : 0xe09f54e9u); // 55 add
|
||||
r2 = r2 ^ ds[r4 & mask]; // 56 load
|
||||
r0 = r1 * r4 + r0; // 57 mad
|
||||
r3 = r3 ^ ds[r5 & mask]; // 58 load
|
||||
r5 = r5 | r6; // 59 or
|
||||
r6 = r5 * r7 + r6; // 60 mad
|
||||
r4 = rotl_imm(r4, 28u); // 61 rotl
|
||||
r5 = mul_hi(r5, r0); // 62 mulhi
|
||||
r3 = r3 ^ ds[r6 & mask]; // 63 load
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
|
||||
#if IGNEUM_EXCHANGE != 0
|
||||
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
|
||||
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
|
||||
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
|
||||
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
|
||||
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
|
||||
}
|
||||
#endif
|
||||
164
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/kernel.cu
Normal file
164
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/kernel.cu
Normal file
|
|
@ -0,0 +1,164 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000". Do not edit by hand.
|
||||
// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal).
|
||||
// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC.
|
||||
#include <cuda_runtime.h>
|
||||
#include <cstdint>
|
||||
#include "program.h"
|
||||
#include "memhard.h"
|
||||
|
||||
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31.
|
||||
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
|
||||
// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x.
|
||||
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) {
|
||||
uint32_t x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item.
|
||||
// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host.
|
||||
__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) {
|
||||
uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
|
||||
uint32_t t = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (t < nItems) {
|
||||
uint32_t s[16];
|
||||
mh_item(cache, t, s);
|
||||
uint32_t* d = ds + (size_t)t * 16u;
|
||||
for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
|
||||
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
|
||||
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
|
||||
__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask) {
|
||||
uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
uint32_t nonce = baseNonce + gid;
|
||||
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint32_t x = nonce ^ 0x667d0fbdu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x7b8e5963u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint32_t x = nonce ^ 0x7b8e5963u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0x31c67e5eu; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint32_t x = nonce ^ 0x31c67e5eu; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4529ddc6u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint32_t x = nonce ^ 0x4529ddc6u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xef19d6d8u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint32_t x = nonce ^ 0xef19d6d8u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xaccf6211u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint32_t x = nonce ^ 0xaccf6211u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0xda0aed32u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint32_t x = nonce ^ 0xda0aed32u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0xabc6df31u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint32_t x = nonce ^ 0xabc6df31u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x667d0fbdu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint32_t it = 0u; it < 8u; ++it) {
|
||||
uint32_t sel = r0;
|
||||
r4 = r4 + r5 + ((((sel >> 13u) & 1u) != 0u) ? 0x5810667au : 0xea86e152u); // 0 add
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r0, 4); // 1 shfl
|
||||
r3 = r3 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0x642e66dbu : 0x2cccb6cau); // 2 add
|
||||
r0 = rotl_imm(r0, 19u); // 3 rotl
|
||||
r7 = rotr_var(r7, r6); // 4 rotr
|
||||
r7 = r7 + r4 + ((((sel >> 21u) & 1u) != 0u) ? 0xc1535555u : 0xee02465fu); // 5 add
|
||||
r1 = __umulhi(r1, r7); // 6 mulhi
|
||||
r4 = r4 ^ ds[r2 & mask]; // 7 load
|
||||
r7 = r7 ^ ds[r4 & mask]; // 8 load
|
||||
r0 = r0 ^ ds[r3 & mask]; // 9 load
|
||||
r5 = r5 ^ ds[r1 & mask]; // 10 load
|
||||
r1 = r1 ^ ds[r5 & mask]; // 11 load
|
||||
r3 = __umulhi(r3, r5); // 12 mulhi
|
||||
r1 = r1 ^ ds[r3 & mask]; // 13 load
|
||||
r0 = r0 - r3; // 14 sub
|
||||
r5 = r1 * r3 + r5; // 15 mad
|
||||
r6 = __umulhi(r6, r1); // 16 mulhi
|
||||
r5 = r5 + r2 + ((((sel >> 28u) & 1u) != 0u) ? 0x8b965b57u : 0x697b3d00u); // 17 add
|
||||
r0 = __umulhi(r0, r6); // 18 mulhi
|
||||
r5 = rotr_var(r5, r3); // 19 rotr
|
||||
r5 = __umulhi(r5, r2); // 20 mulhi
|
||||
r1 = r1 + r0 + ((((sel >> 1u) & 1u) != 0u) ? 0x6d7e8d05u : 0xebcf247au); // 21 add
|
||||
r7 = r7 + r5 + ((((sel >> 12u) & 1u) != 0u) ? 0xb9e3577eu : 0xf66e7017u); // 22 add
|
||||
r1 = __umulhi(r1, r5); // 23 mulhi
|
||||
r2 = r2 - r5; // 24 sub
|
||||
r7 = r7 + r4 + ((((sel >> 2u) & 1u) != 0u) ? 0x699ef1bbu : 0x08ffa6c7u); // 25 add
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 26 shfl
|
||||
r7 = r7 + r1 + ((((sel >> 14u) & 1u) != 0u) ? 0xb4ead2fbu : 0xe60fea84u); // 27 add
|
||||
r3 = r3 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x8f30d21du : 0x65c76dabu); // 28 add
|
||||
r2 = r2 ^ ds[r1 & mask]; // 29 load
|
||||
r5 = r5 ^ ds[r7 & mask]; // 30 load
|
||||
r2 = r2 ^ ds[r5 & mask]; // 31 load
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r7, 4); // 32 shfl
|
||||
r4 = r5 * r7 + r4; // 33 mad
|
||||
r4 = r4 + r2 + ((((sel >> 21u) & 1u) != 0u) ? 0xc7ce690cu : 0x0480debeu); // 34 add
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r7, 8); // 35 shfl
|
||||
r7 = r7 + r1 + ((((sel >> 2u) & 1u) != 0u) ? 0xe10c2c95u : 0xc53b542eu); // 36 add
|
||||
r5 = r5 ^ r7; // 37 xor
|
||||
r2 = r2 | r1; // 38 or
|
||||
r1 = __umulhi(r1, r0); // 39 mulhi
|
||||
r6 = rotl_imm(r6, 19u); // 40 rotl
|
||||
r4 = __umulhi(r4, r6); // 41 mulhi
|
||||
r6 = r6 - r0; // 42 sub
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 43 shfl
|
||||
r4 = r4 ^ ds[r2 & mask]; // 44 load
|
||||
r1 = r1 ^ r3; // 45 xor
|
||||
r7 = r7 ^ ds[r0 & mask]; // 46 load
|
||||
r3 = r3 ^ ds[r1 & mask]; // 47 load
|
||||
r5 = r5 * r3; // 48 mul
|
||||
r1 = r1 - r5; // 49 sub
|
||||
r2 = rotl_imm(r2, 8u); // 50 rotl
|
||||
r1 = r1 + r5 + ((((sel >> 23u) & 1u) != 0u) ? 0x77b9bd43u : 0xa900fec4u); // 51 add
|
||||
r4 = r4 ^ ds[r7 & mask]; // 52 load
|
||||
r2 = r2 - r7; // 53 sub
|
||||
r4 = r4 ^ r0; // 54 xor
|
||||
r1 = r1 + r6 + ((((sel >> 14u) & 1u) != 0u) ? 0x83e825bfu : 0xe09f54e9u); // 55 add
|
||||
r2 = r2 ^ ds[r4 & mask]; // 56 load
|
||||
r0 = r1 * r4 + r0; // 57 mad
|
||||
r3 = r3 ^ ds[r5 & mask]; // 58 load
|
||||
r5 = r5 | r6; // 59 or
|
||||
r6 = r5 * r7 + r6; // 60 mad
|
||||
r4 = rotl_imm(r4, 28u); // 61 rotl
|
||||
r5 = __umulhi(r5, r0); // 62 mulhi
|
||||
r3 = r3 ^ ds[r6 & mask]; // 63 load
|
||||
}
|
||||
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
|
||||
}
|
||||
|
||||
// Host-side launch wrappers. Declared in program.h, called from host.cu.
|
||||
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) {
|
||||
if (nSegments == 0u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 256u;
|
||||
uint32_t grid = (nSegments + block - 1u) / block;
|
||||
igneum_cache_fill<<<grid, block>>>(cache, nSegments);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
|
||||
if (nItems == 0u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 256u;
|
||||
uint32_t grid = (nItems + block - 1u) / block;
|
||||
igneum_build<<<grid, block>>>(ds, cache, nItems);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
uint32_t nonces, uint32_t blockWarps) {
|
||||
if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 32u * blockWarps;
|
||||
if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;
|
||||
igneum_hash<<<nonces / block, block>>>(ds, out, baseNonce, mask);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
|
||||
cudaFuncAttributes attr;
|
||||
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash);
|
||||
if (e != cudaSuccess) return e;
|
||||
*numRegs = attr.numRegs;
|
||||
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0);
|
||||
}
|
||||
373
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/kernel_bound.cl
Normal file
373
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/kernel_bound.cl
Normal file
|
|
@ -0,0 +1,373 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000". Do not edit by hand.
|
||||
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
|
||||
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
|
||||
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
|
||||
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
|
||||
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
|
||||
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
|
||||
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
|
||||
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
|
||||
#ifndef IGNEUM_GROUP
|
||||
#define IGNEUM_GROUP 32
|
||||
#endif
|
||||
#ifndef IGNEUM_EXCHANGE
|
||||
#define IGNEUM_EXCHANGE 0
|
||||
#endif
|
||||
#ifdef __OPENCL_VERSION__
|
||||
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
|
||||
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#ifdef cl_khr_subgroups
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
|
||||
#endif
|
||||
#ifdef cl_khr_subgroup_shuffle
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
|
||||
#endif
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
|
||||
#endif
|
||||
#else
|
||||
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
|
||||
#include "emu_opencl.h"
|
||||
#endif
|
||||
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#else
|
||||
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
|
||||
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
|
||||
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
|
||||
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
|
||||
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
|
||||
#endif
|
||||
|
||||
static inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
|
||||
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
|
||||
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
|
||||
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
|
||||
static inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of 4 x seed-parameterised mixer + one 64-byte cache read, then 4 x final mixer (class v3, mixer multiplier 4,
|
||||
// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
static inline void mh_chacha_block(const uint* x, uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
static inline void mh_cache_segment(__global uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0xceed56d7u ^ prev[4];
|
||||
x[5] = 0x9ba270d2u ^ prev[5];
|
||||
x[6] = 0x82caab2du ^ prev[6];
|
||||
x[7] = 0x81ebce0eu ^ prev[7];
|
||||
x[8] = 0x12b6ecf1u ^ prev[8];
|
||||
x[9] = 0xd0f3fd7cu ^ prev[9];
|
||||
x[10] = 0xd872eefeu ^ prev[10];
|
||||
x[11] = 0xc158c7bdu ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
static inline void mh_mixer(uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xc6892460u + rk)) * 0xf351d601u;
|
||||
s[1] = (s[1] ^ (0x25b7228au + rk)) * 0xa3bb398fu;
|
||||
s[2] = (s[2] ^ (0xcd515004u + rk)) * 0xb5a09e35u;
|
||||
s[3] = (s[3] ^ (0x2846527au + rk)) * 0x7509c9c1u;
|
||||
s[4] = (s[4] ^ (0xa6324241u + rk)) * 0x6bbf31e9u;
|
||||
s[5] = (s[5] ^ (0x36e3ec53u + rk)) * 0xfc849a79u;
|
||||
s[6] = (s[6] ^ (0x82961bacu + rk)) * 0xded91851u;
|
||||
s[7] = (s[7] ^ (0x0f97ba7du + rk)) * 0x8d9113d1u;
|
||||
s[8] = (s[8] ^ (0xb6f921a9u + rk)) * 0x0ff15225u;
|
||||
s[9] = (s[9] ^ (0x3ada24e5u + rk)) * 0x3a5bdd41u;
|
||||
s[10] = (s[10] ^ (0xde20ab91u + rk)) * 0xab533435u;
|
||||
s[11] = (s[11] ^ (0x5378eeb2u + rk)) * 0xe1c55ad5u;
|
||||
s[12] = (s[12] ^ (0x7d161662u + rk)) * 0xe6d3bd0du;
|
||||
s[13] = (s[13] ^ (0x89353cc1u + rk)) * 0x9d9ffbbdu;
|
||||
s[14] = (s[14] ^ (0xb1aa03a2u + rk)) * 0xbb2a3cf3u;
|
||||
s[15] = (s[15] ^ (0x788acae6u + rk)) * 0x50a7c08du;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 17u, 12u, 20u, 23u) MH_QR(s[1], s[5], s[9], s[13], 17u, 12u, 20u, 23u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 17u, 12u, 20u, 23u) MH_QR(s[3], s[7], s[11], s[15], 17u, 12u, 20u, 23u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 7u, 3u, 27u, 16u) MH_QR(s[1], s[6], s[11], s[12], 7u, 3u, 27u, 16u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 7u, 3u, 27u, 16u) MH_QR(s[3], s[4], s[9], s[14], 7u, 3u, 27u, 16u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of 4 x mixer + cache line s[0] & mask; 4 x final mixer.
|
||||
static inline void mh_item(__global const uint* cache, uint t, uint* s) {
|
||||
s[0] = 0xceed56d7u;
|
||||
s[1] = 0x9ba270d2u;
|
||||
s[2] = 0x82caab2du;
|
||||
s[3] = 0x81ebce0eu;
|
||||
s[4] = 0x12b6ecf1u;
|
||||
s[5] = 0xd0f3fd7cu;
|
||||
s[6] = 0xd872eefeu;
|
||||
s[7] = 0xc158c7bdu;
|
||||
s[8] = t * 0xf351d601u + 0xc6892460u;
|
||||
s[9] = t * 0xa3bb398fu + 0x25b7228au;
|
||||
s[10] = t * 0xb5a09e35u + 0xcd515004u;
|
||||
s[11] = t * 0x7509c9c1u + 0x2846527au;
|
||||
s[12] = t * 0x6bbf31e9u + 0xa6324241u;
|
||||
s[13] = t * 0xfc849a79u + 0x36e3ec53u;
|
||||
s[14] = t * 0xded91851u + 0x82961bacu;
|
||||
s[15] = t * 0x8d9113d1u + 0x0f97ba7du;
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
for (uint j = 0u; j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 4u + j + 1u));
|
||||
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
for (uint j = 0u; j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (32u + j + 1u));
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
|
||||
// The same constants as memhard.h in this pack (one emitter, three dialects).
|
||||
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
|
||||
uint seg = (uint)get_global_id(0);
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
|
||||
uint t = (uint)get_global_id(0);
|
||||
if (t < nItems) {
|
||||
uint s[16];
|
||||
mh_item(cache, t, s);
|
||||
__global uint* d = ds + ((ulong)t * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask) {
|
||||
uint gid = (uint)get_global_id(0);
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ 0x667d0fbdu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x7b8e5963u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint x = nonce ^ 0x7b8e5963u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0x31c67e5eu; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint x = nonce ^ 0x31c67e5eu; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4529ddc6u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint x = nonce ^ 0x4529ddc6u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xef19d6d8u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint x = nonce ^ 0xef19d6d8u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xaccf6211u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint x = nonce ^ 0xaccf6211u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0xda0aed32u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint x = nonce ^ 0xda0aed32u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0xabc6df31u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint x = nonce ^ 0xabc6df31u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x667d0fbdu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r4 = r4 + r5 + ((((sel >> 13u) & 1u) != 0u) ? 0x5810667au : 0xea86e152u); // 0 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 4u); r2 = r2 ^ t_; } // 1 shfl
|
||||
r3 = r3 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0x642e66dbu : 0x2cccb6cau); // 2 add
|
||||
r0 = rotl_imm(r0, 19u); // 3 rotl
|
||||
r7 = rotr_var(r7, r6); // 4 rotr
|
||||
r7 = r7 + r4 + ((((sel >> 21u) & 1u) != 0u) ? 0xc1535555u : 0xee02465fu); // 5 add
|
||||
r1 = mul_hi(r1, r7); // 6 mulhi
|
||||
r4 = r4 ^ ds[r2 & mask]; // 7 load
|
||||
r7 = r7 ^ ds[r4 & mask]; // 8 load
|
||||
r0 = r0 ^ ds[r3 & mask]; // 9 load
|
||||
r5 = r5 ^ ds[r1 & mask]; // 10 load
|
||||
r1 = r1 ^ ds[r5 & mask]; // 11 load
|
||||
r3 = mul_hi(r3, r5); // 12 mulhi
|
||||
r1 = r1 ^ ds[r3 & mask]; // 13 load
|
||||
r0 = r0 - r3; // 14 sub
|
||||
r5 = r1 * r3 + r5; // 15 mad
|
||||
r6 = mul_hi(r6, r1); // 16 mulhi
|
||||
r5 = r5 + r2 + ((((sel >> 28u) & 1u) != 0u) ? 0x8b965b57u : 0x697b3d00u); // 17 add
|
||||
r0 = mul_hi(r0, r6); // 18 mulhi
|
||||
r5 = rotr_var(r5, r3); // 19 rotr
|
||||
r5 = mul_hi(r5, r2); // 20 mulhi
|
||||
r1 = r1 + r0 + ((((sel >> 1u) & 1u) != 0u) ? 0x6d7e8d05u : 0xebcf247au); // 21 add
|
||||
r7 = r7 + r5 + ((((sel >> 12u) & 1u) != 0u) ? 0xb9e3577eu : 0xf66e7017u); // 22 add
|
||||
r1 = mul_hi(r1, r5); // 23 mulhi
|
||||
r2 = r2 - r5; // 24 sub
|
||||
r7 = r7 + r4 + ((((sel >> 2u) & 1u) != 0u) ? 0x699ef1bbu : 0x08ffa6c7u); // 25 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 26 shfl
|
||||
r7 = r7 + r1 + ((((sel >> 14u) & 1u) != 0u) ? 0xb4ead2fbu : 0xe60fea84u); // 27 add
|
||||
r3 = r3 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x8f30d21du : 0x65c76dabu); // 28 add
|
||||
r2 = r2 ^ ds[r1 & mask]; // 29 load
|
||||
r5 = r5 ^ ds[r7 & mask]; // 30 load
|
||||
r2 = r2 ^ ds[r5 & mask]; // 31 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 4u); r1 = r1 ^ t_; } // 32 shfl
|
||||
r4 = r5 * r7 + r4; // 33 mad
|
||||
r4 = r4 + r2 + ((((sel >> 21u) & 1u) != 0u) ? 0xc7ce690cu : 0x0480debeu); // 34 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 8u); r3 = r3 ^ t_; } // 35 shfl
|
||||
r7 = r7 + r1 + ((((sel >> 2u) & 1u) != 0u) ? 0xe10c2c95u : 0xc53b542eu); // 36 add
|
||||
r5 = r5 ^ r7; // 37 xor
|
||||
r2 = r2 | r1; // 38 or
|
||||
r1 = mul_hi(r1, r0); // 39 mulhi
|
||||
r6 = rotl_imm(r6, 19u); // 40 rotl
|
||||
r4 = mul_hi(r4, r6); // 41 mulhi
|
||||
r6 = r6 - r0; // 42 sub
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 43 shfl
|
||||
r4 = r4 ^ ds[r2 & mask]; // 44 load
|
||||
r1 = r1 ^ r3; // 45 xor
|
||||
r7 = r7 ^ ds[r0 & mask]; // 46 load
|
||||
r3 = r3 ^ ds[r1 & mask]; // 47 load
|
||||
r5 = r5 * r3; // 48 mul
|
||||
r1 = r1 - r5; // 49 sub
|
||||
r2 = rotl_imm(r2, 8u); // 50 rotl
|
||||
r1 = r1 + r5 + ((((sel >> 23u) & 1u) != 0u) ? 0x77b9bd43u : 0xa900fec4u); // 51 add
|
||||
r4 = r4 ^ ds[r7 & mask]; // 52 load
|
||||
r2 = r2 - r7; // 53 sub
|
||||
r4 = r4 ^ r0; // 54 xor
|
||||
r1 = r1 + r6 + ((((sel >> 14u) & 1u) != 0u) ? 0x83e825bfu : 0xe09f54e9u); // 55 add
|
||||
r2 = r2 ^ ds[r4 & mask]; // 56 load
|
||||
r0 = r1 * r4 + r0; // 57 mad
|
||||
r3 = r3 ^ ds[r5 & mask]; // 58 load
|
||||
r5 = r5 | r6; // 59 or
|
||||
r6 = r5 * r7 + r6; // 60 mad
|
||||
r4 = rotl_imm(r4, 28u); // 61 rotl
|
||||
r5 = mul_hi(r5, r0); // 62 mulhi
|
||||
r3 = r3 ^ ds[r6 & mask]; // 63 load
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
|
||||
#if IGNEUM_EXCHANGE != 0
|
||||
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
|
||||
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
|
||||
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
|
||||
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
|
||||
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
|
||||
}
|
||||
#endif
|
||||
|
||||
// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash.
|
||||
IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw) {
|
||||
uint gid = (uint)get_global_id(0);
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; }
|
||||
{ uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; }
|
||||
{ uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; }
|
||||
{ uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; }
|
||||
{ uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; }
|
||||
{ uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; }
|
||||
{ uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; }
|
||||
{ uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r4 = r4 + r5 + ((((sel >> 13u) & 1u) != 0u) ? 0x5810667au : 0xea86e152u); // 0 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r0, 4u); r2 = r2 ^ t_; } // 1 shfl
|
||||
r3 = r3 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0x642e66dbu : 0x2cccb6cau); // 2 add
|
||||
r0 = rotl_imm(r0, 19u); // 3 rotl
|
||||
r7 = rotr_var(r7, r6); // 4 rotr
|
||||
r7 = r7 + r4 + ((((sel >> 21u) & 1u) != 0u) ? 0xc1535555u : 0xee02465fu); // 5 add
|
||||
r1 = mul_hi(r1, r7); // 6 mulhi
|
||||
r4 = r4 ^ ds[r2 & mask]; // 7 load
|
||||
r7 = r7 ^ ds[r4 & mask]; // 8 load
|
||||
r0 = r0 ^ ds[r3 & mask]; // 9 load
|
||||
r5 = r5 ^ ds[r1 & mask]; // 10 load
|
||||
r1 = r1 ^ ds[r5 & mask]; // 11 load
|
||||
r3 = mul_hi(r3, r5); // 12 mulhi
|
||||
r1 = r1 ^ ds[r3 & mask]; // 13 load
|
||||
r0 = r0 - r3; // 14 sub
|
||||
r5 = r1 * r3 + r5; // 15 mad
|
||||
r6 = mul_hi(r6, r1); // 16 mulhi
|
||||
r5 = r5 + r2 + ((((sel >> 28u) & 1u) != 0u) ? 0x8b965b57u : 0x697b3d00u); // 17 add
|
||||
r0 = mul_hi(r0, r6); // 18 mulhi
|
||||
r5 = rotr_var(r5, r3); // 19 rotr
|
||||
r5 = mul_hi(r5, r2); // 20 mulhi
|
||||
r1 = r1 + r0 + ((((sel >> 1u) & 1u) != 0u) ? 0x6d7e8d05u : 0xebcf247au); // 21 add
|
||||
r7 = r7 + r5 + ((((sel >> 12u) & 1u) != 0u) ? 0xb9e3577eu : 0xf66e7017u); // 22 add
|
||||
r1 = mul_hi(r1, r5); // 23 mulhi
|
||||
r2 = r2 - r5; // 24 sub
|
||||
r7 = r7 + r4 + ((((sel >> 2u) & 1u) != 0u) ? 0x699ef1bbu : 0x08ffa6c7u); // 25 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 26 shfl
|
||||
r7 = r7 + r1 + ((((sel >> 14u) & 1u) != 0u) ? 0xb4ead2fbu : 0xe60fea84u); // 27 add
|
||||
r3 = r3 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x8f30d21du : 0x65c76dabu); // 28 add
|
||||
r2 = r2 ^ ds[r1 & mask]; // 29 load
|
||||
r5 = r5 ^ ds[r7 & mask]; // 30 load
|
||||
r2 = r2 ^ ds[r5 & mask]; // 31 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 4u); r1 = r1 ^ t_; } // 32 shfl
|
||||
r4 = r5 * r7 + r4; // 33 mad
|
||||
r4 = r4 + r2 + ((((sel >> 21u) & 1u) != 0u) ? 0xc7ce690cu : 0x0480debeu); // 34 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 8u); r3 = r3 ^ t_; } // 35 shfl
|
||||
r7 = r7 + r1 + ((((sel >> 2u) & 1u) != 0u) ? 0xe10c2c95u : 0xc53b542eu); // 36 add
|
||||
r5 = r5 ^ r7; // 37 xor
|
||||
r2 = r2 | r1; // 38 or
|
||||
r1 = mul_hi(r1, r0); // 39 mulhi
|
||||
r6 = rotl_imm(r6, 19u); // 40 rotl
|
||||
r4 = mul_hi(r4, r6); // 41 mulhi
|
||||
r6 = r6 - r0; // 42 sub
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 43 shfl
|
||||
r4 = r4 ^ ds[r2 & mask]; // 44 load
|
||||
r1 = r1 ^ r3; // 45 xor
|
||||
r7 = r7 ^ ds[r0 & mask]; // 46 load
|
||||
r3 = r3 ^ ds[r1 & mask]; // 47 load
|
||||
r5 = r5 * r3; // 48 mul
|
||||
r1 = r1 - r5; // 49 sub
|
||||
r2 = rotl_imm(r2, 8u); // 50 rotl
|
||||
r1 = r1 + r5 + ((((sel >> 23u) & 1u) != 0u) ? 0x77b9bd43u : 0xa900fec4u); // 51 add
|
||||
r4 = r4 ^ ds[r7 & mask]; // 52 load
|
||||
r2 = r2 - r7; // 53 sub
|
||||
r4 = r4 ^ r0; // 54 xor
|
||||
r1 = r1 + r6 + ((((sel >> 14u) & 1u) != 0u) ? 0x83e825bfu : 0xe09f54e9u); // 55 add
|
||||
r2 = r2 ^ ds[r4 & mask]; // 56 load
|
||||
r0 = r1 * r4 + r0; // 57 mad
|
||||
r3 = r3 ^ ds[r5 & mask]; // 58 load
|
||||
r5 = r5 | r6; // 59 or
|
||||
r6 = r5 * r7 + r6; // 60 mad
|
||||
r4 = rotl_imm(r4, 28u); // 61 rotl
|
||||
r5 = mul_hi(r5, r0); // 62 mulhi
|
||||
r3 = r3 ^ ds[r6 & mask]; // 63 load
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
123
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/kernel_bound.cu
Normal file
123
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/kernel_bound.cu
Normal file
|
|
@ -0,0 +1,123 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000". Do not edit by hand.
|
||||
// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW.
|
||||
// Host declarations (also in program_bound.h if present):
|
||||
// struct IgneumInitWords { uint32_t w[8]; };
|
||||
// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps);
|
||||
// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
|
||||
#include <cuda_runtime.h>
|
||||
#include <cstdint>
|
||||
#include "program.h"
|
||||
|
||||
struct IgneumInitWords { uint32_t w[8]; };
|
||||
|
||||
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
|
||||
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
|
||||
__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw) {
|
||||
uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
uint32_t nonce = baseNonce + gid;
|
||||
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; }
|
||||
{ uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; }
|
||||
{ uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; }
|
||||
{ uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; }
|
||||
{ uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; }
|
||||
{ uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; }
|
||||
{ uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; }
|
||||
{ uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; }
|
||||
|
||||
for (uint32_t it = 0u; it < 8u; ++it) {
|
||||
uint32_t sel = r0;
|
||||
r4 = r4 + r5 + ((((sel >> 13u) & 1u) != 0u) ? 0x5810667au : 0xea86e152u); // 0 add
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r0, 4); // 1 shfl
|
||||
r3 = r3 + r2 + ((((sel >> 10u) & 1u) != 0u) ? 0x642e66dbu : 0x2cccb6cau); // 2 add
|
||||
r0 = rotl_imm(r0, 19u); // 3 rotl
|
||||
r7 = rotr_var(r7, r6); // 4 rotr
|
||||
r7 = r7 + r4 + ((((sel >> 21u) & 1u) != 0u) ? 0xc1535555u : 0xee02465fu); // 5 add
|
||||
r1 = __umulhi(r1, r7); // 6 mulhi
|
||||
r4 = r4 ^ ds[r2 & mask]; // 7 load
|
||||
r7 = r7 ^ ds[r4 & mask]; // 8 load
|
||||
r0 = r0 ^ ds[r3 & mask]; // 9 load
|
||||
r5 = r5 ^ ds[r1 & mask]; // 10 load
|
||||
r1 = r1 ^ ds[r5 & mask]; // 11 load
|
||||
r3 = __umulhi(r3, r5); // 12 mulhi
|
||||
r1 = r1 ^ ds[r3 & mask]; // 13 load
|
||||
r0 = r0 - r3; // 14 sub
|
||||
r5 = r1 * r3 + r5; // 15 mad
|
||||
r6 = __umulhi(r6, r1); // 16 mulhi
|
||||
r5 = r5 + r2 + ((((sel >> 28u) & 1u) != 0u) ? 0x8b965b57u : 0x697b3d00u); // 17 add
|
||||
r0 = __umulhi(r0, r6); // 18 mulhi
|
||||
r5 = rotr_var(r5, r3); // 19 rotr
|
||||
r5 = __umulhi(r5, r2); // 20 mulhi
|
||||
r1 = r1 + r0 + ((((sel >> 1u) & 1u) != 0u) ? 0x6d7e8d05u : 0xebcf247au); // 21 add
|
||||
r7 = r7 + r5 + ((((sel >> 12u) & 1u) != 0u) ? 0xb9e3577eu : 0xf66e7017u); // 22 add
|
||||
r1 = __umulhi(r1, r5); // 23 mulhi
|
||||
r2 = r2 - r5; // 24 sub
|
||||
r7 = r7 + r4 + ((((sel >> 2u) & 1u) != 0u) ? 0x699ef1bbu : 0x08ffa6c7u); // 25 add
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 26 shfl
|
||||
r7 = r7 + r1 + ((((sel >> 14u) & 1u) != 0u) ? 0xb4ead2fbu : 0xe60fea84u); // 27 add
|
||||
r3 = r3 + r1 + ((((sel >> 6u) & 1u) != 0u) ? 0x8f30d21du : 0x65c76dabu); // 28 add
|
||||
r2 = r2 ^ ds[r1 & mask]; // 29 load
|
||||
r5 = r5 ^ ds[r7 & mask]; // 30 load
|
||||
r2 = r2 ^ ds[r5 & mask]; // 31 load
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r7, 4); // 32 shfl
|
||||
r4 = r5 * r7 + r4; // 33 mad
|
||||
r4 = r4 + r2 + ((((sel >> 21u) & 1u) != 0u) ? 0xc7ce690cu : 0x0480debeu); // 34 add
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r7, 8); // 35 shfl
|
||||
r7 = r7 + r1 + ((((sel >> 2u) & 1u) != 0u) ? 0xe10c2c95u : 0xc53b542eu); // 36 add
|
||||
r5 = r5 ^ r7; // 37 xor
|
||||
r2 = r2 | r1; // 38 or
|
||||
r1 = __umulhi(r1, r0); // 39 mulhi
|
||||
r6 = rotl_imm(r6, 19u); // 40 rotl
|
||||
r4 = __umulhi(r4, r6); // 41 mulhi
|
||||
r6 = r6 - r0; // 42 sub
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 43 shfl
|
||||
r4 = r4 ^ ds[r2 & mask]; // 44 load
|
||||
r1 = r1 ^ r3; // 45 xor
|
||||
r7 = r7 ^ ds[r0 & mask]; // 46 load
|
||||
r3 = r3 ^ ds[r1 & mask]; // 47 load
|
||||
r5 = r5 * r3; // 48 mul
|
||||
r1 = r1 - r5; // 49 sub
|
||||
r2 = rotl_imm(r2, 8u); // 50 rotl
|
||||
r1 = r1 + r5 + ((((sel >> 23u) & 1u) != 0u) ? 0x77b9bd43u : 0xa900fec4u); // 51 add
|
||||
r4 = r4 ^ ds[r7 & mask]; // 52 load
|
||||
r2 = r2 - r7; // 53 sub
|
||||
r4 = r4 ^ r0; // 54 xor
|
||||
r1 = r1 + r6 + ((((sel >> 14u) & 1u) != 0u) ? 0x83e825bfu : 0xe09f54e9u); // 55 add
|
||||
r2 = r2 ^ ds[r4 & mask]; // 56 load
|
||||
r0 = r1 * r4 + r0; // 57 mad
|
||||
r3 = r3 ^ ds[r5 & mask]; // 58 load
|
||||
r5 = r5 | r6; // 59 or
|
||||
r6 = r5 * r7 + r6; // 60 mad
|
||||
r4 = rotl_imm(r4, 28u); // 61 rotl
|
||||
r5 = __umulhi(r5, r0); // 62 mulhi
|
||||
r3 = r3 ^ ds[r6 & mask]; // 63 load
|
||||
}
|
||||
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
|
||||
}
|
||||
|
||||
cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps) {
|
||||
if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 32u * blockWarps;
|
||||
if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;
|
||||
igneum_hash_bound<<<nonces / block, block>>>(ds, out, baseNonce, mask, iw);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
|
||||
cudaFuncAttributes attr;
|
||||
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound);
|
||||
if (e != cudaSuccess) return e;
|
||||
*numRegs = attr.numRegs;
|
||||
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0);
|
||||
}
|
||||
109
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/memhard.h
Normal file
109
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/memhard.h
Normal file
|
|
@ -0,0 +1,109 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000". Do not edit by hand.
|
||||
// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against.
|
||||
// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference).
|
||||
// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C.
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
#if defined(__CUDACC__)
|
||||
#define IGNEUM_HD __host__ __device__ __forceinline__
|
||||
#elif defined(_MSC_VER) && !defined(__cplusplus)
|
||||
#define IGNEUM_HD static __inline
|
||||
#else
|
||||
#define IGNEUM_HD static inline
|
||||
#endif
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of 4 x seed-parameterised mixer + one 64-byte cache read, then 4 x final mixer (class v3, mixer multiplier 4,
|
||||
// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) {
|
||||
for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint32_t r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) {
|
||||
uint32_t prev[16]; uint32_t x[16]; uint32_t y[16];
|
||||
for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0xceed56d7u ^ prev[4];
|
||||
x[5] = 0x9ba270d2u ^ prev[5];
|
||||
x[6] = 0x82caab2du ^ prev[6];
|
||||
x[7] = 0x81ebce0eu ^ prev[7];
|
||||
x[8] = 0x12b6ecf1u ^ prev[8];
|
||||
x[9] = 0xd0f3fd7cu ^ prev[9];
|
||||
x[10] = 0xd872eefeu ^ prev[10];
|
||||
x[11] = 0xc158c7bdu ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) {
|
||||
s[0] = (s[0] ^ (0xc6892460u + rk)) * 0xf351d601u;
|
||||
s[1] = (s[1] ^ (0x25b7228au + rk)) * 0xa3bb398fu;
|
||||
s[2] = (s[2] ^ (0xcd515004u + rk)) * 0xb5a09e35u;
|
||||
s[3] = (s[3] ^ (0x2846527au + rk)) * 0x7509c9c1u;
|
||||
s[4] = (s[4] ^ (0xa6324241u + rk)) * 0x6bbf31e9u;
|
||||
s[5] = (s[5] ^ (0x36e3ec53u + rk)) * 0xfc849a79u;
|
||||
s[6] = (s[6] ^ (0x82961bacu + rk)) * 0xded91851u;
|
||||
s[7] = (s[7] ^ (0x0f97ba7du + rk)) * 0x8d9113d1u;
|
||||
s[8] = (s[8] ^ (0xb6f921a9u + rk)) * 0x0ff15225u;
|
||||
s[9] = (s[9] ^ (0x3ada24e5u + rk)) * 0x3a5bdd41u;
|
||||
s[10] = (s[10] ^ (0xde20ab91u + rk)) * 0xab533435u;
|
||||
s[11] = (s[11] ^ (0x5378eeb2u + rk)) * 0xe1c55ad5u;
|
||||
s[12] = (s[12] ^ (0x7d161662u + rk)) * 0xe6d3bd0du;
|
||||
s[13] = (s[13] ^ (0x89353cc1u + rk)) * 0x9d9ffbbdu;
|
||||
s[14] = (s[14] ^ (0xb1aa03a2u + rk)) * 0xbb2a3cf3u;
|
||||
s[15] = (s[15] ^ (0x788acae6u + rk)) * 0x50a7c08du;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 17u, 12u, 20u, 23u) MH_QR(s[1], s[5], s[9], s[13], 17u, 12u, 20u, 23u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 17u, 12u, 20u, 23u) MH_QR(s[3], s[7], s[11], s[15], 17u, 12u, 20u, 23u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 7u, 3u, 27u, 16u) MH_QR(s[1], s[6], s[11], s[12], 7u, 3u, 27u, 16u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 7u, 3u, 27u, 16u) MH_QR(s[3], s[4], s[9], s[14], 7u, 3u, 27u, 16u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of 4 x mixer + cache line s[0] & mask; 4 x final mixer.
|
||||
IGNEUM_HD void mh_item(const uint32_t* cache, uint32_t t, uint32_t* s) {
|
||||
s[0] = 0xceed56d7u;
|
||||
s[1] = 0x9ba270d2u;
|
||||
s[2] = 0x82caab2du;
|
||||
s[3] = 0x81ebce0eu;
|
||||
s[4] = 0x12b6ecf1u;
|
||||
s[5] = 0xd0f3fd7cu;
|
||||
s[6] = 0xd872eefeu;
|
||||
s[7] = 0xc158c7bdu;
|
||||
s[8] = t * 0xf351d601u + 0xc6892460u;
|
||||
s[9] = t * 0xa3bb398fu + 0x25b7228au;
|
||||
s[10] = t * 0xb5a09e35u + 0xcd515004u;
|
||||
s[11] = t * 0x7509c9c1u + 0x2846527au;
|
||||
s[12] = t * 0x6bbf31e9u + 0xa6324241u;
|
||||
s[13] = t * 0xfc849a79u + 0x36e3ec53u;
|
||||
s[14] = t * 0xded91851u + 0x82961bacu;
|
||||
s[15] = t * 0x8d9113d1u + 0x0f97ba7du;
|
||||
for (uint32_t r = 0u; r < 8u; ++r) {
|
||||
for (uint32_t j = 0u; j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 4u + j + 1u));
|
||||
const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
for (uint32_t j = 0u; j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (32u + j + 1u));
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
IGNEUM_HD uint32_t mh_word(const uint32_t* cache, uint32_t w) { uint32_t s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
107
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/memhard.metal
Normal file
107
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/memhard.metal
Normal file
|
|
@ -0,0 +1,107 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of 4 x seed-parameterised mixer + one 64-byte cache read, then 4 x final mixer (class v3, mixer multiplier 4,
|
||||
// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
inline void mh_chacha_block(const thread uint* x, thread uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
inline void mh_cache_segment(device uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0xceed56d7u ^ prev[4];
|
||||
x[5] = 0x9ba270d2u ^ prev[5];
|
||||
x[6] = 0x82caab2du ^ prev[6];
|
||||
x[7] = 0x81ebce0eu ^ prev[7];
|
||||
x[8] = 0x12b6ecf1u ^ prev[8];
|
||||
x[9] = 0xd0f3fd7cu ^ prev[9];
|
||||
x[10] = 0xd872eefeu ^ prev[10];
|
||||
x[11] = 0xc158c7bdu ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
inline void mh_mixer(thread uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xc6892460u + rk)) * 0xf351d601u;
|
||||
s[1] = (s[1] ^ (0x25b7228au + rk)) * 0xa3bb398fu;
|
||||
s[2] = (s[2] ^ (0xcd515004u + rk)) * 0xb5a09e35u;
|
||||
s[3] = (s[3] ^ (0x2846527au + rk)) * 0x7509c9c1u;
|
||||
s[4] = (s[4] ^ (0xa6324241u + rk)) * 0x6bbf31e9u;
|
||||
s[5] = (s[5] ^ (0x36e3ec53u + rk)) * 0xfc849a79u;
|
||||
s[6] = (s[6] ^ (0x82961bacu + rk)) * 0xded91851u;
|
||||
s[7] = (s[7] ^ (0x0f97ba7du + rk)) * 0x8d9113d1u;
|
||||
s[8] = (s[8] ^ (0xb6f921a9u + rk)) * 0x0ff15225u;
|
||||
s[9] = (s[9] ^ (0x3ada24e5u + rk)) * 0x3a5bdd41u;
|
||||
s[10] = (s[10] ^ (0xde20ab91u + rk)) * 0xab533435u;
|
||||
s[11] = (s[11] ^ (0x5378eeb2u + rk)) * 0xe1c55ad5u;
|
||||
s[12] = (s[12] ^ (0x7d161662u + rk)) * 0xe6d3bd0du;
|
||||
s[13] = (s[13] ^ (0x89353cc1u + rk)) * 0x9d9ffbbdu;
|
||||
s[14] = (s[14] ^ (0xb1aa03a2u + rk)) * 0xbb2a3cf3u;
|
||||
s[15] = (s[15] ^ (0x788acae6u + rk)) * 0x50a7c08du;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 17u, 12u, 20u, 23u) MH_QR(s[1], s[5], s[9], s[13], 17u, 12u, 20u, 23u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 17u, 12u, 20u, 23u) MH_QR(s[3], s[7], s[11], s[15], 17u, 12u, 20u, 23u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 7u, 3u, 27u, 16u) MH_QR(s[1], s[6], s[11], s[12], 7u, 3u, 27u, 16u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 7u, 3u, 27u, 16u) MH_QR(s[3], s[4], s[9], s[14], 7u, 3u, 27u, 16u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of 4 x mixer + cache line s[0] & mask; 4 x final mixer.
|
||||
inline void mh_item(device const uint* cache, uint t, thread uint* s) {
|
||||
s[0] = 0xceed56d7u;
|
||||
s[1] = 0x9ba270d2u;
|
||||
s[2] = 0x82caab2du;
|
||||
s[3] = 0x81ebce0eu;
|
||||
s[4] = 0x12b6ecf1u;
|
||||
s[5] = 0xd0f3fd7cu;
|
||||
s[6] = 0xd872eefeu;
|
||||
s[7] = 0xc158c7bdu;
|
||||
s[8] = t * 0xf351d601u + 0xc6892460u;
|
||||
s[9] = t * 0xa3bb398fu + 0x25b7228au;
|
||||
s[10] = t * 0xb5a09e35u + 0xcd515004u;
|
||||
s[11] = t * 0x7509c9c1u + 0x2846527au;
|
||||
s[12] = t * 0x6bbf31e9u + 0xa6324241u;
|
||||
s[13] = t * 0xfc849a79u + 0x36e3ec53u;
|
||||
s[14] = t * 0xded91851u + 0x82961bacu;
|
||||
s[15] = t * 0x8d9113d1u + 0x0f97ba7du;
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
for (uint j = 0u; j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 4u + j + 1u));
|
||||
device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
for (uint j = 0u; j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (32u + j + 1u));
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
inline uint mh_word(device const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
|
||||
// One thread per segment (2^16 threads).
|
||||
kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) {
|
||||
mh_cache_segment(cache, gid);
|
||||
}
|
||||
// One thread per 64-byte item (dataset words / 16 threads).
|
||||
kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],
|
||||
uint gid [[thread_position_in_grid]]) {
|
||||
uint s[16];
|
||||
mh_item(cache, gid, s);
|
||||
device uint* d = dataset + gid * 16u;
|
||||
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
67
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/program.h
Normal file
67
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/program.h
Normal file
|
|
@ -0,0 +1,67 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000". Do not edit by hand.
|
||||
// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.
|
||||
// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros.
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
#ifndef IGNEUM_NO_CUDA
|
||||
#include <cuda_runtime.h>
|
||||
#endif
|
||||
|
||||
#define IGNEUM_SEED_STRING "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000"
|
||||
#define IGNEUM_SEED_BYTES_HEX "edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07"
|
||||
#define IGNEUM_GENERATOR 3
|
||||
#define IGNEUM_PROGRAM_ATTEMPT 0
|
||||
#define IGNEUM_PROGRAM_ID 0x73bcbfe8ccf988f1ull
|
||||
#define IGNEUM_DAY_STRING "bytes:69676e65756d2d6461792ffa50000000000000"
|
||||
#define IGNEUM_DAY_BYTES_HEX "69676e65756d2d6461792ffa50000000000000"
|
||||
#define IGNEUM_DAY0 0xceed56d7u
|
||||
#define IGNEUM_DAY1 0x9ba270d2u
|
||||
#define IGNEUM_DATASET_LOG2 28
|
||||
#define IGNEUM_MASK 0x0fffffffu
|
||||
#define IGNEUM_LANES 32
|
||||
#define IGNEUM_ITERATIONS 8
|
||||
#define IGNEUM_INSTR_COUNT 64
|
||||
#define IGNEUM_LOADS_PER_HASH 128
|
||||
#define IGNEUM_WIDE_LOADS_PER_HASH 0
|
||||
#define IGNEUM_OP_MIX "load=16 add=13 mulhi=9 shfl=5 sub=5 mad=4 rotl=4 xor=3 or=2 rotr=2 mul=1"
|
||||
// Program class v3 (Counter ASIC 2.0, docs/plans/counter-asic-2-rollout.md): generator version 3; a worker that
|
||||
// runs another class refuses this pack, and a job line names the class it wants (class=v3 era=<hex>).
|
||||
#define IGNEUM_PROGRAM_CLASS "v3"
|
||||
#define IGNEUM_ERA_SEED_HEX "edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07"
|
||||
// Class v3 construction (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): version 2 loads; the dataset item
|
||||
// derivation applies the mixer IGNEUM_MIXER_MULT times per round (memhard.h), and the cache follows the growth rule.
|
||||
#define IGNEUM_LOAD_CLASS "mx4"
|
||||
#define IGNEUM_CLASS_MIXER_MULT 4
|
||||
#define IGNEUM_CACHE_GROWTH 1 // 1: cache words = 2^(26 + doublings(day)), doublings = floor(log2(1 + day / 1460))
|
||||
#define IGNEUM_LOAD_SLOTS 16
|
||||
#define IGNEUM_LOAD_MIX { 100, 0, 0 }
|
||||
#define IGNEUM_LOAD_WIDTH_COUNTS { 16, 0, 0 } // loads of 4, 16, 64 bytes per program
|
||||
#define IGNEUM_BYTES_PER_HASH 512
|
||||
#define IGNEUM_FOLD_ROT 11
|
||||
#define IGNEUM_FOLD_MUL 0x9e3779b1u
|
||||
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
|
||||
#define IGNEUM_DATASET_MODE 1
|
||||
|
||||
#define IGNEUM_SEEDW_INIT { 0x667d0fbdu, 0x7b8e5963u, 0x31c67e5eu, 0x4529ddc6u, 0xef19d6d8u, 0xaccf6211u, 0xda0aed32u, 0xabc6df31u }
|
||||
#define IGNEUM_KEY_INIT { 0xceed56d7u, 0x9ba270d2u, 0x82caab2du, 0x81ebce0eu, 0x12b6ecf1u, 0xd0f3fd7cu, 0xd872eefeu, 0xc158c7bdu }
|
||||
#define IGNEUM_CACHE_LOG2_WORDS 26
|
||||
#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6
|
||||
#define IGNEUM_CACHE_SEGMENTS 65536u
|
||||
#define IGNEUM_ITEM_ROUNDS 8
|
||||
#define IGNEUM_MIXER_MULT 4 // mixer applications per round and after the last read (class v3, docs/plans/mixer-x4.md)
|
||||
#define IGNEUM_MIX_ROT_INIT { 17u, 12u, 20u, 23u, 7u, 3u, 27u, 16u }
|
||||
#define IGNEUM_MIX_MUL_INIT { 0xf351d601u, 0xa3bb398fu, 0xb5a09e35u, 0x7509c9c1u, 0x6bbf31e9u, 0xfc849a79u, 0xded91851u, 0x8d9113d1u, 0x0ff15225u, 0x3a5bdd41u, 0xab533435u, 0xe1c55ad5u, 0xe6d3bd0du, 0x9d9ffbbdu, 0xbb2a3cf3u, 0x50a7c08du }
|
||||
#define IGNEUM_MIX_RC_INIT { 0xc6892460u, 0x25b7228au, 0xcd515004u, 0x2846527au, 0xa6324241u, 0x36e3ec53u, 0x82961bacu, 0x0f97ba7du, 0xb6f921a9u, 0x3ada24e5u, 0xde20ab91u, 0x5378eeb2u, 0x7d161662u, 0x89353cc1u, 0xb1aa03a2u, 0x788acae6u }
|
||||
|
||||
#ifndef IGNEUM_NO_CUDA
|
||||
// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().
|
||||
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);
|
||||
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);
|
||||
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
uint32_t nonces, uint32_t blockWarps);
|
||||
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
|
||||
#endif
|
||||
133
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/program.json
Normal file
133
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/program.json
Normal file
|
|
@ -0,0 +1,133 @@
|
|||
{
|
||||
"format": "igneum-program-pack-3",
|
||||
"generator": 3,
|
||||
"attempt": 0,
|
||||
"program_id": "0x73bcbfe8ccf988f1",
|
||||
"program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32",
|
||||
"dataset_mode": "memory-hard",
|
||||
"seed": "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000",
|
||||
"seed_bytes": "edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07",
|
||||
"seed_words": ["0x667d0fbd", "0x7b8e5963", "0x31c67e5e", "0x4529ddc6", "0xef19d6d8", "0xaccf6211", "0xda0aed32", "0xabc6df31"],
|
||||
"seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32",
|
||||
"generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried",
|
||||
"lanes": 32,
|
||||
"registers": 8,
|
||||
"iterations": 8,
|
||||
"instruction_count": 64,
|
||||
"loads_per_hash": 128,
|
||||
"program_class": "v3",
|
||||
"era_seed_bytes": "edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07",
|
||||
"load_class": "mx4",
|
||||
"mixer_mult": 4,
|
||||
"cache_growth": true,
|
||||
"mixer": "class v3 (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): every mixer application of the item derivation is 4 applications with round keys (r * 4 + j + 1) * 0x9E3779B9, the 8 dependent cache reads per item unchanged; cache growth rule option C: cache words = 2^(26 + doublings(day)), dataset words = 2^(genesis_log2 + doublings(day)), doublings(day) = floor(log2(1 + day / 1460)) for day = days since genesis",
|
||||
"load_slots": 16,
|
||||
"load_mix_percent_4_16_64": [100, 0, 0],
|
||||
"load_width_counts_4_16_64": [16, 0, 0],
|
||||
"bytes_per_hash": 512,
|
||||
"wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots",
|
||||
"op_mix": {"load": 16, "add": 13, "mulhi": 9, "shfl": 5, "sub": 5, "mad": 4, "rotl": 4, "xor": 3, "or": 2, "rotr": 2, "mul": 1},
|
||||
"register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]",
|
||||
"splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16",
|
||||
"iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order",
|
||||
"output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo",
|
||||
"op_semantics": {
|
||||
"add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)",
|
||||
"sub": "dst = dst - src",
|
||||
"mul": "dst = dst * src (low 32)",
|
||||
"mulhi": "dst = high 32 bits of dst * src",
|
||||
"xor": "dst = dst ^ src",
|
||||
"or": "dst = dst | src",
|
||||
"rotl": "dst = rotl(dst, rot), rot in 1..31",
|
||||
"rotr": "dst = rotr(dst, src & 31)",
|
||||
"mad": "dst = src * src2 + dst",
|
||||
"shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp",
|
||||
"load": "dst = dst ^ dataset[src & dataset.mask]",
|
||||
"wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)"
|
||||
},
|
||||
"dataset": {
|
||||
"log2_words": 28,
|
||||
"bytes": 1073741824,
|
||||
"mask": "0x0fffffff",
|
||||
"day": "bytes:69676e65756d2d6461792ffa50000000000000",
|
||||
"day_bytes": "69676e65756d2d6461792ffa50000000000000",
|
||||
"day_words_from": "seed_words_from_bytes(day_bytes)",
|
||||
"d0": "0xceed56d7",
|
||||
"d1": "0x9ba270d2",
|
||||
"mode": "memory-hard",
|
||||
"spec": "proto-metal/MEMHARD.md",
|
||||
"key": ["0xceed56d7", "0x9ba270d2", "0x82caab2d", "0x81ebce0e", "0x12b6ecf1", "0xd0f3fd7c", "0xd872eefe", "0xc158c7bd"],
|
||||
"key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]",
|
||||
"cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"},
|
||||
"mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [17, 12, 20, 23, 7, 3, 27, 16], "mul": ["0xf351d601", "0xa3bb398f", "0xb5a09e35", "0x7509c9c1", "0x6bbf31e9", "0xfc849a79", "0xded91851", "0x8d9113d1", "0x0ff15225", "0x3a5bdd41", "0xab533435", "0xe1c55ad5", "0xe6d3bd0d", "0x9d9ffbbd", "0xbb2a3cf3", "0x50a7c08d"], "rc": ["0xc6892460", "0x25b7228a", "0xcd515004", "0x2846527a", "0xa6324241", "0x36e3ec53", "0x82961bac", "0x0f97ba7d", "0xb6f921a9", "0x3ada24e5", "0xde20ab91", "0x5378eeb2", "0x7d161662", "0x89353cc1", "0xb1aa03a2", "0x788acae6"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"},
|
||||
"mixer_mult": 4,
|
||||
"item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: for j in 0..3: s = M(s, rk = (r * 4 + j + 1) * 0x9E3779B9); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then for j in 0..3: s = M(s, rk = (32 + j + 1) * 0x9E3779B9); item(t) = s",
|
||||
"word": "dataset[w] = item(w >> 4)[w & 15]"
|
||||
},
|
||||
"instructions": [
|
||||
{"i": 0, "op": "add", "dst": 4, "src": 5, "src2": 7, "imm": "0xea86e152", "imm2": "0x5810667a", "rot": 27, "bit": 13, "mask": 2, "width": 1},
|
||||
{"i": 1, "op": "shfl", "dst": 2, "src": 0, "src2": 7, "imm": "0xe3c2f9cb", "imm2": "0xde0bea4e", "rot": 6, "bit": 9, "mask": 4, "width": 1},
|
||||
{"i": 2, "op": "add", "dst": 3, "src": 2, "src2": 0, "imm": "0x2cccb6ca", "imm2": "0x642e66db", "rot": 27, "bit": 10, "mask": 2, "width": 1},
|
||||
{"i": 3, "op": "rotl", "dst": 0, "src": 2, "src2": 1, "imm": "0xb59e83b2", "imm2": "0x19c26fb9", "rot": 19, "bit": 20, "mask": 4, "width": 1},
|
||||
{"i": 4, "op": "rotr", "dst": 7, "src": 6, "src2": 5, "imm": "0x6f055f55", "imm2": "0x550e4ea1", "rot": 31, "bit": 20, "mask": 4, "width": 1},
|
||||
{"i": 5, "op": "add", "dst": 7, "src": 4, "src2": 7, "imm": "0xee02465f", "imm2": "0xc1535555", "rot": 31, "bit": 21, "mask": 4, "width": 1},
|
||||
{"i": 6, "op": "mulhi", "dst": 1, "src": 7, "src2": 5, "imm": "0x7826a6a7", "imm2": "0x946f7818", "rot": 18, "bit": 21, "mask": 8, "width": 1},
|
||||
{"i": 7, "op": "load", "dst": 4, "src": 2, "src2": 0, "imm": "0x5d080878", "imm2": "0xdf885578", "rot": 5, "bit": 18, "mask": 4, "width": 1},
|
||||
{"i": 8, "op": "load", "dst": 7, "src": 4, "src2": 1, "imm": "0x875bbb36", "imm2": "0x594a838f", "rot": 24, "bit": 4, "mask": 2, "width": 1},
|
||||
{"i": 9, "op": "load", "dst": 0, "src": 3, "src2": 1, "imm": "0xe5e607c2", "imm2": "0xa5cd9f75", "rot": 26, "bit": 28, "mask": 4, "width": 1},
|
||||
{"i": 10, "op": "load", "dst": 5, "src": 1, "src2": 3, "imm": "0xea3f7b43", "imm2": "0x10c8d4e7", "rot": 5, "bit": 29, "mask": 8, "width": 1},
|
||||
{"i": 11, "op": "load", "dst": 1, "src": 5, "src2": 4, "imm": "0x4454980f", "imm2": "0xebd31581", "rot": 10, "bit": 28, "mask": 16, "width": 1},
|
||||
{"i": 12, "op": "mulhi", "dst": 3, "src": 5, "src2": 7, "imm": "0xbfd5615c", "imm2": "0xd8224ae2", "rot": 21, "bit": 12, "mask": 2, "width": 1},
|
||||
{"i": 13, "op": "load", "dst": 1, "src": 3, "src2": 5, "imm": "0x35c07cc5", "imm2": "0xe82db54f", "rot": 7, "bit": 6, "mask": 8, "width": 1},
|
||||
{"i": 14, "op": "sub", "dst": 0, "src": 3, "src2": 0, "imm": "0x587ee0f1", "imm2": "0xfd23eefd", "rot": 16, "bit": 21, "mask": 16, "width": 1},
|
||||
{"i": 15, "op": "mad", "dst": 5, "src": 1, "src2": 3, "imm": "0x6974dd29", "imm2": "0xc8148960", "rot": 3, "bit": 11, "mask": 2, "width": 1},
|
||||
{"i": 16, "op": "mulhi", "dst": 6, "src": 1, "src2": 7, "imm": "0x3072c3c6", "imm2": "0x55ee21f8", "rot": 26, "bit": 1, "mask": 1, "width": 1},
|
||||
{"i": 17, "op": "add", "dst": 5, "src": 2, "src2": 2, "imm": "0x697b3d00", "imm2": "0x8b965b57", "rot": 9, "bit": 28, "mask": 1, "width": 1},
|
||||
{"i": 18, "op": "mulhi", "dst": 0, "src": 6, "src2": 3, "imm": "0x2910cacb", "imm2": "0x6ac79431", "rot": 7, "bit": 6, "mask": 4, "width": 1},
|
||||
{"i": 19, "op": "rotr", "dst": 5, "src": 3, "src2": 0, "imm": "0xed96a94a", "imm2": "0x4c988c10", "rot": 24, "bit": 21, "mask": 8, "width": 1},
|
||||
{"i": 20, "op": "mulhi", "dst": 5, "src": 2, "src2": 7, "imm": "0x40d2fc76", "imm2": "0x2f7c7eca", "rot": 13, "bit": 8, "mask": 4, "width": 1},
|
||||
{"i": 21, "op": "add", "dst": 1, "src": 0, "src2": 5, "imm": "0xebcf247a", "imm2": "0x6d7e8d05", "rot": 31, "bit": 1, "mask": 16, "width": 1},
|
||||
{"i": 22, "op": "add", "dst": 7, "src": 5, "src2": 7, "imm": "0xf66e7017", "imm2": "0xb9e3577e", "rot": 9, "bit": 12, "mask": 16, "width": 1},
|
||||
{"i": 23, "op": "mulhi", "dst": 1, "src": 5, "src2": 7, "imm": "0xfdfe72fd", "imm2": "0x735eeb8d", "rot": 30, "bit": 25, "mask": 8, "width": 1},
|
||||
{"i": 24, "op": "sub", "dst": 2, "src": 5, "src2": 0, "imm": "0x32cf1258", "imm2": "0xd813deb6", "rot": 30, "bit": 6, "mask": 16, "width": 1},
|
||||
{"i": 25, "op": "add", "dst": 7, "src": 4, "src2": 2, "imm": "0x08ffa6c7", "imm2": "0x699ef1bb", "rot": 7, "bit": 2, "mask": 16, "width": 1},
|
||||
{"i": 26, "op": "shfl", "dst": 3, "src": 4, "src2": 5, "imm": "0x6f53c70d", "imm2": "0x3357513f", "rot": 3, "bit": 26, "mask": 2, "width": 1},
|
||||
{"i": 27, "op": "add", "dst": 7, "src": 1, "src2": 4, "imm": "0xe60fea84", "imm2": "0xb4ead2fb", "rot": 14, "bit": 14, "mask": 1, "width": 1},
|
||||
{"i": 28, "op": "add", "dst": 3, "src": 1, "src2": 0, "imm": "0x65c76dab", "imm2": "0x8f30d21d", "rot": 24, "bit": 6, "mask": 1, "width": 1},
|
||||
{"i": 29, "op": "load", "dst": 2, "src": 1, "src2": 2, "imm": "0x82fad9a6", "imm2": "0x8c6358db", "rot": 7, "bit": 31, "mask": 2, "width": 1},
|
||||
{"i": 30, "op": "load", "dst": 5, "src": 7, "src2": 3, "imm": "0x6e947ee0", "imm2": "0xaf9a2dda", "rot": 2, "bit": 30, "mask": 4, "width": 1},
|
||||
{"i": 31, "op": "load", "dst": 2, "src": 5, "src2": 1, "imm": "0x608bb7ce", "imm2": "0x4be663db", "rot": 19, "bit": 1, "mask": 16, "width": 1},
|
||||
{"i": 32, "op": "shfl", "dst": 1, "src": 7, "src2": 6, "imm": "0x88e52e20", "imm2": "0x77647269", "rot": 20, "bit": 25, "mask": 4, "width": 1},
|
||||
{"i": 33, "op": "mad", "dst": 4, "src": 5, "src2": 7, "imm": "0xb48420ae", "imm2": "0x3d1f2485", "rot": 14, "bit": 28, "mask": 1, "width": 1},
|
||||
{"i": 34, "op": "add", "dst": 4, "src": 2, "src2": 3, "imm": "0x0480debe", "imm2": "0xc7ce690c", "rot": 12, "bit": 21, "mask": 16, "width": 1},
|
||||
{"i": 35, "op": "shfl", "dst": 3, "src": 7, "src2": 5, "imm": "0xa73f59de", "imm2": "0x84b9e329", "rot": 21, "bit": 27, "mask": 8, "width": 1},
|
||||
{"i": 36, "op": "add", "dst": 7, "src": 1, "src2": 7, "imm": "0xc53b542e", "imm2": "0xe10c2c95", "rot": 21, "bit": 2, "mask": 4, "width": 1},
|
||||
{"i": 37, "op": "xor", "dst": 5, "src": 7, "src2": 4, "imm": "0x81cd7b0e", "imm2": "0x21a51823", "rot": 12, "bit": 10, "mask": 2, "width": 1},
|
||||
{"i": 38, "op": "or", "dst": 2, "src": 1, "src2": 3, "imm": "0x7894e657", "imm2": "0xf8e4b972", "rot": 18, "bit": 9, "mask": 4, "width": 1},
|
||||
{"i": 39, "op": "mulhi", "dst": 1, "src": 0, "src2": 4, "imm": "0xbee8421f", "imm2": "0x070888a8", "rot": 20, "bit": 28, "mask": 4, "width": 1},
|
||||
{"i": 40, "op": "rotl", "dst": 6, "src": 1, "src2": 4, "imm": "0x4609857a", "imm2": "0xaeecb156", "rot": 19, "bit": 21, "mask": 1, "width": 1},
|
||||
{"i": 41, "op": "mulhi", "dst": 4, "src": 6, "src2": 0, "imm": "0x1a84e1e9", "imm2": "0x9b26bb72", "rot": 28, "bit": 19, "mask": 1, "width": 1},
|
||||
{"i": 42, "op": "sub", "dst": 6, "src": 0, "src2": 6, "imm": "0xbf62908e", "imm2": "0xdf03ea88", "rot": 11, "bit": 27, "mask": 16, "width": 1},
|
||||
{"i": 43, "op": "shfl", "dst": 6, "src": 3, "src2": 4, "imm": "0xa966241c", "imm2": "0x9c639efa", "rot": 12, "bit": 4, "mask": 4, "width": 1},
|
||||
{"i": 44, "op": "load", "dst": 4, "src": 2, "src2": 3, "imm": "0x7b5b5474", "imm2": "0x45cfc5dd", "rot": 17, "bit": 18, "mask": 8, "width": 1},
|
||||
{"i": 45, "op": "xor", "dst": 1, "src": 3, "src2": 6, "imm": "0x006193d0", "imm2": "0xfc2acc3f", "rot": 25, "bit": 1, "mask": 1, "width": 1},
|
||||
{"i": 46, "op": "load", "dst": 7, "src": 0, "src2": 6, "imm": "0x4777c4f8", "imm2": "0x3cf0a02f", "rot": 11, "bit": 14, "mask": 4, "width": 1},
|
||||
{"i": 47, "op": "load", "dst": 3, "src": 1, "src2": 3, "imm": "0x10692532", "imm2": "0x1a292ea5", "rot": 20, "bit": 23, "mask": 1, "width": 1},
|
||||
{"i": 48, "op": "mul", "dst": 5, "src": 3, "src2": 7, "imm": "0xadf5bd13", "imm2": "0xb999de2e", "rot": 23, "bit": 10, "mask": 4, "width": 1},
|
||||
{"i": 49, "op": "sub", "dst": 1, "src": 5, "src2": 4, "imm": "0x68ff101e", "imm2": "0xbdaaf46a", "rot": 25, "bit": 23, "mask": 16, "width": 1},
|
||||
{"i": 50, "op": "rotl", "dst": 2, "src": 6, "src2": 4, "imm": "0x92d9a412", "imm2": "0x0daf96ea", "rot": 8, "bit": 4, "mask": 4, "width": 1},
|
||||
{"i": 51, "op": "add", "dst": 1, "src": 5, "src2": 0, "imm": "0xa900fec4", "imm2": "0x77b9bd43", "rot": 6, "bit": 23, "mask": 2, "width": 1},
|
||||
{"i": 52, "op": "load", "dst": 4, "src": 7, "src2": 1, "imm": "0x51392a72", "imm2": "0x99e8bb36", "rot": 11, "bit": 9, "mask": 1, "width": 1},
|
||||
{"i": 53, "op": "sub", "dst": 2, "src": 7, "src2": 2, "imm": "0x0ffe2ac7", "imm2": "0x030743df", "rot": 9, "bit": 30, "mask": 16, "width": 1},
|
||||
{"i": 54, "op": "xor", "dst": 4, "src": 0, "src2": 4, "imm": "0x9123ff15", "imm2": "0x10c329a7", "rot": 28, "bit": 15, "mask": 2, "width": 1},
|
||||
{"i": 55, "op": "add", "dst": 1, "src": 6, "src2": 5, "imm": "0xe09f54e9", "imm2": "0x83e825bf", "rot": 23, "bit": 14, "mask": 8, "width": 1},
|
||||
{"i": 56, "op": "load", "dst": 2, "src": 4, "src2": 3, "imm": "0x239de52c", "imm2": "0xf80bae18", "rot": 15, "bit": 24, "mask": 1, "width": 1},
|
||||
{"i": 57, "op": "mad", "dst": 0, "src": 1, "src2": 4, "imm": "0x31dede8e", "imm2": "0xd3f619e6", "rot": 29, "bit": 7, "mask": 2, "width": 1},
|
||||
{"i": 58, "op": "load", "dst": 3, "src": 5, "src2": 3, "imm": "0x206437d6", "imm2": "0x28d1c290", "rot": 17, "bit": 28, "mask": 4, "width": 1},
|
||||
{"i": 59, "op": "or", "dst": 5, "src": 6, "src2": 3, "imm": "0x8fffd674", "imm2": "0x0507903a", "rot": 26, "bit": 27, "mask": 2, "width": 1},
|
||||
{"i": 60, "op": "mad", "dst": 6, "src": 5, "src2": 7, "imm": "0xf572bdb9", "imm2": "0xeda2af31", "rot": 21, "bit": 8, "mask": 2, "width": 1},
|
||||
{"i": 61, "op": "rotl", "dst": 4, "src": 2, "src2": 7, "imm": "0x84f12ddf", "imm2": "0x81ef22e1", "rot": 28, "bit": 30, "mask": 1, "width": 1},
|
||||
{"i": 62, "op": "mulhi", "dst": 5, "src": 0, "src2": 6, "imm": "0xf6bb45ee", "imm2": "0x6bfb632d", "rot": 22, "bit": 0, "mask": 4, "width": 1},
|
||||
{"i": 63, "op": "load", "dst": 3, "src": 6, "src2": 6, "imm": "0x50ec702a", "imm2": "0xae6ee96e", "rot": 20, "bit": 25, "mask": 2, "width": 1}
|
||||
]
|
||||
}
|
||||
109
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/program.metal
Normal file
109
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/program.metal
Normal file
|
|
@ -0,0 +1,109 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
|
||||
#define MASK 0x0fffffffu
|
||||
constant uint SEEDW[8] = { 0x667d0fbdu, 0x7b8e5963u, 0x31c67e5eu, 0x4529ddc6u, 0xef19d6d8u, 0xaccf6211u, 0xda0aed32u, 0xabc6df31u };
|
||||
|
||||
inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
|
||||
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
|
||||
device ulong* out [[buffer(1)]],
|
||||
constant uint& baseNonce [[buffer(2)]],
|
||||
uint gid [[thread_position_in_grid]]) {
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint x = nonce ^ SEEDW[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ SEEDW[1]; }
|
||||
{ uint x = nonce ^ SEEDW[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ SEEDW[2]; }
|
||||
{ uint x = nonce ^ SEEDW[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ SEEDW[3]; }
|
||||
{ uint x = nonce ^ SEEDW[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ SEEDW[4]; }
|
||||
{ uint x = nonce ^ SEEDW[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ SEEDW[5]; }
|
||||
{ uint x = nonce ^ SEEDW[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ SEEDW[6]; }
|
||||
{ uint x = nonce ^ SEEDW[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ SEEDW[7]; }
|
||||
{ uint x = nonce ^ SEEDW[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ SEEDW[0]; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r4 = r4 + r5 + select(0xea86e152u, 0x5810667au, ((sel >> 13u) & 1u) != 0u); // 0
|
||||
r2 = r2 ^ simd_shuffle_xor(r0, (ushort)4); // 1
|
||||
r3 = r3 + r2 + select(0x2cccb6cau, 0x642e66dbu, ((sel >> 10u) & 1u) != 0u); // 2
|
||||
r0 = rotl_imm(r0, 19u); // 3
|
||||
r7 = rotr_var(r7, r6); // 4
|
||||
r7 = r7 + r4 + select(0xee02465fu, 0xc1535555u, ((sel >> 21u) & 1u) != 0u); // 5
|
||||
r1 = mulhi(r1, r7); // 6
|
||||
r4 = r4 ^ dataset[r2 & MASK]; // 7
|
||||
r7 = r7 ^ dataset[r4 & MASK]; // 8
|
||||
r0 = r0 ^ dataset[r3 & MASK]; // 9
|
||||
r5 = r5 ^ dataset[r1 & MASK]; // 10
|
||||
r1 = r1 ^ dataset[r5 & MASK]; // 11
|
||||
r3 = mulhi(r3, r5); // 12
|
||||
r1 = r1 ^ dataset[r3 & MASK]; // 13
|
||||
r0 = r0 - r3; // 14
|
||||
r5 = r1 * r3 + r5; // 15
|
||||
r6 = mulhi(r6, r1); // 16
|
||||
r5 = r5 + r2 + select(0x697b3d00u, 0x8b965b57u, ((sel >> 28u) & 1u) != 0u); // 17
|
||||
r0 = mulhi(r0, r6); // 18
|
||||
r5 = rotr_var(r5, r3); // 19
|
||||
r5 = mulhi(r5, r2); // 20
|
||||
r1 = r1 + r0 + select(0xebcf247au, 0x6d7e8d05u, ((sel >> 1u) & 1u) != 0u); // 21
|
||||
r7 = r7 + r5 + select(0xf66e7017u, 0xb9e3577eu, ((sel >> 12u) & 1u) != 0u); // 22
|
||||
r1 = mulhi(r1, r5); // 23
|
||||
r2 = r2 - r5; // 24
|
||||
r7 = r7 + r4 + select(0x08ffa6c7u, 0x699ef1bbu, ((sel >> 2u) & 1u) != 0u); // 25
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 26
|
||||
r7 = r7 + r1 + select(0xe60fea84u, 0xb4ead2fbu, ((sel >> 14u) & 1u) != 0u); // 27
|
||||
r3 = r3 + r1 + select(0x65c76dabu, 0x8f30d21du, ((sel >> 6u) & 1u) != 0u); // 28
|
||||
r2 = r2 ^ dataset[r1 & MASK]; // 29
|
||||
r5 = r5 ^ dataset[r7 & MASK]; // 30
|
||||
r2 = r2 ^ dataset[r5 & MASK]; // 31
|
||||
r1 = r1 ^ simd_shuffle_xor(r7, (ushort)4); // 32
|
||||
r4 = r5 * r7 + r4; // 33
|
||||
r4 = r4 + r2 + select(0x0480debeu, 0xc7ce690cu, ((sel >> 21u) & 1u) != 0u); // 34
|
||||
r3 = r3 ^ simd_shuffle_xor(r7, (ushort)8); // 35
|
||||
r7 = r7 + r1 + select(0xc53b542eu, 0xe10c2c95u, ((sel >> 2u) & 1u) != 0u); // 36
|
||||
r5 = r5 ^ r7; // 37
|
||||
r2 = r2 | r1; // 38
|
||||
r1 = mulhi(r1, r0); // 39
|
||||
r6 = rotl_imm(r6, 19u); // 40
|
||||
r4 = mulhi(r4, r6); // 41
|
||||
r6 = r6 - r0; // 42
|
||||
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 43
|
||||
r4 = r4 ^ dataset[r2 & MASK]; // 44
|
||||
r1 = r1 ^ r3; // 45
|
||||
r7 = r7 ^ dataset[r0 & MASK]; // 46
|
||||
r3 = r3 ^ dataset[r1 & MASK]; // 47
|
||||
r5 = r5 * r3; // 48
|
||||
r1 = r1 - r5; // 49
|
||||
r2 = rotl_imm(r2, 8u); // 50
|
||||
r1 = r1 + r5 + select(0xa900fec4u, 0x77b9bd43u, ((sel >> 23u) & 1u) != 0u); // 51
|
||||
r4 = r4 ^ dataset[r7 & MASK]; // 52
|
||||
r2 = r2 - r7; // 53
|
||||
r4 = r4 ^ r0; // 54
|
||||
r1 = r1 + r6 + select(0xe09f54e9u, 0x83e825bfu, ((sel >> 14u) & 1u) != 0u); // 55
|
||||
r2 = r2 ^ dataset[r4 & MASK]; // 56
|
||||
r0 = r1 * r4 + r0; // 57
|
||||
r3 = r3 ^ dataset[r5 & MASK]; // 58
|
||||
r5 = r5 | r6; // 59
|
||||
r6 = r5 * r7 + r6; // 60
|
||||
r4 = rotl_imm(r4, 28u); // 61
|
||||
r5 = mulhi(r5, r0); // 62
|
||||
r3 = r3 ^ dataset[r6 & MASK]; // 63
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
111
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/program_bound.metal
Normal file
111
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/program_bound.metal
Normal file
|
|
@ -0,0 +1,111 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
|
||||
#define MASK 0x0fffffffu
|
||||
constant uint SEEDW[8] = { 0x667d0fbdu, 0x7b8e5963u, 0x31c67e5eu, 0x4529ddc6u, 0xef19d6d8u, 0xaccf6211u, 0xda0aed32u, 0xabc6df31u };
|
||||
|
||||
inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
|
||||
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW.
|
||||
kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
|
||||
device ulong* out [[buffer(1)]],
|
||||
constant uint& baseNonce [[buffer(2)]],
|
||||
constant uint* initw [[buffer(3)]],
|
||||
uint gid [[thread_position_in_grid]]) {
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint x = nonce ^ initw[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ initw[1]; }
|
||||
{ uint x = nonce ^ initw[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ initw[2]; }
|
||||
{ uint x = nonce ^ initw[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ initw[3]; }
|
||||
{ uint x = nonce ^ initw[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ initw[4]; }
|
||||
{ uint x = nonce ^ initw[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ initw[5]; }
|
||||
{ uint x = nonce ^ initw[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ initw[6]; }
|
||||
{ uint x = nonce ^ initw[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ initw[7]; }
|
||||
{ uint x = nonce ^ initw[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ initw[0]; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r4 = r4 + r5 + select(0xea86e152u, 0x5810667au, ((sel >> 13u) & 1u) != 0u); // 0
|
||||
r2 = r2 ^ simd_shuffle_xor(r0, (ushort)4); // 1
|
||||
r3 = r3 + r2 + select(0x2cccb6cau, 0x642e66dbu, ((sel >> 10u) & 1u) != 0u); // 2
|
||||
r0 = rotl_imm(r0, 19u); // 3
|
||||
r7 = rotr_var(r7, r6); // 4
|
||||
r7 = r7 + r4 + select(0xee02465fu, 0xc1535555u, ((sel >> 21u) & 1u) != 0u); // 5
|
||||
r1 = mulhi(r1, r7); // 6
|
||||
r4 = r4 ^ dataset[r2 & MASK]; // 7
|
||||
r7 = r7 ^ dataset[r4 & MASK]; // 8
|
||||
r0 = r0 ^ dataset[r3 & MASK]; // 9
|
||||
r5 = r5 ^ dataset[r1 & MASK]; // 10
|
||||
r1 = r1 ^ dataset[r5 & MASK]; // 11
|
||||
r3 = mulhi(r3, r5); // 12
|
||||
r1 = r1 ^ dataset[r3 & MASK]; // 13
|
||||
r0 = r0 - r3; // 14
|
||||
r5 = r1 * r3 + r5; // 15
|
||||
r6 = mulhi(r6, r1); // 16
|
||||
r5 = r5 + r2 + select(0x697b3d00u, 0x8b965b57u, ((sel >> 28u) & 1u) != 0u); // 17
|
||||
r0 = mulhi(r0, r6); // 18
|
||||
r5 = rotr_var(r5, r3); // 19
|
||||
r5 = mulhi(r5, r2); // 20
|
||||
r1 = r1 + r0 + select(0xebcf247au, 0x6d7e8d05u, ((sel >> 1u) & 1u) != 0u); // 21
|
||||
r7 = r7 + r5 + select(0xf66e7017u, 0xb9e3577eu, ((sel >> 12u) & 1u) != 0u); // 22
|
||||
r1 = mulhi(r1, r5); // 23
|
||||
r2 = r2 - r5; // 24
|
||||
r7 = r7 + r4 + select(0x08ffa6c7u, 0x699ef1bbu, ((sel >> 2u) & 1u) != 0u); // 25
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 26
|
||||
r7 = r7 + r1 + select(0xe60fea84u, 0xb4ead2fbu, ((sel >> 14u) & 1u) != 0u); // 27
|
||||
r3 = r3 + r1 + select(0x65c76dabu, 0x8f30d21du, ((sel >> 6u) & 1u) != 0u); // 28
|
||||
r2 = r2 ^ dataset[r1 & MASK]; // 29
|
||||
r5 = r5 ^ dataset[r7 & MASK]; // 30
|
||||
r2 = r2 ^ dataset[r5 & MASK]; // 31
|
||||
r1 = r1 ^ simd_shuffle_xor(r7, (ushort)4); // 32
|
||||
r4 = r5 * r7 + r4; // 33
|
||||
r4 = r4 + r2 + select(0x0480debeu, 0xc7ce690cu, ((sel >> 21u) & 1u) != 0u); // 34
|
||||
r3 = r3 ^ simd_shuffle_xor(r7, (ushort)8); // 35
|
||||
r7 = r7 + r1 + select(0xc53b542eu, 0xe10c2c95u, ((sel >> 2u) & 1u) != 0u); // 36
|
||||
r5 = r5 ^ r7; // 37
|
||||
r2 = r2 | r1; // 38
|
||||
r1 = mulhi(r1, r0); // 39
|
||||
r6 = rotl_imm(r6, 19u); // 40
|
||||
r4 = mulhi(r4, r6); // 41
|
||||
r6 = r6 - r0; // 42
|
||||
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 43
|
||||
r4 = r4 ^ dataset[r2 & MASK]; // 44
|
||||
r1 = r1 ^ r3; // 45
|
||||
r7 = r7 ^ dataset[r0 & MASK]; // 46
|
||||
r3 = r3 ^ dataset[r1 & MASK]; // 47
|
||||
r5 = r5 * r3; // 48
|
||||
r1 = r1 - r5; // 49
|
||||
r2 = rotl_imm(r2, 8u); // 50
|
||||
r1 = r1 + r5 + select(0xa900fec4u, 0x77b9bd43u, ((sel >> 23u) & 1u) != 0u); // 51
|
||||
r4 = r4 ^ dataset[r7 & MASK]; // 52
|
||||
r2 = r2 - r7; // 53
|
||||
r4 = r4 ^ r0; // 54
|
||||
r1 = r1 + r6 + select(0xe09f54e9u, 0x83e825bfu, ((sel >> 14u) & 1u) != 0u); // 55
|
||||
r2 = r2 ^ dataset[r4 & MASK]; // 56
|
||||
r0 = r1 * r4 + r0; // 57
|
||||
r3 = r3 ^ dataset[r5 & MASK]; // 58
|
||||
r5 = r5 | r6; // 59
|
||||
r6 = r5 * r7 + r6; // 60
|
||||
r4 = rotl_imm(r4, 28u); // 61
|
||||
r5 = mulhi(r5, r0); // 62
|
||||
r3 = r3 ^ dataset[r6 & MASK]; // 63
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
57
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/vectors.h
Normal file
57
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/vectors.h
Normal file
|
|
@ -0,0 +1,57 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000". Do not edit by hand.
|
||||
// Expected outputs: igneum-pow (Rust) CPU interpreter, generator v3, memory-hard dataset
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
|
||||
#define IGNEUM_VEC_WARPS 3
|
||||
static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u };
|
||||
static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {
|
||||
{ // base nonce 0
|
||||
0x212c6442b51e87aeull, 0xc374795c00839331ull, 0xb6036a220a98f4b3ull, 0xeb8b8013e637367bull, 0xdb5866e9b73930fdull, 0xf3f3d01f46e90333ull, 0x9d913991ab8ed428ull, 0x7ccb1d8fa100a800ull,
|
||||
0x3cf45ba44f09a0feull, 0x91acf48ef1a63082ull, 0x6ea46c69fb082f99ull, 0x581f0218977a9d72ull, 0x9a4623a5c62ddf2dull, 0xab6eb5e768f0feb4ull, 0x07b70bdccf8aca12ull, 0xd666311ae5e4311eull,
|
||||
0x53114757d669f0a4ull, 0xbd5d6ace87ce2ce4ull, 0xfb712015e8189192ull, 0xa32cec81103e134bull, 0x83f3d18c3289c124ull, 0xfe29f1984b132b3dull, 0xc9ffcf4e3774497aull, 0xac99c9243dc63809ull,
|
||||
0xd78a7e8217a32f3cull, 0xab81ad63d242fc31ull, 0x0e4c30b7e00024afull, 0xce014289fff6778dull, 0x64a1292e2a8b4a91ull, 0xd5b8c90e681d7e3aull, 0x06078117673030fdull, 0x51bf77b280173930ull
|
||||
},
|
||||
{ // base nonce 4096
|
||||
0x3c797978566b5950ull, 0x7c759e60185b6411ull, 0x199136b84028d653ull, 0x0ad7b60cb722da50ull, 0x93e5fd443b85acc5ull, 0x4a0fadca2f0b7dc6ull, 0x91b5f2478c1fbd4full, 0xe3debc1f1cdc889aull,
|
||||
0x9c39b1635a6aec09ull, 0xbeaea0ab408df45aull, 0x080283b6f52a494aull, 0x9acae9a878c16f2eull, 0xc37dc4e098fee807ull, 0xf3daf9ff09a2c0e5ull, 0xfff82c3572c8f362ull, 0xd96414dfa0b4bd51ull,
|
||||
0xecf77498c26cc5abull, 0x72945c7a2e52a90cull, 0xe5b6749102690fa4ull, 0x90eddb84d13a2b9full, 0x6e9186a56dc002c1ull, 0xfa2047ec0ae61fe3ull, 0xa8cd4efd68b7eff5ull, 0xd23b650a6345bf73ull,
|
||||
0xe1141a33ea09bae1ull, 0x0e9aae812f5d762aull, 0x9ae3811dcc68797full, 0x0a4dab24c1efff6full, 0xa22d476bc42b9064ull, 0xc7a16d6f59414a2eull, 0x5aff92ecae652329ull, 0x96a903eb9a0ca390ull
|
||||
},
|
||||
{ // base nonce 1000000
|
||||
0xd5a8da0568df8ee7ull, 0x68cb69c04208285aull, 0xe3e0c57190a809dbull, 0x87ba27d76aa356d5ull, 0x5a30616d4b8bfcf2ull, 0x66f5d2c31997fd24ull, 0x0447779e26a2854aull, 0x1225eb67933b2faeull,
|
||||
0xde78d94be53c669bull, 0x02a3fe00eb77c8cdull, 0x4fd029ba4cfc526aull, 0x2ef8ea603a0463d6ull, 0x345cff9b465b6f12ull, 0x03bb80d5a700b7feull, 0xf7da33ca2094ae26ull, 0xb3adaaa8383ad2e9ull,
|
||||
0x68d5eafde2924d62ull, 0xeab53a1b039835e0ull, 0xfd04e678835cb646ull, 0xf25b86a707f333c0ull, 0xa3bb24e6981917deull, 0x791d38b29dd6055aull, 0xfd18f01f064bdbcaull, 0xee051af1de1b70f9ull,
|
||||
0x246bfe79acd966eaull, 0xbcb8105b8a73efedull, 0x9b7e8c6fcc25fa92ull, 0xa0ffe2d9e4717892ull, 0x22f52e2324035114ull, 0xcdac38103479920dull, 0x49477ffa08ac4a6cull, 0xf8ca84a1a5d78cf5ull
|
||||
}
|
||||
};
|
||||
|
||||
// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (268435455).
|
||||
static const uint32_t IGNEUM_DS_HEAD[16] = {
|
||||
0xafe80d67u, 0xb9fbd029u, 0x6c79f193u, 0x95139ad9u, 0x96310affu, 0x4609f8b1u, 0x75279e63u, 0x28235be1u,
|
||||
0x47b17dcbu, 0x718e0ef2u, 0xa52588c8u, 0xa8bf49d5u, 0x19cf243eu, 0x5ec8905eu, 0xa4851f66u, 0xaf9cd9f3u
|
||||
};
|
||||
static const uint32_t IGNEUM_DS_LAST_INDEX = 268435455u;
|
||||
static const uint32_t IGNEUM_DS_LAST = 0xe6a99c7au;
|
||||
// 64 sampled dataset words (index, value) computed on the Mac.
|
||||
#define IGNEUM_DS_SAMPLES 64
|
||||
static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = {
|
||||
59471966u, 217795994u, 208353206u, 42483309u, 172547758u, 148076330u, 183853158u, 214389424u, 267488061u, 169781097u, 184093494u, 153880993u, 84977930u, 46426879u, 3093825u, 225364072u, 44593546u, 260713159u, 168250303u, 52384140u, 223401610u, 45554030u, 95410555u, 175039924u, 79171087u, 267580473u, 24168642u, 37981670u, 171551130u, 195559979u, 204611762u, 140997658u, 138925853u, 86637313u, 20736778u, 219665210u, 160430336u, 264654675u, 8013395u, 228945585u, 213884386u, 104419827u, 44185464u, 142737231u, 99284897u, 132475900u, 61861762u, 132056166u, 262388043u, 91878046u, 117353561u, 124768597u, 71352993u, 190698941u, 46055428u, 55281366u, 165145231u, 106810753u, 171985651u, 232085256u, 159510492u, 40072060u, 209107596u, 39023794u
|
||||
};
|
||||
static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = {
|
||||
0x57642b58u, 0xc279baddu, 0xcbccbaadu, 0x32cce392u, 0x71fdb4c6u, 0xf5a268ceu, 0x3ca1d676u, 0x7977b03du, 0xb73a6cb1u, 0xe773c5c9u, 0x2533a3f1u, 0xd7c48638u, 0xfc80830au, 0xab3874e9u, 0xc093812eu, 0xb71e49d6u, 0x0e151e7bu, 0x395af161u, 0x7156ef4eu, 0x9202463fu, 0x562e88c8u, 0xb9b0c5b3u, 0xe3f6102fu, 0xe3a94cbbu, 0x9e26c494u, 0x47ace326u, 0xe20ea113u, 0xce4eefa1u, 0x97a41cc9u, 0xa6d51cb5u, 0x71f3df27u, 0x115df551u, 0x20e273b8u, 0x50bf99ffu, 0xc9b1ae13u, 0x342160c6u, 0xc1b8eb36u, 0x9b11737bu, 0x4b73d48fu, 0x455e9b9eu, 0xe5213a1cu, 0xf7049bc3u, 0x0a026f93u, 0x156a8a99u, 0x1029fe6cu, 0x4f29d205u, 0x273315edu, 0x483e85a2u, 0xec802349u, 0x76023cccu, 0xe37e8135u, 0x0881d277u, 0x56ba9c69u, 0x31ad3b78u, 0xc8ad9363u, 0x8647e3c0u, 0x12e78b02u, 0x2d3a1b5fu, 0xa75e9390u, 0xbe0cebd8u, 0x759af86du, 0x9b4bd567u, 0xd272d0aeu, 0x7a18d82eu
|
||||
};
|
||||
// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^26 words.
|
||||
static const uint32_t IGNEUM_CACHE_HEAD[16] = {
|
||||
0xebd9055cu, 0x7eaf6b21u, 0x9b610855u, 0x134a8562u, 0xb10059bau, 0x7f459e58u, 0x8f3c7873u, 0x3523e609u,
|
||||
0x75b8f4cau, 0x750d31b4u, 0x0dae0782u, 0x26f10015u, 0x7ba87a41u, 0x0efd8543u, 0x691d6368u, 0x8929d967u
|
||||
};
|
||||
static const uint32_t IGNEUM_CACHE_LAST[16] = {
|
||||
0x51474edfu, 0xc3cfba93u, 0xf21454cfu, 0x9b79baacu, 0x4d4fce23u, 0xd701dfd5u, 0x37357ab3u, 0x1be693fau,
|
||||
0xa701cb3bu, 0x7467c620u, 0x428184e8u, 0xf4010df0u, 0xe33aa88fu, 0xc78d6d62u, 0xae9ba9c0u, 0xf4bba97eu
|
||||
};
|
||||
static const uint64_t IGNEUM_CACHE_FNV64 = 0x448274a57f508cbcull;
|
||||
36
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/vectors.json
Normal file
36
proto-cuda/packs-ca2-mixer/mx4-devnet-epoch0/vectors.json
Normal file
|
|
@ -0,0 +1,36 @@
|
|||
{
|
||||
"seed": "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000",
|
||||
"day": "bytes:69676e65756d2d6461792ffa50000000000000",
|
||||
"dataset_mode": "memory-hard",
|
||||
"dataset_log2_words": 28,
|
||||
"mask": "0x0fffffff",
|
||||
"lanes": 32,
|
||||
"source": "igneum-pow (Rust) CPU interpreter, generator v3, memory-hard dataset",
|
||||
"warps": [
|
||||
{"base_nonce": 0, "expected": [
|
||||
"0x212c6442b51e87ae", "0xc374795c00839331", "0xb6036a220a98f4b3", "0xeb8b8013e637367b", "0xdb5866e9b73930fd", "0xf3f3d01f46e90333", "0x9d913991ab8ed428", "0x7ccb1d8fa100a800",
|
||||
"0x3cf45ba44f09a0fe", "0x91acf48ef1a63082", "0x6ea46c69fb082f99", "0x581f0218977a9d72", "0x9a4623a5c62ddf2d", "0xab6eb5e768f0feb4", "0x07b70bdccf8aca12", "0xd666311ae5e4311e",
|
||||
"0x53114757d669f0a4", "0xbd5d6ace87ce2ce4", "0xfb712015e8189192", "0xa32cec81103e134b", "0x83f3d18c3289c124", "0xfe29f1984b132b3d", "0xc9ffcf4e3774497a", "0xac99c9243dc63809",
|
||||
"0xd78a7e8217a32f3c", "0xab81ad63d242fc31", "0x0e4c30b7e00024af", "0xce014289fff6778d", "0x64a1292e2a8b4a91", "0xd5b8c90e681d7e3a", "0x06078117673030fd", "0x51bf77b280173930"
|
||||
]},
|
||||
{"base_nonce": 4096, "expected": [
|
||||
"0x3c797978566b5950", "0x7c759e60185b6411", "0x199136b84028d653", "0x0ad7b60cb722da50", "0x93e5fd443b85acc5", "0x4a0fadca2f0b7dc6", "0x91b5f2478c1fbd4f", "0xe3debc1f1cdc889a",
|
||||
"0x9c39b1635a6aec09", "0xbeaea0ab408df45a", "0x080283b6f52a494a", "0x9acae9a878c16f2e", "0xc37dc4e098fee807", "0xf3daf9ff09a2c0e5", "0xfff82c3572c8f362", "0xd96414dfa0b4bd51",
|
||||
"0xecf77498c26cc5ab", "0x72945c7a2e52a90c", "0xe5b6749102690fa4", "0x90eddb84d13a2b9f", "0x6e9186a56dc002c1", "0xfa2047ec0ae61fe3", "0xa8cd4efd68b7eff5", "0xd23b650a6345bf73",
|
||||
"0xe1141a33ea09bae1", "0x0e9aae812f5d762a", "0x9ae3811dcc68797f", "0x0a4dab24c1efff6f", "0xa22d476bc42b9064", "0xc7a16d6f59414a2e", "0x5aff92ecae652329", "0x96a903eb9a0ca390"
|
||||
]},
|
||||
{"base_nonce": 1000000, "expected": [
|
||||
"0xd5a8da0568df8ee7", "0x68cb69c04208285a", "0xe3e0c57190a809db", "0x87ba27d76aa356d5", "0x5a30616d4b8bfcf2", "0x66f5d2c31997fd24", "0x0447779e26a2854a", "0x1225eb67933b2fae",
|
||||
"0xde78d94be53c669b", "0x02a3fe00eb77c8cd", "0x4fd029ba4cfc526a", "0x2ef8ea603a0463d6", "0x345cff9b465b6f12", "0x03bb80d5a700b7fe", "0xf7da33ca2094ae26", "0xb3adaaa8383ad2e9",
|
||||
"0x68d5eafde2924d62", "0xeab53a1b039835e0", "0xfd04e678835cb646", "0xf25b86a707f333c0", "0xa3bb24e6981917de", "0x791d38b29dd6055a", "0xfd18f01f064bdbca", "0xee051af1de1b70f9",
|
||||
"0x246bfe79acd966ea", "0xbcb8105b8a73efed", "0x9b7e8c6fcc25fa92", "0xa0ffe2d9e4717892", "0x22f52e2324035114", "0xcdac38103479920d", "0x49477ffa08ac4a6c", "0xf8ca84a1a5d78cf5"
|
||||
]}
|
||||
],
|
||||
"dataset_head": ["0xafe80d67", "0xb9fbd029", "0x6c79f193", "0x95139ad9", "0x96310aff", "0x4609f8b1", "0x75279e63", "0x28235be1", "0x47b17dcb", "0x718e0ef2", "0xa52588c8", "0xa8bf49d5", "0x19cf243e", "0x5ec8905e", "0xa4851f66", "0xaf9cd9f3"],
|
||||
"dataset_last_index": 268435455,
|
||||
"dataset_last": "0xe6a99c7a",
|
||||
"dataset_samples": [{"index": 59471966, "value": "0x57642b58"}, {"index": 217795994, "value": "0xc279badd"}, {"index": 208353206, "value": "0xcbccbaad"}, {"index": 42483309, "value": "0x32cce392"}, {"index": 172547758, "value": "0x71fdb4c6"}, {"index": 148076330, "value": "0xf5a268ce"}, {"index": 183853158, "value": "0x3ca1d676"}, {"index": 214389424, "value": "0x7977b03d"}, {"index": 267488061, "value": "0xb73a6cb1"}, {"index": 169781097, "value": "0xe773c5c9"}, {"index": 184093494, "value": "0x2533a3f1"}, {"index": 153880993, "value": "0xd7c48638"}, {"index": 84977930, "value": "0xfc80830a"}, {"index": 46426879, "value": "0xab3874e9"}, {"index": 3093825, "value": "0xc093812e"}, {"index": 225364072, "value": "0xb71e49d6"}, {"index": 44593546, "value": "0x0e151e7b"}, {"index": 260713159, "value": "0x395af161"}, {"index": 168250303, "value": "0x7156ef4e"}, {"index": 52384140, "value": "0x9202463f"}, {"index": 223401610, "value": "0x562e88c8"}, {"index": 45554030, "value": "0xb9b0c5b3"}, {"index": 95410555, "value": "0xe3f6102f"}, {"index": 175039924, "value": "0xe3a94cbb"}, {"index": 79171087, "value": "0x9e26c494"}, {"index": 267580473, "value": "0x47ace326"}, {"index": 24168642, "value": "0xe20ea113"}, {"index": 37981670, "value": "0xce4eefa1"}, {"index": 171551130, "value": "0x97a41cc9"}, {"index": 195559979, "value": "0xa6d51cb5"}, {"index": 204611762, "value": "0x71f3df27"}, {"index": 140997658, "value": "0x115df551"}, {"index": 138925853, "value": "0x20e273b8"}, {"index": 86637313, "value": "0x50bf99ff"}, {"index": 20736778, "value": "0xc9b1ae13"}, {"index": 219665210, "value": "0x342160c6"}, {"index": 160430336, "value": "0xc1b8eb36"}, {"index": 264654675, "value": "0x9b11737b"}, {"index": 8013395, "value": "0x4b73d48f"}, {"index": 228945585, "value": "0x455e9b9e"}, {"index": 213884386, "value": "0xe5213a1c"}, {"index": 104419827, "value": "0xf7049bc3"}, {"index": 44185464, "value": "0x0a026f93"}, {"index": 142737231, "value": "0x156a8a99"}, {"index": 99284897, "value": "0x1029fe6c"}, {"index": 132475900, "value": "0x4f29d205"}, {"index": 61861762, "value": "0x273315ed"}, {"index": 132056166, "value": "0x483e85a2"}, {"index": 262388043, "value": "0xec802349"}, {"index": 91878046, "value": "0x76023ccc"}, {"index": 117353561, "value": "0xe37e8135"}, {"index": 124768597, "value": "0x0881d277"}, {"index": 71352993, "value": "0x56ba9c69"}, {"index": 190698941, "value": "0x31ad3b78"}, {"index": 46055428, "value": "0xc8ad9363"}, {"index": 55281366, "value": "0x8647e3c0"}, {"index": 165145231, "value": "0x12e78b02"}, {"index": 106810753, "value": "0x2d3a1b5f"}, {"index": 171985651, "value": "0xa75e9390"}, {"index": 232085256, "value": "0xbe0cebd8"}, {"index": 159510492, "value": "0x759af86d"}, {"index": 40072060, "value": "0x9b4bd567"}, {"index": 209107596, "value": "0xd272d0ae"}, {"index": 39023794, "value": "0x7a18d82e"}],
|
||||
"cache_head": ["0xebd9055c", "0x7eaf6b21", "0x9b610855", "0x134a8562", "0xb10059ba", "0x7f459e58", "0x8f3c7873", "0x3523e609", "0x75b8f4ca", "0x750d31b4", "0x0dae0782", "0x26f10015", "0x7ba87a41", "0x0efd8543", "0x691d6368", "0x8929d967"],
|
||||
"cache_last_line": ["0x51474edf", "0xc3cfba93", "0xf21454cf", "0x9b79baac", "0x4d4fce23", "0xd701dfd5", "0x37357ab3", "0x1be693fa", "0xa701cb3b", "0x7467c620", "0x428184e8", "0xf4010df0", "0xe33aa88f", "0xc78d6d62", "0xae9ba9c0", "0xf4bba97e"],
|
||||
"cache_fnv1a64": "0x448274a57f508cbc"
|
||||
}
|
||||
279
proto-cuda/packs-ca2-mixer/mx4-genesis/kernel.cl
Normal file
279
proto-cuda/packs-ca2-mixer/mx4-genesis/kernel.cl
Normal file
|
|
@ -0,0 +1,279 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
|
||||
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
|
||||
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
|
||||
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
|
||||
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
|
||||
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
|
||||
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
|
||||
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
|
||||
#ifndef IGNEUM_GROUP
|
||||
#define IGNEUM_GROUP 32
|
||||
#endif
|
||||
#ifndef IGNEUM_EXCHANGE
|
||||
#define IGNEUM_EXCHANGE 0
|
||||
#endif
|
||||
#ifdef __OPENCL_VERSION__
|
||||
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
|
||||
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#ifdef cl_khr_subgroups
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
|
||||
#endif
|
||||
#ifdef cl_khr_subgroup_shuffle
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
|
||||
#endif
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
|
||||
#endif
|
||||
#else
|
||||
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
|
||||
#include "emu_opencl.h"
|
||||
#endif
|
||||
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#else
|
||||
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
|
||||
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
|
||||
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
|
||||
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
|
||||
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
|
||||
#endif
|
||||
|
||||
static inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
|
||||
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
|
||||
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
|
||||
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
|
||||
static inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of 4 x seed-parameterised mixer + one 64-byte cache read, then 4 x final mixer (class v3, mixer multiplier 4,
|
||||
// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
static inline void mh_chacha_block(const uint* x, uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
static inline void mh_cache_segment(__global uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
static inline void mh_mixer(uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of 4 x mixer + cache line s[0] & mask; 4 x final mixer.
|
||||
static inline void mh_item(__global const uint* cache, uint t, uint* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
for (uint j = 0u; j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 4u + j + 1u));
|
||||
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
for (uint j = 0u; j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (32u + j + 1u));
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
|
||||
// The same constants as memhard.h in this pack (one emitter, three dialects).
|
||||
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
|
||||
uint seg = (uint)get_global_id(0);
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
|
||||
uint t = (uint)get_global_id(0);
|
||||
if (t < nItems) {
|
||||
uint s[16];
|
||||
mh_item(cache, t, s);
|
||||
__global uint* d = ds + ((ulong)t * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask) {
|
||||
uint gid = (uint)get_global_id(0);
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r2 = r1 * r1 + r2; // 1 mad
|
||||
r2 = r3 * r2 + r2; // 2 mad
|
||||
r3 = r3 ^ r5; // 3 xor
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r5 = r5 ^ ds[r7 & mask]; // 5 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl
|
||||
r1 = mul_hi(r1, r5); // 8 mulhi
|
||||
r6 = rotr_var(r6, r3); // 9 rotr
|
||||
r3 = r3 | r4; // 10 or
|
||||
r4 = r4 ^ ds[r3 & mask]; // 11 load
|
||||
r0 = mul_hi(r0, r4); // 12 mulhi
|
||||
r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add
|
||||
r0 = r0 ^ ds[r4 & mask]; // 14 load
|
||||
r2 = r2 - r4; // 15 sub
|
||||
r2 = r2 ^ ds[r0 & mask]; // 16 load
|
||||
r7 = r7 ^ ds[r2 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl
|
||||
r5 = r5 * r0; // 19 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl
|
||||
r6 = mul_hi(r6, r2); // 22 mulhi
|
||||
r6 = r6 ^ ds[r1 & mask]; // 23 load
|
||||
r5 = r5 * r0; // 24 mul
|
||||
r5 = rotl_imm(r5, 19u); // 25 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl
|
||||
r0 = r0 ^ r5; // 27 xor
|
||||
r0 = r0 ^ r4; // 28 xor
|
||||
r3 = r3 - r0; // 29 sub
|
||||
r5 = r5 * r1; // 30 mul
|
||||
r7 = r7 ^ ds[r2 & mask]; // 31 load
|
||||
r1 = r1 ^ ds[r0 & mask]; // 32 load
|
||||
r5 = r5 ^ r6; // 33 xor
|
||||
r5 = r5 ^ ds[r1 & mask]; // 34 load
|
||||
r0 = mul_hi(r0, r5); // 35 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl
|
||||
r7 = r7 ^ ds[r0 & mask]; // 37 load
|
||||
r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl
|
||||
r2 = r2 ^ r5; // 40 xor
|
||||
r3 = r6 * r3 + r3; // 41 mad
|
||||
r6 = r6 - r7; // 42 sub
|
||||
r7 = r7 ^ r0; // 43 xor
|
||||
r1 = r1 ^ ds[r7 & mask]; // 44 load
|
||||
r2 = r2 * r3; // 45 mul
|
||||
r1 = mul_hi(r1, r5); // 46 mulhi
|
||||
r4 = r4 - r3; // 47 sub
|
||||
r2 = rotr_var(r2, r6); // 48 rotr
|
||||
r3 = r3 ^ ds[r5 & mask]; // 49 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add
|
||||
r0 = r0 * r2; // 51 mul
|
||||
r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add
|
||||
r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add
|
||||
r7 = rotl_imm(r7, 14u); // 54 rotl
|
||||
r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add
|
||||
r6 = r6 ^ ds[r7 & mask]; // 56 load
|
||||
r1 = rotr_var(r1, r5); // 57 rotr
|
||||
r5 = r5 ^ ds[r4 & mask]; // 58 load
|
||||
r6 = r6 ^ ds[r2 & mask]; // 59 load
|
||||
r3 = r5 * r0 + r3; // 60 mad
|
||||
r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add
|
||||
r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add
|
||||
r5 = rotl_imm(r5, 19u); // 63 rotl
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
|
||||
#if IGNEUM_EXCHANGE != 0
|
||||
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
|
||||
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
|
||||
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
|
||||
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
|
||||
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
|
||||
}
|
||||
#endif
|
||||
164
proto-cuda/packs-ca2-mixer/mx4-genesis/kernel.cu
Normal file
164
proto-cuda/packs-ca2-mixer/mx4-genesis/kernel.cu
Normal file
|
|
@ -0,0 +1,164 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal).
|
||||
// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC.
|
||||
#include <cuda_runtime.h>
|
||||
#include <cstdint>
|
||||
#include "program.h"
|
||||
#include "memhard.h"
|
||||
|
||||
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31.
|
||||
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
|
||||
// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x.
|
||||
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) {
|
||||
uint32_t x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item.
|
||||
// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host.
|
||||
__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) {
|
||||
uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
|
||||
uint32_t t = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (t < nItems) {
|
||||
uint32_t s[16];
|
||||
mh_item(cache, t, s);
|
||||
uint32_t* d = ds + (size_t)t * 16u;
|
||||
for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
|
||||
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
|
||||
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
|
||||
__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask) {
|
||||
uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
uint32_t nonce = baseNonce + gid;
|
||||
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint32_t x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint32_t x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint32_t x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint32_t x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint32_t x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint32_t x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint32_t x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint32_t x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint32_t it = 0u; it < 8u; ++it) {
|
||||
uint32_t sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r2 = r1 * r1 + r2; // 1 mad
|
||||
r2 = r3 * r2 + r2; // 2 mad
|
||||
r3 = r3 ^ r5; // 3 xor
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r5 = r5 ^ ds[r7 & mask]; // 5 load
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 6 shfl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 8); // 7 shfl
|
||||
r1 = __umulhi(r1, r5); // 8 mulhi
|
||||
r6 = rotr_var(r6, r3); // 9 rotr
|
||||
r3 = r3 | r4; // 10 or
|
||||
r4 = r4 ^ ds[r3 & mask]; // 11 load
|
||||
r0 = __umulhi(r0, r4); // 12 mulhi
|
||||
r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add
|
||||
r0 = r0 ^ ds[r4 & mask]; // 14 load
|
||||
r2 = r2 - r4; // 15 sub
|
||||
r2 = r2 ^ ds[r0 & mask]; // 16 load
|
||||
r7 = r7 ^ ds[r2 & mask]; // 17 load
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 18 shfl
|
||||
r5 = r5 * r0; // 19 mul
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 20 shfl
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 16); // 21 shfl
|
||||
r6 = __umulhi(r6, r2); // 22 mulhi
|
||||
r6 = r6 ^ ds[r1 & mask]; // 23 load
|
||||
r5 = r5 * r0; // 24 mul
|
||||
r5 = rotl_imm(r5, 19u); // 25 rotl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r6, 2); // 26 shfl
|
||||
r0 = r0 ^ r5; // 27 xor
|
||||
r0 = r0 ^ r4; // 28 xor
|
||||
r3 = r3 - r0; // 29 sub
|
||||
r5 = r5 * r1; // 30 mul
|
||||
r7 = r7 ^ ds[r2 & mask]; // 31 load
|
||||
r1 = r1 ^ ds[r0 & mask]; // 32 load
|
||||
r5 = r5 ^ r6; // 33 xor
|
||||
r5 = r5 ^ ds[r1 & mask]; // 34 load
|
||||
r0 = __umulhi(r0, r5); // 35 mulhi
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // 36 shfl
|
||||
r7 = r7 ^ ds[r0 & mask]; // 37 load
|
||||
r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 4); // 39 shfl
|
||||
r2 = r2 ^ r5; // 40 xor
|
||||
r3 = r6 * r3 + r3; // 41 mad
|
||||
r6 = r6 - r7; // 42 sub
|
||||
r7 = r7 ^ r0; // 43 xor
|
||||
r1 = r1 ^ ds[r7 & mask]; // 44 load
|
||||
r2 = r2 * r3; // 45 mul
|
||||
r1 = __umulhi(r1, r5); // 46 mulhi
|
||||
r4 = r4 - r3; // 47 sub
|
||||
r2 = rotr_var(r2, r6); // 48 rotr
|
||||
r3 = r3 ^ ds[r5 & mask]; // 49 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add
|
||||
r0 = r0 * r2; // 51 mul
|
||||
r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add
|
||||
r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add
|
||||
r7 = rotl_imm(r7, 14u); // 54 rotl
|
||||
r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add
|
||||
r6 = r6 ^ ds[r7 & mask]; // 56 load
|
||||
r1 = rotr_var(r1, r5); // 57 rotr
|
||||
r5 = r5 ^ ds[r4 & mask]; // 58 load
|
||||
r6 = r6 ^ ds[r2 & mask]; // 59 load
|
||||
r3 = r5 * r0 + r3; // 60 mad
|
||||
r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add
|
||||
r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add
|
||||
r5 = rotl_imm(r5, 19u); // 63 rotl
|
||||
}
|
||||
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
|
||||
}
|
||||
|
||||
// Host-side launch wrappers. Declared in program.h, called from host.cu.
|
||||
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) {
|
||||
if (nSegments == 0u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 256u;
|
||||
uint32_t grid = (nSegments + block - 1u) / block;
|
||||
igneum_cache_fill<<<grid, block>>>(cache, nSegments);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
|
||||
if (nItems == 0u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 256u;
|
||||
uint32_t grid = (nItems + block - 1u) / block;
|
||||
igneum_build<<<grid, block>>>(ds, cache, nItems);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
uint32_t nonces, uint32_t blockWarps) {
|
||||
if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 32u * blockWarps;
|
||||
if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;
|
||||
igneum_hash<<<nonces / block, block>>>(ds, out, baseNonce, mask);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
|
||||
cudaFuncAttributes attr;
|
||||
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash);
|
||||
if (e != cudaSuccess) return e;
|
||||
*numRegs = attr.numRegs;
|
||||
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0);
|
||||
}
|
||||
373
proto-cuda/packs-ca2-mixer/mx4-genesis/kernel_bound.cl
Normal file
373
proto-cuda/packs-ca2-mixer/mx4-genesis/kernel_bound.cl
Normal file
|
|
@ -0,0 +1,373 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
|
||||
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
|
||||
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
|
||||
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
|
||||
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
|
||||
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
|
||||
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
|
||||
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
|
||||
#ifndef IGNEUM_GROUP
|
||||
#define IGNEUM_GROUP 32
|
||||
#endif
|
||||
#ifndef IGNEUM_EXCHANGE
|
||||
#define IGNEUM_EXCHANGE 0
|
||||
#endif
|
||||
#ifdef __OPENCL_VERSION__
|
||||
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
|
||||
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#ifdef cl_khr_subgroups
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
|
||||
#endif
|
||||
#ifdef cl_khr_subgroup_shuffle
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
|
||||
#endif
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
|
||||
#endif
|
||||
#else
|
||||
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
|
||||
#include "emu_opencl.h"
|
||||
#endif
|
||||
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#else
|
||||
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
|
||||
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
|
||||
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
|
||||
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
|
||||
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
|
||||
#endif
|
||||
|
||||
static inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
|
||||
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
|
||||
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
|
||||
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
|
||||
static inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of 4 x seed-parameterised mixer + one 64-byte cache read, then 4 x final mixer (class v3, mixer multiplier 4,
|
||||
// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
static inline void mh_chacha_block(const uint* x, uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
static inline void mh_cache_segment(__global uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
static inline void mh_mixer(uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of 4 x mixer + cache line s[0] & mask; 4 x final mixer.
|
||||
static inline void mh_item(__global const uint* cache, uint t, uint* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
for (uint j = 0u; j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 4u + j + 1u));
|
||||
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
for (uint j = 0u; j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (32u + j + 1u));
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
|
||||
// The same constants as memhard.h in this pack (one emitter, three dialects).
|
||||
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
|
||||
uint seg = (uint)get_global_id(0);
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
|
||||
uint t = (uint)get_global_id(0);
|
||||
if (t < nItems) {
|
||||
uint s[16];
|
||||
mh_item(cache, t, s);
|
||||
__global uint* d = ds + ((ulong)t * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask) {
|
||||
uint gid = (uint)get_global_id(0);
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r2 = r1 * r1 + r2; // 1 mad
|
||||
r2 = r3 * r2 + r2; // 2 mad
|
||||
r3 = r3 ^ r5; // 3 xor
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r5 = r5 ^ ds[r7 & mask]; // 5 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl
|
||||
r1 = mul_hi(r1, r5); // 8 mulhi
|
||||
r6 = rotr_var(r6, r3); // 9 rotr
|
||||
r3 = r3 | r4; // 10 or
|
||||
r4 = r4 ^ ds[r3 & mask]; // 11 load
|
||||
r0 = mul_hi(r0, r4); // 12 mulhi
|
||||
r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add
|
||||
r0 = r0 ^ ds[r4 & mask]; // 14 load
|
||||
r2 = r2 - r4; // 15 sub
|
||||
r2 = r2 ^ ds[r0 & mask]; // 16 load
|
||||
r7 = r7 ^ ds[r2 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl
|
||||
r5 = r5 * r0; // 19 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl
|
||||
r6 = mul_hi(r6, r2); // 22 mulhi
|
||||
r6 = r6 ^ ds[r1 & mask]; // 23 load
|
||||
r5 = r5 * r0; // 24 mul
|
||||
r5 = rotl_imm(r5, 19u); // 25 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl
|
||||
r0 = r0 ^ r5; // 27 xor
|
||||
r0 = r0 ^ r4; // 28 xor
|
||||
r3 = r3 - r0; // 29 sub
|
||||
r5 = r5 * r1; // 30 mul
|
||||
r7 = r7 ^ ds[r2 & mask]; // 31 load
|
||||
r1 = r1 ^ ds[r0 & mask]; // 32 load
|
||||
r5 = r5 ^ r6; // 33 xor
|
||||
r5 = r5 ^ ds[r1 & mask]; // 34 load
|
||||
r0 = mul_hi(r0, r5); // 35 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl
|
||||
r7 = r7 ^ ds[r0 & mask]; // 37 load
|
||||
r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl
|
||||
r2 = r2 ^ r5; // 40 xor
|
||||
r3 = r6 * r3 + r3; // 41 mad
|
||||
r6 = r6 - r7; // 42 sub
|
||||
r7 = r7 ^ r0; // 43 xor
|
||||
r1 = r1 ^ ds[r7 & mask]; // 44 load
|
||||
r2 = r2 * r3; // 45 mul
|
||||
r1 = mul_hi(r1, r5); // 46 mulhi
|
||||
r4 = r4 - r3; // 47 sub
|
||||
r2 = rotr_var(r2, r6); // 48 rotr
|
||||
r3 = r3 ^ ds[r5 & mask]; // 49 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add
|
||||
r0 = r0 * r2; // 51 mul
|
||||
r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add
|
||||
r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add
|
||||
r7 = rotl_imm(r7, 14u); // 54 rotl
|
||||
r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add
|
||||
r6 = r6 ^ ds[r7 & mask]; // 56 load
|
||||
r1 = rotr_var(r1, r5); // 57 rotr
|
||||
r5 = r5 ^ ds[r4 & mask]; // 58 load
|
||||
r6 = r6 ^ ds[r2 & mask]; // 59 load
|
||||
r3 = r5 * r0 + r3; // 60 mad
|
||||
r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add
|
||||
r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add
|
||||
r5 = rotl_imm(r5, 19u); // 63 rotl
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
|
||||
#if IGNEUM_EXCHANGE != 0
|
||||
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
|
||||
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
|
||||
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
|
||||
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
|
||||
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
|
||||
}
|
||||
#endif
|
||||
|
||||
// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash.
|
||||
IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw) {
|
||||
uint gid = (uint)get_global_id(0);
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; }
|
||||
{ uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; }
|
||||
{ uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; }
|
||||
{ uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; }
|
||||
{ uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; }
|
||||
{ uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; }
|
||||
{ uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; }
|
||||
{ uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r2 = r1 * r1 + r2; // 1 mad
|
||||
r2 = r3 * r2 + r2; // 2 mad
|
||||
r3 = r3 ^ r5; // 3 xor
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r5 = r5 ^ ds[r7 & mask]; // 5 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r1 = r1 ^ t_; } // 6 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 8u); r7 = r7 ^ t_; } // 7 shfl
|
||||
r1 = mul_hi(r1, r5); // 8 mulhi
|
||||
r6 = rotr_var(r6, r3); // 9 rotr
|
||||
r3 = r3 | r4; // 10 or
|
||||
r4 = r4 ^ ds[r3 & mask]; // 11 load
|
||||
r0 = mul_hi(r0, r4); // 12 mulhi
|
||||
r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add
|
||||
r0 = r0 ^ ds[r4 & mask]; // 14 load
|
||||
r2 = r2 - r4; // 15 sub
|
||||
r2 = r2 ^ ds[r0 & mask]; // 16 load
|
||||
r7 = r7 ^ ds[r2 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 18 shfl
|
||||
r5 = r5 * r0; // 19 mul
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 20 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 16u); r2 = r2 ^ t_; } // 21 shfl
|
||||
r6 = mul_hi(r6, r2); // 22 mulhi
|
||||
r6 = r6 ^ ds[r1 & mask]; // 23 load
|
||||
r5 = r5 * r0; // 24 mul
|
||||
r5 = rotl_imm(r5, 19u); // 25 rotl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 2u); r7 = r7 ^ t_; } // 26 shfl
|
||||
r0 = r0 ^ r5; // 27 xor
|
||||
r0 = r0 ^ r4; // 28 xor
|
||||
r3 = r3 - r0; // 29 sub
|
||||
r5 = r5 * r1; // 30 mul
|
||||
r7 = r7 ^ ds[r2 & mask]; // 31 load
|
||||
r1 = r1 ^ ds[r0 & mask]; // 32 load
|
||||
r5 = r5 ^ r6; // 33 xor
|
||||
r5 = r5 ^ ds[r1 & mask]; // 34 load
|
||||
r0 = mul_hi(r0, r5); // 35 mulhi
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r2, 4u); r5 = r5 ^ t_; } // 36 shfl
|
||||
r7 = r7 ^ ds[r0 & mask]; // 37 load
|
||||
r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 4u); r1 = r1 ^ t_; } // 39 shfl
|
||||
r2 = r2 ^ r5; // 40 xor
|
||||
r3 = r6 * r3 + r3; // 41 mad
|
||||
r6 = r6 - r7; // 42 sub
|
||||
r7 = r7 ^ r0; // 43 xor
|
||||
r1 = r1 ^ ds[r7 & mask]; // 44 load
|
||||
r2 = r2 * r3; // 45 mul
|
||||
r1 = mul_hi(r1, r5); // 46 mulhi
|
||||
r4 = r4 - r3; // 47 sub
|
||||
r2 = rotr_var(r2, r6); // 48 rotr
|
||||
r3 = r3 ^ ds[r5 & mask]; // 49 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add
|
||||
r0 = r0 * r2; // 51 mul
|
||||
r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add
|
||||
r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add
|
||||
r7 = rotl_imm(r7, 14u); // 54 rotl
|
||||
r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add
|
||||
r6 = r6 ^ ds[r7 & mask]; // 56 load
|
||||
r1 = rotr_var(r1, r5); // 57 rotr
|
||||
r5 = r5 ^ ds[r4 & mask]; // 58 load
|
||||
r6 = r6 ^ ds[r2 & mask]; // 59 load
|
||||
r3 = r5 * r0 + r3; // 60 mad
|
||||
r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add
|
||||
r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add
|
||||
r5 = rotl_imm(r5, 19u); // 63 rotl
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
123
proto-cuda/packs-ca2-mixer/mx4-genesis/kernel_bound.cu
Normal file
123
proto-cuda/packs-ca2-mixer/mx4-genesis/kernel_bound.cu
Normal file
|
|
@ -0,0 +1,123 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW.
|
||||
// Host declarations (also in program_bound.h if present):
|
||||
// struct IgneumInitWords { uint32_t w[8]; };
|
||||
// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps);
|
||||
// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
|
||||
#include <cuda_runtime.h>
|
||||
#include <cstdint>
|
||||
#include "program.h"
|
||||
|
||||
struct IgneumInitWords { uint32_t w[8]; };
|
||||
|
||||
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
|
||||
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
|
||||
__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw) {
|
||||
uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
uint32_t nonce = baseNonce + gid;
|
||||
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; }
|
||||
{ uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; }
|
||||
{ uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; }
|
||||
{ uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; }
|
||||
{ uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; }
|
||||
{ uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; }
|
||||
{ uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; }
|
||||
{ uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; }
|
||||
|
||||
for (uint32_t it = 0u; it < 8u; ++it) {
|
||||
uint32_t sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r2 = r1 * r1 + r2; // 1 mad
|
||||
r2 = r3 * r2 + r2; // 2 mad
|
||||
r3 = r3 ^ r5; // 3 xor
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r5 = r5 ^ ds[r7 & mask]; // 5 load
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 6 shfl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 8); // 7 shfl
|
||||
r1 = __umulhi(r1, r5); // 8 mulhi
|
||||
r6 = rotr_var(r6, r3); // 9 rotr
|
||||
r3 = r3 | r4; // 10 or
|
||||
r4 = r4 ^ ds[r3 & mask]; // 11 load
|
||||
r0 = __umulhi(r0, r4); // 12 mulhi
|
||||
r5 = r5 + r1 + ((((sel >> 30u) & 1u) != 0u) ? 0xd3177981u : 0xc7934706u); // 13 add
|
||||
r0 = r0 ^ ds[r4 & mask]; // 14 load
|
||||
r2 = r2 - r4; // 15 sub
|
||||
r2 = r2 ^ ds[r0 & mask]; // 16 load
|
||||
r7 = r7 ^ ds[r2 & mask]; // 17 load
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 18 shfl
|
||||
r5 = r5 * r0; // 19 mul
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 20 shfl
|
||||
r2 = r2 ^ __shfl_xor_sync(0xffffffffu, r4, 16); // 21 shfl
|
||||
r6 = __umulhi(r6, r2); // 22 mulhi
|
||||
r6 = r6 ^ ds[r1 & mask]; // 23 load
|
||||
r5 = r5 * r0; // 24 mul
|
||||
r5 = rotl_imm(r5, 19u); // 25 rotl
|
||||
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r6, 2); // 26 shfl
|
||||
r0 = r0 ^ r5; // 27 xor
|
||||
r0 = r0 ^ r4; // 28 xor
|
||||
r3 = r3 - r0; // 29 sub
|
||||
r5 = r5 * r1; // 30 mul
|
||||
r7 = r7 ^ ds[r2 & mask]; // 31 load
|
||||
r1 = r1 ^ ds[r0 & mask]; // 32 load
|
||||
r5 = r5 ^ r6; // 33 xor
|
||||
r5 = r5 ^ ds[r1 & mask]; // 34 load
|
||||
r0 = __umulhi(r0, r5); // 35 mulhi
|
||||
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r2, 4); // 36 shfl
|
||||
r7 = r7 ^ ds[r0 & mask]; // 37 load
|
||||
r3 = r3 + r1 + ((((sel >> 27u) & 1u) != 0u) ? 0x230c005cu : 0x75ba2fadu); // 38 add
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 4); // 39 shfl
|
||||
r2 = r2 ^ r5; // 40 xor
|
||||
r3 = r6 * r3 + r3; // 41 mad
|
||||
r6 = r6 - r7; // 42 sub
|
||||
r7 = r7 ^ r0; // 43 xor
|
||||
r1 = r1 ^ ds[r7 & mask]; // 44 load
|
||||
r2 = r2 * r3; // 45 mul
|
||||
r1 = __umulhi(r1, r5); // 46 mulhi
|
||||
r4 = r4 - r3; // 47 sub
|
||||
r2 = rotr_var(r2, r6); // 48 rotr
|
||||
r3 = r3 ^ ds[r5 & mask]; // 49 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 50 add
|
||||
r0 = r0 * r2; // 51 mul
|
||||
r0 = r0 + r2 + ((((sel >> 6u) & 1u) != 0u) ? 0x699fd448u : 0x4f92b968u); // 52 add
|
||||
r1 = r1 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0x77b1520du : 0x2bb965afu); // 53 add
|
||||
r7 = rotl_imm(r7, 14u); // 54 rotl
|
||||
r3 = r3 + r7 + ((((sel >> 1u) & 1u) != 0u) ? 0xa54c55a0u : 0x7b0fe07au); // 55 add
|
||||
r6 = r6 ^ ds[r7 & mask]; // 56 load
|
||||
r1 = rotr_var(r1, r5); // 57 rotr
|
||||
r5 = r5 ^ ds[r4 & mask]; // 58 load
|
||||
r6 = r6 ^ ds[r2 & mask]; // 59 load
|
||||
r3 = r5 * r0 + r3; // 60 mad
|
||||
r5 = r5 + r7 + ((((sel >> 31u) & 1u) != 0u) ? 0xad7493e7u : 0xaf9dd72du); // 61 add
|
||||
r4 = r4 + r6 + ((((sel >> 27u) & 1u) != 0u) ? 0x1e07c3d9u : 0x89841d87u); // 62 add
|
||||
r5 = rotl_imm(r5, 19u); // 63 rotl
|
||||
}
|
||||
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
|
||||
}
|
||||
|
||||
cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps) {
|
||||
if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 32u * blockWarps;
|
||||
if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;
|
||||
igneum_hash_bound<<<nonces / block, block>>>(ds, out, baseNonce, mask, iw);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
|
||||
cudaFuncAttributes attr;
|
||||
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound);
|
||||
if (e != cudaSuccess) return e;
|
||||
*numRegs = attr.numRegs;
|
||||
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0);
|
||||
}
|
||||
109
proto-cuda/packs-ca2-mixer/mx4-genesis/memhard.h
Normal file
109
proto-cuda/packs-ca2-mixer/mx4-genesis/memhard.h
Normal file
|
|
@ -0,0 +1,109 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against.
|
||||
// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference).
|
||||
// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C.
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
#if defined(__CUDACC__)
|
||||
#define IGNEUM_HD __host__ __device__ __forceinline__
|
||||
#elif defined(_MSC_VER) && !defined(__cplusplus)
|
||||
#define IGNEUM_HD static __inline
|
||||
#else
|
||||
#define IGNEUM_HD static inline
|
||||
#endif
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of 4 x seed-parameterised mixer + one 64-byte cache read, then 4 x final mixer (class v3, mixer multiplier 4,
|
||||
// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) {
|
||||
for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint32_t r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) {
|
||||
uint32_t prev[16]; uint32_t x[16]; uint32_t y[16];
|
||||
for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of 4 x mixer + cache line s[0] & mask; 4 x final mixer.
|
||||
IGNEUM_HD void mh_item(const uint32_t* cache, uint32_t t, uint32_t* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint32_t r = 0u; r < 8u; ++r) {
|
||||
for (uint32_t j = 0u; j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 4u + j + 1u));
|
||||
const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
for (uint32_t j = 0u; j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (32u + j + 1u));
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
IGNEUM_HD uint32_t mh_word(const uint32_t* cache, uint32_t w) { uint32_t s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
107
proto-cuda/packs-ca2-mixer/mx4-genesis/memhard.metal
Normal file
107
proto-cuda/packs-ca2-mixer/mx4-genesis/memhard.metal
Normal file
|
|
@ -0,0 +1,107 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of 4 x seed-parameterised mixer + one 64-byte cache read, then 4 x final mixer (class v3, mixer multiplier 4,
|
||||
// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
inline void mh_chacha_block(const thread uint* x, thread uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
inline void mh_cache_segment(device uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
inline void mh_mixer(thread uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of 4 x mixer + cache line s[0] & mask; 4 x final mixer.
|
||||
inline void mh_item(device const uint* cache, uint t, thread uint* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
for (uint j = 0u; j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 4u + j + 1u));
|
||||
device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
for (uint j = 0u; j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (32u + j + 1u));
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
inline uint mh_word(device const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
|
||||
// One thread per segment (2^16 threads).
|
||||
kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) {
|
||||
mh_cache_segment(cache, gid);
|
||||
}
|
||||
// One thread per 64-byte item (dataset words / 16 threads).
|
||||
kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],
|
||||
uint gid [[thread_position_in_grid]]) {
|
||||
uint s[16];
|
||||
mh_item(cache, gid, s);
|
||||
device uint* d = dataset + gid * 16u;
|
||||
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
66
proto-cuda/packs-ca2-mixer/mx4-genesis/program.h
Normal file
66
proto-cuda/packs-ca2-mixer/mx4-genesis/program.h
Normal file
|
|
@ -0,0 +1,66 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.
|
||||
// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros.
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
#ifndef IGNEUM_NO_CUDA
|
||||
#include <cuda_runtime.h>
|
||||
#endif
|
||||
|
||||
#define IGNEUM_SEED_STRING "igneum-genesis"
|
||||
#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973"
|
||||
#define IGNEUM_GENERATOR 3
|
||||
#define IGNEUM_PROGRAM_ATTEMPT 0
|
||||
#define IGNEUM_PROGRAM_ID 0xe323b9dcaf283a6full
|
||||
#define IGNEUM_DAY_STRING "2026-10-03"
|
||||
#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033"
|
||||
#define IGNEUM_DAY0 0x3067619fu
|
||||
#define IGNEUM_DAY1 0x3c269176u
|
||||
#define IGNEUM_DATASET_LOG2 28
|
||||
#define IGNEUM_MASK 0x0fffffffu
|
||||
#define IGNEUM_LANES 32
|
||||
#define IGNEUM_ITERATIONS 8
|
||||
#define IGNEUM_INSTR_COUNT 64
|
||||
#define IGNEUM_LOADS_PER_HASH 128
|
||||
#define IGNEUM_WIDE_LOADS_PER_HASH 0
|
||||
#define IGNEUM_OP_MIX "load=16 add=8 shfl=8 xor=6 mad=5 mul=5 mulhi=5 sub=4 rotl=3 rotr=3 or=1"
|
||||
// Program class v3 (Counter ASIC 2.0, docs/plans/counter-asic-2-rollout.md): generator version 3; a worker that
|
||||
// runs another class refuses this pack, and a job line names the class it wants (class=v3 era=<hex>).
|
||||
#define IGNEUM_PROGRAM_CLASS "v3"
|
||||
// Class v3 construction (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): version 2 loads; the dataset item
|
||||
// derivation applies the mixer IGNEUM_MIXER_MULT times per round (memhard.h), and the cache follows the growth rule.
|
||||
#define IGNEUM_LOAD_CLASS "mx4"
|
||||
#define IGNEUM_CLASS_MIXER_MULT 4
|
||||
#define IGNEUM_CACHE_GROWTH 1 // 1: cache words = 2^(26 + doublings(day)), doublings = floor(log2(1 + day / 1460))
|
||||
#define IGNEUM_LOAD_SLOTS 16
|
||||
#define IGNEUM_LOAD_MIX { 100, 0, 0 }
|
||||
#define IGNEUM_LOAD_WIDTH_COUNTS { 16, 0, 0 } // loads of 4, 16, 64 bytes per program
|
||||
#define IGNEUM_BYTES_PER_HASH 512
|
||||
#define IGNEUM_FOLD_ROT 11
|
||||
#define IGNEUM_FOLD_MUL 0x9e3779b1u
|
||||
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
|
||||
#define IGNEUM_DATASET_MODE 1
|
||||
|
||||
#define IGNEUM_SEEDW_INIT { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }
|
||||
#define IGNEUM_KEY_INIT { 0x3067619fu, 0x3c269176u, 0x84a03b03u, 0xf8c63294u, 0xff977c5bu, 0xe60def3eu, 0x63630141u, 0xb8fbcb58u }
|
||||
#define IGNEUM_CACHE_LOG2_WORDS 26
|
||||
#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6
|
||||
#define IGNEUM_CACHE_SEGMENTS 65536u
|
||||
#define IGNEUM_ITEM_ROUNDS 8
|
||||
#define IGNEUM_MIXER_MULT 4 // mixer applications per round and after the last read (class v3, docs/plans/mixer-x4.md)
|
||||
#define IGNEUM_MIX_ROT_INIT { 20u, 20u, 19u, 4u, 26u, 3u, 3u, 27u }
|
||||
#define IGNEUM_MIX_MUL_INIT { 0x42146205u, 0x52cbe0fbu, 0x7ecf4a03u, 0x6728907fu, 0xd81d9751u, 0x132952c3u, 0xf60de277u, 0x05358035u, 0xbaf6499du, 0xe4db9667u, 0x3e98f45du, 0xd0004eddu, 0x2691630du, 0x9beb3bcfu, 0xab310379u, 0x99cfb423u }
|
||||
#define IGNEUM_MIX_RC_INIT { 0xbab68293u, 0xcc162340u, 0x6ce151ccu, 0xe62b8997u, 0xc9c80297u, 0xf74a1654u, 0x3d704af5u, 0x3cf522b7u, 0x2b9cac04u, 0xa880ac10u, 0x13e5dd1du, 0x6fc3e233u, 0x2d83eeacu, 0x9006e8bfu, 0x2c4b5362u, 0x31b49ee2u }
|
||||
|
||||
#ifndef IGNEUM_NO_CUDA
|
||||
// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().
|
||||
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);
|
||||
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);
|
||||
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
uint32_t nonces, uint32_t blockWarps);
|
||||
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
|
||||
#endif
|
||||
132
proto-cuda/packs-ca2-mixer/mx4-genesis/program.json
Normal file
132
proto-cuda/packs-ca2-mixer/mx4-genesis/program.json
Normal file
|
|
@ -0,0 +1,132 @@
|
|||
{
|
||||
"format": "igneum-program-pack-3",
|
||||
"generator": 3,
|
||||
"attempt": 0,
|
||||
"program_id": "0xe323b9dcaf283a6f",
|
||||
"program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32",
|
||||
"dataset_mode": "memory-hard",
|
||||
"seed": "igneum-genesis",
|
||||
"seed_bytes": "69676e65756d2d67656e65736973",
|
||||
"seed_words": ["0x67a9a7be", "0x1a155b25", "0xfddfb732", "0x4b5af2e8", "0xc55caf33", "0xa27c13b7", "0x06628a48", "0x03852469"],
|
||||
"seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32",
|
||||
"generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried",
|
||||
"lanes": 32,
|
||||
"registers": 8,
|
||||
"iterations": 8,
|
||||
"instruction_count": 64,
|
||||
"loads_per_hash": 128,
|
||||
"program_class": "v3",
|
||||
"load_class": "mx4",
|
||||
"mixer_mult": 4,
|
||||
"cache_growth": true,
|
||||
"mixer": "class v3 (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): every mixer application of the item derivation is 4 applications with round keys (r * 4 + j + 1) * 0x9E3779B9, the 8 dependent cache reads per item unchanged; cache growth rule option C: cache words = 2^(26 + doublings(day)), dataset words = 2^(genesis_log2 + doublings(day)), doublings(day) = floor(log2(1 + day / 1460)) for day = days since genesis",
|
||||
"load_slots": 16,
|
||||
"load_mix_percent_4_16_64": [100, 0, 0],
|
||||
"load_width_counts_4_16_64": [16, 0, 0],
|
||||
"bytes_per_hash": 512,
|
||||
"wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots",
|
||||
"op_mix": {"load": 16, "add": 8, "shfl": 8, "xor": 6, "mad": 5, "mul": 5, "mulhi": 5, "sub": 4, "rotl": 3, "rotr": 3, "or": 1},
|
||||
"register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]",
|
||||
"splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16",
|
||||
"iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order",
|
||||
"output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo",
|
||||
"op_semantics": {
|
||||
"add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)",
|
||||
"sub": "dst = dst - src",
|
||||
"mul": "dst = dst * src (low 32)",
|
||||
"mulhi": "dst = high 32 bits of dst * src",
|
||||
"xor": "dst = dst ^ src",
|
||||
"or": "dst = dst | src",
|
||||
"rotl": "dst = rotl(dst, rot), rot in 1..31",
|
||||
"rotr": "dst = rotr(dst, src & 31)",
|
||||
"mad": "dst = src * src2 + dst",
|
||||
"shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp",
|
||||
"load": "dst = dst ^ dataset[src & dataset.mask]",
|
||||
"wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)"
|
||||
},
|
||||
"dataset": {
|
||||
"log2_words": 28,
|
||||
"bytes": 1073741824,
|
||||
"mask": "0x0fffffff",
|
||||
"day": "2026-10-03",
|
||||
"day_bytes": "6461792f323032362d31302d3033",
|
||||
"day_words_from": "seed_words_from_bytes(day_bytes)",
|
||||
"d0": "0x3067619f",
|
||||
"d1": "0x3c269176",
|
||||
"mode": "memory-hard",
|
||||
"spec": "proto-metal/MEMHARD.md",
|
||||
"key": ["0x3067619f", "0x3c269176", "0x84a03b03", "0xf8c63294", "0xff977c5b", "0xe60def3e", "0x63630141", "0xb8fbcb58"],
|
||||
"key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]",
|
||||
"cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"},
|
||||
"mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [20, 20, 19, 4, 26, 3, 3, 27], "mul": ["0x42146205", "0x52cbe0fb", "0x7ecf4a03", "0x6728907f", "0xd81d9751", "0x132952c3", "0xf60de277", "0x05358035", "0xbaf6499d", "0xe4db9667", "0x3e98f45d", "0xd0004edd", "0x2691630d", "0x9beb3bcf", "0xab310379", "0x99cfb423"], "rc": ["0xbab68293", "0xcc162340", "0x6ce151cc", "0xe62b8997", "0xc9c80297", "0xf74a1654", "0x3d704af5", "0x3cf522b7", "0x2b9cac04", "0xa880ac10", "0x13e5dd1d", "0x6fc3e233", "0x2d83eeac", "0x9006e8bf", "0x2c4b5362", "0x31b49ee2"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"},
|
||||
"mixer_mult": 4,
|
||||
"item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: for j in 0..3: s = M(s, rk = (r * 4 + j + 1) * 0x9E3779B9); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then for j in 0..3: s = M(s, rk = (32 + j + 1) * 0x9E3779B9); item(t) = s",
|
||||
"word": "dataset[w] = item(w >> 4)[w & 15]"
|
||||
},
|
||||
"instructions": [
|
||||
{"i": 0, "op": "mad", "dst": 2, "src": 3, "src2": 4, "imm": "0xbf7b174d", "imm2": "0x337b762e", "rot": 17, "bit": 2, "mask": 2, "width": 1},
|
||||
{"i": 1, "op": "mad", "dst": 2, "src": 1, "src2": 1, "imm": "0xdd04a5da", "imm2": "0x42da7657", "rot": 15, "bit": 30, "mask": 16, "width": 1},
|
||||
{"i": 2, "op": "mad", "dst": 2, "src": 3, "src2": 2, "imm": "0x734003fa", "imm2": "0x5bb67700", "rot": 3, "bit": 20, "mask": 1, "width": 1},
|
||||
{"i": 3, "op": "xor", "dst": 3, "src": 5, "src2": 5, "imm": "0xc55a1b1c", "imm2": "0xa19720f3", "rot": 7, "bit": 8, "mask": 1, "width": 1},
|
||||
{"i": 4, "op": "load", "dst": 7, "src": 2, "src2": 5, "imm": "0xad572dd7", "imm2": "0x9ceb3ea7", "rot": 18, "bit": 30, "mask": 2, "width": 1},
|
||||
{"i": 5, "op": "load", "dst": 5, "src": 7, "src2": 3, "imm": "0x769a53be", "imm2": "0x80f9067e", "rot": 12, "bit": 22, "mask": 1, "width": 1},
|
||||
{"i": 6, "op": "shfl", "dst": 1, "src": 4, "src2": 0, "imm": "0xd3613d88", "imm2": "0x262fb219", "rot": 10, "bit": 30, "mask": 8, "width": 1},
|
||||
{"i": 7, "op": "shfl", "dst": 7, "src": 3, "src2": 4, "imm": "0xce38e42f", "imm2": "0xb868b818", "rot": 11, "bit": 8, "mask": 8, "width": 1},
|
||||
{"i": 8, "op": "mulhi", "dst": 1, "src": 5, "src2": 2, "imm": "0xa5eebca5", "imm2": "0x5703a72b", "rot": 13, "bit": 13, "mask": 16, "width": 1},
|
||||
{"i": 9, "op": "rotr", "dst": 6, "src": 3, "src2": 4, "imm": "0x17a5a9c7", "imm2": "0xdcfb93a1", "rot": 20, "bit": 27, "mask": 2, "width": 1},
|
||||
{"i": 10, "op": "or", "dst": 3, "src": 4, "src2": 1, "imm": "0xccb7d785", "imm2": "0xc335364c", "rot": 14, "bit": 12, "mask": 4, "width": 1},
|
||||
{"i": 11, "op": "load", "dst": 4, "src": 3, "src2": 2, "imm": "0x88cb9af3", "imm2": "0x4e7dc10d", "rot": 24, "bit": 17, "mask": 4, "width": 1},
|
||||
{"i": 12, "op": "mulhi", "dst": 0, "src": 4, "src2": 4, "imm": "0x45374321", "imm2": "0x3cd91989", "rot": 11, "bit": 4, "mask": 2, "width": 1},
|
||||
{"i": 13, "op": "add", "dst": 5, "src": 1, "src2": 2, "imm": "0xc7934706", "imm2": "0xd3177981", "rot": 16, "bit": 30, "mask": 2, "width": 1},
|
||||
{"i": 14, "op": "load", "dst": 0, "src": 4, "src2": 3, "imm": "0xfd7f56bb", "imm2": "0x65e14f52", "rot": 13, "bit": 22, "mask": 2, "width": 1},
|
||||
{"i": 15, "op": "sub", "dst": 2, "src": 4, "src2": 4, "imm": "0x35a80b49", "imm2": "0x060f2d13", "rot": 16, "bit": 20, "mask": 16, "width": 1},
|
||||
{"i": 16, "op": "load", "dst": 2, "src": 0, "src2": 2, "imm": "0xae0a32c2", "imm2": "0x4c2a4cfe", "rot": 8, "bit": 31, "mask": 16, "width": 1},
|
||||
{"i": 17, "op": "load", "dst": 7, "src": 2, "src2": 6, "imm": "0x82a84cc3", "imm2": "0x21a38d68", "rot": 15, "bit": 21, "mask": 2, "width": 1},
|
||||
{"i": 18, "op": "shfl", "dst": 7, "src": 3, "src2": 3, "imm": "0xa3818806", "imm2": "0x8f66b5c8", "rot": 14, "bit": 6, "mask": 4, "width": 1},
|
||||
{"i": 19, "op": "mul", "dst": 5, "src": 0, "src2": 1, "imm": "0xa00de107", "imm2": "0x77bfcaa5", "rot": 3, "bit": 10, "mask": 2, "width": 1},
|
||||
{"i": 20, "op": "shfl", "dst": 3, "src": 4, "src2": 7, "imm": "0x1d2b8cab", "imm2": "0x80b4f9a2", "rot": 14, "bit": 25, "mask": 2, "width": 1},
|
||||
{"i": 21, "op": "shfl", "dst": 2, "src": 4, "src2": 5, "imm": "0x3ac915d2", "imm2": "0x5fba7bc2", "rot": 16, "bit": 1, "mask": 16, "width": 1},
|
||||
{"i": 22, "op": "mulhi", "dst": 6, "src": 2, "src2": 0, "imm": "0xdc3ec8fd", "imm2": "0x599e2fa3", "rot": 22, "bit": 3, "mask": 2, "width": 1},
|
||||
{"i": 23, "op": "load", "dst": 6, "src": 1, "src2": 5, "imm": "0x2a6b16d5", "imm2": "0xd73e396f", "rot": 28, "bit": 29, "mask": 2, "width": 1},
|
||||
{"i": 24, "op": "mul", "dst": 5, "src": 0, "src2": 7, "imm": "0x376d0223", "imm2": "0xe1c2169a", "rot": 4, "bit": 16, "mask": 16, "width": 1},
|
||||
{"i": 25, "op": "rotl", "dst": 5, "src": 7, "src2": 3, "imm": "0x78ad8c60", "imm2": "0x6f5b77d5", "rot": 19, "bit": 11, "mask": 16, "width": 1},
|
||||
{"i": 26, "op": "shfl", "dst": 7, "src": 6, "src2": 5, "imm": "0x93915b9f", "imm2": "0x1e61fb6b", "rot": 28, "bit": 23, "mask": 2, "width": 1},
|
||||
{"i": 27, "op": "xor", "dst": 0, "src": 5, "src2": 7, "imm": "0x6378fe15", "imm2": "0x66c78f42", "rot": 12, "bit": 31, "mask": 8, "width": 1},
|
||||
{"i": 28, "op": "xor", "dst": 0, "src": 4, "src2": 7, "imm": "0x20a57fda", "imm2": "0x088c848e", "rot": 16, "bit": 13, "mask": 4, "width": 1},
|
||||
{"i": 29, "op": "sub", "dst": 3, "src": 0, "src2": 2, "imm": "0x49d95fd5", "imm2": "0x1a5f946a", "rot": 6, "bit": 12, "mask": 1, "width": 1},
|
||||
{"i": 30, "op": "mul", "dst": 5, "src": 1, "src2": 7, "imm": "0x0a816217", "imm2": "0x405c4f73", "rot": 13, "bit": 27, "mask": 4, "width": 1},
|
||||
{"i": 31, "op": "load", "dst": 7, "src": 2, "src2": 2, "imm": "0x09ed045e", "imm2": "0xd69c4715", "rot": 5, "bit": 9, "mask": 2, "width": 1},
|
||||
{"i": 32, "op": "load", "dst": 1, "src": 0, "src2": 6, "imm": "0xeb79ea49", "imm2": "0xcc587f5a", "rot": 6, "bit": 8, "mask": 16, "width": 1},
|
||||
{"i": 33, "op": "xor", "dst": 5, "src": 6, "src2": 1, "imm": "0x3027401e", "imm2": "0x5f20c27e", "rot": 18, "bit": 9, "mask": 2, "width": 1},
|
||||
{"i": 34, "op": "load", "dst": 5, "src": 1, "src2": 3, "imm": "0x0e1cab07", "imm2": "0x09356c5b", "rot": 19, "bit": 31, "mask": 1, "width": 1},
|
||||
{"i": 35, "op": "mulhi", "dst": 0, "src": 5, "src2": 2, "imm": "0x90e31357", "imm2": "0xabd32484", "rot": 26, "bit": 5, "mask": 8, "width": 1},
|
||||
{"i": 36, "op": "shfl", "dst": 5, "src": 2, "src2": 5, "imm": "0xee9a955f", "imm2": "0x31b3faed", "rot": 8, "bit": 24, "mask": 4, "width": 1},
|
||||
{"i": 37, "op": "load", "dst": 7, "src": 0, "src2": 7, "imm": "0x3ba2f832", "imm2": "0x1160dcd3", "rot": 4, "bit": 29, "mask": 1, "width": 1},
|
||||
{"i": 38, "op": "add", "dst": 3, "src": 1, "src2": 7, "imm": "0x75ba2fad", "imm2": "0x230c005c", "rot": 4, "bit": 27, "mask": 1, "width": 1},
|
||||
{"i": 39, "op": "shfl", "dst": 1, "src": 5, "src2": 3, "imm": "0xdbf37e75", "imm2": "0xb5ac1969", "rot": 30, "bit": 13, "mask": 4, "width": 1},
|
||||
{"i": 40, "op": "xor", "dst": 2, "src": 5, "src2": 5, "imm": "0x47f136c5", "imm2": "0x06ce9153", "rot": 19, "bit": 10, "mask": 2, "width": 1},
|
||||
{"i": 41, "op": "mad", "dst": 3, "src": 6, "src2": 3, "imm": "0xce13eff8", "imm2": "0x04cc1d55", "rot": 3, "bit": 1, "mask": 4, "width": 1},
|
||||
{"i": 42, "op": "sub", "dst": 6, "src": 7, "src2": 1, "imm": "0x6a65ab71", "imm2": "0x8fbc1bcd", "rot": 4, "bit": 1, "mask": 8, "width": 1},
|
||||
{"i": 43, "op": "xor", "dst": 7, "src": 0, "src2": 7, "imm": "0xdaeb4928", "imm2": "0xc0423027", "rot": 24, "bit": 11, "mask": 8, "width": 1},
|
||||
{"i": 44, "op": "load", "dst": 1, "src": 7, "src2": 7, "imm": "0x778f01c9", "imm2": "0x28cedcea", "rot": 12, "bit": 4, "mask": 16, "width": 1},
|
||||
{"i": 45, "op": "mul", "dst": 2, "src": 3, "src2": 2, "imm": "0xf4264f1b", "imm2": "0x0f627d56", "rot": 5, "bit": 28, "mask": 8, "width": 1},
|
||||
{"i": 46, "op": "mulhi", "dst": 1, "src": 5, "src2": 1, "imm": "0xffb2147a", "imm2": "0xccde9b05", "rot": 13, "bit": 9, "mask": 2, "width": 1},
|
||||
{"i": 47, "op": "sub", "dst": 4, "src": 3, "src2": 5, "imm": "0x73b36234", "imm2": "0x3f5d5997", "rot": 7, "bit": 18, "mask": 2, "width": 1},
|
||||
{"i": 48, "op": "rotr", "dst": 2, "src": 6, "src2": 3, "imm": "0x3a4d9aa9", "imm2": "0x212bec7b", "rot": 4, "bit": 29, "mask": 16, "width": 1},
|
||||
{"i": 49, "op": "load", "dst": 3, "src": 5, "src2": 2, "imm": "0x626f11df", "imm2": "0x56cd5bfd", "rot": 7, "bit": 1, "mask": 1, "width": 1},
|
||||
{"i": 50, "op": "add", "dst": 1, "src": 5, "src2": 6, "imm": "0x81b8bc2c", "imm2": "0x1907970c", "rot": 28, "bit": 7, "mask": 4, "width": 1},
|
||||
{"i": 51, "op": "mul", "dst": 0, "src": 2, "src2": 2, "imm": "0xa8848b30", "imm2": "0xef6ac348", "rot": 9, "bit": 15, "mask": 8, "width": 1},
|
||||
{"i": 52, "op": "add", "dst": 0, "src": 2, "src2": 0, "imm": "0x4f92b968", "imm2": "0x699fd448", "rot": 22, "bit": 6, "mask": 4, "width": 1},
|
||||
{"i": 53, "op": "add", "dst": 1, "src": 0, "src2": 2, "imm": "0x2bb965af", "imm2": "0x77b1520d", "rot": 2, "bit": 12, "mask": 8, "width": 1},
|
||||
{"i": 54, "op": "rotl", "dst": 7, "src": 1, "src2": 0, "imm": "0x553e678b", "imm2": "0x3cc8eae0", "rot": 14, "bit": 20, "mask": 2, "width": 1},
|
||||
{"i": 55, "op": "add", "dst": 3, "src": 7, "src2": 2, "imm": "0x7b0fe07a", "imm2": "0xa54c55a0", "rot": 10, "bit": 1, "mask": 1, "width": 1},
|
||||
{"i": 56, "op": "load", "dst": 6, "src": 7, "src2": 4, "imm": "0x01eba9aa", "imm2": "0x2758c0f7", "rot": 14, "bit": 15, "mask": 4, "width": 1},
|
||||
{"i": 57, "op": "rotr", "dst": 1, "src": 5, "src2": 2, "imm": "0x1f5267b3", "imm2": "0x236f5a27", "rot": 2, "bit": 31, "mask": 16, "width": 1},
|
||||
{"i": 58, "op": "load", "dst": 5, "src": 4, "src2": 3, "imm": "0xa9954a9b", "imm2": "0x6a54d4e8", "rot": 11, "bit": 10, "mask": 16, "width": 1},
|
||||
{"i": 59, "op": "load", "dst": 6, "src": 2, "src2": 4, "imm": "0x9923ff88", "imm2": "0x9357254e", "rot": 16, "bit": 1, "mask": 16, "width": 1},
|
||||
{"i": 60, "op": "mad", "dst": 3, "src": 5, "src2": 0, "imm": "0xc0cc51a6", "imm2": "0x3fd7701b", "rot": 20, "bit": 1, "mask": 4, "width": 1},
|
||||
{"i": 61, "op": "add", "dst": 5, "src": 7, "src2": 7, "imm": "0xaf9dd72d", "imm2": "0xad7493e7", "rot": 7, "bit": 31, "mask": 16, "width": 1},
|
||||
{"i": 62, "op": "add", "dst": 4, "src": 6, "src2": 2, "imm": "0x89841d87", "imm2": "0x1e07c3d9", "rot": 6, "bit": 27, "mask": 1, "width": 1},
|
||||
{"i": 63, "op": "rotl", "dst": 5, "src": 4, "src2": 2, "imm": "0xae210f8d", "imm2": "0x8e499ba4", "rot": 19, "bit": 9, "mask": 1, "width": 1}
|
||||
]
|
||||
}
|
||||
109
proto-cuda/packs-ca2-mixer/mx4-genesis/program.metal
Normal file
109
proto-cuda/packs-ca2-mixer/mx4-genesis/program.metal
Normal file
|
|
@ -0,0 +1,109 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
|
||||
#define MASK 0x0fffffffu
|
||||
constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u };
|
||||
|
||||
inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
|
||||
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
|
||||
device ulong* out [[buffer(1)]],
|
||||
constant uint& baseNonce [[buffer(2)]],
|
||||
uint gid [[thread_position_in_grid]]) {
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint x = nonce ^ SEEDW[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ SEEDW[1]; }
|
||||
{ uint x = nonce ^ SEEDW[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ SEEDW[2]; }
|
||||
{ uint x = nonce ^ SEEDW[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ SEEDW[3]; }
|
||||
{ uint x = nonce ^ SEEDW[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ SEEDW[4]; }
|
||||
{ uint x = nonce ^ SEEDW[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ SEEDW[5]; }
|
||||
{ uint x = nonce ^ SEEDW[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ SEEDW[6]; }
|
||||
{ uint x = nonce ^ SEEDW[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ SEEDW[7]; }
|
||||
{ uint x = nonce ^ SEEDW[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ SEEDW[0]; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0
|
||||
r2 = r1 * r1 + r2; // 1
|
||||
r2 = r3 * r2 + r2; // 2
|
||||
r3 = r3 ^ r5; // 3
|
||||
r7 = r7 ^ dataset[r2 & MASK]; // 4
|
||||
r5 = r5 ^ dataset[r7 & MASK]; // 5
|
||||
r1 = r1 ^ simd_shuffle_xor(r4, (ushort)8); // 6
|
||||
r7 = r7 ^ simd_shuffle_xor(r3, (ushort)8); // 7
|
||||
r1 = mulhi(r1, r5); // 8
|
||||
r6 = rotr_var(r6, r3); // 9
|
||||
r3 = r3 | r4; // 10
|
||||
r4 = r4 ^ dataset[r3 & MASK]; // 11
|
||||
r0 = mulhi(r0, r4); // 12
|
||||
r5 = r5 + r1 + select(0xc7934706u, 0xd3177981u, ((sel >> 30u) & 1u) != 0u); // 13
|
||||
r0 = r0 ^ dataset[r4 & MASK]; // 14
|
||||
r2 = r2 - r4; // 15
|
||||
r2 = r2 ^ dataset[r0 & MASK]; // 16
|
||||
r7 = r7 ^ dataset[r2 & MASK]; // 17
|
||||
r7 = r7 ^ simd_shuffle_xor(r3, (ushort)4); // 18
|
||||
r5 = r5 * r0; // 19
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 20
|
||||
r2 = r2 ^ simd_shuffle_xor(r4, (ushort)16); // 21
|
||||
r6 = mulhi(r6, r2); // 22
|
||||
r6 = r6 ^ dataset[r1 & MASK]; // 23
|
||||
r5 = r5 * r0; // 24
|
||||
r5 = rotl_imm(r5, 19u); // 25
|
||||
r7 = r7 ^ simd_shuffle_xor(r6, (ushort)2); // 26
|
||||
r0 = r0 ^ r5; // 27
|
||||
r0 = r0 ^ r4; // 28
|
||||
r3 = r3 - r0; // 29
|
||||
r5 = r5 * r1; // 30
|
||||
r7 = r7 ^ dataset[r2 & MASK]; // 31
|
||||
r1 = r1 ^ dataset[r0 & MASK]; // 32
|
||||
r5 = r5 ^ r6; // 33
|
||||
r5 = r5 ^ dataset[r1 & MASK]; // 34
|
||||
r0 = mulhi(r0, r5); // 35
|
||||
r5 = r5 ^ simd_shuffle_xor(r2, (ushort)4); // 36
|
||||
r7 = r7 ^ dataset[r0 & MASK]; // 37
|
||||
r3 = r3 + r1 + select(0x75ba2fadu, 0x230c005cu, ((sel >> 27u) & 1u) != 0u); // 38
|
||||
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)4); // 39
|
||||
r2 = r2 ^ r5; // 40
|
||||
r3 = r6 * r3 + r3; // 41
|
||||
r6 = r6 - r7; // 42
|
||||
r7 = r7 ^ r0; // 43
|
||||
r1 = r1 ^ dataset[r7 & MASK]; // 44
|
||||
r2 = r2 * r3; // 45
|
||||
r1 = mulhi(r1, r5); // 46
|
||||
r4 = r4 - r3; // 47
|
||||
r2 = rotr_var(r2, r6); // 48
|
||||
r3 = r3 ^ dataset[r5 & MASK]; // 49
|
||||
r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 50
|
||||
r0 = r0 * r2; // 51
|
||||
r0 = r0 + r2 + select(0x4f92b968u, 0x699fd448u, ((sel >> 6u) & 1u) != 0u); // 52
|
||||
r1 = r1 + r0 + select(0x2bb965afu, 0x77b1520du, ((sel >> 12u) & 1u) != 0u); // 53
|
||||
r7 = rotl_imm(r7, 14u); // 54
|
||||
r3 = r3 + r7 + select(0x7b0fe07au, 0xa54c55a0u, ((sel >> 1u) & 1u) != 0u); // 55
|
||||
r6 = r6 ^ dataset[r7 & MASK]; // 56
|
||||
r1 = rotr_var(r1, r5); // 57
|
||||
r5 = r5 ^ dataset[r4 & MASK]; // 58
|
||||
r6 = r6 ^ dataset[r2 & MASK]; // 59
|
||||
r3 = r5 * r0 + r3; // 60
|
||||
r5 = r5 + r7 + select(0xaf9dd72du, 0xad7493e7u, ((sel >> 31u) & 1u) != 0u); // 61
|
||||
r4 = r4 + r6 + select(0x89841d87u, 0x1e07c3d9u, ((sel >> 27u) & 1u) != 0u); // 62
|
||||
r5 = rotl_imm(r5, 19u); // 63
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
111
proto-cuda/packs-ca2-mixer/mx4-genesis/program_bound.metal
Normal file
111
proto-cuda/packs-ca2-mixer/mx4-genesis/program_bound.metal
Normal file
|
|
@ -0,0 +1,111 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
|
||||
#define MASK 0x0fffffffu
|
||||
constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u };
|
||||
|
||||
inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
|
||||
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW.
|
||||
kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
|
||||
device ulong* out [[buffer(1)]],
|
||||
constant uint& baseNonce [[buffer(2)]],
|
||||
constant uint* initw [[buffer(3)]],
|
||||
uint gid [[thread_position_in_grid]]) {
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint x = nonce ^ initw[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ initw[1]; }
|
||||
{ uint x = nonce ^ initw[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ initw[2]; }
|
||||
{ uint x = nonce ^ initw[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ initw[3]; }
|
||||
{ uint x = nonce ^ initw[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ initw[4]; }
|
||||
{ uint x = nonce ^ initw[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ initw[5]; }
|
||||
{ uint x = nonce ^ initw[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ initw[6]; }
|
||||
{ uint x = nonce ^ initw[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ initw[7]; }
|
||||
{ uint x = nonce ^ initw[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ initw[0]; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0
|
||||
r2 = r1 * r1 + r2; // 1
|
||||
r2 = r3 * r2 + r2; // 2
|
||||
r3 = r3 ^ r5; // 3
|
||||
r7 = r7 ^ dataset[r2 & MASK]; // 4
|
||||
r5 = r5 ^ dataset[r7 & MASK]; // 5
|
||||
r1 = r1 ^ simd_shuffle_xor(r4, (ushort)8); // 6
|
||||
r7 = r7 ^ simd_shuffle_xor(r3, (ushort)8); // 7
|
||||
r1 = mulhi(r1, r5); // 8
|
||||
r6 = rotr_var(r6, r3); // 9
|
||||
r3 = r3 | r4; // 10
|
||||
r4 = r4 ^ dataset[r3 & MASK]; // 11
|
||||
r0 = mulhi(r0, r4); // 12
|
||||
r5 = r5 + r1 + select(0xc7934706u, 0xd3177981u, ((sel >> 30u) & 1u) != 0u); // 13
|
||||
r0 = r0 ^ dataset[r4 & MASK]; // 14
|
||||
r2 = r2 - r4; // 15
|
||||
r2 = r2 ^ dataset[r0 & MASK]; // 16
|
||||
r7 = r7 ^ dataset[r2 & MASK]; // 17
|
||||
r7 = r7 ^ simd_shuffle_xor(r3, (ushort)4); // 18
|
||||
r5 = r5 * r0; // 19
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 20
|
||||
r2 = r2 ^ simd_shuffle_xor(r4, (ushort)16); // 21
|
||||
r6 = mulhi(r6, r2); // 22
|
||||
r6 = r6 ^ dataset[r1 & MASK]; // 23
|
||||
r5 = r5 * r0; // 24
|
||||
r5 = rotl_imm(r5, 19u); // 25
|
||||
r7 = r7 ^ simd_shuffle_xor(r6, (ushort)2); // 26
|
||||
r0 = r0 ^ r5; // 27
|
||||
r0 = r0 ^ r4; // 28
|
||||
r3 = r3 - r0; // 29
|
||||
r5 = r5 * r1; // 30
|
||||
r7 = r7 ^ dataset[r2 & MASK]; // 31
|
||||
r1 = r1 ^ dataset[r0 & MASK]; // 32
|
||||
r5 = r5 ^ r6; // 33
|
||||
r5 = r5 ^ dataset[r1 & MASK]; // 34
|
||||
r0 = mulhi(r0, r5); // 35
|
||||
r5 = r5 ^ simd_shuffle_xor(r2, (ushort)4); // 36
|
||||
r7 = r7 ^ dataset[r0 & MASK]; // 37
|
||||
r3 = r3 + r1 + select(0x75ba2fadu, 0x230c005cu, ((sel >> 27u) & 1u) != 0u); // 38
|
||||
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)4); // 39
|
||||
r2 = r2 ^ r5; // 40
|
||||
r3 = r6 * r3 + r3; // 41
|
||||
r6 = r6 - r7; // 42
|
||||
r7 = r7 ^ r0; // 43
|
||||
r1 = r1 ^ dataset[r7 & MASK]; // 44
|
||||
r2 = r2 * r3; // 45
|
||||
r1 = mulhi(r1, r5); // 46
|
||||
r4 = r4 - r3; // 47
|
||||
r2 = rotr_var(r2, r6); // 48
|
||||
r3 = r3 ^ dataset[r5 & MASK]; // 49
|
||||
r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 50
|
||||
r0 = r0 * r2; // 51
|
||||
r0 = r0 + r2 + select(0x4f92b968u, 0x699fd448u, ((sel >> 6u) & 1u) != 0u); // 52
|
||||
r1 = r1 + r0 + select(0x2bb965afu, 0x77b1520du, ((sel >> 12u) & 1u) != 0u); // 53
|
||||
r7 = rotl_imm(r7, 14u); // 54
|
||||
r3 = r3 + r7 + select(0x7b0fe07au, 0xa54c55a0u, ((sel >> 1u) & 1u) != 0u); // 55
|
||||
r6 = r6 ^ dataset[r7 & MASK]; // 56
|
||||
r1 = rotr_var(r1, r5); // 57
|
||||
r5 = r5 ^ dataset[r4 & MASK]; // 58
|
||||
r6 = r6 ^ dataset[r2 & MASK]; // 59
|
||||
r3 = r5 * r0 + r3; // 60
|
||||
r5 = r5 + r7 + select(0xaf9dd72du, 0xad7493e7u, ((sel >> 31u) & 1u) != 0u); // 61
|
||||
r4 = r4 + r6 + select(0x89841d87u, 0x1e07c3d9u, ((sel >> 27u) & 1u) != 0u); // 62
|
||||
r5 = rotl_imm(r5, 19u); // 63
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
57
proto-cuda/packs-ca2-mixer/mx4-genesis/vectors.h
Normal file
57
proto-cuda/packs-ca2-mixer/mx4-genesis/vectors.h
Normal file
|
|
@ -0,0 +1,57 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Expected outputs: igneum-pow (Rust) CPU interpreter, generator v3, memory-hard dataset
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
|
||||
#define IGNEUM_VEC_WARPS 3
|
||||
static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u };
|
||||
static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {
|
||||
{ // base nonce 0
|
||||
0x63acd2d273f475baull, 0xe929c78b34b80d4bull, 0x0b1011cb19982558ull, 0x1457a0df5497aa11ull, 0x957d0f3bb71d98fbull, 0xac16901e6e6f6057ull, 0x8ea1c6279f4b177aull, 0xf28146e60bd08ba9ull,
|
||||
0xfe5b8cfe87f8e65bull, 0x49f87240566ace62ull, 0x6ef6d6b7bdea8e41ull, 0x46d9c0dc29a97b9cull, 0x111fe30128db9398ull, 0x66dc39084f0946d4ull, 0x8ee11bdfd35fecf2ull, 0x2861fcfc75db6677ull,
|
||||
0x31c7667d4bde8556ull, 0xc5989c48858b4ce0ull, 0x276395e734a9d30dull, 0x84217b41e91368ffull, 0x3604861e34d9f697ull, 0x9f51d8ee16bf3639ull, 0xc89e47bafa84401cull, 0x7ae78c1f10b70e19ull,
|
||||
0x0b8c947157a29a48ull, 0xd67192e8cfb43842ull, 0x05a4c6d182c8c675ull, 0x188e2661f3263f2eull, 0xa2df24238f7fea2eull, 0xed69ea7e13ad3a48ull, 0x2d44ae509bab91b8ull, 0xadad61931ea4fb70ull
|
||||
},
|
||||
{ // base nonce 4096
|
||||
0xedd508ac57e5699aull, 0x4eefd56d526cdaebull, 0x6c6407d53b9ade77ull, 0x8aaddb277f4d7ea4ull, 0xa1268b3328ec5c7eull, 0x3046863cc08f10b6ull, 0x47fe3bb47491f11cull, 0x9892e81a319f05b0ull,
|
||||
0x2c977f8db84667f0ull, 0x3751e42afdf7d37full, 0xf7c4efcd768d3da5ull, 0x79b00b8156bd1981ull, 0x5983f862fb97ee2full, 0xb34e9a2d9e810a26ull, 0x39a71549ca7948b8ull, 0x37c86a24538629b9ull,
|
||||
0x4e519f3ad2615633ull, 0xe1393fce43030a86ull, 0x46802bbd8f7913edull, 0x7116a52e1e51c9a0ull, 0x2b966fdbbc23bf85ull, 0x718ec9de9c49f734ull, 0x6d770291c4b420d0ull, 0x9779a328a8bea8b1ull,
|
||||
0xc5af3efc143bd2c2ull, 0x0931d55121d12a16ull, 0x8f985babebd420caull, 0x3a4ce5974e9a6e42ull, 0xd2d8644ee07ac277ull, 0x3633f2b2d5ae6edeull, 0xe84b2890e1648233ull, 0x8892f8604733b1e0ull
|
||||
},
|
||||
{ // base nonce 1000000
|
||||
0x8b3183778a49f59cull, 0x1831b72a8797e895ull, 0x6288de49326df031ull, 0xdf8d589babc64daaull, 0x3110afa71db82c11ull, 0x3ca14a14afa49b89ull, 0xa14b46a71bcfb2e1ull, 0xf148f5ef363e111bull,
|
||||
0xc62d5fb72f53e8adull, 0x8f218b5e9ee8ac70ull, 0xc9683c2cebe6f13eull, 0xc0539e433297efc5ull, 0xf72639c8336113f7ull, 0xedfee376a9fde5d7ull, 0x4d2f3bb8b7c96dc5ull, 0x5a2c68ee48a669d4ull,
|
||||
0x0e50f07d454f804full, 0x4c53ea83ac05243eull, 0xd98d409876d1843dull, 0xfec5272f4d75b773ull, 0xbdc1031f6698693aull, 0xb4bc7d47d1e9aabeull, 0x1c7524d48c985b71ull, 0x65ae6f47ef1a0a46ull,
|
||||
0x15137619fbf9f945ull, 0x7a73e6324619d0a4ull, 0xcd76884a72b4e50full, 0xcc58c48cf2fabeb9ull, 0x8c7e0c672424b507ull, 0x30dc7c5a2186db58ull, 0xb5322548a60418d6ull, 0x75eae55eba53a506ull
|
||||
}
|
||||
};
|
||||
|
||||
// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (268435455).
|
||||
static const uint32_t IGNEUM_DS_HEAD[16] = {
|
||||
0x61ff2180u, 0x0d4c7e6cu, 0x2177d443u, 0x60df9025u, 0xcf8b2e10u, 0x63675bfbu, 0x25289e58u, 0x9c45dc42u,
|
||||
0x2d271c54u, 0x9652369bu, 0x2dd77508u, 0x5921392cu, 0x3afa60eeu, 0xc640ad68u, 0xf2bb56ffu, 0xcfa46438u
|
||||
};
|
||||
static const uint32_t IGNEUM_DS_LAST_INDEX = 268435455u;
|
||||
static const uint32_t IGNEUM_DS_LAST = 0x5020180eu;
|
||||
// 64 sampled dataset words (index, value) computed on the Mac.
|
||||
#define IGNEUM_DS_SAMPLES 64
|
||||
static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = {
|
||||
59471966u, 217795994u, 208353206u, 42483309u, 172547758u, 148076330u, 183853158u, 214389424u, 267488061u, 169781097u, 184093494u, 153880993u, 84977930u, 46426879u, 3093825u, 225364072u, 44593546u, 260713159u, 168250303u, 52384140u, 223401610u, 45554030u, 95410555u, 175039924u, 79171087u, 267580473u, 24168642u, 37981670u, 171551130u, 195559979u, 204611762u, 140997658u, 138925853u, 86637313u, 20736778u, 219665210u, 160430336u, 264654675u, 8013395u, 228945585u, 213884386u, 104419827u, 44185464u, 142737231u, 99284897u, 132475900u, 61861762u, 132056166u, 262388043u, 91878046u, 117353561u, 124768597u, 71352993u, 190698941u, 46055428u, 55281366u, 165145231u, 106810753u, 171985651u, 232085256u, 159510492u, 40072060u, 209107596u, 39023794u
|
||||
};
|
||||
static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = {
|
||||
0xde85726du, 0x7cfc31c7u, 0xd3cd5289u, 0x6ebbeef1u, 0x9858413bu, 0xe786f141u, 0x64f13833u, 0xca229d04u, 0xd6e9303eu, 0xac7c7b85u, 0x1c5098bcu, 0xfa0f23c8u, 0xb6fad4fdu, 0xd17f0c5fu, 0xd6b7abc9u, 0xd5ee83ccu, 0x53306f26u, 0x3319d1b6u, 0x3513a069u, 0x9f07da0au, 0xef99d8efu, 0x0ba1c662u, 0x3f18c586u, 0x80727509u, 0xb68425f9u, 0x8a1f253au, 0x40cf9312u, 0x84e16868u, 0x00d870dfu, 0x3a61e8a0u, 0x53f22acau, 0xfbdd07c8u, 0x07932a9du, 0xb22c4cbcu, 0x711e42d3u, 0xa0176126u, 0x6d8823aau, 0x20a0e0b0u, 0x41ba7beau, 0xcaafcc88u, 0x7fdb6389u, 0x06759acau, 0x237fa289u, 0x2f4e42d3u, 0x55c22a13u, 0x005fc019u, 0xf6030945u, 0x17795f6bu, 0x0c816b99u, 0x62ea510du, 0x838557cbu, 0x3546a21fu, 0x804664f0u, 0x266f4dc5u, 0xb7d87ff0u, 0x17ee0753u, 0x30e21993u, 0x464e7559u, 0xdff3b3ceu, 0x3af3bf66u, 0x5552e478u, 0xa42607edu, 0x970cb406u, 0xf44efe28u
|
||||
};
|
||||
// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^26 words.
|
||||
static const uint32_t IGNEUM_CACHE_HEAD[16] = {
|
||||
0x355a86d2u, 0x7957db1cu, 0xd21772afu, 0x6fc1e09bu, 0xd55ce61du, 0x6e6a278bu, 0xd3f543ceu, 0x223d8e82u,
|
||||
0x143ab337u, 0x2e9f05bdu, 0x2eb389bfu, 0x0c6e449eu, 0x5cfa4222u, 0xba6560feu, 0x8e3e1aa4u, 0xdbcc1d53u
|
||||
};
|
||||
static const uint32_t IGNEUM_CACHE_LAST[16] = {
|
||||
0x41190d91u, 0xbd277957u, 0x22ddbb49u, 0x6986f207u, 0xdf69a4d6u, 0x26401a3au, 0x818230fbu, 0xc417122du,
|
||||
0x3597b211u, 0xb553ce55u, 0xcf39cc0du, 0x3b7fc43au, 0x3fd43b00u, 0x67e1c80eu, 0xffa7ea7du, 0xca2960abu
|
||||
};
|
||||
static const uint64_t IGNEUM_CACHE_FNV64 = 0x48c4f5bf24166b2eull;
|
||||
36
proto-cuda/packs-ca2-mixer/mx4-genesis/vectors.json
Normal file
36
proto-cuda/packs-ca2-mixer/mx4-genesis/vectors.json
Normal file
|
|
@ -0,0 +1,36 @@
|
|||
{
|
||||
"seed": "igneum-genesis",
|
||||
"day": "2026-10-03",
|
||||
"dataset_mode": "memory-hard",
|
||||
"dataset_log2_words": 28,
|
||||
"mask": "0x0fffffff",
|
||||
"lanes": 32,
|
||||
"source": "igneum-pow (Rust) CPU interpreter, generator v3, memory-hard dataset",
|
||||
"warps": [
|
||||
{"base_nonce": 0, "expected": [
|
||||
"0x63acd2d273f475ba", "0xe929c78b34b80d4b", "0x0b1011cb19982558", "0x1457a0df5497aa11", "0x957d0f3bb71d98fb", "0xac16901e6e6f6057", "0x8ea1c6279f4b177a", "0xf28146e60bd08ba9",
|
||||
"0xfe5b8cfe87f8e65b", "0x49f87240566ace62", "0x6ef6d6b7bdea8e41", "0x46d9c0dc29a97b9c", "0x111fe30128db9398", "0x66dc39084f0946d4", "0x8ee11bdfd35fecf2", "0x2861fcfc75db6677",
|
||||
"0x31c7667d4bde8556", "0xc5989c48858b4ce0", "0x276395e734a9d30d", "0x84217b41e91368ff", "0x3604861e34d9f697", "0x9f51d8ee16bf3639", "0xc89e47bafa84401c", "0x7ae78c1f10b70e19",
|
||||
"0x0b8c947157a29a48", "0xd67192e8cfb43842", "0x05a4c6d182c8c675", "0x188e2661f3263f2e", "0xa2df24238f7fea2e", "0xed69ea7e13ad3a48", "0x2d44ae509bab91b8", "0xadad61931ea4fb70"
|
||||
]},
|
||||
{"base_nonce": 4096, "expected": [
|
||||
"0xedd508ac57e5699a", "0x4eefd56d526cdaeb", "0x6c6407d53b9ade77", "0x8aaddb277f4d7ea4", "0xa1268b3328ec5c7e", "0x3046863cc08f10b6", "0x47fe3bb47491f11c", "0x9892e81a319f05b0",
|
||||
"0x2c977f8db84667f0", "0x3751e42afdf7d37f", "0xf7c4efcd768d3da5", "0x79b00b8156bd1981", "0x5983f862fb97ee2f", "0xb34e9a2d9e810a26", "0x39a71549ca7948b8", "0x37c86a24538629b9",
|
||||
"0x4e519f3ad2615633", "0xe1393fce43030a86", "0x46802bbd8f7913ed", "0x7116a52e1e51c9a0", "0x2b966fdbbc23bf85", "0x718ec9de9c49f734", "0x6d770291c4b420d0", "0x9779a328a8bea8b1",
|
||||
"0xc5af3efc143bd2c2", "0x0931d55121d12a16", "0x8f985babebd420ca", "0x3a4ce5974e9a6e42", "0xd2d8644ee07ac277", "0x3633f2b2d5ae6ede", "0xe84b2890e1648233", "0x8892f8604733b1e0"
|
||||
]},
|
||||
{"base_nonce": 1000000, "expected": [
|
||||
"0x8b3183778a49f59c", "0x1831b72a8797e895", "0x6288de49326df031", "0xdf8d589babc64daa", "0x3110afa71db82c11", "0x3ca14a14afa49b89", "0xa14b46a71bcfb2e1", "0xf148f5ef363e111b",
|
||||
"0xc62d5fb72f53e8ad", "0x8f218b5e9ee8ac70", "0xc9683c2cebe6f13e", "0xc0539e433297efc5", "0xf72639c8336113f7", "0xedfee376a9fde5d7", "0x4d2f3bb8b7c96dc5", "0x5a2c68ee48a669d4",
|
||||
"0x0e50f07d454f804f", "0x4c53ea83ac05243e", "0xd98d409876d1843d", "0xfec5272f4d75b773", "0xbdc1031f6698693a", "0xb4bc7d47d1e9aabe", "0x1c7524d48c985b71", "0x65ae6f47ef1a0a46",
|
||||
"0x15137619fbf9f945", "0x7a73e6324619d0a4", "0xcd76884a72b4e50f", "0xcc58c48cf2fabeb9", "0x8c7e0c672424b507", "0x30dc7c5a2186db58", "0xb5322548a60418d6", "0x75eae55eba53a506"
|
||||
]}
|
||||
],
|
||||
"dataset_head": ["0x61ff2180", "0x0d4c7e6c", "0x2177d443", "0x60df9025", "0xcf8b2e10", "0x63675bfb", "0x25289e58", "0x9c45dc42", "0x2d271c54", "0x9652369b", "0x2dd77508", "0x5921392c", "0x3afa60ee", "0xc640ad68", "0xf2bb56ff", "0xcfa46438"],
|
||||
"dataset_last_index": 268435455,
|
||||
"dataset_last": "0x5020180e",
|
||||
"dataset_samples": [{"index": 59471966, "value": "0xde85726d"}, {"index": 217795994, "value": "0x7cfc31c7"}, {"index": 208353206, "value": "0xd3cd5289"}, {"index": 42483309, "value": "0x6ebbeef1"}, {"index": 172547758, "value": "0x9858413b"}, {"index": 148076330, "value": "0xe786f141"}, {"index": 183853158, "value": "0x64f13833"}, {"index": 214389424, "value": "0xca229d04"}, {"index": 267488061, "value": "0xd6e9303e"}, {"index": 169781097, "value": "0xac7c7b85"}, {"index": 184093494, "value": "0x1c5098bc"}, {"index": 153880993, "value": "0xfa0f23c8"}, {"index": 84977930, "value": "0xb6fad4fd"}, {"index": 46426879, "value": "0xd17f0c5f"}, {"index": 3093825, "value": "0xd6b7abc9"}, {"index": 225364072, "value": "0xd5ee83cc"}, {"index": 44593546, "value": "0x53306f26"}, {"index": 260713159, "value": "0x3319d1b6"}, {"index": 168250303, "value": "0x3513a069"}, {"index": 52384140, "value": "0x9f07da0a"}, {"index": 223401610, "value": "0xef99d8ef"}, {"index": 45554030, "value": "0x0ba1c662"}, {"index": 95410555, "value": "0x3f18c586"}, {"index": 175039924, "value": "0x80727509"}, {"index": 79171087, "value": "0xb68425f9"}, {"index": 267580473, "value": "0x8a1f253a"}, {"index": 24168642, "value": "0x40cf9312"}, {"index": 37981670, "value": "0x84e16868"}, {"index": 171551130, "value": "0x00d870df"}, {"index": 195559979, "value": "0x3a61e8a0"}, {"index": 204611762, "value": "0x53f22aca"}, {"index": 140997658, "value": "0xfbdd07c8"}, {"index": 138925853, "value": "0x07932a9d"}, {"index": 86637313, "value": "0xb22c4cbc"}, {"index": 20736778, "value": "0x711e42d3"}, {"index": 219665210, "value": "0xa0176126"}, {"index": 160430336, "value": "0x6d8823aa"}, {"index": 264654675, "value": "0x20a0e0b0"}, {"index": 8013395, "value": "0x41ba7bea"}, {"index": 228945585, "value": "0xcaafcc88"}, {"index": 213884386, "value": "0x7fdb6389"}, {"index": 104419827, "value": "0x06759aca"}, {"index": 44185464, "value": "0x237fa289"}, {"index": 142737231, "value": "0x2f4e42d3"}, {"index": 99284897, "value": "0x55c22a13"}, {"index": 132475900, "value": "0x005fc019"}, {"index": 61861762, "value": "0xf6030945"}, {"index": 132056166, "value": "0x17795f6b"}, {"index": 262388043, "value": "0x0c816b99"}, {"index": 91878046, "value": "0x62ea510d"}, {"index": 117353561, "value": "0x838557cb"}, {"index": 124768597, "value": "0x3546a21f"}, {"index": 71352993, "value": "0x804664f0"}, {"index": 190698941, "value": "0x266f4dc5"}, {"index": 46055428, "value": "0xb7d87ff0"}, {"index": 55281366, "value": "0x17ee0753"}, {"index": 165145231, "value": "0x30e21993"}, {"index": 106810753, "value": "0x464e7559"}, {"index": 171985651, "value": "0xdff3b3ce"}, {"index": 232085256, "value": "0x3af3bf66"}, {"index": 159510492, "value": "0x5552e478"}, {"index": 40072060, "value": "0xa42607ed"}, {"index": 209107596, "value": "0x970cb406"}, {"index": 39023794, "value": "0xf44efe28"}],
|
||||
"cache_head": ["0x355a86d2", "0x7957db1c", "0xd21772af", "0x6fc1e09b", "0xd55ce61d", "0x6e6a278b", "0xd3f543ce", "0x223d8e82", "0x143ab337", "0x2e9f05bd", "0x2eb389bf", "0x0c6e449e", "0x5cfa4222", "0xba6560fe", "0x8e3e1aa4", "0xdbcc1d53"],
|
||||
"cache_last_line": ["0x41190d91", "0xbd277957", "0x22ddbb49", "0x6986f207", "0xdf69a4d6", "0x26401a3a", "0x818230fb", "0xc417122d", "0x3597b211", "0xb553ce55", "0xcf39cc0d", "0x3b7fc43a", "0x3fd43b00", "0x67e1c80e", "0xffa7ea7d", "0xca2960ab"],
|
||||
"cache_fnv1a64": "0x48c4f5bf24166b2e"
|
||||
}
|
||||
Loading…
Reference in a new issue