Merge origin/miner-perf into release-0.3.5
Conflicts resolved: state.rs keeps both the sweep fields (miner-eff) and the race fields (miner-perf); bench-log.md
keeps both appended entries; publish-manifest.sh keeps master's --override implementation (8082576, the "every
height switch" rule, --verify-only, --tries, the retrying live check) and adds miner-perf's --tuning / --no-tuning
with the carry-over of consensus.override and tuning from the current manifest. One --override case, one parser.
This commit is contained in:
commit
ca2ac6dfa8
13 changed files with 1411 additions and 48 deletions
|
|
@ -1004,7 +1004,7 @@ impl Engine {
|
|||
let seg = if self.node_starts > 1 { format!("-r{}", self.node_starts) } else { String::new() };
|
||||
let log = self.shared.runtime.log_dir.join(format!("node-{}{seg}.log", self.stamp));
|
||||
let args = self.node_args();
|
||||
match procs::spawn(Source::Node, &self.bins.node, &args, None, &log, &self.lines_tx) {
|
||||
match procs::spawn(Source::Node, &self.bins.node, &args, None, &log, &self.lines_tx, &[]) {
|
||||
Ok(p) => {
|
||||
self.shared.log(&format!("igneumd started (pid {}): {}", p.pid(), p.cmdline));
|
||||
let mut st = self.st();
|
||||
|
|
@ -1044,7 +1044,7 @@ impl Engine {
|
|||
fn start_watch(&mut self) {
|
||||
let args = vec!["watch".to_string(), "1000000000".to_string(), self.shared.runtime.rpc_url()];
|
||||
let log = self.shared.runtime.log_dir.join(format!("watch-{}.log", self.stamp));
|
||||
if let Ok(p) = procs::spawn(Source::Watch, &self.bins.miner, &args, None, &log, &self.lines_tx) {
|
||||
if let Ok(p) = procs::spawn(Source::Watch, &self.bins.miner, &args, None, &log, &self.lines_tx, &[]) {
|
||||
self.watch = Some(p);
|
||||
}
|
||||
self.watch_retry_at = Instant::now() + Duration::from_secs(4);
|
||||
|
|
@ -1168,7 +1168,9 @@ impl Engine {
|
|||
let log = self.shared.runtime.log_dir.join(format!("miner-{}-{}{seg}.log", self.miners[i].label, self.stamp));
|
||||
let cwd = if card.worker == "Metal" { None } else { Some(self.shared.runtime.app_dir.clone()) };
|
||||
self.miners[i].prepared = false;
|
||||
match procs::spawn(Source::Miner(card_idx), &self.bins.miner, &args, cwd.as_deref(), &log, &self.lines_tx) {
|
||||
// the fleet's per-card kernel tuning (from the signed manifest) reaches the GPU worker through the miner's environment
|
||||
let envs: Vec<(String, String)> = self.ota.tuning_path().map(|p| vec![("IGNEUM_TUNING_FILE".to_string(), p.display().to_string())]).unwrap_or_default();
|
||||
match procs::spawn(Source::Miner(card_idx), &self.bins.miner, &args, cwd.as_deref(), &log, &self.lines_tx, &envs) {
|
||||
Ok(p) => {
|
||||
self.shared.log(&format!("miner {} started (pid {}): {}", self.miners[i].label, p.pid(), p.cmdline));
|
||||
let mut st = self.st();
|
||||
|
|
@ -1409,7 +1411,7 @@ impl Engine {
|
|||
if self.telemetry.is_none() && now >= self.telemetry_retry_at {
|
||||
let args: Vec<String> = ["--query-gpu=index,power.draw,temperature.gpu,temperature.memory,power.limit", "--format=csv,noheader,nounits", "-l", "5"].iter().map(|s| s.to_string()).collect();
|
||||
let log = self.shared.runtime.log_dir.join(format!("gpu-{}.log", self.stamp));
|
||||
match procs::spawn(Source::Telemetry, &crate::platform::tool("nvidia-smi"), &args, None, &log, &self.lines_tx) {
|
||||
match procs::spawn(Source::Telemetry, &crate::platform::tool("nvidia-smi"), &args, None, &log, &self.lines_tx, &[]) {
|
||||
Ok(p) => self.telemetry = Some(p),
|
||||
Err(_) => self.telemetry_retry_at = now + Duration::from_secs(300),
|
||||
}
|
||||
|
|
@ -1942,6 +1944,12 @@ impl Engine {
|
|||
self.shared.log(&format!("consensus override changed ({}); the node restarts with it at a safe moment", p.display()));
|
||||
self.node_override_restart = true;
|
||||
}
|
||||
if let Some(p) = self.ota.take_tuning_change() {
|
||||
// no restart: a worker started before this reads the file at its next prepare only if it was started with
|
||||
// the path, so a miner without it is restarted at the next hour boundary by the usual path (exit 42 or
|
||||
// restart); a running worker that has the path picks the change up at the next prepare by itself
|
||||
self.shared.log(&format!("kernel tuning changed ({}); the workers read it at their next hourly prepare", p.display()));
|
||||
}
|
||||
if self.node_override_restart && self.node.is_some() && !self.node_external {
|
||||
// between hourly boundaries unless the switch is close
|
||||
let (eta, daa) = { let st = self.st(); (st.program.eta_s, st.node.daa) };
|
||||
|
|
@ -2562,6 +2570,8 @@ impl Engine {
|
|||
c.message = String::new();
|
||||
}
|
||||
}
|
||||
} else if text.contains(" worker: race ") {
|
||||
self.race_line(card, &card_name, text);
|
||||
} else if text.contains(" worker: ready ") {
|
||||
let prepare = text.contains(" prepare 1");
|
||||
let mut st = self.st();
|
||||
|
|
@ -2630,6 +2640,59 @@ impl Engine {
|
|||
}
|
||||
}
|
||||
|
||||
/// A worker's race line (one per hourly prepare, docs/design/miner-tuning.md):
|
||||
/// `race <epoch16> device <name> driver <d> arch <a> loads <n> wide <n> variants <k> <name>=<MH/s>/<regs>r/<warps>w ...
|
||||
/// winner <name> <MH/s> base <MH/s> gain <pct>% compile <ms> bench <ms> total <ms> ms [| <variant>: <why>]`.
|
||||
/// Becomes the fleet record: one `TUNING {json}` line in the app log (uploaded to the intake, aggregated by
|
||||
/// tools/tuning.mjs) with the card's power figures, plus the card state and one event.
|
||||
fn race_line(&mut self, card: usize, card_name: &str, text: &str) {
|
||||
let Some(body) = text.split(" worker: race ").nth(1) else { return };
|
||||
let Some(r) = parse_race(body) else { return };
|
||||
let (vendor, worker, power_limit_w, power_w, power_pct) = {
|
||||
let mut st = self.st();
|
||||
match st.mining.cards.get_mut(card) {
|
||||
Some(c) => {
|
||||
c.variant = r.winner.clone();
|
||||
c.race_mhs = r.mhs;
|
||||
c.race_gain_pct = r.gain_pct;
|
||||
c.race_variants = r.variants.len() as u32;
|
||||
(c.vendor.clone(), c.worker.clone(), c.power_limit_w, c.power_w, c.power_pct)
|
||||
}
|
||||
None => (String::new(), String::new(), 0.0, 0.0, 0),
|
||||
}
|
||||
};
|
||||
let record = serde_json::json!({
|
||||
"ts": crate::platform::unix_now_f().round(),
|
||||
"machine": self.shared.runtime.id8(),
|
||||
"app": VERSION,
|
||||
"card": r.device,
|
||||
"vendor": vendor,
|
||||
"worker": worker,
|
||||
"driver": r.driver,
|
||||
"arch": r.arch,
|
||||
"epoch": r.epoch,
|
||||
"loads": r.loads,
|
||||
"wide": r.wide,
|
||||
"variants": r.variants,
|
||||
"winner": r.winner,
|
||||
"mhs": r.mhs,
|
||||
"base_mhs": r.base_mhs,
|
||||
"gain_pct": r.gain_pct,
|
||||
"power_limit_w": power_limit_w,
|
||||
"power_w": power_w,
|
||||
"power_pct": power_pct,
|
||||
"mh_per_w": if power_w > 0.0 { (r.mhs / power_w * 1000.0).round() / 1000.0 } else { 0.0 },
|
||||
"total_ms": r.total_ms,
|
||||
"pinned": r.pinned,
|
||||
"tuned": r.tuned,
|
||||
"notes": r.notes,
|
||||
});
|
||||
self.shared.log(&format!("TUNING {record}"));
|
||||
if !r.winner.is_empty() {
|
||||
self.shared.event("build", &format!("{card_name} kernel race: {} at {:.1} MH/s ({:+.1}% over base, {} variants, {:.0} s)", r.winner, r.mhs, r.gain_pct, r.variants.len(), r.total_ms / 1000.0));
|
||||
}
|
||||
}
|
||||
|
||||
// ---- status, uploads, shutdown ---------------------------------------------------------------------------
|
||||
|
||||
fn status_line(&mut self) {
|
||||
|
|
@ -2785,9 +2848,82 @@ fn behind_from_warning(text: &str) -> Option<f64> {
|
|||
if behind > 0.0 { Some(behind) } else { None }
|
||||
}
|
||||
|
||||
/// A worker's race line after "race " (docs/design/miner-tuning.md section 3), parsed into the record's fields.
|
||||
#[derive(Debug, Default, PartialEq)]
|
||||
pub struct RaceParsed {
|
||||
pub epoch: String,
|
||||
pub device: String,
|
||||
pub driver: String,
|
||||
pub arch: String,
|
||||
pub loads: u64,
|
||||
pub wide: u64,
|
||||
/// variant name -> MH/s (null when the variant was discarded or not timed)
|
||||
pub variants: serde_json::Map<String, Value>,
|
||||
pub winner: String,
|
||||
pub mhs: f64,
|
||||
pub base_mhs: f64,
|
||||
pub gain_pct: f64,
|
||||
pub total_ms: f64,
|
||||
pub pinned: bool,
|
||||
pub tuned: bool,
|
||||
pub notes: String,
|
||||
}
|
||||
|
||||
pub fn parse_race(body: &str) -> Option<RaceParsed> {
|
||||
let (main, notes) = match body.split_once(" | ") {
|
||||
Some((m, n)) => (m, n),
|
||||
None => (body, ""),
|
||||
};
|
||||
let t: Vec<&str> = main.split_whitespace().collect();
|
||||
let after = |key: &str| t.iter().position(|x| *x == key).and_then(|i| t.get(i + 1)).map(|s| s.to_string());
|
||||
let num = |key: &str| after(key).and_then(|v| v.trim_end_matches('%').parse::<f64>().ok());
|
||||
let epoch = t.first()?.to_string();
|
||||
if epoch.len() < 8 || !epoch.chars().all(|c| c.is_ascii_hexdigit()) {
|
||||
return None;
|
||||
}
|
||||
let mut r = RaceParsed { epoch, device: after("device").unwrap_or_default(), driver: after("driver").unwrap_or_default(), arch: after("arch").unwrap_or_default(), loads: num("loads").unwrap_or(0.0) as u64, wide: num("wide").unwrap_or(0.0) as u64, ..Default::default() };
|
||||
let n = num("variants").unwrap_or(0.0) as usize;
|
||||
if let Some(i) = t.iter().position(|x| *x == "variants") {
|
||||
for tok in t.iter().skip(i + 2).take(n) {
|
||||
if let Some((name, rest)) = tok.split_once('=') {
|
||||
let mhs = rest.split('/').next().and_then(|m| m.parse::<f64>().ok());
|
||||
r.variants.insert(name.to_string(), mhs.map(Value::from).unwrap_or(Value::Null));
|
||||
}
|
||||
}
|
||||
}
|
||||
r.winner = after("winner").unwrap_or_default();
|
||||
r.mhs = t.iter().position(|x| *x == "winner").and_then(|i| t.get(i + 2)).and_then(|v| v.parse::<f64>().ok()).unwrap_or(0.0);
|
||||
r.base_mhs = num("base").unwrap_or(0.0);
|
||||
r.gain_pct = num("gain").unwrap_or(0.0);
|
||||
r.total_ms = num("total").unwrap_or(0.0);
|
||||
r.pinned = main.contains(" pinned by tuning");
|
||||
r.tuned = main.contains(" tuned order");
|
||||
r.notes = notes.to_string();
|
||||
Some(r)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{sync_decision, Reading};
|
||||
use super::{parse_race, sync_decision, Reading};
|
||||
|
||||
#[test]
|
||||
fn race_line_parses() {
|
||||
let line = "a1b2c3d4e5f60718 device NVIDIA_GeForce_RTX_5090 driver 581.4 arch sm_120 loads 128 wide 0 variants 4 base=115.900/40r/1w u2=117.200/40r/1w ldg=- w4=118.300/40r/4w winner w4 118.300 base 115.900 gain +2.07% compile 1234 bench 9876 total 11110 ms tuned order | ldg: compile: identifier __ldg undefined";
|
||||
let r = parse_race(line).unwrap();
|
||||
assert_eq!(r.epoch, "a1b2c3d4e5f60718");
|
||||
assert_eq!(r.device, "NVIDIA_GeForce_RTX_5090");
|
||||
assert_eq!((r.driver.as_str(), r.arch.as_str(), r.loads, r.wide), ("581.4", "sm_120", 128, 0));
|
||||
assert_eq!(r.variants.len(), 4);
|
||||
assert_eq!(r.variants["w4"], 118.3);
|
||||
assert!(r.variants["ldg"].is_null());
|
||||
assert_eq!((r.winner.as_str(), r.mhs, r.base_mhs, r.gain_pct, r.total_ms), ("w4", 118.3, 115.9, 2.07, 11110.0));
|
||||
assert!(r.tuned && !r.pinned);
|
||||
assert_eq!(r.notes, "ldg: compile: identifier __ldg undefined");
|
||||
// the Metal shape, pinned, no notes
|
||||
let r = parse_race("0000000000000000 device Apple_M5_Max driver macos-26.0.1 arch metal loads 128 wide 0 variants 2 base=-/1024t/32w u2=9.800/1024t/32w winner u2 9.800 base 0.000 gain +0.00% compile 300 bench 10 total 310 ms pinned by tuning").unwrap();
|
||||
assert!(r.pinned && r.winner == "u2" && r.variants["base"].is_null());
|
||||
assert!(parse_race("not a race line").is_none());
|
||||
}
|
||||
|
||||
fn r(blocks: u64, headers: u64, peers: u64, flag: Option<bool>) -> Reading {
|
||||
Reading { blocks, headers, peers, flag }
|
||||
|
|
|
|||
|
|
@ -11,9 +11,12 @@
|
|||
//! "version": "0.3.1", "published_at": "2026-10-04T13:00:00Z", "channel": "devnet",
|
||||
//! "platforms": { "mac": {"url","sha256","size","kind":"dmg"|"zip"}, "windows": {"url","sha256","size","kind":"inno-setup"} },
|
||||
//! "min_supported_version": "0.3.0", "notes": "one line",
|
||||
//! "consensus": { "activation_height": null|number, "deadline_note": "" }
|
||||
//! "consensus": { "activation_height": null|number, "deadline_note": "", "override": {...} },
|
||||
//! "tuning": { "updated": "...", "cards": { "<card model>": { "variant": "u2", "race": true, "candidates": [..] } } }
|
||||
//! }
|
||||
//! A platform that is missing is not updated (the Windows build lands later than the Mac one).
|
||||
//! `tuning` (4 October 2026, docs/design/miner-tuning.md) is the fleet's per-card kernel tuning: the engine writes it
|
||||
//! to <app data>/tuning.json and every GPU worker reads it at its next prepare (IGNEUM_TUNING_FILE).
|
||||
|
||||
#![allow(dead_code)]
|
||||
|
||||
|
|
@ -53,6 +56,9 @@ pub struct Manifest {
|
|||
/// consensus.override: the exact object the engine writes to <app data>/override.json for igneumd's
|
||||
/// --override-params-file (for example {"difficulty_v2_activation_daa": N}); signed with the rest of the manifest.
|
||||
pub override_params: Option<serde_json::Value>,
|
||||
/// tuning: the per-card kernel tuning object (tools/tuning.mjs writes it, publish-manifest.sh --tuning carries
|
||||
/// it), written as is to <app data>/tuning.json for the GPU workers; signed with the rest of the manifest.
|
||||
pub tuning: Option<serde_json::Value>,
|
||||
}
|
||||
|
||||
impl Manifest {
|
||||
|
|
@ -157,6 +163,11 @@ pub fn parse(text: &str) -> Result<Manifest, String> {
|
|||
Some(o) if !o.is_null() => return Err("consensus.override must be an object".into()),
|
||||
_ => None,
|
||||
},
|
||||
tuning: match v.get("tuning") {
|
||||
Some(t) if t.is_object() && t.get("cards").map(|c| c.is_object()).unwrap_or(false) => Some(t.clone()),
|
||||
Some(t) if !t.is_null() => return Err("tuning must be an object with a cards object".into()),
|
||||
_ => None,
|
||||
},
|
||||
})
|
||||
}
|
||||
|
||||
|
|
@ -371,6 +382,16 @@ mod tests {
|
|||
assert_eq!(ours, theirs);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tuning_parses() {
|
||||
let m = parse(r#"{"version":"0.3.4","platforms":{},"tuning":{"updated":"2026-10-04T20:00:00Z","cards":{"NVIDIA_GeForce_RTX_5090":{"variant":"u2-ldg","race":true,"candidates":["u2-ldg","ldg","base"]}}}}"#).unwrap();
|
||||
assert_eq!(m.tuning.as_ref().unwrap()["cards"]["NVIDIA_GeForce_RTX_5090"]["variant"], "u2-ldg");
|
||||
assert!(parse(r#"{"version":"0.3.4","platforms":{},"tuning":null}"#).unwrap().tuning.is_none());
|
||||
assert!(parse(r#"{"version":"0.3.4","platforms":{}}"#).unwrap().tuning.is_none());
|
||||
assert!(parse(r#"{"version":"0.3.4","platforms":{},"tuning":"fast"}"#).is_err());
|
||||
assert!(parse(r#"{"version":"0.3.4","platforms":{},"tuning":{"cards":[]}}"#).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn consensus_override_parses() {
|
||||
let m = parse(r#"{"version":"0.3.2","platforms":{},"consensus":{"activation_height":5000,"override":{"difficulty_v2_activation_daa":5000}}}"#).unwrap();
|
||||
|
|
|
|||
|
|
@ -109,6 +109,8 @@ pub struct Updater {
|
|||
/// the consensus override file written from the manifest, when it changed since the last take
|
||||
override_changed: Option<PathBuf>,
|
||||
override_daa: u64,
|
||||
/// the per-card tuning file written from the manifest, when it changed since the last take
|
||||
tuning_changed: Option<PathBuf>,
|
||||
/// Windows: the installer was started and the engine is still up (it stops us when it may run)
|
||||
apply_launched: Option<Instant>,
|
||||
/// the administrator prompt was not answered: no automatic retry before this (Install now still works)
|
||||
|
|
@ -157,6 +159,7 @@ impl Updater {
|
|||
staged_digest: String::new(),
|
||||
override_changed: None,
|
||||
override_daa: 0,
|
||||
tuning_changed: None,
|
||||
apply_launched: None,
|
||||
deferred_until: None,
|
||||
slot: manifest::slot_minute(&shared.runtime.id8()),
|
||||
|
|
@ -180,6 +183,7 @@ impl Updater {
|
|||
if let Ok(m) = manifest::parse(&text) {
|
||||
u.min_supported = m.min_supported_version.clone();
|
||||
u.write_override(shared, &m);
|
||||
u.write_tuning(shared, &m);
|
||||
}
|
||||
}
|
||||
u.settle_previous(shared);
|
||||
|
|
@ -224,6 +228,45 @@ impl Updater {
|
|||
shared.state.lock().unwrap().node.consensus_switch_daa = self.override_daa;
|
||||
}
|
||||
|
||||
/// The per-card tuning from the manifest (docs/design/miner-tuning.md): `tuning` written as is to
|
||||
/// <app data>/tuning.json; the GPU workers read it at their next prepare through IGNEUM_TUNING_FILE (the engine
|
||||
/// passes the path to every miner it starts). A manifest without tuning removes the file: the workers race
|
||||
/// every variant again.
|
||||
fn write_tuning(&mut self, shared: &Arc<Shared>, m: &Manifest) {
|
||||
let path = self.app_dir.join("tuning.json");
|
||||
match &m.tuning {
|
||||
Some(t) => {
|
||||
let text = t.to_string();
|
||||
if std::fs::read_to_string(&path).ok().as_deref() == Some(text.as_str()) {
|
||||
return;
|
||||
}
|
||||
if let Err(e) = std::fs::write(&path, &text) {
|
||||
shared.log(&format!("could not write {}: {e}", path.display()));
|
||||
return;
|
||||
}
|
||||
let cards = t.get("cards").and_then(|c| c.as_object()).map(|c| c.len()).unwrap_or(0);
|
||||
shared.event("info", &format!("kernel tuning from the signed manifest: {cards} card model(s), updated {}", t.get("updated").and_then(|u| u.as_str()).unwrap_or("?")));
|
||||
self.tuning_changed = Some(path);
|
||||
}
|
||||
None => {
|
||||
if path.is_file() && std::fs::remove_file(&path).is_ok() {
|
||||
shared.log("the manifest carries no kernel tuning any more; tuning.json removed (the workers race every variant again)");
|
||||
self.tuning_changed = Some(path);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The tuning file, once per change (the engine only logs it: the workers read the file at their next prepare).
|
||||
pub fn take_tuning_change(&mut self) -> Option<PathBuf> {
|
||||
self.tuning_changed.take()
|
||||
}
|
||||
|
||||
pub fn tuning_path(&self) -> Option<PathBuf> {
|
||||
let p = self.app_dir.join("tuning.json");
|
||||
if p.is_file() { Some(p) } else { None }
|
||||
}
|
||||
|
||||
/// The override file to start the node with, once per change.
|
||||
pub fn take_override_change(&mut self) -> Option<PathBuf> {
|
||||
self.override_changed.take()
|
||||
|
|
@ -616,6 +659,7 @@ impl Updater {
|
|||
self.manifest = Some(m.clone());
|
||||
self.min_supported = m.min_supported_version.clone();
|
||||
self.write_override(shared, &m);
|
||||
self.write_tuning(shared, &m);
|
||||
if let Some(e) = entry {
|
||||
if changed {
|
||||
self.file = None;
|
||||
|
|
|
|||
|
|
@ -40,10 +40,14 @@ pub struct Proc {
|
|||
pub exit_code: Option<i32>,
|
||||
}
|
||||
|
||||
/// Starts a program with stdout and stderr piped; every line goes to `tx` and to the log file.
|
||||
pub fn spawn(src: Source, exe: &Path, args: &[String], cwd: Option<&Path>, log_path: &Path, tx: &Sender<Line>) -> std::io::Result<Proc> {
|
||||
/// Starts a program with stdout and stderr piped; every line goes to `tx` and to the log file. `envs` are added to
|
||||
/// the child's environment (a miner passes IGNEUM_TUNING_FILE on to its GPU worker).
|
||||
pub fn spawn(src: Source, exe: &Path, args: &[String], cwd: Option<&Path>, log_path: &Path, tx: &Sender<Line>, envs: &[(String, String)]) -> std::io::Result<Proc> {
|
||||
let mut cmd = Command::new(exe);
|
||||
cmd.args(args).stdin(Stdio::piped()).stdout(Stdio::piped()).stderr(Stdio::piped());
|
||||
for (k, v) in envs {
|
||||
cmd.env(k, v);
|
||||
}
|
||||
if let Some(d) = cwd {
|
||||
cmd.current_dir(d);
|
||||
}
|
||||
|
|
|
|||
|
|
@ -81,6 +81,11 @@ pub struct CardState {
|
|||
pub sweep_mhs: f64,
|
||||
pub sweep_at: f64, // unix s of the last sweep
|
||||
pub pinned: bool, // the user set the cap by hand; the sweep records but does not change it
|
||||
// the kernel variant race (docs/design/miner-tuning.md): what the worker's last race chose
|
||||
pub variant: String,
|
||||
pub race_mhs: f64,
|
||||
pub race_gain_pct: f64,
|
||||
pub race_variants: u32,
|
||||
}
|
||||
|
||||
#[derive(Clone, Serialize, Default)]
|
||||
|
|
|
|||
|
|
@ -1095,3 +1095,84 @@ the project lead, 4 October 2026 evening: "we need to fix these serious issues b
|
|||
| split70, v3: 4/2 keys, the 4 side at 70% of weight (shares 0.175 x 4 against 0.15 x 2), 150 s | the 4 side locks during the split, the 2 side does not; 0 conflicts | 4 side: 4 new locks, the first 30 s after the cut; 2 side: 0; heal: all three at 17; 0 conflicting certificates; 0 disagreeing indices | PASS (6B at 70/30; exactly 4/6 is a knife edge under both rules, simulator row above) |
|
||||
|
||||
**What remains uncertain.** (1) The v3 split's hold was observed over the last 24 s of a 150-s split (the control crossed at 126 s); a longer split under the 240-s expiry (say 200 s) would show more held checkpoints, and the "held by the frozen table" line is logged at debug, which the runs did not enable. (2) No cloud rehearsal: the 12-node Hetzner network was destroyed at 15:30 UTC, so the 95% target is shown on three nodes with emulated 300-ms links and on the cloud logs' arithmetic, not on the cloud topology itself; the rollout plan names the re-creation and the partition experiment to run first. (3) The frozen reference is the node's own highest lock on the chain, not the certificate carried in C_i's past, so two honest nodes can test one checkpoint against tables 30 s apart; in a connected network those tables differ by a minute of blocks, under a partition both are pre-split, and no run showed a disagreement, but it is a property argued, not proved. (4) The price: a sudden departure of a third or more now pauses finality for a full window (30 days on mainnet) instead of 1.4 to 10 days; a gradual one costs nothing. the project lead asked for the pause over the fork; the number is stated in spec 3.7 item 2. (5) The fold clock is in memory: a restarted node folds from `daa(C_i) + depth`, a few seconds late at worst. (6) Binaries, all from `finality-fixes` 6aa69a45, hashes and checks in the rollout plan's section 2: Mac native `fe982a1d...` (verified running), Linux `7c100fc2...` (cargo-zigbuild, 34 min, not run on a Linux host), Windows `cc1d1001...` (mingw, 12 min 28 s, the v2 exe's DLL set, cannot run here); the Windows payload inputs were staged with `push-inputs.sh --no-deploy` into a scratch folder and NOT deployed (plan 7a).
|
||||
|
||||
## 4 October 2026, miner performance: variant racing (Metal worker on the M5 Max; the RTX 5090 job is ready, not run)
|
||||
|
||||
Method (`docs/design/miner-tuning.md`): at every hourly prepare the worker compiles the bound kernel in several
|
||||
variants (unroll, load path, register budget, threads per group, combinations), checks each bit for bit against the
|
||||
base kernel, times each for 2 s with the job loop paused, and keeps the fastest for the hour. Base is the kernel as
|
||||
it has always shipped. Code: `proto-metal/main.swift` (`raceProgram`, `--race-test`), `proto-cuda/nvrtc/worker.cpp`
|
||||
(`racePair`, `--race`), branch `miner-perf`, commit 460a99a.
|
||||
|
||||
Machine: Apple M5 Max, Darwin 25.6.0 (macOS 26.6.2), 64 GiB. CONDITIONS: the live Igneum Miner app's own Metal
|
||||
worker (`igneum-bench --serve`, pid 14687) was mining on the same GPU throughout, and the load average was 130 at the
|
||||
build and 14 to 67 during the races (other agents' cargo builds). The absolute MH/s below are therefore about half
|
||||
of the card's (the app reported 26.7 MH/s on 4 October with the GPU to itself) and each window was contended; the
|
||||
numbers to read are the ratios, taken as the best of three interleaved rounds per variant so the contention hits
|
||||
every variant alike. A re-run with the Mac card paused is listed under "next".
|
||||
|
||||
Command (under the measure lock, which holds the build lock too):
|
||||
|
||||
tools/lock/with-lock.sh measure bash scratchpad/metal/measure.sh
|
||||
= swiftc -O -target arm64-apple-macos11 -o igneum-bench main.swift -framework Metal (47 s under load 130)
|
||||
igneum-bench --race-test --seed igneum-genesis --day 2026-10-04 --race-rounds 3 --race-bench-ms 2000
|
||||
igneum-bench --race-test --seed igneum-hourly --day 2026-10-04 --race-rounds 3 --race-bench-ms 2000
|
||||
|
||||
Dataset 2^28 words (1 GiB, memory-hard, built in 285 and 295 ms), batch 2^22 nonces per launch, programs by the
|
||||
version-2 generator (128 loads per hash, no wide loads). 14 variants, every one bit-exact with base over 2^16 nonces
|
||||
(no variant discarded). MH/s = best of 3 rounds, 2 s windows, first launch of each window not counted.
|
||||
|
||||
| variant | threads/group | max threads/group | seed igneum-genesis MH/s | vs base | seed igneum-hourly MH/s | vs base |
|
||||
|---|---|---|---|---|---|---|
|
||||
| base | 32 | 1024 | 11.180 | 0 | 10.800 | 0 |
|
||||
| g64 | 64 | 1024 | 11.582 | +3.6% | 11.801 | +9.3% |
|
||||
| g128 | 128 | 1024 | 12.736 | +13.9% | 12.306 | +14.0% |
|
||||
| g256 | 256 | 1024 | **13.114** | **+17.3%** | **13.089** | **+21.2%** |
|
||||
| u2 | 32 | 1024 | 10.683 | -4.4% | 10.596 | -1.9% |
|
||||
| u8 | 32 | 1024 | 10.823 | -3.2% | 10.280 | -4.8% |
|
||||
| mt256 | 32 | 256 | 10.953 | -2.0% | 10.664 | -1.3% |
|
||||
| mt512 | 32 | 512 | 11.022 | -1.4% | 10.636 | -1.5% |
|
||||
| mt1024 | 32 | 1024 | 10.666 | -4.6% | 11.150 | +3.2% |
|
||||
| osize | 32 | 1024 | 10.833 | -3.1% | 10.437 | -3.4% |
|
||||
| u2-g128 | 128 | 1024 | 12.793 | +14.4% | 12.928 | +19.7% |
|
||||
| u8-g128 | 128 | 1024 | 12.496 | +11.8% | 11.923 | +10.4% |
|
||||
| mt256-g128 | 128 | 256 | 12.628 | +13.0% | 12.675 | +17.4% |
|
||||
| mt512-g256 | 256 | 512 | 13.101 | +17.2% | 12.817 | +18.7% |
|
||||
|
||||
Race cost: compile 1,798 ms (first seed; the Metal compiler cold) and 267 ms, timing 108 s for 14 variants x 3
|
||||
rounds (2 s windows plus the 2^16-nonce check); in `--serve` the race runs one round, about 40 s, inside a 600-DAA
|
||||
lead, with mining paused only inside the windows.
|
||||
|
||||
Reading. On Apple silicon the win is threads per threadgroup: the Metal worker has dispatched one 32-thread group
|
||||
per 32 nonces since 3 October, and 256-thread groups are 17 to 21% faster on both programs under these conditions,
|
||||
with 128 close behind; the unroll, register-budget and size-optimisation knobs are within noise or worse on their
|
||||
own. The winner agrees across the two programs, so a tuning entry `Apple_M5_Max: g256` would be the first
|
||||
fleet default; the race itself finds it in one round. These two programs are two points, under contention; the
|
||||
figure for the Mac's own card with the GPU to itself is still to take. Nothing here says anything about NVIDIA:
|
||||
`w8` (8 warps per block) is the CUDA cousin of `g256`, and whether the 5090 moves at all is what the PC job
|
||||
(`docs/plans/miner-perf.md`) measures. Range to measure there: from no gain to what the block-size and load-path
|
||||
variants give on a 1 GiB random-read kernel; no claim.
|
||||
|
||||
Serve-protocol check (the same binary, `--serve --race-rounds 1`, scripted stdin: two inline jobs on pair A, the
|
||||
deferred race on A, `prepare` of pair B with its race, jobs on A meanwhile, the swap to B, a job across the 32-bit
|
||||
nonce boundary): 44 jobs done, 0 errors, no found line missed; the inline compile of pair A 220 ms, the deferred
|
||||
race on A `winner g256 14.207 base 11.404 gain +24.58%` (compile 563 ms, 36 s of windows); `prepared` for B after
|
||||
35,554 ms = program 58 ms, dataset 524 ms, race 34,972 ms (`winner mt512-g256 12.807 base 10.346 gain +23.79%`),
|
||||
the swap to B in 0.01 ms, the 64-nonce job across the 32-bit boundary 14.8 ms. Found by this check: the job queued
|
||||
during a race waited for the whole race (job 2 done after 35,946 ms; the mutex is not fair), so the race now
|
||||
pauses 150 ms after every window (commit 32d1c01). Re-check with the pause (`--serve`, a job every 3 s through
|
||||
both races, load average 134 to 183): every job during the deferred race on A and the prepare race on B finished
|
||||
in 0.17 to 2.8 s (33 jobs, none over 2,831 ms, versus 35,946 ms before), the race on B 39.8 s inside a
|
||||
40.2 s prepare, winner g256 both times, swap 0.01 ms, 0 errors; the race's own windows were 2 to 3 s longer
|
||||
in total than without the pause, as expected.
|
||||
|
||||
NVIDIA side, what the Mac could check: `proto-cuda/nvrtc/emu/test.sh` PASS on the race build (the race off under
|
||||
emulation, "variants 1 base only, no race (emulation)" logged per pair; 9 source checks PASS, the --serve protocol
|
||||
with prepare, swap and self-heal unchanged, 17 sampled hashes equal to `igneum-pow hash-bound`); mingw cross-compile of
|
||||
`igneum-worker-cuda.exe` with the race (`build-windows.sh`, mingw, static): 1,509,376 bytes, the same imports as the
|
||||
shipped worker (KERNEL32 and the Universal CRT), icon and version block verified; zipped as
|
||||
`~/Desktop/igneum-worker-cuda-race.zip` (429,387 bytes, sha256 321a086e...c4c049) for the PC 1 job. The race has
|
||||
not run on a GPU.
|
||||
|
||||
Next: the PC 1 job (ready in `docs/plans/miner-perf.md`); the Mac card paused for a clean absolute table; the
|
||||
Mac app's own worker on this build (its hourly prepare then races by itself and logs the TUNING record).
|
||||
|
|
|
|||
156
docs/design/miner-tuning.md
Normal file
156
docs/design/miner-tuning.md
Normal file
|
|
@ -0,0 +1,156 @@
|
|||
# Miner tuning: variant racing and fleet learning
|
||||
|
||||
4 October 2026, evening. the project lead: "we need to make our miner better than anything else can be". Two levers, both
|
||||
measured: a race between kernel variants at every hourly swap, and a fleet that remembers which variant each card
|
||||
model likes. Numbers live in `docs/bench-log.md` ("miner performance: variant racing"); the PC job is in
|
||||
`docs/plans/miner-perf.md`. Nothing here changes the hash: every variant is the same instruction text in a
|
||||
different shape for the compiler, and a variant that is not bit-exact is discarded before it is timed.
|
||||
|
||||
## 1. Why a race
|
||||
|
||||
The lottery program changes every hour. The generator draws 64 instructions with 16 loads; the compiler sees a
|
||||
different straight-line body each time, and what suits one body (full unrolling, a read-only load path, more
|
||||
threads per block) does not suit the next. A fixed compile is a guess. The compile-ahead pipeline already builds
|
||||
the next hour's kernel one lead (600 DAA, about 600 s) before the boundary, so there is time to build several
|
||||
and let the card pick.
|
||||
|
||||
## 2. The variants
|
||||
|
||||
| Knob | NVIDIA (NVRTC worker, `proto-cuda/nvrtc/worker.cpp`) | Apple (Metal worker, `proto-metal/main.swift`) |
|
||||
|---|---|---|
|
||||
| Unroll | `#pragma unroll 2` or `8` before the iteration loop (`u2`, `u8`) | the same pragma (`u2`, `u8`) |
|
||||
| Load path | `ds[i]` as shipped; `__ldg` read-only path (`ldg`); `__ldcg` L2 only (`ldcg`); `__ldcs` streaming (`ldcs`) | none (one address space on Apple silicon) |
|
||||
| Register budget | `--maxrregcount=32` or `64` (`r32`, `r64`); `__launch_bounds__(128, 4)` (`lb4-w4`), `(64, 8)` (`lb8-w2`) | `[[max_total_threads_per_threadgroup(N)]]` 256, 512, 1024 (`mt256` ...) |
|
||||
| Threads per block | 2, 4, 8 warps (`w2`, `w4`, `w8`) | 64, 128, 256 threads per threadgroup (`g64`, `g128`, `g256`) |
|
||||
| Compiler | | `optimizationLevel = .size` (`osize`) |
|
||||
| Combinations | `u2-ldg`, `u2-w4`, `ldg-w4`, `ldcg-w4` | `u2-g128`, `u8-g128`, `mt256-g128`, `mt512-g256` |
|
||||
|
||||
17 names on NVIDIA, 14 on Metal. `base` is always the pack's text as shipped with the worker's default block:
|
||||
the kernel every machine ran before this change. Names are stable; the tuning file and the fleet records use them.
|
||||
|
||||
NVIDIA rewrites are textual, on the pack's own `kernel_bound.cu`, with exact anchors from `igneum-pow`'s emitter
|
||||
(`"\n for (uint32_t it = 0u; it < "`, `" ^ ds["`, `"__global__ void igneum_hash_bound("`); a text without the
|
||||
anchor refuses the variant instead of guessing. The pack format, the miner and `igneum-pow` are untouched, so a
|
||||
new worker races old packs. OpenCL (AMD, Intel) is not raced yet: `proto-opencl/host.c` builds through
|
||||
`clBuildProgram` with `-D IGNEUM_GROUP` already, so the same catalogue (unroll pragma, group size, `-cl-` options)
|
||||
is the next step; see "open".
|
||||
|
||||
## 3. The race inside the prepare
|
||||
|
||||
```
|
||||
prepare <epoch> <day> <pack> (the miner, one lead before the boundary)
|
||||
compile kernel.cu, build cache + dataset, self-test base (as before)
|
||||
compile the variants (NVRTC: 4 threads; Metal: in turn) budget: --race-budget-s, default 120
|
||||
for each round (1 in --serve):
|
||||
for each variant: lock the card, self-test, time ~2 s, unlock
|
||||
winner = fastest; base keeps its place unless beaten by 0.5%
|
||||
one "race ..." line, then "prepared ..." as before
|
||||
job on the new pair at the boundary (swaps as before; the winner serves the hour)
|
||||
```
|
||||
|
||||
Rules that keep the swap safe:
|
||||
|
||||
| Rule | Where |
|
||||
|---|---|
|
||||
| Base is the first entry and is never discarded; a race that runs out of budget keeps the best so far | `racePair`, `raceProgram` |
|
||||
| Every variant must reproduce the pack's vector warps (NVIDIA) or the base kernel's output over 2^16 nonces (Metal), bit for bit, or it is out | `raceTime`, `raceProgram` |
|
||||
| The card is exclusive while a variant is timed: one mutex, held per chunk by the job loop and per window by the race. Mining pauses about 2 s per variant and resumes between variants | `gpuMutex`, `gpuLock` |
|
||||
| `--race-budget-s` is capped at 540 (the lead is 600 DAA); default 120 | option parsing |
|
||||
| A pair compiled inline (nobody prepared it) races after its first job, in the background | Metal `raceDue`; NVIDIA self-heal path |
|
||||
| `--race off` restores the old behaviour; `--race a,b,c` limits the catalogue | both workers |
|
||||
| Under `IGNEUM_EMU` (the Mac's emulation test) the race is off: the stand-in checks that the handed-over text is the pack's | `racePair` |
|
||||
|
||||
Cost per hour: on the 5090 about 17 variants x (2 s window + a self-test) of paused mining, under 1% of the hour,
|
||||
plus the compiles on the CPU. The expected gain is what the race measures; nothing is claimed for it.
|
||||
|
||||
The line, one per race (the miner logs it as `worker: race ...`):
|
||||
|
||||
```
|
||||
race <epoch16> device <name> driver <d> arch <a> loads <n> wide <n> variants <k> base=<MH/s>/<regs>r/<warps>w u2=... ldg=-
|
||||
winner <name> <MH/s> base <MH/s> gain <+pct>% compile <ms> bench <ms> total <ms> ms [pinned by tuning|tuned order]
|
||||
[| <variant>: <why it is out>]
|
||||
```
|
||||
|
||||
`--race --pack <dir>` (NVIDIA) and `--race-test --seed <s> --day <d>` (Metal) run the race alone, three rounds,
|
||||
and print a table; that is what the bench log and the PC job use.
|
||||
|
||||
## 4. Fleet learning
|
||||
|
||||
### 4.1 The record
|
||||
|
||||
The app's engine reads the race line (`engine.rs` `race_line`) and writes one `TUNING {json}` line to the app
|
||||
log, which the existing intake receives with every upload (Neon `miner_logs`, the same table `tools/logs.mjs`
|
||||
reads). Fields:
|
||||
|
||||
| Field | From |
|
||||
|---|---|
|
||||
| `ts`, `machine` (id8), `app` (version) | the engine |
|
||||
| `card` (the worker's device name, spaces as underscores: the key everything else uses), `vendor`, `worker` (CUDA, Metal, OpenCL) | the race line, the card state |
|
||||
| `driver`, `arch` (sm_120, metal) | the race line |
|
||||
| `epoch` (16 hex), `loads`, `wide` (the program class features the generator fixes: loads per hash and wide loads per hash) | the race line |
|
||||
| `variants` {name: MH/s or null} | the race line |
|
||||
| `winner`, `mhs`, `base_mhs`, `gain_pct`, `total_ms`, `pinned`, `tuned`, `notes` | the race line |
|
||||
| `power_limit_w`, `power_w`, `power_pct`, `mh_per_w` (= mhs / power_w, 0 when the card reports no draw) | the card state (nvidia-smi telemetry; Apple reports none) |
|
||||
|
||||
The card state also carries `variant`, `race_mhs`, `race_gain_pct`, `race_variants` for the dashboard, and one
|
||||
event per race ("RTX 5090 kernel race: u2-ldg at 118.3 MH/s (+2.1% over base, 17 variants, 41 s)").
|
||||
|
||||
### 4.2 The aggregation
|
||||
|
||||
`tools/tuning.mjs` on the Mac (or a small job): every app-log upload of the window (default 7 days) with a
|
||||
TUNING line, de-duplicated on (machine, card, epoch) because the log is re-sent every minute, then per card
|
||||
model and variant the sample count, the median MH/s and the median MH per watt. The winner is the best median
|
||||
with at least `--min-samples` (3) samples; `--by mhw` ranks by MH per watt instead. `--write tuning.json` writes:
|
||||
|
||||
```json
|
||||
{"updated": "2026-10-04T21:00:00Z", "window_days": 7,
|
||||
"cards": {"NVIDIA_GeForce_RTX_5090": {"variant": "u2-ldg", "race": true, "candidates": ["u2-ldg", "ldg", "base"],
|
||||
"samples": 41, "races": 41, "machines": 2, "mhs": 118.3, "base_mhs": 115.9,
|
||||
"gain_pct": 2.07, "mh_per_w": 0.254, "worker": "CUDA", "by": "mhs"}}}
|
||||
```
|
||||
|
||||
A program class split (by `loads`, `wide`) is in the record and not yet in the aggregation: the generator fixes
|
||||
16 loads per instruction block, so every program has 128 loads per hash today; the split starts to matter when the
|
||||
era draw changes the mix.
|
||||
|
||||
### 4.3 The way back: the manifest
|
||||
|
||||
`packaging/ota/publish-manifest.sh --tuning tuning.json` puts the object under `tuning` in the signed
|
||||
`igneum-app-latest.json` (carried over from the current manifest when not given; `--no-tuning` drops it; the
|
||||
same script now also takes `--override '{json}'` for `consensus.override` and carries that over too).
|
||||
`manifest.rs` parses `tuning` (an object with a `cards` object, else the manifest is refused). The updater writes
|
||||
it as is to `<app data>/app/tuning.json` (`ota.rs` `write_tuning`, next to `override.json`), logs one event, and
|
||||
the engine starts every miner with `IGNEUM_TUNING_FILE=<that path>` (`procs::spawn` gained an environment
|
||||
parameter); the miner's child, the worker, reads it at every prepare. No restart for a change: a worker that has
|
||||
the path reads the file again at its next prepare; a miner started before the file existed is restarted by the
|
||||
usual hourly path.
|
||||
|
||||
What a worker does with its entry (`readTuning`, `readMetalTuning`):
|
||||
|
||||
| Entry | Behaviour |
|
||||
|---|---|
|
||||
| none for this card model | the full race |
|
||||
| `race: true` with `candidates` | the candidates are raced first, then the rest as the budget allows (the fleet keeps learning; the card starts from the known best) |
|
||||
| `race: false` with `variant` | the variant is compiled and self-tested, no timing (about 1 s); a failed self-test falls back to the full race |
|
||||
| `--variant <name>` on the worker | the same as a pinned entry, for tests |
|
||||
|
||||
### 4.4 What is implemented and what is not
|
||||
|
||||
| Piece | State |
|
||||
|---|---|
|
||||
| NVRTC worker race, `--race` mode, tuning file | code, syntax-checked for mingw and the emulation build; emulation suite; **not yet run on a GPU** (the PC job does that) |
|
||||
| Metal worker race, `--race-test`, deferred race, tuning file | code and the Mac measurement (bench log) |
|
||||
| OpenCL worker race | not implemented (open) |
|
||||
| Engine: race line to TUNING record, card state, event, `IGNEUM_TUNING_FILE` | code, unit tests of the manifest parse |
|
||||
| `publish-manifest.sh --tuning`, `--override`, carry-over | code (dry run against a scratch folder in the plan) |
|
||||
| `tools/tuning.mjs` | code; needs records, so no table yet |
|
||||
| Aggregation as a job on a PC or the observer | not needed yet: the Mac script reads the intake |
|
||||
| Dashboard field for the variant | state only; the UI does not show it yet |
|
||||
|
||||
## 5. Open
|
||||
|
||||
- OpenCL: the same catalogue through `clBuildProgram` options and the pragma; `--group-warps` exists already.
|
||||
- The program class split in the aggregation once eras change the instruction mix.
|
||||
- A per-card cap on how long a race may pause mining (today the 2 s windows plus the budget); and whether to race
|
||||
only every N hours once a card's winner is stable (the pinned entry does that by hand).
|
||||
- Power: the record has MH per watt, the race does not touch the power cap; a race across caps is a later lever.
|
||||
123
docs/plans/miner-perf.md
Normal file
123
docs/plans/miner-perf.md
Normal file
|
|
@ -0,0 +1,123 @@
|
|||
# Miner performance: the variant race on PC 1's RTX 5090
|
||||
|
||||
4 October 2026, evening. The race is built and measured on the Mac's Metal worker (`docs/bench-log.md`, "miner
|
||||
performance: variant racing"); the design is `docs/design/miner-tuning.md`. This plan is the same race on the RTX
|
||||
5090 with the NVRTC worker, as a signed job for PC 1 (machine `ae432dc7`), ready for the main session to publish.
|
||||
Nothing here was published by the agent: PC 1 mines, PC 2 runs a shard job.
|
||||
|
||||
## What the job does
|
||||
|
||||
Two jobs in the signed file, in this order (the app runs them one at a time, in file order):
|
||||
|
||||
1. `fetch-race-worker-20261004` (kind `fetch`, `--dir jobs --extract`): the race build of
|
||||
`igneum-worker-cuda.exe` (cross-compiled on the Mac with mingw from the `miner-perf` branch, the same
|
||||
`build-windows.sh` the package uses) lands in `%LOCALAPPDATA%\igneum\app\jobs\fetch-race-worker-20261004\`.
|
||||
2. `run-race-5090-20261004` (kind `run`, `--stop-miners`, cap 20 min): `relay/playbooks/race-5090.ps1`. The
|
||||
miners are stopped first, so the 5090 is the race's alone. The script copies the NVRTC DLLs from the installed
|
||||
app next to the fetched exe, takes this hour's prepared pack (`packs\prepare\<epoch16>-<day>`, else
|
||||
`packs\devnet`), prints the card's state from `nvidia-smi`, and runs the race twice
|
||||
(`--race --pack <dir> --race-rounds 3 --race-bench-ms 2000 --race-budget-s 400`): 17 variants, each
|
||||
self-tested against the pack's vectors and timed for 2 s, three interleaved rounds, best per variant. Every
|
||||
line starts with `RESULT`, so the dashboard's job strip and `tools/jobs.mjs` show them. The miners restart
|
||||
when the job ends (the engine's job path).
|
||||
|
||||
Expected wall time: a 1 GiB dataset build (about 10 s on the 5090 from the 4 October numbers), 17 NVRTC
|
||||
compiles in four threads, then 17 x 3 x about 2.3 s of timing, twice: about 5 min with the miners stopped.
|
||||
|
||||
## The script
|
||||
|
||||
`relay/playbooks/race-5090.ps1` (in the repo, parse-checked by `windows.yml` with the other playbooks):
|
||||
|
||||
```powershell
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$app = $env:IGNEUM_APP_DIR
|
||||
$jobs = Split-Path $env:IGNEUM_JOB_DIR
|
||||
$fetched = Join-Path $jobs 'fetch-race-worker-20261004'
|
||||
$exe = Join-Path $fetched 'igneum-worker-cuda.exe'
|
||||
if (-not (Test-Path $exe)) { Write-Output "RESULT race worker missing at $exe (the fetch job runs first)"; exit 2 }
|
||||
$inst = @("$env:LOCALAPPDATA\Programs\Igneum Miner", "$env:ProgramFiles\Igneum Miner") | Where-Object { Test-Path (Join-Path $_ 'igneum-worker-cuda.exe') } | Select-Object -First 1
|
||||
if (-not $inst) { Write-Output "RESULT no installed igneum-worker-cuda.exe found (the NVRTC DLLs come from there)"; exit 2 }
|
||||
Get-ChildItem $inst -Filter 'nvrtc*.dll' | Copy-Item -Destination $fetched -Force
|
||||
Write-Output "RESULT worker $exe with $((Get-ChildItem $fetched -Filter 'nvrtc*.dll').Count) NVRTC DLL(s) from $inst"
|
||||
$pack = Get-ChildItem "$app\packs\prepare" -Directory -ErrorAction SilentlyContinue | Sort-Object LastWriteTime -Descending | Select-Object -First 1
|
||||
if ($pack) { $pack = $pack.FullName } else { $pack = "$app\packs\devnet" }
|
||||
if (-not (Test-Path "$pack\seeds.txt")) { Write-Output "RESULT no pack with seeds.txt under $app\packs"; exit 2 }
|
||||
Write-Output "RESULT pack $pack"
|
||||
Get-Content "$pack\seeds.txt" | ForEach-Object { "RESULT seeds $_" }
|
||||
& nvidia-smi --query-gpu=name,driver_version,power.limit,power.default_limit,clocks.sm,clocks.mem,temperature.gpu --format=csv,noheader | ForEach-Object { "RESULT gpu $_" }
|
||||
foreach ($run in 1..2) {
|
||||
Write-Output "RESULT run $run start $(Get-Date -Format HH:mm:ss)"
|
||||
& $exe --race --pack $pack --race-rounds 3 --race-bench-ms 2000 --race-budget-s 400 2>&1 | ForEach-Object { "RESULT $_" }
|
||||
Write-Output "RESULT run $run exit $LASTEXITCODE"
|
||||
}
|
||||
& nvidia-smi --query-gpu=power.draw,clocks.sm,temperature.gpu --format=csv,noheader | ForEach-Object { "RESULT gpu-after $_" }
|
||||
exit 0
|
||||
```
|
||||
|
||||
Quoting, as in the jobs that worked on PC 2 (`run-20261004-173115`): a plain PowerShell file, no nested
|
||||
`bash -c` here because the NVRTC worker is a native Windows exe (no WSL); `$env:IGNEUM_*` come from the app
|
||||
(`jobrun.rs` `job_env`); the script runs with its job folder as the working directory. `& $exe ... 2>&1 |
|
||||
ForEach-Object { "RESULT $_" }` keeps stderr in the report.
|
||||
|
||||
## The publish commands (main session)
|
||||
|
||||
The zip: `~/Desktop/igneum-worker-cuda-race.zip` (429,387 bytes, sha256
|
||||
`321a086ef47af05421d530ab251d170443cbafb917093113a316f617a7c4c049`; one file, `igneum-worker-cuda.exe`, 1,509,376
|
||||
bytes, built by `proto-cuda/nvrtc/build-windows.sh` from `miner-perf` at 32d1c01 plus the base-only short-circuit,
|
||||
icon and version block verified). `publish-jobs.sh add` hashes the copy it puts in the downloads folder; it must
|
||||
print this sha:
|
||||
|
||||
```bash
|
||||
cd ~/Projects/igneum # master or the miner-perf worktree: the script is the same
|
||||
# 1. the race worker (the fetch copies the zip into dl/<token>/ and hashes it)
|
||||
packaging/ota/publish-jobs.sh add --kind fetch --target ae432dc7 --platform windows \
|
||||
--file ~/Desktop/igneum-worker-cuda-race.zip --dir jobs --extract \
|
||||
--id fetch-race-worker-20261004 --title "Race build of the NVRTC worker for PC 1 (variant racing)" --expires-hours 48
|
||||
# 2. the race, miners stopped for its duration (about 5 min)
|
||||
packaging/ota/publish-jobs.sh add --kind run --target ae432dc7 --platform windows \
|
||||
--script relay/playbooks/race-5090.ps1 --stop-miners --timeout-minutes 20 \
|
||||
--id run-race-5090-20261004 --title "Variant race on the RTX 5090 (17 NVRTC variants, 3 rounds, twice)" --expires-hours 48 \
|
||||
--deploy
|
||||
# read back (the app polls every 10 minutes; Settings > remote jobs > Check now at once)
|
||||
node tools/jobs.mjs watch run-race-5090-20261004
|
||||
node tools/jobs.mjs run-race-5090-20261004 | grep '^RESULT'
|
||||
```
|
||||
|
||||
`--deploy` on the second `add` ships both (the first one leaves the file written, not deployed). The
|
||||
`--target` is PC 1 only. If `relay/playbooks/race-5090.ps1` is not on master yet, give its path in the
|
||||
`miner-perf` worktree (`~/Projects/igneum-wt-perf/relay/playbooks/race-5090.ps1`).
|
||||
|
||||
## What to read in the result
|
||||
|
||||
- `RESULT race <epoch16> device NVIDIA_GeForce_RTX_5090 driver ... variants 17 base=<MH/s>/<regs>r/1w w2=... winner <name> <MH/s> base <MH/s> gain <pct>% ...`,
|
||||
twice (run 1, run 2). The two winners should agree; the gain is the number for the bench log.
|
||||
- `| <variant>: <why>` after the line names any variant that was discarded (a compile error on sm_120, a vector
|
||||
mismatch) or not timed (budget).
|
||||
- `RESULT winner <name>: <regs> registers, <n> blocks/SM at <w> warp(s)/block`.
|
||||
- `RESULT gpu ...`: power limit (80% cap by the app's default) and clocks before; `gpu-after` after.
|
||||
|
||||
The gain on the 5090 is unknown until this runs. The Mac's Metal race (bench log) is a different compiler and a
|
||||
different memory system; its table says what the method finds there and nothing about NVIDIA. The range to
|
||||
measure on the 5090: from no gain (base stays, the compiler was already right for a 128-load random kernel) to
|
||||
whatever the load path and block variants give on a 1 GiB random-read kernel; the memory-hard kernel is
|
||||
bandwidth-bound by design, so a double-digit gain would be a surprise to check, not a claim.
|
||||
|
||||
## After the result
|
||||
|
||||
1. Add the two race lines to `docs/bench-log.md` under "miner performance: variant racing" (the PC 1 rows).
|
||||
2. If a variant wins twice: the race build of the worker goes into the next app version (it is the same
|
||||
`worker.cpp`; `build-windows.sh` then `packaging/windows/push-inputs.sh` as for any worker change), so every
|
||||
machine races at every prepare and logs the `TUNING` record; `node tools/tuning.mjs` shows the fleet table
|
||||
after a day; `node tools/tuning.mjs --write tuning.json` and `packaging/ota/publish-manifest.sh --version
|
||||
<current> --tuning tuning.json --deploy` send the per-card defaults back.
|
||||
3. If base wins both runs: the race costs the 5090 about 40 s of paused mining per hour for nothing on this
|
||||
program class; keep `--race on` for a day of records before deciding, since programs differ hour to hour.
|
||||
|
||||
## Untested
|
||||
|
||||
- The NVRTC worker's race has never run on a GPU (the Mac has none): the emulation suite covers the protocol and
|
||||
the base path only; the rewrites (`#pragma unroll`, `__ldg`/`__ldcg`/`__ldcs`, `__launch_bounds__`,
|
||||
`--maxrregcount`) are first compiled by the real NVRTC in this job. A variant NVRTC refuses is discarded with
|
||||
its reason in the line; the race still ends with a winner (base at worst).
|
||||
- The job's folder assumptions (`jobs\fetch-race-worker-20261004`, the per-user install folder for the DLLs) follow
|
||||
`jobrun.rs` and the 0.3.3 installer; PC 1 has not run a `fetch --dir jobs --extract` job before.
|
||||
|
|
@ -6,7 +6,14 @@
|
|||
#
|
||||
# packaging/ota/publish-manifest.sh --version 0.3.1 --mac packaging/mac/dist/Igneum-Miner-0.3.1.dmg \
|
||||
# [--win packaging/windows/dist/Igneum-Miner-Setup-0.3.1.exe] --notes "one line of what changed" \
|
||||
# [--activation-height 120000 --deadline-note "difficulty v2"] [--override '{"difficulty_v2_activation_daa":33000,"proving_v0_activation_daa":120000}'] [--min-supported 0.3.0] [--channel devnet] [--deploy]
|
||||
# [--activation-height 120000 --deadline-note "difficulty v2"] [--min-supported 0.3.0] [--channel devnet] [--deploy]
|
||||
# [--override '{"difficulty_v2_activation_daa":33000,"proving_v0_activation_daa":120000}']
|
||||
# consensus.override: the exact object every app writes to its override.json (the node's
|
||||
# --override-params-file), so it carries EVERY height switch, not just the new one;
|
||||
# carried over from the current manifest when not given
|
||||
# [--tuning tuning.json | --no-tuning] the fleet's per-card kernel tuning (tools/tuning.mjs writes it;
|
||||
# docs/design/miner-tuning.md); carried over from the current
|
||||
# manifest when not given, as is consensus.override
|
||||
#
|
||||
# A platform you do not pass is carried over from the manifest already in the folder when that one has the same
|
||||
# version (the Windows build lands later than the Mac one: publish the Mac entry first, add the Windows entry when
|
||||
|
|
@ -28,7 +35,8 @@ PUB="$HOME/.config/igneum/ota-signing-key.pub"
|
|||
TOKEN_FILE="$HOME/.config/igneum/dl-token"
|
||||
SIGNER="$ROOT/app/igneum-app/target/release/igneum-ota-sign"
|
||||
|
||||
VERSION="" MAC="" WIN="" NOTES="" ACTIVATION="" DEADLINE=""; OVERRIDE="" MIN_SUPPORTED="" CHANNEL="devnet" BASE="" DEST="" DEPLOY=0 VERIFY_ONLY=0 TRIES=12
|
||||
VERSION="" MAC="" WIN="" NOTES="" ACTIVATION="" DEADLINE="" MIN_SUPPORTED="" CHANNEL="devnet" BASE="" DEST="" DEPLOY=0 VERIFY_ONLY=0 TRIES=12
|
||||
OVERRIDE="" TUNING_FILE="" NO_TUNING=0
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--version) VERSION="$2"; shift 2 ;;
|
||||
|
|
@ -40,6 +48,8 @@ while [ $# -gt 0 ]; do
|
|||
--override) OVERRIDE="$2"; shift 2 ;; # consensus.override: the exact JSON object every app writes to its override.json (all height switches, not just the new one)
|
||||
--min-supported) MIN_SUPPORTED="$2"; shift 2 ;;
|
||||
--channel) CHANNEL="$2"; shift 2 ;;
|
||||
--tuning) TUNING_FILE="$2"; shift 2 ;;
|
||||
--no-tuning) NO_TUNING=1; shift ;;
|
||||
--base-url) BASE="$2"; shift 2 ;;
|
||||
--dest) DEST="$2"; shift 2 ;;
|
||||
--deploy) DEPLOY=1; shift ;;
|
||||
|
|
@ -143,14 +153,27 @@ fi
|
|||
if [ -z "$MIN_SUPPORTED" ] && [ -f "$OLD" ]; then
|
||||
MIN_SUPPORTED="$(python3 -c 'import json,sys; print(json.load(open(sys.argv[1])).get("min_supported_version",""))' "$OLD" 2>/dev/null || true)"
|
||||
fi
|
||||
# consensus.override and tuning: given here, else carried over from the current manifest (whatever its version)
|
||||
if [ -z "$OVERRIDE" ] && [ -f "$OLD" ]; then
|
||||
OVERRIDE="$(python3 -c 'import json,sys; o=json.load(open(sys.argv[1])).get("consensus",{}).get("override"); print(json.dumps(o, sort_keys=True, separators=(",",":")) if isinstance(o, dict) and o else "")' "$OLD" 2>/dev/null || true)"
|
||||
[ -n "$OVERRIDE" ] && echo "consensus.override: carried over from the current manifest: $OVERRIDE"
|
||||
fi
|
||||
TUNING=""
|
||||
if [ -n "$TUNING_FILE" ]; then
|
||||
[ -f "$TUNING_FILE" ] || { echo "missing: $TUNING_FILE" >&2; exit 1; }
|
||||
TUNING="$(python3 -c 'import json,sys; t=json.load(open(sys.argv[1])); assert isinstance(t.get("cards"), dict), "tuning needs a cards object"; print(json.dumps(t, sort_keys=True, separators=(",",":")))' "$TUNING_FILE")"
|
||||
elif [ "$NO_TUNING" = 0 ] && [ -f "$OLD" ]; then
|
||||
TUNING="$(python3 -c 'import json,sys; t=json.load(open(sys.argv[1])).get("tuning"); print(json.dumps(t, sort_keys=True, separators=(",",":")) if isinstance(t, dict) and isinstance(t.get("cards"), dict) else "")' "$OLD" 2>/dev/null || true)"
|
||||
[ -n "$TUNING" ] && echo "tuning: carried over from the current manifest ($(python3 -c 'import json,sys; print(len(json.loads(sys.argv[1])["cards"]))' "$TUNING") card model(s))"
|
||||
fi
|
||||
|
||||
# canonical JSON: sorted keys, no whitespace; the signature is over these exact bytes
|
||||
NEW="$DEST/igneum-app-latest.json.new"
|
||||
python3 - "$NEW" "$VERSION" "$CHANNEL" "$NOTES" "$MIN_SUPPORTED" "$ACTIVATION" "$DEADLINE" "$MAC_ENTRY" "$WIN_ENTRY" "${OVERRIDE:-}" <<'PY'
|
||||
python3 - "$NEW" "$VERSION" "$CHANNEL" "$NOTES" "$MIN_SUPPORTED" "$ACTIVATION" "$DEADLINE" "$MAC_ENTRY" "$WIN_ENTRY" "${OVERRIDE:-}" "${TUNING:-}" <<'PY'
|
||||
import json, sys, datetime
|
||||
out, version, channel, notes, min_supported, activation, deadline, mac, win = sys.argv[1:10]
|
||||
override = json.loads(sys.argv[10]) if len(sys.argv) > 10 and sys.argv[10] else None
|
||||
if override is not None and not isinstance(override, dict): raise SystemExit("--override must be a JSON object")
|
||||
out, version, channel, notes, min_supported, activation, deadline, mac, win, override, tuning = sys.argv[1:12]
|
||||
override = json.loads(override) if override else None
|
||||
if override is not None and (not isinstance(override, dict) or not override): raise SystemExit("--override must be a non-empty JSON object")
|
||||
def entry(s):
|
||||
if not s: return None
|
||||
url, sha, size, kind = s.split()
|
||||
|
|
@ -164,6 +187,8 @@ m = {
|
|||
"notes": notes,
|
||||
"consensus": {"activation_height": int(activation) if activation else None, "deadline_note": deadline, **({"override": override} if override is not None else {})},
|
||||
}
|
||||
if tuning:
|
||||
m["tuning"] = json.loads(tuning)
|
||||
open(out, "w").write(json.dumps(m, sort_keys=True, separators=(",", ":"), ensure_ascii=False))
|
||||
PY
|
||||
"$SIGNER" sign "$KEY" "$NEW" > "$NEW.sig"
|
||||
|
|
|
|||
|
|
@ -30,8 +30,23 @@
|
|||
// and the three vector warps of vectors.h through the bound kernel with the pack's own seed words. A pack that fails
|
||||
// is refused.
|
||||
//
|
||||
// Variant racing (4 October 2026, evening; docs/design/miner-tuning.md): every pack's bound kernel is compiled in
|
||||
// several variants (loop unrolling, the dataset load path: plain, __ldg, __ldcg, __ldcs; a register budget through
|
||||
// -maxrregcount or __launch_bounds__; threads per block), each self-tested against the pack's vectors (bit-exact or
|
||||
// discarded) and run for about two seconds on the card; the fastest serves the hour. A race runs inside the prepare
|
||||
// (the hourly compile-ahead, one lead before the boundary) and never delays the swap: it has a time budget, "base"
|
||||
// (the pack's text as shipped) is always the first entry, and a prepare that runs out of budget keeps the best so far.
|
||||
// While a variant is timed the job loop pauses (one mutex): the numbers are exclusive, mining resumes between
|
||||
// variants. One line per race: `race <epoch16> device <name> ... variants N a=MH/s b=MH/s ... winner <name> <MH/s>
|
||||
// gain <pct> ...`. A tuning file (--tuning, or IGNEUM_TUNING_FILE from the app) may pin a variant for this card
|
||||
// model or order the candidates; the app's over-the-air manifest carries it (fleet learning). Under IGNEUM_EMU the
|
||||
// race is off (the stand-in checks that the handed-over text equals the pack's).
|
||||
//
|
||||
// Usage: igneum-worker-cuda --serve --pack <dir> [--device D] [--batch-log2 22] [--block-warps 1] [--arch sm_120|auto]
|
||||
// [--race on|off|<name,name,...>] [--race-bench-ms 2000] [--race-budget-s 120] [--race-rounds 1]
|
||||
// [--variant <name>] [--tuning <file>]
|
||||
// igneum-worker-cuda --check --pack <dir> [--device D] compile, build, self-test, print timings, exit 0/1
|
||||
// igneum-worker-cuda --race --pack <dir> [--device D] [--race-rounds 3] the race alone: one line per variant, exit 0/1
|
||||
|
||||
#include <cstdint>
|
||||
#include <cstdarg>
|
||||
|
|
@ -43,6 +58,8 @@
|
|||
#include <vector>
|
||||
#include <thread>
|
||||
#include <atomic>
|
||||
#include <mutex>
|
||||
#include <algorithm>
|
||||
#include <iostream>
|
||||
|
||||
#include "cuda_api.h"
|
||||
|
|
@ -233,6 +250,11 @@ struct Ctx {
|
|||
bool ptx = false; // true when archOpt is compute_XY (PTX, driver JIT)
|
||||
std::string why; // how archOpt was chosen
|
||||
int blockWarps = 1;
|
||||
// variant racing (see the header): which variants, how long each is timed, the budget of a race, rounds
|
||||
std::string race = "on"; // on | off | comma list of variant names
|
||||
int raceBenchMs = 2000, raceBudgetS = 120, raceRounds = 1, batchLog2 = 22;
|
||||
std::string pinned; // --variant: use this variant, no race
|
||||
std::string tuning; // the tuning file's text ("" = none)
|
||||
std::string err(CUresult r) { const char* s = nullptr; if (drv.getErrorString) drv.getErrorString(r, &s); return s ? s : "CUDA driver error"; }
|
||||
};
|
||||
|
||||
|
|
@ -317,7 +339,7 @@ struct Compiled {
|
|||
};
|
||||
|
||||
static bool rtcCompile(Ctx& c, const std::string& src, const char* name, const std::string& programH, const std::string& memhardH,
|
||||
const std::vector<std::string>& nameExprs, Compiled& out, std::string& err) {
|
||||
const std::vector<std::string>& nameExprs, Compiled& out, std::string& err, const std::vector<std::string>& extraOpts = {}) {
|
||||
double t0 = wallMs();
|
||||
const char* headers[4] = { STUB_CUDA_RUNTIME, STUB_CSTDINT, programH.c_str(), memhardH.c_str() };
|
||||
const char* names[4] = { "cuda_runtime.h", "cstdint", "program.h", "memhard.h" };
|
||||
|
|
@ -331,8 +353,9 @@ static bool rtcCompile(Ctx& c, const std::string& src, const char* name, const s
|
|||
std::string archOpt = "--gpu-architecture=" + c.archOpt;
|
||||
// -default-device: NVRTC rejects unannotated functions as host code (nvcc treats them as host and discards them);
|
||||
// the pack headers (program.h, memhard.h) carry plain inline helpers, so every unannotated function is device code here.
|
||||
const char* opts[3] = { archOpt.c_str(), "--std=c++17", "-default-device" };
|
||||
r = c.rtc.compileProgram(prog, 3, opts);
|
||||
std::vector<const char*> opts = { archOpt.c_str(), "--std=c++17", "-default-device" };
|
||||
for (const std::string& o : extraOpts) opts.push_back(o.c_str());
|
||||
r = c.rtc.compileProgram(prog, (int)opts.size(), opts.data());
|
||||
{
|
||||
size_t logSize = 0;
|
||||
if (c.rtc.getProgramLogSize(prog, &logSize) == NVRTC_SUCCESS && logSize > 1) {
|
||||
|
|
@ -382,8 +405,16 @@ struct Pair {
|
|||
std::string check;
|
||||
bool checkPass = false, checked = false;
|
||||
int regs = 0, blocksPerSM = 0;
|
||||
int blockWarps = 1; // threads per block = 32 x this (the winning variant's, else the worker's default)
|
||||
std::string variant = "base"; // the bound kernel in service: a variant name (see allVariants)
|
||||
std::string raceLine; // the race's one-line report, emitted by the main thread with "prepared"
|
||||
double raceMs = 0;
|
||||
};
|
||||
|
||||
// The job loop and a race take turns on the card: a variant is timed with no job running (exclusive numbers), and
|
||||
// mining resumes between variants. Held per chunk by the job loop, per variant by the race.
|
||||
static std::mutex gpuMutex;
|
||||
|
||||
static void releasePair(Ctx& c, Pair* p) {
|
||||
if (!p) return;
|
||||
if (p->ds) c.drv.memFree(p->ds);
|
||||
|
|
@ -404,9 +435,309 @@ static bool launchHash(Ctx& c, Pair* p, CUdeviceptr out, uint32_t baseNonce, con
|
|||
return true;
|
||||
}
|
||||
|
||||
// Compiles the pack in `dir`, builds its cache and dataset on stream `s`, runs the self-test. Returns the pair or
|
||||
// null with `err` set. Runs on the main thread for --pack and --check, on the prepare thread for `prepare`.
|
||||
static Pair* buildPair(Ctx& c, const std::string& dir, CUstream s, std::string& err) {
|
||||
|
||||
// ---------------------------------------------------------------------------------------------
|
||||
// Variant racing
|
||||
|
||||
struct Variant {
|
||||
std::string name;
|
||||
int unroll = 0; // 0: the iteration loop as emitted; N: "#pragma unroll N" before it (8 = fully unrolled)
|
||||
int load = 0; // 0: plain ds[i]; 1: __ldg (read-only data path); 2: __ldcg (L2 only, no L1); 3: __ldcs (streaming)
|
||||
int maxrreg = 0; // 0: none; N: --maxrregcount=N (registers per thread, occupancy against spills)
|
||||
int blockWarps = 0; // 0: the worker's --block-warps; N: 32 x N threads per block
|
||||
int minBlocks = 0; // N > 0: __launch_bounds__(32 x blockWarps, N) (the compiler fits N blocks per SM)
|
||||
};
|
||||
|
||||
// The catalogue. Names are stable: the tuning file and the fleet records use them. "base" is the pack's text as
|
||||
// shipped with the worker's default block and is always the first entry of a race.
|
||||
static std::vector<Variant> allVariants() {
|
||||
std::vector<Variant> v;
|
||||
auto add = [&](const char* n, int unroll, int load, int maxrreg, int bw, int minBlocks) { Variant x; x.name = n; x.unroll = unroll; x.load = load; x.maxrreg = maxrreg; x.blockWarps = bw; x.minBlocks = minBlocks; v.push_back(x); };
|
||||
add("base", 0, 0, 0, 0, 0);
|
||||
add("w2", 0, 0, 0, 2, 0);
|
||||
add("w4", 0, 0, 0, 4, 0);
|
||||
add("w8", 0, 0, 0, 8, 0);
|
||||
add("u2", 2, 0, 0, 0, 0);
|
||||
add("u8", 8, 0, 0, 0, 0);
|
||||
add("ldg", 0, 1, 0, 0, 0);
|
||||
add("ldcg", 0, 2, 0, 0, 0);
|
||||
add("ldcs", 0, 3, 0, 0, 0);
|
||||
add("r32", 0, 0, 32, 0, 0);
|
||||
add("r64", 0, 0, 64, 0, 0);
|
||||
add("lb4-w4", 0, 0, 0, 4, 4);
|
||||
add("lb8-w2", 0, 0, 0, 2, 8);
|
||||
add("u2-ldg", 2, 1, 0, 0, 0);
|
||||
add("u2-w4", 2, 0, 0, 4, 0);
|
||||
add("ldg-w4", 0, 1, 0, 4, 0);
|
||||
add("ldcg-w4", 0, 2, 0, 4, 0);
|
||||
return v;
|
||||
}
|
||||
|
||||
static const Variant* findVariant(const std::vector<Variant>& all, const std::string& name) {
|
||||
for (const Variant& v : all) if (v.name == name) return &v;
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
// The variant's source: the pack's bound-kernel text with the variant's rewrites. Every rewrite has an exact anchor
|
||||
// in the text igneum-pow emits; a text without the anchor refuses the variant (why), it is never guessed.
|
||||
static bool variantSource(const std::string& base, const Variant& v, int blockWarps, std::string& out, std::string& why) {
|
||||
out = base;
|
||||
if (v.unroll > 0) {
|
||||
const char* anchor = "\n for (uint32_t it = 0u; it < ";
|
||||
size_t p = out.find(anchor);
|
||||
if (p == std::string::npos) { why = "no iteration loop in the bound kernel text"; return false; }
|
||||
out.insert(p + 1, fmt("#pragma unroll %d\n", v.unroll));
|
||||
}
|
||||
if (v.load > 0) {
|
||||
const char* fn = v.load == 1 ? "__ldg" : v.load == 2 ? "__ldcg" : "__ldcs";
|
||||
size_t body = out.find("igneum_hash_bound(");
|
||||
if (body == std::string::npos) { why = "no igneum_hash_bound in the text"; return false; }
|
||||
size_t p = body; int n = 0;
|
||||
while ((p = out.find(" ^ ds[", p)) != std::string::npos) {
|
||||
size_t close = out.find(']', p);
|
||||
if (close == std::string::npos) { why = "an unterminated dataset load"; return false; }
|
||||
out.insert(close + 1, ")"); // " ^ ds[idx]" -> " ^ __ldg(&ds[idx])"
|
||||
out.insert(p + 3, std::string(fn) + "(&");
|
||||
p += 6; ++n;
|
||||
}
|
||||
if (n == 0) { why = "no dataset loads in the bound kernel"; return false; }
|
||||
}
|
||||
if (v.minBlocks > 0) {
|
||||
const char* a = "__global__ void igneum_hash_bound(";
|
||||
size_t p = out.find(a);
|
||||
if (p == std::string::npos) { why = "no kernel declaration anchor"; return false; }
|
||||
out.replace(p, std::strlen(a), fmt("__global__ void __launch_bounds__(%d, %d) igneum_hash_bound(", 32 * blockWarps, v.minBlocks));
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
// The tuning file: {"cards": {"<device name as this worker prints it>": {"variant": "u2-ldg", "race": false,
|
||||
// "candidates": ["u2-ldg", "ldg", "base"]}}, ...}. Read with plain string scanning (no JSON library in this exe);
|
||||
// a file that does not parse means no tuning. Keys and names are [A-Za-z0-9_.-].
|
||||
struct Tuning {
|
||||
bool found = false;
|
||||
std::string variant; // pinned variant ("" = none)
|
||||
bool race = true; // false: use the pinned variant without a race
|
||||
std::vector<std::string> candidates;
|
||||
};
|
||||
|
||||
static std::string jsonStringAfter(const std::string& t, size_t from, const char* key, size_t limit) {
|
||||
size_t k = t.find(std::string("\"") + key + "\"", from);
|
||||
if (k == std::string::npos || k > limit) return "";
|
||||
size_t q = t.find('"', t.find(':', k) + 1);
|
||||
if (q == std::string::npos) return "";
|
||||
size_t e = t.find('"', q + 1);
|
||||
return e == std::string::npos ? "" : t.substr(q + 1, e - q - 1);
|
||||
}
|
||||
|
||||
static Tuning readTuning(const std::string& text, const std::string& device) {
|
||||
Tuning tu;
|
||||
if (text.empty()) return tu;
|
||||
size_t cards = text.find("\"cards\"");
|
||||
if (cards == std::string::npos) return tu;
|
||||
size_t k = text.find("\"" + device + "\"", cards);
|
||||
if (k == std::string::npos) return tu;
|
||||
size_t open = text.find('{', k);
|
||||
if (open == std::string::npos) return tu;
|
||||
size_t close = open; int depth = 0;
|
||||
for (; close < text.size(); ++close) { if (text[close] == '{') ++depth; else if (text[close] == '}' && --depth == 0) break; }
|
||||
if (close >= text.size()) return tu;
|
||||
tu.found = true;
|
||||
tu.variant = jsonStringAfter(text, open, "variant", close);
|
||||
size_t r = text.find("\"race\"", open);
|
||||
if (r != std::string::npos && r < close) { size_t c = text.find(':', r); tu.race = text.compare(text.find_first_not_of(" \t\r\n", c + 1), 5, "false") != 0; }
|
||||
size_t cand = text.find("\"candidates\"", open);
|
||||
if (cand != std::string::npos && cand < close) {
|
||||
size_t a = text.find('[', cand), b = text.find(']', a == std::string::npos ? cand : a);
|
||||
if (a != std::string::npos && b != std::string::npos && b < close) {
|
||||
size_t i = a;
|
||||
while ((i = text.find('"', i + 1)) != std::string::npos && i < b) { size_t e = text.find('"', i + 1); if (e == std::string::npos || e > b) break; tu.candidates.push_back(text.substr(i + 1, e - i - 1)); i = e; }
|
||||
}
|
||||
}
|
||||
return tu;
|
||||
}
|
||||
|
||||
struct RaceEntry {
|
||||
Variant v;
|
||||
int blockWarps = 1; // the block this entry runs with
|
||||
Compiled cb;
|
||||
CUmodule mod = nullptr;
|
||||
CUfunction fn = nullptr;
|
||||
int regs = 0, blocksPerSM = 0;
|
||||
double mhs = 0; // best round
|
||||
bool ok = false; // compiled, loaded, self-tested
|
||||
std::string note; // why not, or a detail
|
||||
std::string src;
|
||||
};
|
||||
|
||||
// One timed window on the card for an entry: the pack's vector warps (bit-exact or the entry is out), then
|
||||
// launches of `batch` nonces until benchMs elapsed (the first launch warms up and is not counted). Holds gpuMutex.
|
||||
static bool raceTime(Ctx& c, Pair* p, const PfPack& pk, RaceEntry& e, CUdeviceptr dOut, uint32_t batch, int benchMs, CUstream s, bool selfTest) {
|
||||
std::lock_guard<std::mutex> hold(gpuMutex);
|
||||
CUfunction keep = p->fHashBound;
|
||||
p->fHashBound = e.fn;
|
||||
std::string err;
|
||||
uint32_t block = 32u * (uint32_t)e.blockWarps;
|
||||
bool ok = true;
|
||||
if (selfTest && pk.haveVectors) {
|
||||
std::vector<uint64_t> vec(32);
|
||||
for (int w = 0; w < pk.vecWarps && ok; ++w) {
|
||||
if (!launchHash(c, p, dOut, pk.vecBase[w], p->sw, block, block, s, err)) { e.note = "launch: " + err; ok = false; break; }
|
||||
CUresult r = c.drv.streamSynchronize(s);
|
||||
if (r == CUDA_SUCCESS) r = c.drv.memcpyDtoH(vec.data(), dOut, 32u * 8u);
|
||||
if (r != CUDA_SUCCESS) { e.note = "vector warp: " + c.err(r); ok = false; break; }
|
||||
for (int l = 0; l < 32; ++l) if (vec[(size_t)l] != pk.vecOut[w][l]) { e.note = fmt("vector warp %d lane %d: device %016llx expected %016llx (discarded)", w, l, (unsigned long long)vec[(size_t)l], (unsigned long long)pk.vecOut[w][l]); ok = false; break; }
|
||||
}
|
||||
}
|
||||
if (ok) {
|
||||
uint32_t iw[8]; std::memcpy(iw, p->sw, 32);
|
||||
uint32_t n = batch - (batch % block);
|
||||
if (n == 0) n = block;
|
||||
double t0 = 0; uint64_t hashes = 0; int launches = 0;
|
||||
while (true) {
|
||||
if (!launchHash(c, p, dOut, 0x10000000u + (uint32_t)launches * n, iw, n, block, s, err)) { e.note = "launch: " + err; ok = false; break; }
|
||||
CUresult r = c.drv.streamSynchronize(s);
|
||||
if (r != CUDA_SUCCESS) { e.note = "bench: " + c.err(r); ok = false; break; }
|
||||
double now = wallMs();
|
||||
if (launches == 0) t0 = now; else hashes += n;
|
||||
++launches;
|
||||
if (launches >= 3 && now - t0 >= benchMs) { double mhs = (double)hashes / (now - t0) / 1000.0; if (mhs > e.mhs) e.mhs = mhs; break; }
|
||||
}
|
||||
}
|
||||
p->fHashBound = keep;
|
||||
return ok;
|
||||
}
|
||||
|
||||
// Races the bound kernel of `p` (its cache and dataset are built, its base kernel self-tested) and installs the
|
||||
// winner: p->modBound, fHashBound, regs, blockWarps, variant. The pair keeps serving its base kernel if every other
|
||||
// entry fails. `boundDev`, `programH`, `memhardH` are the texts the base was compiled from. Sets p->raceLine.
|
||||
static void racePair(Ctx& c, Pair* p, const PfPack& pk, const std::string& boundDev, const std::string& programH, const std::string& memhardH, CUstream s) {
|
||||
double t0 = wallMs();
|
||||
std::vector<Variant> all = allVariants();
|
||||
Tuning tu = readTuning(c.tuning, c.name);
|
||||
std::string pinned = !c.pinned.empty() ? c.pinned : (tu.found && !tu.race ? tu.variant : "");
|
||||
// the order: base first, then the pinned or tuned candidates, then the rest (or the --race list only)
|
||||
std::vector<Variant> order;
|
||||
auto push = [&](const std::string& n) { const Variant* v = findVariant(all, n); if (v && !findVariant(order, n)) order.push_back(*v); };
|
||||
push("base");
|
||||
if (!pinned.empty()) push(pinned);
|
||||
else {
|
||||
for (const std::string& n : tu.candidates) push(n);
|
||||
if (c.race != "on" && c.race != "off") { std::string rest = c.race; size_t i = 0; while (i <= rest.size()) { size_t j = rest.find(',', i); if (j == std::string::npos) j = rest.size(); if (j > i) push(rest.substr(i, j - i)); i = j + 1; } }
|
||||
else if (c.race == "on") for (const Variant& v : all) push(v.name);
|
||||
}
|
||||
#ifdef IGNEUM_EMU
|
||||
order.resize(1); // the stand-in checks that the handed-over text is the pack's; no rewrites under emulation
|
||||
#endif
|
||||
if (order.size() < 2) {
|
||||
// nothing to race against base (emulation, or --race with no known name): no timing, the base kernel serves
|
||||
p->blockWarps = c.blockWarps; p->variant = "base"; p->raceMs = wallMs() - t0;
|
||||
p->raceLine = fmt("race %.16s device %s variants 1 base only, no race (%s)", p->epochHex.c_str(), c.name.c_str(),
|
||||
#ifdef IGNEUM_EMU
|
||||
"emulation");
|
||||
#else
|
||||
"no other variant named");
|
||||
#endif
|
||||
return;
|
||||
}
|
||||
const bool pinnedOnly = !pinned.empty() && order.size() == 2;
|
||||
uint32_t batch = 1u << c.batchLog2;
|
||||
int benchMs = c.raceBenchMs;
|
||||
double deadline = t0 + c.raceBudgetS * 1000.0;
|
||||
std::vector<RaceEntry> entries;
|
||||
for (const Variant& v : order) {
|
||||
RaceEntry e; e.v = v; e.blockWarps = v.blockWarps > 0 ? v.blockWarps : c.blockWarps;
|
||||
if (v.name == "base") { e.mod = p->modBound; e.fn = p->fHashBound; e.regs = p->regs; e.blocksPerSM = p->blocksPerSM; e.ok = true; }
|
||||
else if (!variantSource(boundDev, v, e.blockWarps, e.src, e.note)) e.ok = false;
|
||||
else e.ok = true; // compiled below
|
||||
entries.push_back(std::move(e));
|
||||
}
|
||||
// Compile the variants, up to four at a time (NVRTC is thread-safe; the compile is CPU work)
|
||||
{
|
||||
std::vector<size_t> todo;
|
||||
for (size_t i = 1; i < entries.size(); ++i) if (entries[i].ok) todo.push_back(i);
|
||||
size_t next = 0;
|
||||
std::mutex m;
|
||||
auto work = [&]() {
|
||||
while (true) {
|
||||
size_t i;
|
||||
{ std::lock_guard<std::mutex> g(m); if (next >= todo.size() || wallMs() > deadline - benchMs) return; i = todo[next++]; }
|
||||
RaceEntry& e = entries[i];
|
||||
std::vector<std::string> extra;
|
||||
if (e.v.maxrreg > 0) extra.push_back(fmt("--maxrregcount=%d", e.v.maxrreg));
|
||||
std::string err;
|
||||
if (!rtcCompile(c, e.src, "kernel_bound.cu", programH, memhardH, { "igneum_hash_bound" }, e.cb, err, extra)) { e.ok = false; e.note = "compile: " + err.substr(0, 200); }
|
||||
}
|
||||
};
|
||||
int threads = (int)std::min<size_t>(4, std::max<size_t>(1, todo.size()));
|
||||
std::vector<std::thread> ts;
|
||||
for (int t = 0; t < threads; ++t) ts.emplace_back(work);
|
||||
for (std::thread& t : ts) t.join();
|
||||
for (size_t i = 1; i < entries.size(); ++i) if (entries[i].ok && entries[i].cb.image.empty()) { entries[i].ok = false; entries[i].note = "not compiled: the race budget ran out"; }
|
||||
}
|
||||
// Load the modules (the context is current on this thread)
|
||||
for (size_t i = 1; i < entries.size(); ++i) {
|
||||
RaceEntry& e = entries[i];
|
||||
if (!e.ok) continue;
|
||||
CUresult r = c.drv.moduleLoadData(&e.mod, e.cb.image.data());
|
||||
if (r != CUDA_SUCCESS) { e.ok = false; e.note = "cuModuleLoadData: " + c.err(r); e.mod = nullptr; continue; }
|
||||
if (c.drv.moduleGetFunction(&e.fn, e.mod, e.cb.lowered[0].c_str()) != CUDA_SUCCESS) { e.ok = false; e.note = "function not in the module"; continue; }
|
||||
c.drv.funcGetAttribute(&e.regs, CU_FUNC_ATTRIBUTE_NUM_REGS, e.fn);
|
||||
c.drv.occupancy(&e.blocksPerSM, e.fn, 32 * e.blockWarps, 0);
|
||||
}
|
||||
double compileMs = wallMs() - t0;
|
||||
// Time them: rounds over the entries, interleaved, best per entry. A pinned variant is only self-tested.
|
||||
CUdeviceptr dOut = 0;
|
||||
std::string benchErr;
|
||||
if (c.drv.memAlloc(&dOut, (size_t)batch * 8u) != CUDA_SUCCESS) { benchErr = "cuMemAlloc for the race"; for (RaceEntry& e : entries) if (e.v.name != "base") e.ok = false; }
|
||||
int rounds = pinnedOnly ? 1 : std::max(1, c.raceRounds);
|
||||
for (int round = 0; round < rounds && benchErr.empty(); ++round) {
|
||||
for (size_t i = 0; i < entries.size(); ++i) {
|
||||
RaceEntry& e = entries[i];
|
||||
if (!e.ok) continue;
|
||||
if (i > 0 && round == 0 && wallMs() > deadline) { e.ok = false; e.note = "not timed: the race budget ran out"; continue; }
|
||||
if (pinnedOnly && i == 0) continue;
|
||||
if (!raceTime(c, p, pk, e, dOut, batch, pinnedOnly ? 0 : benchMs, s, round == 0)) e.ok = false;
|
||||
// the mutex is not fair: give the job loop the card between windows (measured on the Mac, 4 October
|
||||
// 2026: without this a queued job waited the whole race, 36 s)
|
||||
std::this_thread::sleep_for(std::chrono::milliseconds(150));
|
||||
}
|
||||
}
|
||||
if (dOut) c.drv.memFree(dOut);
|
||||
// The winner: the fastest entry; base keeps its place unless a variant is at least 0.5% faster (noise guard).
|
||||
size_t win = 0;
|
||||
if (pinnedOnly && entries.size() == 2 && entries[1].ok) win = 1;
|
||||
else for (size_t i = 1; i < entries.size(); ++i) if (entries[i].ok && entries[i].mhs > entries[win].mhs * (win == 0 ? 1.005 : 1.0)) win = i;
|
||||
double baseMhs = entries[0].mhs, winMhs = entries[win].mhs;
|
||||
if (win != 0) {
|
||||
RaceEntry& w = entries[win];
|
||||
c.drv.moduleUnload(p->modBound);
|
||||
p->modBound = w.mod; p->fHashBound = w.fn; p->regs = w.regs; p->blocksPerSM = w.blocksPerSM; p->blockWarps = w.blockWarps; p->variant = w.v.name;
|
||||
w.mod = nullptr;
|
||||
} else {
|
||||
p->blockWarps = c.blockWarps; p->variant = "base";
|
||||
}
|
||||
for (size_t i = 1; i < entries.size(); ++i) if (entries[i].mod) c.drv.moduleUnload(entries[i].mod);
|
||||
p->raceMs = wallMs() - t0;
|
||||
// The one line. Variants in race order: name=MH/s (regs), or name=- (why).
|
||||
uint32_t loads = 0, wide = 0;
|
||||
pf_define_u32(programH.c_str(), "IGNEUM_LOADS_PER_HASH", &loads);
|
||||
pf_define_u32(programH.c_str(), "IGNEUM_WIDE_LOADS_PER_HASH", &wide);
|
||||
std::string line = fmt("race %.16s device %s driver %d.%d arch %s loads %u wide %u variants %zu", p->epochHex.c_str(), c.name.c_str(), c.driverVersion / 1000, (c.driverVersion % 100) / 10, c.archOpt.c_str(), loads, wide, entries.size());
|
||||
for (const RaceEntry& e : entries) {
|
||||
if (e.ok && (e.mhs > 0 || pinnedOnly)) line += fmt(" %s=%.3f/%dr/%dw", e.v.name.c_str(), e.mhs, e.regs, e.blockWarps);
|
||||
else line += fmt(" %s=-", e.v.name.c_str());
|
||||
}
|
||||
line += fmt(" winner %s %.3f base %.3f gain %+.2f%% compile %.0f bench %.0f total %.0f ms%s%s", p->variant.c_str(), winMhs, baseMhs, baseMhs > 0 ? (winMhs / baseMhs - 1.0) * 100.0 : 0.0, compileMs, p->raceMs - compileMs, p->raceMs,
|
||||
pinnedOnly ? " pinned by tuning" : (tu.found ? " tuned order" : ""), benchErr.empty() ? "" : (" " + benchErr).c_str());
|
||||
for (const RaceEntry& e : entries) if (!e.ok && !e.note.empty()) line += " | " + e.v.name + ": " + e.note;
|
||||
p->raceLine = line;
|
||||
}
|
||||
|
||||
// Compiles the pack in `dir`, builds its cache and dataset on stream `s`, runs the self-test, races the variants
|
||||
// (`race`). Returns the pair or null with `err` set. Runs on the main thread for --pack and --check, on the prepare
|
||||
// thread for `prepare`.
|
||||
static Pair* buildPair(Ctx& c, const std::string& dir, CUstream s, std::string& err, bool race) {
|
||||
PfPack pk;
|
||||
char perr[512];
|
||||
if (!pf_load(dir.c_str(), &pk, perr, sizeof(perr))) { err = std::string("pack ") + dir + ": " + perr; return nullptr; }
|
||||
|
|
@ -504,11 +835,13 @@ static Pair* buildPair(Ctx& c, const std::string& dir, CUstream s, std::string&
|
|||
}
|
||||
p->checkMs = wallMs() - t0;
|
||||
if (!p->checkPass) { err = p->check; releasePair(c, p); return nullptr; }
|
||||
p->blockWarps = c.blockWarps;
|
||||
if (race && c.race != "off") racePair(c, p, pk, boundDev, programH, memhardH, s);
|
||||
return p;
|
||||
}
|
||||
|
||||
static std::string pairSummary(const Pair* p) {
|
||||
return fmt("nvrtc %.0f cache %.0f dataset %.0f check %.0f ms; %s", p->compileMs, p->cacheMs, p->dsMs, p->checkMs, p->check.c_str());
|
||||
return fmt("nvrtc %.0f cache %.0f dataset %.0f check %.0f race %.0f ms variant %s; %s", p->compileMs, p->cacheMs, p->dsMs, p->checkMs, p->raceMs, p->variant.c_str(), p->check.c_str());
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------------------------
|
||||
|
|
@ -527,7 +860,7 @@ static void prepareRun(Ctx* c, PrepareTask* t) {
|
|||
std::string err;
|
||||
if (c->drv.ctxSetCurrent(c->ctx) != CUDA_SUCCESS) { t->error = "cuCtxSetCurrent on the prepare thread"; t->done = true; return; }
|
||||
if (c->drv.streamCreate(&s, CU_STREAM_NON_BLOCKING) != CUDA_SUCCESS) { t->error = "cuStreamCreate on the prepare thread"; t->done = true; return; }
|
||||
Pair* p = buildPair(*c, t->dir, s, err);
|
||||
Pair* p = buildPair(*c, t->dir, s, err, true);
|
||||
c->drv.streamDestroy(s);
|
||||
if (p && (p->epochHex != t->epochHex || p->dayHex != t->dayHex)) {
|
||||
err = "the pack in " + t->dir + " is for epoch " + p->epochHex.substr(0, 16) + " day " + p->dayHex + ", not the prepared seeds";
|
||||
|
|
@ -542,9 +875,11 @@ static void prepareRun(Ctx* c, PrepareTask* t) {
|
|||
// Serve
|
||||
|
||||
struct Options {
|
||||
bool serve = false, check = false;
|
||||
bool serve = false, check = false, raceOnly = false;
|
||||
int device = 0, batchLog2 = 22, blockWarps = 1;
|
||||
std::string pack, arch = "auto";
|
||||
std::string race = "on", pinned, tuningPath;
|
||||
int raceBenchMs = 2000, raceBudgetS = 120, raceRounds = 0; // rounds 0 = 1 in --serve, 3 in --race
|
||||
};
|
||||
|
||||
static void usage() {
|
||||
|
|
@ -554,7 +889,14 @@ static void usage() {
|
|||
" --device D CUDA device index (default 0)\n"
|
||||
" --batch-log2 B nonces per dispatch = 2^B (default 22)\n"
|
||||
" --block-warps W warps per thread block (default 1)\n"
|
||||
" --arch sm_XY|compute_XY|auto NVRTC target (default auto: the device's architecture)\n", WORKER_VERSION);
|
||||
" --arch sm_XY|compute_XY|auto NVRTC target (default auto: the device's architecture)\n"
|
||||
" --race --pack <dir> the variant race alone (3 rounds): one line per variant, the race line, exit 0 or 1\n"
|
||||
" --race on|off|a,b,c in --serve: race every variant (default), none, or these names\n"
|
||||
" --race-bench-ms N timed window per variant (default 2000)\n"
|
||||
" --race-budget-s N a race stops compiling and timing after this (default 120; base is kept)\n"
|
||||
" --race-rounds N interleaved rounds, best per variant (default 1 in --serve, 3 in --race)\n"
|
||||
" --variant <name> use this variant without a race (also from the tuning file)\n"
|
||||
" --tuning <file> the per-card tuning file (default: IGNEUM_TUNING_FILE from the environment)\n", WORKER_VERSION);
|
||||
}
|
||||
|
||||
static Options parseArgs(int argc, char** argv) {
|
||||
|
|
@ -564,6 +906,13 @@ static Options parseArgs(int argc, char** argv) {
|
|||
auto next = [&]() -> std::string { if (i + 1 >= argc) { usage(); std::exit(2); } return argv[++i]; };
|
||||
if (a == "--serve") o.serve = true;
|
||||
else if (a == "--check") o.check = true;
|
||||
else if (a == "--race" && (i + 1 >= argc || std::string(argv[i + 1]).rfind("--", 0) == 0)) o.raceOnly = true;
|
||||
else if (a == "--race") o.race = next();
|
||||
else if (a == "--race-bench-ms") o.raceBenchMs = std::atoi(next().c_str());
|
||||
else if (a == "--race-budget-s") o.raceBudgetS = std::atoi(next().c_str());
|
||||
else if (a == "--race-rounds") o.raceRounds = std::atoi(next().c_str());
|
||||
else if (a == "--variant") o.pinned = next();
|
||||
else if (a == "--tuning") o.tuningPath = next();
|
||||
else if (a == "--pack") o.pack = next();
|
||||
else if (a == "--device") o.device = std::atoi(next().c_str());
|
||||
else if (a == "--batch-log2") o.batchLog2 = std::atoi(next().c_str());
|
||||
|
|
@ -575,7 +924,11 @@ static Options parseArgs(int argc, char** argv) {
|
|||
}
|
||||
if (o.batchLog2 < 10 || o.batchLog2 > 28) { std::printf("--batch-log2 must be between 10 and 28\n"); std::exit(2); }
|
||||
if (o.blockWarps < 1 || o.blockWarps > 32) { std::printf("--block-warps must be between 1 and 32\n"); std::exit(2); }
|
||||
if (!o.serve && !o.check) { usage(); std::exit(2); }
|
||||
if (!o.serve && !o.check && !o.raceOnly) { usage(); std::exit(2); }
|
||||
if (o.raceBenchMs < 200 || o.raceBenchMs > 20000) { std::printf("--race-bench-ms must be between 200 and 20000\n"); std::exit(2); }
|
||||
if (o.raceBudgetS < 5 || o.raceBudgetS > 540) { std::printf("--race-budget-s must be between 5 and 540 (the prepare lead is 600 DAA)\n"); std::exit(2); }
|
||||
if (o.raceRounds == 0) o.raceRounds = o.raceOnly ? 3 : 1;
|
||||
if (o.tuningPath.empty()) if (const char* t = std::getenv("IGNEUM_TUNING_FILE")) o.tuningPath = t;
|
||||
if (o.pack.empty()) { std::printf("--pack <dir> is required (igneum-miner export-pack <node> <dir> writes one)\n"); std::exit(2); }
|
||||
while (o.pack.size() > 1 && (o.pack.back() == '/' || o.pack.back() == '\\')) o.pack.pop_back();
|
||||
return o;
|
||||
|
|
@ -632,9 +985,10 @@ static int runServe(Ctx& c, const Options& o, Pair* cur) {
|
|||
Pair* old = nullptr;
|
||||
PrepareTask* task = nullptr;
|
||||
std::string prepareRoot; // the parent of the last prepare's pack directory: where the miner writes its packs
|
||||
emit(fmt("ready cuda %s pack %s dataset-log2 %u batch %u regs %d prepare 1 path nvrtc %d.%d driver %d.%d arch %s worker %s",
|
||||
c.name.c_str(), cur->seedString.c_str(), cur->datasetLog2, batch, cur->regs, c.rtcMajor, c.rtcMinor, c.driverVersion / 1000, (c.driverVersion % 100) / 10, c.archOpt.c_str(), WORKER_VERSION));
|
||||
emit(fmt("ready cuda %s pack %s dataset-log2 %u batch %u regs %d prepare 1 path nvrtc %d.%d driver %d.%d arch %s variant %s race %s worker %s",
|
||||
c.name.c_str(), cur->seedString.c_str(), cur->datasetLog2, batch, cur->regs, c.rtcMajor, c.rtcMinor, c.driverVersion / 1000, (c.driverVersion % 100) / 10, c.archOpt.c_str(), cur->variant.c_str(), c.race.c_str(), WORKER_VERSION));
|
||||
info(fmt("first pack %s: %s", cur->dir.c_str(), pairSummary(cur).c_str()));
|
||||
if (!cur->raceLine.empty()) emit(cur->raceLine);
|
||||
std::string line;
|
||||
while (std::getline(std::cin, line)) {
|
||||
if (line == "quit") break;
|
||||
|
|
@ -646,6 +1000,7 @@ static int runServe(Ctx& c, const Options& o, Pair* cur) {
|
|||
if (task->result) {
|
||||
if (prepared) releasePair(c, prepared);
|
||||
prepared = task->result;
|
||||
if (!prepared->raceLine.empty()) emit(prepared->raceLine);
|
||||
emit(fmt("prepared %s %s %.1f %s resident 2 programs 2 datasets", prepared->epochHex.c_str(), prepared->dayHex.c_str(), wallMs() - task->t0, pairSummary(prepared).c_str()));
|
||||
} else {
|
||||
emit(fmt("prepare-failed %s %s %s", task->epochHex.c_str(), task->dayHex.c_str(), task->error.c_str()));
|
||||
|
|
@ -695,9 +1050,9 @@ static int runServe(Ctx& c, const Options& o, Pair* cur) {
|
|||
if (!dir.empty()) {
|
||||
info(fmt("job %s is for epoch %.16s day %s, which is not resident; building its pack %s now (foreground)", jobId.c_str(), f[6].c_str(), f[7].c_str(), dir.c_str()));
|
||||
std::string berr;
|
||||
Pair* p = buildPair(c, dir, nullptr, berr);
|
||||
Pair* p = buildPair(c, dir, nullptr, berr, true);
|
||||
if (p && (p->epochHex != f[6] || p->dayHex != f[7])) { berr = "the pack in " + dir + " is for other seeds"; releasePair(c, p); p = nullptr; }
|
||||
if (p) { if (prepared) releasePair(c, prepared); prepared = p; info(fmt("built %s: %s", dir.c_str(), pairSummary(p).c_str())); }
|
||||
if (p) { if (prepared) releasePair(c, prepared); prepared = p; info(fmt("built %s: %s", dir.c_str(), pairSummary(p).c_str())); if (!p->raceLine.empty()) emit(p->raceLine); }
|
||||
else emit("error " + jobId + " could not build " + dir + ": " + berr);
|
||||
}
|
||||
}
|
||||
|
|
@ -734,8 +1089,10 @@ static int runServe(Ctx& c, const Options& o, Pair* cur) {
|
|||
b[45] = (uint8_t)hi; b[46] = (uint8_t)(hi >> 8); b[47] = (uint8_t)(hi >> 16); b[48] = (uint8_t)(hi >> 24);
|
||||
pf_seed_words_from_bytes(b, 49, iw);
|
||||
}
|
||||
// A chunk that is not a multiple of the block is finished one 32-lane block at a time
|
||||
uint32_t block = 32u * (uint32_t)o.blockWarps;
|
||||
// A chunk that is not a multiple of the block is finished one 32-lane block at a time. The block is the
|
||||
// pair's (its winning variant's). The mutex gives a race its exclusive windows between chunks.
|
||||
std::lock_guard<std::mutex> hold(gpuMutex);
|
||||
uint32_t block = 32u * (uint32_t)cur->blockWarps;
|
||||
uint32_t main = chunk - (chunk % block);
|
||||
CUresult r = CUDA_SUCCESS;
|
||||
if (main > 0 && !launchHash(c, cur, dOut, lo, iw, main, block, nullptr, err)) { emit("error " + jobId + " dispatch failed: " + err); failed = true; break; }
|
||||
|
|
@ -774,15 +1131,25 @@ int main(int argc, char** argv) {
|
|||
Options o = parseArgs(argc, argv);
|
||||
Ctx c;
|
||||
c.blockWarps = o.blockWarps;
|
||||
c.race = o.race; c.raceBenchMs = o.raceBenchMs; c.raceBudgetS = o.raceBudgetS; c.raceRounds = o.raceRounds; c.batchLog2 = o.batchLog2; c.pinned = o.pinned;
|
||||
if (!o.tuningPath.empty()) { bool ok = false; c.tuning = readText(o.tuningPath, ok); if (!ok) c.tuning.clear(); }
|
||||
std::string err, drvLib, rtcLib;
|
||||
if (!loadDriver(c.drv, err, drvLib)) { emit("error 0 " + err); return 2; }
|
||||
if (!loadNvrtc(c.rtc, err, rtcLib)) { emit("error 0 " + err); return 2; }
|
||||
if (!openDevice(c, o.device, o.arch, err)) { emit("error 0 " + err); return 2; }
|
||||
info(fmt("igneum-worker-cuda %s: device %d %s (sm_%d%d, %d SMs), driver %d.%d from %s, NVRTC %d.%d from %s, target %s (%s)",
|
||||
WORKER_VERSION, o.device, c.name.c_str(), c.major, c.minor, c.sms, c.driverVersion / 1000, (c.driverVersion % 100) / 10, drvLib.c_str(), c.rtcMajor, c.rtcMinor, rtcLib.c_str(), c.archOpt.c_str(), c.why.c_str()));
|
||||
if (!c.tuning.empty()) info(fmt("tuning file %s (%zu bytes): %s", o.tuningPath.c_str(), c.tuning.size(), readTuning(c.tuning, c.name).found ? "has an entry for this card" : "no entry for this card"));
|
||||
double t0 = wallMs();
|
||||
Pair* cur = buildPair(c, o.pack, nullptr, err);
|
||||
Pair* cur = buildPair(c, o.pack, nullptr, err, !o.check);
|
||||
if (!cur) { emit("error 0 " + err); return 1; }
|
||||
if (o.raceOnly) {
|
||||
std::printf("race %s on %s (%s, %d SMs, driver %d.%d, NVRTC %d.%d, %s): %s\n", o.pack.c_str(), c.name.c_str(), c.archOpt.c_str(), c.sms, c.driverVersion / 1000, (c.driverVersion % 100) / 10, c.rtcMajor, c.rtcMinor, c.why.c_str(), pairSummary(cur).c_str());
|
||||
std::printf("%s\n", cur->raceLine.c_str());
|
||||
std::printf("winner %s: %d registers, %d blocks/SM at %d warp(s)/block\n", cur->variant.c_str(), cur->regs, cur->blocksPerSM, cur->blockWarps);
|
||||
releasePair(c, cur);
|
||||
return 0;
|
||||
}
|
||||
if (o.check) {
|
||||
std::printf("check PASS %s in %.0f ms: %s\n", o.pack.c_str(), wallMs() - t0, pairSummary(cur).c_str());
|
||||
std::printf(" epoch %s day %s, dataset 2^%u words, cache 2^%u words in %u segments, %d registers, %d blocks/SM at %d warp(s)/block, target %s\n",
|
||||
|
|
|
|||
|
|
@ -35,6 +35,16 @@ struct Options {
|
|||
// Generator levers (MEMHARD.md section 7). Defaults reproduce the original generator exactly.
|
||||
var loadWeight = 25 // --load-weight W: percent weight of the load op (default 25)
|
||||
var wideFrac = 0 // --wide-frac P: percent of load instructions emitted as warp-coalesced wide loads
|
||||
// Variant racing (4 October 2026, evening; docs/design/miner-tuning.md): in --serve every prepared program is
|
||||
// compiled in several variants, each checked bit for bit against the base kernel and timed for about two
|
||||
// seconds with the job loop paused; the fastest serves the hour. --race-test runs the race alone and prints it.
|
||||
var race = "on" // --race on|off|a,b,c
|
||||
var raceBenchMs = 2000 // --race-bench-ms
|
||||
var raceBudgetS = 120 // --race-budget-s (the prepare lead is 600 DAA; base is kept when the budget runs out)
|
||||
var raceRounds = 0 // --race-rounds (0 = 1 in --serve, 3 in --race-test)
|
||||
var pinnedVariant: String? = nil // --variant: this one, no race
|
||||
var tuningPath: String? = nil // --tuning <file> (default IGNEUM_TUNING_FILE)
|
||||
var raceTest = false // --race-test: the race for --seed on --day, rounds, table, exit
|
||||
var anyTest: Bool { fuzz != nil || edge || stats || determinism || memcheck }
|
||||
}
|
||||
|
||||
|
|
@ -66,6 +76,13 @@ func parseArgs() -> Options {
|
|||
case "--closed-form": o.closedForm = true
|
||||
case "--load-weight": o.loadWeight = Int(take()) ?? o.loadWeight
|
||||
case "--wide-frac": o.wideFrac = Int(take()) ?? o.wideFrac
|
||||
case "--race": o.race = take()
|
||||
case "--race-bench-ms": o.raceBenchMs = Int(take()) ?? o.raceBenchMs
|
||||
case "--race-budget-s": o.raceBudgetS = Int(take()) ?? o.raceBudgetS
|
||||
case "--race-rounds": o.raceRounds = Int(take()) ?? o.raceRounds
|
||||
case "--variant": o.pinnedVariant = take()
|
||||
case "--tuning": o.tuningPath = take()
|
||||
case "--race-test": o.raceTest = true
|
||||
case "-h", "--help":
|
||||
print("""
|
||||
igneum-bench [--seed <string>] [--hours N] [--batch-log2 22] [--batches 4]
|
||||
|
|
@ -84,6 +101,10 @@ func parseArgs() -> Options {
|
|||
[--memcheck] static dataset-index mask check, 4 MiB run with wrapping nonces
|
||||
shortcut measurement:
|
||||
[--inline-dataset] bench variant: every load computes ds_elem(index) inline, no memory read
|
||||
variant racing (4 October 2026; in --serve every prepared program is raced, the fastest variant serves the hour):
|
||||
[--race on|off|a,b,c] [--race-bench-ms 2000] [--race-budget-s 120] [--race-rounds N]
|
||||
[--variant <name>] use this variant, no race [--tuning <file>] per-card tuning (IGNEUM_TUNING_FILE)
|
||||
[--race-test] the race alone for --seed on --day (3 rounds): a table per variant, exit 0/1
|
||||
""")
|
||||
exit(0)
|
||||
default:
|
||||
|
|
@ -782,7 +803,27 @@ enum LoadSource {
|
|||
case inlineMemhard(MixParams) // memory-hard: mh_word(cache, index), 8 dependent cache reads per word
|
||||
}
|
||||
|
||||
func generateMSL(_ p: Program, datasetLog2: Int, source: LoadSource = .stored, bound: Bool = false) -> String {
|
||||
// A kernel variant for the race (4 October 2026): the same instruction text, a different shape for the compiler.
|
||||
// Names are stable: the tuning file and the fleet records use them. "base" is the kernel as it has always shipped.
|
||||
struct MetalVariant {
|
||||
let name: String
|
||||
var unroll = 0 // 0: the iteration loop as emitted; N: "#pragma unroll N" before it (8 = fully unrolled)
|
||||
var maxThreads = 0 // N > 0: [[max_total_threads_per_threadgroup(N)]] (fewer threads per group, more registers per thread)
|
||||
var groupWidth = 32 // threads per threadgroup at dispatch (a multiple of 32; SIMD groups stay 32 wide)
|
||||
var sizeOpt = false // MTLCompileOptions.optimizationLevel = .size
|
||||
}
|
||||
|
||||
func metalVariants() -> [MetalVariant] {
|
||||
[MetalVariant(name: "base"),
|
||||
MetalVariant(name: "g64", groupWidth: 64), MetalVariant(name: "g128", groupWidth: 128), MetalVariant(name: "g256", groupWidth: 256),
|
||||
MetalVariant(name: "u2", unroll: 2), MetalVariant(name: "u8", unroll: 8),
|
||||
MetalVariant(name: "mt256", maxThreads: 256), MetalVariant(name: "mt512", maxThreads: 512), MetalVariant(name: "mt1024", maxThreads: 1024),
|
||||
MetalVariant(name: "osize", sizeOpt: true),
|
||||
MetalVariant(name: "u2-g128", unroll: 2, groupWidth: 128), MetalVariant(name: "u8-g128", unroll: 8, groupWidth: 128),
|
||||
MetalVariant(name: "mt256-g128", maxThreads: 256, groupWidth: 128), MetalVariant(name: "mt512-g256", maxThreads: 512, groupWidth: 256)]
|
||||
}
|
||||
|
||||
func generateMSL(_ p: Program, datasetLog2: Int, source: LoadSource = .stored, bound: Bool = false, variant: MetalVariant? = nil) -> String {
|
||||
let mask = UInt32((1 << datasetLog2) - 1)
|
||||
var s = """
|
||||
#include <metal_stdlib>
|
||||
|
|
@ -820,6 +861,7 @@ func generateMSL(_ p: Program, datasetLog2: Int, source: LoadSource = .stored, b
|
|||
let kernelName = bound ? "igneum_hash_bound" : "igneum_hash"
|
||||
let initArg = bound ? " constant uint* initw [[buffer(3)]],\n" : ""
|
||||
let iw = bound ? "initw" : "SEEDW"
|
||||
if let v = variant, v.maxThreads > 0 { s += "[[max_total_threads_per_threadgroup(\(v.maxThreads))]]\n" }
|
||||
s += """
|
||||
kernel void \(kernelName)(\(buffer0),
|
||||
device ulong* out [[buffer(1)]],
|
||||
|
|
@ -833,6 +875,7 @@ func generateMSL(_ p: Program, datasetLog2: Int, source: LoadSource = .stored, b
|
|||
for i in 0..<8 {
|
||||
s += " { uint x = nonce ^ \(iw)[\(i)]; x += 0x9e3779b9u * \(i + 1)u; x = splitmix32(x); r\(i) = x ^ \(iw)[\((i + 1) & 7)]; }\n"
|
||||
}
|
||||
if let v = variant, v.unroll > 0 { s += "\n#pragma unroll \(v.unroll)" }
|
||||
s += "\n for (uint it = 0u; it < \(Program.iterations)u; ++it) {\n uint sel = r0;\n"
|
||||
// The word index expression for a load: plain = a & MASK; wide = lane 0's a, aligned to 32 words, plus lane.
|
||||
func wordIndex(_ a: String, wide: Bool) -> String { wide ? "(simd_broadcast(\(a), 0) & WMASK) + lane" : "\(a) & MASK" }
|
||||
|
|
@ -2007,6 +2050,8 @@ struct CompiledHash {
|
|||
let pipeline: MTLComputePipelineState
|
||||
let libraryMs: Double
|
||||
let pipelineMs: Double
|
||||
var groupWidth = 32 // threads per threadgroup at dispatch (the variant's)
|
||||
var variant = "base"
|
||||
var totalMs: Double { libraryMs + pipelineMs }
|
||||
}
|
||||
|
||||
|
|
@ -2737,9 +2782,166 @@ func blockInitWords(prehash: [UInt8], nonceHi: UInt32) -> [UInt32] {
|
|||
final class ServeProgram {
|
||||
let seedHex: String
|
||||
let program: Program
|
||||
let compiled: CompiledHash
|
||||
init(seedHex: String, program: Program, compiled: CompiledHash) { self.seedHex = seedHex; self.program = program; self.compiled = compiled }
|
||||
private var slot: CompiledHash
|
||||
private let lock = NSLock()
|
||||
/// true until a race ran for this program (an inline compile races after the first job on it)
|
||||
var raceDue = true
|
||||
var raceLine = ""
|
||||
init(seedHex: String, program: Program, compiled: CompiledHash) { self.seedHex = seedHex; self.program = program; self.slot = compiled }
|
||||
var compiled: CompiledHash { lock.lock(); defer { lock.unlock() }; return slot }
|
||||
func install(_ c: CompiledHash) { lock.lock(); slot = c; lock.unlock() }
|
||||
}
|
||||
|
||||
// The job loop and a race take turns on the card: a variant is timed with no job running (exclusive numbers), and
|
||||
// mining resumes between variants. Held per chunk by the job loop, per variant window by the race.
|
||||
let gpuLock = NSLock()
|
||||
|
||||
// The tuning file: {"cards": {"<device name with underscores>": {"variant": "u2", "race": false, "candidates": [...]}}}
|
||||
struct MetalTuning {
|
||||
var found = false
|
||||
var variant = ""
|
||||
var race = true
|
||||
var candidates = [String]()
|
||||
}
|
||||
func readMetalTuning(path: String?, device: String) -> MetalTuning {
|
||||
var t = MetalTuning()
|
||||
guard let path = path, let data = FileManager.default.contents(atPath: path),
|
||||
let root = try? JSONSerialization.jsonObject(with: data) as? [String: Any],
|
||||
let cards = root["cards"] as? [String: Any], let e = cards[device] as? [String: Any] else { return t }
|
||||
t.found = true
|
||||
t.variant = e["variant"] as? String ?? ""
|
||||
t.race = e["race"] as? Bool ?? true
|
||||
t.candidates = e["candidates"] as? [String] ?? []
|
||||
return t
|
||||
}
|
||||
|
||||
// Dispatches `count` lane nonces (a multiple of 32) from `base` with the kernel's group width; a tail under the
|
||||
// width goes in 32-wide groups (same pipeline, a threadgroup size is a dispatch parameter in Metal).
|
||||
func encodeHash(_ enc: MTLComputeCommandEncoder, _ k: CompiledHash, dataset: MTLBuffer, out: MTLBuffer, base: UInt32, initw: inout [UInt32], count: Int) {
|
||||
enc.setComputePipelineState(k.pipeline)
|
||||
enc.setBuffer(dataset, offset: 0, index: 0)
|
||||
enc.setBuffer(out, offset: 0, index: 1)
|
||||
var b = base
|
||||
enc.setBytes(&b, length: 4, index: 2)
|
||||
enc.setBytes(&initw, length: 32, index: 3)
|
||||
let w = k.groupWidth
|
||||
let main = count - count % w
|
||||
if main > 0 { enc.dispatchThreadgroups(MTLSize(width: main / w, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: w, height: 1, depth: 1)) }
|
||||
if main < count {
|
||||
var b2 = base &+ UInt32(main)
|
||||
enc.setBytes(&b2, length: 4, index: 2)
|
||||
enc.setBuffer(out, offset: main * 8, index: 1)
|
||||
enc.dispatchThreadgroups(MTLSize(width: (count - main) / 32, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 32, height: 1, depth: 1))
|
||||
}
|
||||
}
|
||||
|
||||
struct RaceEntry {
|
||||
let v: MetalVariant
|
||||
var compiled: CompiledHash? = nil
|
||||
var mhs = 0.0 // best round
|
||||
var note = "" // why it is out, or a detail
|
||||
var ok: Bool { compiled != nil && note.isEmpty }
|
||||
}
|
||||
|
||||
// Races the variants of `program` on `dataset` and returns the race line; the winner is installed in `program`.
|
||||
// Every variant must equal the base kernel bit for bit over 2^16 nonces (the base is the kernel the miner's CPU
|
||||
// re-check has always covered); a window is `benchMs` of launches of `batch` nonces, the first launch warming up.
|
||||
// `rounds` interleaved rounds, best per variant. Stops compiling and timing after `budgetS` (base is always kept).
|
||||
func raceProgram(_ gpu: GPU, _ program: ServeProgram, dataset: MTLBuffer, datasetLog2: Int, opts: Options, rounds: Int, table: Bool = false) -> String {
|
||||
let t0 = nowNs()
|
||||
let device = gpu.device.name.replacingOccurrences(of: " ", with: "_")
|
||||
let tuning = readMetalTuning(path: opts.tuningPath ?? ProcessInfo.processInfo.environment["IGNEUM_TUNING_FILE"], device: device)
|
||||
let all = metalVariants()
|
||||
let pinned = opts.pinnedVariant ?? (tuning.found && !tuning.race ? tuning.variant : "")
|
||||
var order = [MetalVariant]()
|
||||
func push(_ n: String) { if let v = all.first(where: { $0.name == n }), !order.contains(where: { $0.name == n }) { order.append(v) } }
|
||||
push("base")
|
||||
if !pinned.isEmpty { push(pinned) }
|
||||
else {
|
||||
tuning.candidates.forEach(push)
|
||||
if opts.race == "on" { all.forEach { push($0.name) } }
|
||||
else if opts.race != "off" { opts.race.split(separator: ",").forEach { push(String($0)) } }
|
||||
}
|
||||
let pinnedOnly = !pinned.isEmpty && order.count == 2
|
||||
let batch = 1 << opts.batchLog2
|
||||
let checkN = 1 << 16
|
||||
let deadline = t0 + UInt64(opts.raceBudgetS) * 1_000_000_000
|
||||
guard let outBuf = gpu.device.makeBuffer(length: batch * 8, options: .storageModeShared),
|
||||
let refBuf = gpu.device.makeBuffer(length: checkN * 8, options: .storageModeShared) else { return "race \(program.seedHex.prefix(16)) failed: no buffers" }
|
||||
var entries = order.map { RaceEntry(v: $0) }
|
||||
entries[0].compiled = program.compiled
|
||||
// Compile (the base is already compiled)
|
||||
for i in 1..<entries.count {
|
||||
if nowNs() > deadline { entries[i].note = "not compiled: the race budget ran out"; continue }
|
||||
do { entries[i].compiled = try compileBound(gpu, msl: generateMSL(program.program, datasetLog2: datasetLog2, source: .stored, bound: true, variant: entries[i].v), variant: entries[i].v) }
|
||||
catch { entries[i].note = "compile: \(error)".prefix(200).description }
|
||||
}
|
||||
let compileMs = ms(t0, nowNs())
|
||||
var initw = blockInitWords(prehash: [UInt8](repeating: 0x5a, count: 32), nonceHi: 7)
|
||||
func run(_ k: CompiledHash, base: UInt32, count: Int, into: MTLBuffer) -> Bool {
|
||||
let cb = gpu.queue.makeCommandBuffer()!
|
||||
let enc = cb.makeComputeCommandEncoder()!
|
||||
encodeHash(enc, k, dataset: dataset, out: into, base: base, initw: &initw, count: count)
|
||||
enc.endEncoding()
|
||||
cb.commit(); cb.waitUntilCompleted()
|
||||
return cb.error == nil
|
||||
}
|
||||
// Reference output of the base kernel (exclusive window)
|
||||
gpuLock.lock()
|
||||
let refOk = run(entries[0].compiled!, base: 0x1000_0000, count: checkN, into: refBuf)
|
||||
gpuLock.unlock()
|
||||
if !refOk { return "race \(program.seedHex.prefix(16)) failed: the base kernel did not run" }
|
||||
let ref = refBuf.contents().bindMemory(to: UInt64.self, capacity: checkN)
|
||||
let out = outBuf.contents().bindMemory(to: UInt64.self, capacity: batch)
|
||||
for round in 0..<max(1, rounds) {
|
||||
for i in 0..<entries.count {
|
||||
guard entries[i].ok, let k = entries[i].compiled else { continue }
|
||||
if pinnedOnly && i == 0 { continue }
|
||||
if round == 0 && i > 0 && nowNs() > deadline { entries[i].note = "not timed: the race budget ran out"; continue }
|
||||
gpuLock.lock()
|
||||
// NSLock is not fair: after the window the job loop gets the card (measured 4 October 2026: without the
|
||||
// pause a queued job waited the whole race, 36 s)
|
||||
defer { gpuLock.unlock(); Thread.sleep(forTimeInterval: 0.15) }
|
||||
if round == 0 && i > 0 {
|
||||
if !run(k, base: 0x1000_0000, count: checkN, into: outBuf) { entries[i].note = "did not run"; continue }
|
||||
var bad = -1
|
||||
for j in 0..<checkN where out[j] != ref[j] { bad = j; break }
|
||||
if bad >= 0 { entries[i].note = String(format: "lane %d: %016llx, base %016llx (discarded)", bad, out[bad], ref[bad]); continue }
|
||||
}
|
||||
if pinnedOnly { continue }
|
||||
var launches = 0, hashes = 0, tStart: UInt64 = 0
|
||||
while true {
|
||||
if !run(k, base: 0x2000_0000 &+ UInt32(launches * batch), count: batch, into: outBuf) { entries[i].note = "bench: did not run"; break }
|
||||
let now = nowNs()
|
||||
if launches == 0 { tStart = now } else { hashes += batch }
|
||||
launches += 1
|
||||
if launches >= 3 && ms(tStart, now) >= Double(opts.raceBenchMs) { entries[i].mhs = max(entries[i].mhs, Double(hashes) / ms(tStart, now) / 1000.0); break }
|
||||
}
|
||||
}
|
||||
}
|
||||
var win = 0
|
||||
if pinnedOnly && entries.count == 2 && entries[1].ok { win = 1 }
|
||||
else { for i in 1..<entries.count where entries[i].ok && entries[i].mhs > entries[win].mhs * (win == 0 ? 1.005 : 1.0) { win = i } }
|
||||
let baseMhs = entries[0].mhs, winMhs = entries[win].mhs
|
||||
if win != 0, let k = entries[win].compiled { program.install(k) }
|
||||
let total = ms(t0, nowNs())
|
||||
let os = ProcessInfo.processInfo.operatingSystemVersion
|
||||
var line = "race \(program.seedHex.prefix(16)) device \(device) driver macos-\(os.majorVersion).\(os.minorVersion).\(os.patchVersion) arch metal loads \(program.program.loadsPerHash) wide \(program.program.wideLoadsPerHash) variants \(entries.count)"
|
||||
for e in entries { line += e.ok && (e.mhs > 0 || pinnedOnly) ? " \(e.v.name)=\(fmt(e.mhs, 3))/\(e.compiled!.pipeline.maxTotalThreadsPerThreadgroup)t/\(e.v.groupWidth)w" : " \(e.v.name)=-" }
|
||||
let gain = baseMhs > 0 ? (winMhs / baseMhs - 1) * 100 : 0
|
||||
line += " winner \(entries[win].v.name) \(fmt(winMhs, 3)) base \(fmt(baseMhs, 3)) gain \(gain >= 0 ? "+" : "")\(fmt(gain, 2))% compile \(fmt(compileMs, 0)) bench \(fmt(total - compileMs, 0)) total \(fmt(total, 0)) ms"
|
||||
if pinnedOnly { line += " pinned by tuning" } else if tuning.found { line += " tuned order" }
|
||||
for e in entries where !e.note.isEmpty { line += " | \(e.v.name): \(e.note)" }
|
||||
program.raceDue = false
|
||||
program.raceLine = line
|
||||
if table {
|
||||
print("| variant | threads/group | max threads | MH/s | vs base | note |")
|
||||
print("|---|---|---|---|---|---|")
|
||||
for e in entries { print("| \(e.v.name) | \(e.v.groupWidth) | \(e.compiled.map { String($0.pipeline.maxTotalThreadsPerThreadgroup) } ?? "-") | \(e.ok ? fmt(e.mhs, 3) : "-") | \(e.ok && baseMhs > 0 ? (e.mhs / baseMhs - 1 >= 0 ? "+" : "") + fmt((e.mhs / baseMhs - 1) * 100, 2) + "%" : "-") | \(e.note) |") }
|
||||
}
|
||||
return line
|
||||
}
|
||||
|
||||
final class ServeDataset {
|
||||
let dayHex: String
|
||||
let ctx: DatasetContext
|
||||
|
|
@ -2782,14 +2984,19 @@ final class ServeStore {
|
|||
}
|
||||
}
|
||||
|
||||
func compileBound(_ gpu: GPU, msl: String) throws -> CompiledHash {
|
||||
func compileBound(_ gpu: GPU, msl: String, variant: MetalVariant? = nil) throws -> CompiledHash {
|
||||
let t0 = nowNs()
|
||||
let lib = try gpu.device.makeLibrary(source: msl, options: MTLCompileOptions())
|
||||
let copts = MTLCompileOptions()
|
||||
if let v = variant, v.sizeOpt { if #available(macOS 13.0, *) { copts.optimizationLevel = .size } }
|
||||
let lib = try gpu.device.makeLibrary(source: msl, options: copts)
|
||||
let t1 = nowNs()
|
||||
guard let fn = lib.makeFunction(name: "igneum_hash_bound") else { throw IgneumError("no igneum_hash_bound function in library") }
|
||||
let pipe = try gpu.device.makeComputePipelineState(function: fn)
|
||||
let t2 = nowNs()
|
||||
return CompiledHash(pipeline: pipe, libraryMs: ms(t0, t1), pipelineMs: ms(t1, t2))
|
||||
var c = CompiledHash(pipeline: pipe, libraryMs: ms(t0, t1), pipelineMs: ms(t1, t2))
|
||||
if let v = variant { c.groupWidth = v.groupWidth; c.variant = v.name }
|
||||
if c.groupWidth > pipe.maxTotalThreadsPerThreadgroup { throw IgneumError("variant \(c.variant): group width \(c.groupWidth) is over the pipeline's maxTotalThreadsPerThreadgroup \(pipe.maxTotalThreadsPerThreadgroup)") }
|
||||
return c
|
||||
}
|
||||
|
||||
// Builds the program for an epoch seed (hex) unless resident. Returns (program, compile ms) or throws.
|
||||
|
|
@ -2823,7 +3030,7 @@ func runServe(_ opts: Options) -> Never {
|
|||
let store = ServeStore()
|
||||
let prepareQueue = DispatchQueue(label: "igneum.prepare") // one prepare at a time, off the job loop
|
||||
guard let outBuf = gpu.device.makeBuffer(length: batch * 8, options: .storageModeShared) else { emit("error 0 cannot allocate the output buffer"); exit(1) }
|
||||
emit("ready metal \(gpu.device.name.replacingOccurrences(of: " ", with: "_")) dataset-log2 \(datasetLog2) batch \(batch) prepare \(opts.noPrepare ? 0 : 1)")
|
||||
emit("ready metal \(gpu.device.name.replacingOccurrences(of: " ", with: "_")) dataset-log2 \(datasetLog2) batch \(batch) prepare \(opts.noPrepare ? 0 : 1) race \(opts.race)")
|
||||
var lastPair: (String, String)? = nil
|
||||
while let line = readLine(strippingNewline: true) {
|
||||
let f = line.split(separator: " ").map(String.init)
|
||||
|
|
@ -2843,8 +3050,15 @@ func runServe(_ opts: Options) -> Never {
|
|||
do {
|
||||
let (sp, progMs) = try serveProgram(gpu, store, seedHex: epochHex, seed: epochSeed, datasetLog2: datasetLog2)
|
||||
let (sd, dsMs) = serveDataset(gpu, store, dayHex: dayHex, day: daySeed, datasetLog2: datasetLog2)
|
||||
// The race: the prepared program against its own dataset, exclusive windows between jobs
|
||||
var raceMs = 0.0
|
||||
if sp.raceDue && opts.race != "off" {
|
||||
let r0 = nowNs()
|
||||
emit(raceProgram(gpu, sp, dataset: sd.buffer, datasetLog2: datasetLog2, opts: opts, rounds: opts.raceRounds))
|
||||
raceMs = ms(r0, nowNs())
|
||||
}
|
||||
let (np, nd) = store.counts()
|
||||
emit("prepared \(epochHex) \(dayHex) \(fmt(ms(t0, nowNs()), 1)) program \(fmt(progMs, 1)) dataset \(fmt(dsMs, 1)) loads/hash \(sp.program.loadsPerHash) cache-fill \(fmt(sd.ctx.cacheFillGPUms, 1)) resident \(np) programs \(nd) datasets")
|
||||
emit("prepared \(epochHex) \(dayHex) \(fmt(ms(t0, nowNs()), 1)) program \(fmt(progMs, 1)) dataset \(fmt(dsMs, 1)) race \(fmt(raceMs, 1)) variant \(sp.compiled.variant) loads/hash \(sp.program.loadsPerHash) cache-fill \(fmt(sd.ctx.cacheFillGPUms, 1)) resident \(np) programs \(nd) datasets")
|
||||
} catch { emit("prepare-failed \(epochHex) \(dayHex) Metal compile failed: \(error)") }
|
||||
}
|
||||
continue
|
||||
|
|
@ -2885,21 +3099,18 @@ func runServe(_ opts: Options) -> Never {
|
|||
var found = 0
|
||||
var failed = false
|
||||
let outPtr = outBuf.contents().bindMemory(to: UInt64.self, capacity: batch)
|
||||
let kernel = program.compiled // the pair's kernel for this job (a race may swap it for the next)
|
||||
while remaining > 0 {
|
||||
let room = UInt64(UInt32.max - lo) + 1 // lane nonces left before the high word steps
|
||||
let chunk = Int(min(min(remaining, UInt64(batch)), room))
|
||||
var initw = blockInitWords(prehash: prehash, nonceHi: hi)
|
||||
gpuLock.lock() // a race's exclusive windows fall between chunks
|
||||
let cb = gpu.queue.makeCommandBuffer()!
|
||||
let enc = cb.makeComputeCommandEncoder()!
|
||||
enc.setComputePipelineState(program.compiled.pipeline)
|
||||
enc.setBuffer(dataset.buffer, offset: 0, index: 0)
|
||||
enc.setBuffer(outBuf, offset: 0, index: 1)
|
||||
var b = lo
|
||||
enc.setBytes(&b, length: 4, index: 2)
|
||||
enc.setBytes(&initw, length: 32, index: 3)
|
||||
enc.dispatchThreadgroups(MTLSize(width: chunk / 32, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 32, height: 1, depth: 1))
|
||||
encodeHash(enc, kernel, dataset: dataset.buffer, out: outBuf, base: lo, initw: &initw, count: chunk)
|
||||
enc.endEncoding()
|
||||
cb.commit(); cb.waitUntilCompleted()
|
||||
gpuLock.unlock()
|
||||
if let e = cb.error { emit("error \(jobId) dispatch failed: \(e)"); failed = true; break }
|
||||
for i in 0..<chunk where outPtr[i] <= target {
|
||||
let nonce = (UInt64(hi) << 32) | UInt64(lo &+ UInt32(i))
|
||||
|
|
@ -2919,6 +3130,11 @@ func runServe(_ opts: Options) -> Never {
|
|||
let (dp, dd) = store.prune(to: pair)
|
||||
if dp + dd > 0 { emit("info dropped \(dp) program(s) and \(dd) dataset(s) of the previous pair") }
|
||||
}
|
||||
if program.raceDue && opts.race != "off" {
|
||||
// A pair compiled inline (nobody prepared it) races now, in the background, exclusive windows between chunks
|
||||
program.raceDue = false
|
||||
prepareQueue.async { emit(raceProgram(gpu, program, dataset: dataset.buffer, datasetLog2: datasetLog2, opts: opts, rounds: opts.raceRounds)) }
|
||||
}
|
||||
lastPair = pair
|
||||
}
|
||||
exit(0)
|
||||
|
|
@ -2926,9 +3142,30 @@ func runServe(_ opts: Options) -> Never {
|
|||
|
||||
// MARK: - Main
|
||||
|
||||
let opts = parseArgs()
|
||||
// The race alone (the Mac measurement, docs/bench-log.md "miner performance: variant racing"): the memory-hard
|
||||
// dataset for --day, the version-2 program for --seed, every variant timed for --race-rounds rounds, a table.
|
||||
func runRaceTest(_ opts: Options) -> Never {
|
||||
let gpu = GPU()
|
||||
print("igneum-bench --race-test on \(gpu.device.name): seed \"\(opts.seed)\", day \"\(opts.day)\", dataset 2^\(opts.datasetLog2) words, batch 2^\(opts.batchLog2), \(opts.raceRounds) rounds, \(opts.raceBenchMs) ms per window")
|
||||
let ctx = DatasetContext(gpu: gpu, closedForm: false, dayString: opts.day)
|
||||
let t0 = nowNs()
|
||||
let dataset = ctx.makeDataset(log2: opts.datasetLog2)
|
||||
print("dataset built in \(fmt(ms(t0, nowNs()), 0)) ms (cache fill \(fmt(ctx.cacheFillGPUms, 0)) ms GPU, build \(fmt(ctx.lastBuildGPUms, 0)) ms GPU)")
|
||||
let p = generateProgramV2(seedString: opts.seed, bytes: Array(opts.seed.utf8))
|
||||
print("program: \(describeProgram(p))")
|
||||
let base: CompiledHash
|
||||
do { base = try compileBound(gpu, msl: generateMSL(p, datasetLog2: opts.datasetLog2, source: .stored, bound: true)) } catch { print("FAIL: \(error)"); exit(1) }
|
||||
let sp = ServeProgram(seedHex: String(repeating: "0", count: 64), program: p, compiled: base)
|
||||
let line = raceProgram(gpu, sp, dataset: dataset, datasetLog2: opts.datasetLog2, opts: opts, rounds: opts.raceRounds, table: true)
|
||||
print(line)
|
||||
exit(line.contains("winner ") ? 0 : 1)
|
||||
}
|
||||
|
||||
var opts = parseArgs()
|
||||
if opts.raceRounds == 0 { opts.raceRounds = opts.raceTest ? 3 : 1 }
|
||||
generatorConfig = GeneratorConfig(loadWeight: opts.loadWeight, wideFrac: opts.wideFrac)
|
||||
if opts.exportPack != nil { exportPack(opts) }
|
||||
if opts.raceTest { runRaceTest(opts) }
|
||||
if opts.serve { runServe(opts) }
|
||||
if opts.anyTest { runTests(opts) }
|
||||
let gpu = GPU()
|
||||
|
|
|
|||
31
relay/playbooks/race-5090.ps1
Normal file
31
relay/playbooks/race-5090.ps1
Normal file
|
|
@ -0,0 +1,31 @@
|
|||
# Variant race on PC 1's RTX 5090 with the NVRTC worker (docs/plans/miner-perf.md). A `run` job: the miners are
|
||||
# stopped first (--stop-miners), so the card is the race's alone. Needs the race worker fetched by the job
|
||||
# fetch-race-worker-20261004 (packaging/ota/publish-jobs.sh add --kind fetch ... --dir jobs --extract --id ...),
|
||||
# which lands next to this job's folder under <app data>\app\jobs\. Every line that matters starts with RESULT so
|
||||
# the dashboard and tools/jobs.mjs show it; the race lines are the same format the workers log in --serve.
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$app = $env:IGNEUM_APP_DIR
|
||||
$jobs = Split-Path $env:IGNEUM_JOB_DIR
|
||||
$fetched = Join-Path $jobs 'fetch-race-worker-20261004'
|
||||
$exe = Join-Path $fetched 'igneum-worker-cuda.exe'
|
||||
if (-not (Test-Path $exe)) { Write-Output "RESULT race worker missing at $exe (the fetch job runs first)"; exit 2 }
|
||||
# NVIDIA's NVRTC DLLs sit next to the installed worker (per-user install since 0.3.3, else Program Files)
|
||||
$inst = @("$env:LOCALAPPDATA\Programs\Igneum Miner", "$env:ProgramFiles\Igneum Miner") | Where-Object { Test-Path (Join-Path $_ 'igneum-worker-cuda.exe') } | Select-Object -First 1
|
||||
if (-not $inst) { Write-Output "RESULT no installed igneum-worker-cuda.exe found (the NVRTC DLLs come from there)"; exit 2 }
|
||||
Get-ChildItem $inst -Filter 'nvrtc*.dll' | Copy-Item -Destination $fetched -Force
|
||||
Write-Output "RESULT worker $exe with $((Get-ChildItem $fetched -Filter 'nvrtc*.dll').Count) NVRTC DLL(s) from $inst"
|
||||
# The pack: this hour's prepared pack (the miner writes one per pair under packs\prepare), else the one exported at the start
|
||||
$pack = Get-ChildItem "$app\packs\prepare" -Directory -ErrorAction SilentlyContinue | Sort-Object LastWriteTime -Descending | Select-Object -First 1
|
||||
if ($pack) { $pack = $pack.FullName } else { $pack = "$app\packs\devnet" }
|
||||
if (-not (Test-Path "$pack\seeds.txt")) { Write-Output "RESULT no pack with seeds.txt under $app\packs"; exit 2 }
|
||||
Write-Output "RESULT pack $pack"
|
||||
Get-Content "$pack\seeds.txt" | ForEach-Object { "RESULT seeds $_" }
|
||||
& nvidia-smi --query-gpu=name,driver_version,power.limit,power.default_limit,clocks.sm,clocks.mem,temperature.gpu --format=csv,noheader | ForEach-Object { "RESULT gpu $_" }
|
||||
# Two independent races of three interleaved rounds each (best of three per variant), 2 s per timed window
|
||||
foreach ($run in 1..2) {
|
||||
Write-Output "RESULT run $run start $(Get-Date -Format HH:mm:ss)"
|
||||
& $exe --race --pack $pack --race-rounds 3 --race-bench-ms 2000 --race-budget-s 400 2>&1 | ForEach-Object { "RESULT $_" }
|
||||
Write-Output "RESULT run $run exit $LASTEXITCODE"
|
||||
}
|
||||
& nvidia-smi --query-gpu=power.draw,clocks.sm,temperature.gpu --format=csv,noheader | ForEach-Object { "RESULT gpu-after $_" }
|
||||
exit 0
|
||||
133
tools/tuning.mjs
Normal file
133
tools/tuning.mjs
Normal file
|
|
@ -0,0 +1,133 @@
|
|||
#!/usr/bin/env node
|
||||
// Fleet learning for the GPU workers' kernel variants (docs/design/miner-tuning.md). Runs on the Mac.
|
||||
//
|
||||
// Every app logs one `TUNING {json}` line per hourly race (src/engine.rs race_line: card model, driver, program
|
||||
// class, every variant's MH/s, the winner, power cap, MH per watt). The app log reaches the intake (Neon table
|
||||
// miner_logs, the same one tools/logs.mjs reads). This script aggregates those records per card model and writes
|
||||
// the tuning object the over-the-air manifest carries back to every machine (publish-manifest.sh --tuning):
|
||||
//
|
||||
// node tools/tuning.mjs the table: per card model, per variant, samples, median MH/s, MH per watt
|
||||
// node tools/tuning.mjs --write tuning.json [--days 7] [--min-samples 3] [--by mhs|mhw] [--pin]
|
||||
// writes {"updated": ..., "cards": {"<card>": {"variant", "race", "candidates", "samples", "mhs", "base_mhs",
|
||||
// "gain_pct", "mh_per_w"}}}. The winner is the variant with the best median over the window; "candidates" are
|
||||
// the top three, which the workers race first at every prepare. --pin sets race false (the worker uses the
|
||||
// variant without a race; the record then carries only the self-test), the default keeps racing (the fleet keeps
|
||||
// learning while the card starts from the known best).
|
||||
// node tools/tuning.mjs --records [--days 7] [--card <model>] the raw records, newest first
|
||||
//
|
||||
// Reads DATABASE_URL from ~/.config/igneum/env. No dependencies: Neon HTTP SQL over fetch.
|
||||
import { readFileSync, writeFileSync } from 'node:fs';
|
||||
import { homedir } from 'node:os';
|
||||
|
||||
process.stdout.on('error', e => { if (e.code === 'EPIPE') process.exit(0); throw e; });
|
||||
|
||||
const env = readFileSync(`${homedir()}/.config/igneum/env`, 'utf8');
|
||||
const m = /^DATABASE_URL=(.*)$/m.exec(env);
|
||||
if (!m) { console.error('DATABASE_URL not found in ~/.config/igneum/env'); process.exit(1); }
|
||||
const url = m[1].trim().replace(/^['"]|['"]$/g, '');
|
||||
const host = new URL(url).hostname.replace('-pooler', '');
|
||||
|
||||
async function sql(query, params = []) {
|
||||
const r = await fetch(`https://${host}/sql`, {
|
||||
method: 'POST',
|
||||
headers: { 'Neon-Connection-String': url, 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({ query, params }),
|
||||
});
|
||||
const j = await r.json();
|
||||
if (!r.ok) throw new Error(j.message || JSON.stringify(j));
|
||||
return j.rows;
|
||||
}
|
||||
|
||||
const args = process.argv.slice(2);
|
||||
const opt = (name, def) => { const i = args.indexOf(name); return i >= 0 && i + 1 < args.length ? args[i + 1] : def; };
|
||||
const flag = name => args.includes(name);
|
||||
const days = Number(opt('--days', 7));
|
||||
const minSamples = Number(opt('--min-samples', 3));
|
||||
const by = opt('--by', 'mhs');
|
||||
const outFile = opt('--write', '');
|
||||
const onlyCard = opt('--card', '');
|
||||
|
||||
// Every app-log upload of the window; the TUNING lines out of them. The same race is uploaded many times (the log
|
||||
// is re-sent every minute), so records are de-duplicated on (machine, card, epoch).
|
||||
const rows = await sql(
|
||||
`SELECT machine, run_id, lines FROM miner_logs WHERE received_at > now() - ($1 || ' days')::interval AND lines LIKE '%TUNING {%' ORDER BY received_at DESC`,
|
||||
[String(days)]);
|
||||
const seen = new Set();
|
||||
const records = [];
|
||||
for (const r of rows) {
|
||||
for (const line of r.lines.split('\n')) {
|
||||
const i = line.indexOf('TUNING {');
|
||||
if (i < 0) continue;
|
||||
let rec;
|
||||
try { rec = JSON.parse(line.slice(i + 7)); } catch { continue; }
|
||||
if (!rec.card || !rec.winner) continue;
|
||||
const key = `${rec.machine}|${rec.card}|${rec.epoch}`;
|
||||
if (seen.has(key)) continue;
|
||||
seen.add(key);
|
||||
rec.run_id = r.run_id;
|
||||
records.push(rec);
|
||||
}
|
||||
}
|
||||
records.sort((a, b) => (b.ts || 0) - (a.ts || 0));
|
||||
|
||||
if (flag('--records')) {
|
||||
for (const r of records) if (!onlyCard || r.card === onlyCard) console.log(JSON.stringify(r));
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
const median = xs => { const s = [...xs].sort((a, b) => a - b); return s.length ? (s.length % 2 ? s[(s.length - 1) / 2] : (s[s.length / 2 - 1] + s[s.length / 2]) / 2) : 0; };
|
||||
|
||||
// Per card model (the worker's device name), per variant: every timed MH/s of every race (a race that could not time a
|
||||
// variant leaves it null), plus MH per watt from the race's power draw when the card reported one.
|
||||
const cards = {};
|
||||
for (const r of records) {
|
||||
if (onlyCard && r.card !== onlyCard) continue;
|
||||
const c = cards[r.card] ??= { worker: r.worker, vendor: r.vendor, machines: new Set(), races: 0, variants: {}, base: [], power: [] };
|
||||
c.machines.add(r.machine);
|
||||
c.races += 1;
|
||||
if (r.base_mhs > 0) c.base.push(r.base_mhs);
|
||||
if (r.power_w > 0) c.power.push(r.power_w);
|
||||
for (const [name, mhs] of Object.entries(r.variants || {})) {
|
||||
if (typeof mhs !== 'number' || mhs <= 0) continue;
|
||||
const v = c.variants[name] ??= { mhs: [], mhw: [] };
|
||||
v.mhs.push(mhs);
|
||||
if (r.power_w > 0) v.mhw.push(mhs / r.power_w);
|
||||
}
|
||||
}
|
||||
|
||||
const tuning = { updated: new Date().toISOString().replace(/\.\d{3}Z$/, 'Z'), window_days: days, cards: {} };
|
||||
const tableRows = [];
|
||||
for (const [card, c] of Object.entries(cards)) {
|
||||
const stats = Object.entries(c.variants).map(([name, v]) => ({ name, samples: v.mhs.length, mhs: median(v.mhs), mhw: v.mhw.length ? median(v.mhw) : 0 }));
|
||||
const baseMhs = stats.find(s => s.name === 'base')?.mhs || median(c.base);
|
||||
const eligible = stats.filter(s => s.samples >= minSamples);
|
||||
const ranked = (eligible.length ? eligible : stats).sort((a, b) => (by === 'mhw' ? b.mhw - a.mhw : b.mhs - a.mhs));
|
||||
for (const s of stats.sort((a, b) => b.mhs - a.mhs)) {
|
||||
tableRows.push({ card, variant: s.name, samples: s.samples, 'median MH/s': Number(s.mhs.toFixed(3)), 'vs base %': baseMhs ? Number(((s.mhs / baseMhs - 1) * 100).toFixed(2)) : null, 'MH/W': s.mhw ? Number(s.mhw.toFixed(3)) : null });
|
||||
}
|
||||
if (!ranked.length) continue;
|
||||
const best = ranked[0];
|
||||
tuning.cards[card] = {
|
||||
variant: best.name,
|
||||
race: !flag('--pin'),
|
||||
candidates: ranked.slice(0, 3).map(s => s.name),
|
||||
samples: best.samples,
|
||||
races: c.races,
|
||||
machines: c.machines.size,
|
||||
mhs: Number(best.mhs.toFixed(3)),
|
||||
base_mhs: Number(baseMhs.toFixed(3)),
|
||||
gain_pct: baseMhs ? Number(((best.mhs / baseMhs - 1) * 100).toFixed(2)) : 0,
|
||||
mh_per_w: best.mhw ? Number(best.mhw.toFixed(3)) : 0,
|
||||
worker: c.worker,
|
||||
by,
|
||||
};
|
||||
}
|
||||
|
||||
if (!records.length) { console.log(`No TUNING records in the last ${days} day(s). The workers log one per hourly race once the app with the race is on the machines.`); process.exit(0); }
|
||||
console.log(`${records.length} race record(s) from ${new Set(records.map(r => r.machine)).size} machine(s), last ${days} day(s), winner by ${by === 'mhw' ? 'MH per watt' : 'median MH/s'}, at least ${minSamples} sample(s) to be eligible`);
|
||||
console.table(tableRows);
|
||||
for (const [card, t] of Object.entries(tuning.cards)) console.log(`${card}: ${t.variant} (${t.samples} samples, ${t.mhs} MH/s, ${t.gain_pct >= 0 ? '+' : ''}${t.gain_pct}% over base ${t.base_mhs}${t.mh_per_w ? `, ${t.mh_per_w} MH/W` : ''}); candidates ${t.candidates.join(', ')}; race ${t.race}`);
|
||||
if (outFile) {
|
||||
writeFileSync(outFile, JSON.stringify(tuning, null, 2) + '\n');
|
||||
console.log(`written ${outFile}; publish with: packaging/ota/publish-manifest.sh --version <current> --tuning ${outFile} [--deploy]`);
|
||||
}
|
||||
Loading…
Reference in a new issue