Fleet learning for the kernel race: the TUNING record, the aggregator, the manifest's tuning object, the PC 1 race job
The engine turns a worker's race line into one TUNING {json} line in the app log (card model as the worker names it,
driver, arch, program class loads and wide loads, every variant's MH/s, winner, gain, the card's power cap and
draw, MH per watt), which the existing intake receives; the card state carries the variant for the dashboard and
one event per race. tools/tuning.mjs aggregates the records from miner_logs per card model (median MH/s or MH per
watt, at least 3 samples, de-duplicated per race) and writes tuning.json; publish-manifest.sh --tuning puts it in
the signed manifest (and now takes --override for consensus.override; both are carried over from the current
manifest when not given, --no-tuning drops it); manifest.rs parses it; ota.rs writes <app data>/tuning.json and
removes it when the manifest drops it; procs::spawn takes an environment and every miner starts with
IGNEUM_TUNING_FILE, which its worker reads at every prepare. Dry run of the publisher against a scratch folder:
tuning and override written, carried over, dropped, signature verified.
docs/plans/miner-perf.md: the signed jobs for PC 1 (fetch the race build of the NVRTC worker, then
relay/playbooks/race-5090.ps1 with the miners stopped: 17 variants, 3 rounds, twice) with the exact publish
commands for the main session; not published by the agent.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
parent
460a99a3a2
commit
dffb556d4a
9 changed files with 533 additions and 10 deletions
|
|
@ -877,7 +877,7 @@ impl Engine {
|
|||
let seg = if self.node_starts > 1 { format!("-r{}", self.node_starts) } else { String::new() };
|
||||
let log = self.shared.runtime.log_dir.join(format!("node-{}{seg}.log", self.stamp));
|
||||
let args = self.node_args();
|
||||
match procs::spawn(Source::Node, &self.bins.node, &args, None, &log, &self.lines_tx) {
|
||||
match procs::spawn(Source::Node, &self.bins.node, &args, None, &log, &self.lines_tx, &[]) {
|
||||
Ok(p) => {
|
||||
self.shared.log(&format!("igneumd started (pid {}): {}", p.pid(), p.cmdline));
|
||||
let mut st = self.st();
|
||||
|
|
@ -917,7 +917,7 @@ impl Engine {
|
|||
fn start_watch(&mut self) {
|
||||
let args = vec!["watch".to_string(), "1000000000".to_string(), self.shared.runtime.rpc_url()];
|
||||
let log = self.shared.runtime.log_dir.join(format!("watch-{}.log", self.stamp));
|
||||
if let Ok(p) = procs::spawn(Source::Watch, &self.bins.miner, &args, None, &log, &self.lines_tx) {
|
||||
if let Ok(p) = procs::spawn(Source::Watch, &self.bins.miner, &args, None, &log, &self.lines_tx, &[]) {
|
||||
self.watch = Some(p);
|
||||
}
|
||||
self.watch_retry_at = Instant::now() + Duration::from_secs(4);
|
||||
|
|
@ -1041,7 +1041,9 @@ impl Engine {
|
|||
let log = self.shared.runtime.log_dir.join(format!("miner-{}-{}{seg}.log", self.miners[i].label, self.stamp));
|
||||
let cwd = if card.worker == "Metal" { None } else { Some(self.shared.runtime.app_dir.clone()) };
|
||||
self.miners[i].prepared = false;
|
||||
match procs::spawn(Source::Miner(card_idx), &self.bins.miner, &args, cwd.as_deref(), &log, &self.lines_tx) {
|
||||
// the fleet's per-card kernel tuning (from the signed manifest) reaches the GPU worker through the miner's environment
|
||||
let envs: Vec<(String, String)> = self.ota.tuning_path().map(|p| vec![("IGNEUM_TUNING_FILE".to_string(), p.display().to_string())]).unwrap_or_default();
|
||||
match procs::spawn(Source::Miner(card_idx), &self.bins.miner, &args, cwd.as_deref(), &log, &self.lines_tx, &envs) {
|
||||
Ok(p) => {
|
||||
self.shared.log(&format!("miner {} started (pid {}): {}", self.miners[i].label, p.pid(), p.cmdline));
|
||||
let mut st = self.st();
|
||||
|
|
@ -1267,7 +1269,7 @@ impl Engine {
|
|||
if self.telemetry.is_none() && now >= self.telemetry_retry_at {
|
||||
let args: Vec<String> = ["--query-gpu=index,power.draw,temperature.gpu,temperature.memory,power.limit", "--format=csv,noheader,nounits", "-l", "5"].iter().map(|s| s.to_string()).collect();
|
||||
let log = self.shared.runtime.log_dir.join(format!("gpu-{}.log", self.stamp));
|
||||
match procs::spawn(Source::Telemetry, &crate::platform::tool("nvidia-smi"), &args, None, &log, &self.lines_tx) {
|
||||
match procs::spawn(Source::Telemetry, &crate::platform::tool("nvidia-smi"), &args, None, &log, &self.lines_tx, &[]) {
|
||||
Ok(p) => self.telemetry = Some(p),
|
||||
Err(_) => self.telemetry_retry_at = now + Duration::from_secs(300),
|
||||
}
|
||||
|
|
@ -1415,6 +1417,12 @@ impl Engine {
|
|||
self.shared.log(&format!("consensus override changed ({}); the node restarts with it at a safe moment", p.display()));
|
||||
self.node_override_restart = true;
|
||||
}
|
||||
if let Some(p) = self.ota.take_tuning_change() {
|
||||
// no restart: a worker started before this reads the file at its next prepare only if it was started with
|
||||
// the path, so a miner without it is restarted at the next hour boundary by the usual path (exit 42 or
|
||||
// restart); a running worker that has the path picks the change up at the next prepare by itself
|
||||
self.shared.log(&format!("kernel tuning changed ({}); the workers read it at their next hourly prepare", p.display()));
|
||||
}
|
||||
if self.node_override_restart && self.node.is_some() && !self.node_external {
|
||||
// between hourly boundaries unless the switch is close
|
||||
let (eta, daa) = { let st = self.st(); (st.program.eta_s, st.node.daa) };
|
||||
|
|
@ -2028,6 +2036,8 @@ impl Engine {
|
|||
c.message = String::new();
|
||||
}
|
||||
}
|
||||
} else if text.contains(" worker: race ") {
|
||||
self.race_line(card, &card_name, text);
|
||||
} else if text.contains(" worker: ready ") {
|
||||
let prepare = text.contains(" prepare 1");
|
||||
let mut st = self.st();
|
||||
|
|
@ -2096,6 +2106,59 @@ impl Engine {
|
|||
}
|
||||
}
|
||||
|
||||
/// A worker's race line (one per hourly prepare, docs/design/miner-tuning.md):
|
||||
/// `race <epoch16> device <name> driver <d> arch <a> loads <n> wide <n> variants <k> <name>=<MH/s>/<regs>r/<warps>w ...
|
||||
/// winner <name> <MH/s> base <MH/s> gain <pct>% compile <ms> bench <ms> total <ms> ms [| <variant>: <why>]`.
|
||||
/// Becomes the fleet record: one `TUNING {json}` line in the app log (uploaded to the intake, aggregated by
|
||||
/// tools/tuning.mjs) with the card's power figures, plus the card state and one event.
|
||||
fn race_line(&mut self, card: usize, card_name: &str, text: &str) {
|
||||
let Some(body) = text.split(" worker: race ").nth(1) else { return };
|
||||
let Some(r) = parse_race(body) else { return };
|
||||
let (vendor, worker, power_limit_w, power_w, power_pct) = {
|
||||
let mut st = self.st();
|
||||
match st.mining.cards.get_mut(card) {
|
||||
Some(c) => {
|
||||
c.variant = r.winner.clone();
|
||||
c.race_mhs = r.mhs;
|
||||
c.race_gain_pct = r.gain_pct;
|
||||
c.race_variants = r.variants.len() as u32;
|
||||
(c.vendor.clone(), c.worker.clone(), c.power_limit_w, c.power_w, c.power_pct)
|
||||
}
|
||||
None => (String::new(), String::new(), 0.0, 0.0, 0),
|
||||
}
|
||||
};
|
||||
let record = serde_json::json!({
|
||||
"ts": crate::platform::unix_now_f().round(),
|
||||
"machine": self.shared.runtime.id8(),
|
||||
"app": VERSION,
|
||||
"card": r.device,
|
||||
"vendor": vendor,
|
||||
"worker": worker,
|
||||
"driver": r.driver,
|
||||
"arch": r.arch,
|
||||
"epoch": r.epoch,
|
||||
"loads": r.loads,
|
||||
"wide": r.wide,
|
||||
"variants": r.variants,
|
||||
"winner": r.winner,
|
||||
"mhs": r.mhs,
|
||||
"base_mhs": r.base_mhs,
|
||||
"gain_pct": r.gain_pct,
|
||||
"power_limit_w": power_limit_w,
|
||||
"power_w": power_w,
|
||||
"power_pct": power_pct,
|
||||
"mh_per_w": if power_w > 0.0 { (r.mhs / power_w * 1000.0).round() / 1000.0 } else { 0.0 },
|
||||
"total_ms": r.total_ms,
|
||||
"pinned": r.pinned,
|
||||
"tuned": r.tuned,
|
||||
"notes": r.notes,
|
||||
});
|
||||
self.shared.log(&format!("TUNING {record}"));
|
||||
if !r.winner.is_empty() {
|
||||
self.shared.event("build", &format!("{card_name} kernel race: {} at {:.1} MH/s ({:+.1}% over base, {} variants, {:.0} s)", r.winner, r.mhs, r.gain_pct, r.variants.len(), r.total_ms / 1000.0));
|
||||
}
|
||||
}
|
||||
|
||||
// ---- status, uploads, shutdown ---------------------------------------------------------------------------
|
||||
|
||||
fn status_line(&mut self) {
|
||||
|
|
@ -2242,9 +2305,82 @@ fn behind_from_warning(text: &str) -> Option<f64> {
|
|||
if behind > 0.0 { Some(behind) } else { None }
|
||||
}
|
||||
|
||||
/// A worker's race line after "race " (docs/design/miner-tuning.md section 3), parsed into the record's fields.
|
||||
#[derive(Debug, Default, PartialEq)]
|
||||
pub struct RaceParsed {
|
||||
pub epoch: String,
|
||||
pub device: String,
|
||||
pub driver: String,
|
||||
pub arch: String,
|
||||
pub loads: u64,
|
||||
pub wide: u64,
|
||||
/// variant name -> MH/s (null when the variant was discarded or not timed)
|
||||
pub variants: serde_json::Map<String, Value>,
|
||||
pub winner: String,
|
||||
pub mhs: f64,
|
||||
pub base_mhs: f64,
|
||||
pub gain_pct: f64,
|
||||
pub total_ms: f64,
|
||||
pub pinned: bool,
|
||||
pub tuned: bool,
|
||||
pub notes: String,
|
||||
}
|
||||
|
||||
pub fn parse_race(body: &str) -> Option<RaceParsed> {
|
||||
let (main, notes) = match body.split_once(" | ") {
|
||||
Some((m, n)) => (m, n),
|
||||
None => (body, ""),
|
||||
};
|
||||
let t: Vec<&str> = main.split_whitespace().collect();
|
||||
let after = |key: &str| t.iter().position(|x| *x == key).and_then(|i| t.get(i + 1)).map(|s| s.to_string());
|
||||
let num = |key: &str| after(key).and_then(|v| v.trim_end_matches('%').parse::<f64>().ok());
|
||||
let epoch = t.first()?.to_string();
|
||||
if epoch.len() < 8 || !epoch.chars().all(|c| c.is_ascii_hexdigit()) {
|
||||
return None;
|
||||
}
|
||||
let mut r = RaceParsed { epoch, device: after("device").unwrap_or_default(), driver: after("driver").unwrap_or_default(), arch: after("arch").unwrap_or_default(), loads: num("loads").unwrap_or(0.0) as u64, wide: num("wide").unwrap_or(0.0) as u64, ..Default::default() };
|
||||
let n = num("variants").unwrap_or(0.0) as usize;
|
||||
if let Some(i) = t.iter().position(|x| *x == "variants") {
|
||||
for tok in t.iter().skip(i + 2).take(n) {
|
||||
if let Some((name, rest)) = tok.split_once('=') {
|
||||
let mhs = rest.split('/').next().and_then(|m| m.parse::<f64>().ok());
|
||||
r.variants.insert(name.to_string(), mhs.map(Value::from).unwrap_or(Value::Null));
|
||||
}
|
||||
}
|
||||
}
|
||||
r.winner = after("winner").unwrap_or_default();
|
||||
r.mhs = t.iter().position(|x| *x == "winner").and_then(|i| t.get(i + 2)).and_then(|v| v.parse::<f64>().ok()).unwrap_or(0.0);
|
||||
r.base_mhs = num("base").unwrap_or(0.0);
|
||||
r.gain_pct = num("gain").unwrap_or(0.0);
|
||||
r.total_ms = num("total").unwrap_or(0.0);
|
||||
r.pinned = main.contains(" pinned by tuning");
|
||||
r.tuned = main.contains(" tuned order");
|
||||
r.notes = notes.to_string();
|
||||
Some(r)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{sync_decision, Reading};
|
||||
use super::{parse_race, sync_decision, Reading};
|
||||
|
||||
#[test]
|
||||
fn race_line_parses() {
|
||||
let line = "a1b2c3d4e5f60718 device NVIDIA_GeForce_RTX_5090 driver 581.4 arch sm_120 loads 128 wide 0 variants 4 base=115.900/40r/1w u2=117.200/40r/1w ldg=- w4=118.300/40r/4w winner w4 118.300 base 115.900 gain +2.07% compile 1234 bench 9876 total 11110 ms tuned order | ldg: compile: identifier __ldg undefined";
|
||||
let r = parse_race(line).unwrap();
|
||||
assert_eq!(r.epoch, "a1b2c3d4e5f60718");
|
||||
assert_eq!(r.device, "NVIDIA_GeForce_RTX_5090");
|
||||
assert_eq!((r.driver.as_str(), r.arch.as_str(), r.loads, r.wide), ("581.4", "sm_120", 128, 0));
|
||||
assert_eq!(r.variants.len(), 4);
|
||||
assert_eq!(r.variants["w4"], 118.3);
|
||||
assert!(r.variants["ldg"].is_null());
|
||||
assert_eq!((r.winner.as_str(), r.mhs, r.base_mhs, r.gain_pct, r.total_ms), ("w4", 118.3, 115.9, 2.07, 11110.0));
|
||||
assert!(r.tuned && !r.pinned);
|
||||
assert_eq!(r.notes, "ldg: compile: identifier __ldg undefined");
|
||||
// the Metal shape, pinned, no notes
|
||||
let r = parse_race("0000000000000000 device Apple_M5_Max driver macos-26.0.1 arch metal loads 128 wide 0 variants 2 base=-/1024t/32w u2=9.800/1024t/32w winner u2 9.800 base 0.000 gain +0.00% compile 300 bench 10 total 310 ms pinned by tuning").unwrap();
|
||||
assert!(r.pinned && r.winner == "u2" && r.variants["base"].is_null());
|
||||
assert!(parse_race("not a race line").is_none());
|
||||
}
|
||||
|
||||
fn r(blocks: u64, headers: u64, peers: u64, flag: Option<bool>) -> Reading {
|
||||
Reading { blocks, headers, peers, flag }
|
||||
|
|
|
|||
|
|
@ -11,9 +11,12 @@
|
|||
//! "version": "0.3.1", "published_at": "2026-10-04T13:00:00Z", "channel": "devnet",
|
||||
//! "platforms": { "mac": {"url","sha256","size","kind":"dmg"|"zip"}, "windows": {"url","sha256","size","kind":"inno-setup"} },
|
||||
//! "min_supported_version": "0.3.0", "notes": "one line",
|
||||
//! "consensus": { "activation_height": null|number, "deadline_note": "" }
|
||||
//! "consensus": { "activation_height": null|number, "deadline_note": "", "override": {...} },
|
||||
//! "tuning": { "updated": "...", "cards": { "<card model>": { "variant": "u2", "race": true, "candidates": [..] } } }
|
||||
//! }
|
||||
//! A platform that is missing is not updated (the Windows build lands later than the Mac one).
|
||||
//! `tuning` (4 October 2026, docs/design/miner-tuning.md) is the fleet's per-card kernel tuning: the engine writes it
|
||||
//! to <app data>/tuning.json and every GPU worker reads it at its next prepare (IGNEUM_TUNING_FILE).
|
||||
|
||||
#![allow(dead_code)]
|
||||
|
||||
|
|
@ -53,6 +56,9 @@ pub struct Manifest {
|
|||
/// consensus.override: the exact object the engine writes to <app data>/override.json for igneumd's
|
||||
/// --override-params-file (for example {"difficulty_v2_activation_daa": N}); signed with the rest of the manifest.
|
||||
pub override_params: Option<serde_json::Value>,
|
||||
/// tuning: the per-card kernel tuning object (tools/tuning.mjs writes it, publish-manifest.sh --tuning carries
|
||||
/// it), written as is to <app data>/tuning.json for the GPU workers; signed with the rest of the manifest.
|
||||
pub tuning: Option<serde_json::Value>,
|
||||
}
|
||||
|
||||
impl Manifest {
|
||||
|
|
@ -157,6 +163,11 @@ pub fn parse(text: &str) -> Result<Manifest, String> {
|
|||
Some(o) if !o.is_null() => return Err("consensus.override must be an object".into()),
|
||||
_ => None,
|
||||
},
|
||||
tuning: match v.get("tuning") {
|
||||
Some(t) if t.is_object() && t.get("cards").map(|c| c.is_object()).unwrap_or(false) => Some(t.clone()),
|
||||
Some(t) if !t.is_null() => return Err("tuning must be an object with a cards object".into()),
|
||||
_ => None,
|
||||
},
|
||||
})
|
||||
}
|
||||
|
||||
|
|
@ -371,6 +382,16 @@ mod tests {
|
|||
assert_eq!(ours, theirs);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tuning_parses() {
|
||||
let m = parse(r#"{"version":"0.3.4","platforms":{},"tuning":{"updated":"2026-10-04T20:00:00Z","cards":{"NVIDIA_GeForce_RTX_5090":{"variant":"u2-ldg","race":true,"candidates":["u2-ldg","ldg","base"]}}}}"#).unwrap();
|
||||
assert_eq!(m.tuning.as_ref().unwrap()["cards"]["NVIDIA_GeForce_RTX_5090"]["variant"], "u2-ldg");
|
||||
assert!(parse(r#"{"version":"0.3.4","platforms":{},"tuning":null}"#).unwrap().tuning.is_none());
|
||||
assert!(parse(r#"{"version":"0.3.4","platforms":{}}"#).unwrap().tuning.is_none());
|
||||
assert!(parse(r#"{"version":"0.3.4","platforms":{},"tuning":"fast"}"#).is_err());
|
||||
assert!(parse(r#"{"version":"0.3.4","platforms":{},"tuning":{"cards":[]}}"#).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn consensus_override_parses() {
|
||||
let m = parse(r#"{"version":"0.3.2","platforms":{},"consensus":{"activation_height":5000,"override":{"difficulty_v2_activation_daa":5000}}}"#).unwrap();
|
||||
|
|
|
|||
|
|
@ -109,6 +109,8 @@ pub struct Updater {
|
|||
/// the consensus override file written from the manifest, when it changed since the last take
|
||||
override_changed: Option<PathBuf>,
|
||||
override_daa: u64,
|
||||
/// the per-card tuning file written from the manifest, when it changed since the last take
|
||||
tuning_changed: Option<PathBuf>,
|
||||
/// Windows: the installer was started and the engine is still up (it stops us when it may run)
|
||||
apply_launched: Option<Instant>,
|
||||
/// the administrator prompt was not answered: no automatic retry before this (Install now still works)
|
||||
|
|
@ -157,6 +159,7 @@ impl Updater {
|
|||
staged_digest: String::new(),
|
||||
override_changed: None,
|
||||
override_daa: 0,
|
||||
tuning_changed: None,
|
||||
apply_launched: None,
|
||||
deferred_until: None,
|
||||
slot: manifest::slot_minute(&shared.runtime.id8()),
|
||||
|
|
@ -180,6 +183,7 @@ impl Updater {
|
|||
if let Ok(m) = manifest::parse(&text) {
|
||||
u.min_supported = m.min_supported_version.clone();
|
||||
u.write_override(shared, &m);
|
||||
u.write_tuning(shared, &m);
|
||||
}
|
||||
}
|
||||
u.settle_previous(shared);
|
||||
|
|
@ -224,6 +228,45 @@ impl Updater {
|
|||
shared.state.lock().unwrap().node.consensus_switch_daa = self.override_daa;
|
||||
}
|
||||
|
||||
/// The per-card tuning from the manifest (docs/design/miner-tuning.md): `tuning` written as is to
|
||||
/// <app data>/tuning.json; the GPU workers read it at their next prepare through IGNEUM_TUNING_FILE (the engine
|
||||
/// passes the path to every miner it starts). A manifest without tuning removes the file: the workers race
|
||||
/// every variant again.
|
||||
fn write_tuning(&mut self, shared: &Arc<Shared>, m: &Manifest) {
|
||||
let path = self.app_dir.join("tuning.json");
|
||||
match &m.tuning {
|
||||
Some(t) => {
|
||||
let text = t.to_string();
|
||||
if std::fs::read_to_string(&path).ok().as_deref() == Some(text.as_str()) {
|
||||
return;
|
||||
}
|
||||
if let Err(e) = std::fs::write(&path, &text) {
|
||||
shared.log(&format!("could not write {}: {e}", path.display()));
|
||||
return;
|
||||
}
|
||||
let cards = t.get("cards").and_then(|c| c.as_object()).map(|c| c.len()).unwrap_or(0);
|
||||
shared.event("info", &format!("kernel tuning from the signed manifest: {cards} card model(s), updated {}", t.get("updated").and_then(|u| u.as_str()).unwrap_or("?")));
|
||||
self.tuning_changed = Some(path);
|
||||
}
|
||||
None => {
|
||||
if path.is_file() && std::fs::remove_file(&path).is_ok() {
|
||||
shared.log("the manifest carries no kernel tuning any more; tuning.json removed (the workers race every variant again)");
|
||||
self.tuning_changed = Some(path);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The tuning file, once per change (the engine only logs it: the workers read the file at their next prepare).
|
||||
pub fn take_tuning_change(&mut self) -> Option<PathBuf> {
|
||||
self.tuning_changed.take()
|
||||
}
|
||||
|
||||
pub fn tuning_path(&self) -> Option<PathBuf> {
|
||||
let p = self.app_dir.join("tuning.json");
|
||||
if p.is_file() { Some(p) } else { None }
|
||||
}
|
||||
|
||||
/// The override file to start the node with, once per change.
|
||||
pub fn take_override_change(&mut self) -> Option<PathBuf> {
|
||||
self.override_changed.take()
|
||||
|
|
@ -616,6 +659,7 @@ impl Updater {
|
|||
self.manifest = Some(m.clone());
|
||||
self.min_supported = m.min_supported_version.clone();
|
||||
self.write_override(shared, &m);
|
||||
self.write_tuning(shared, &m);
|
||||
if let Some(e) = entry {
|
||||
if changed {
|
||||
self.file = None;
|
||||
|
|
|
|||
|
|
@ -40,10 +40,14 @@ pub struct Proc {
|
|||
pub exit_code: Option<i32>,
|
||||
}
|
||||
|
||||
/// Starts a program with stdout and stderr piped; every line goes to `tx` and to the log file.
|
||||
pub fn spawn(src: Source, exe: &Path, args: &[String], cwd: Option<&Path>, log_path: &Path, tx: &Sender<Line>) -> std::io::Result<Proc> {
|
||||
/// Starts a program with stdout and stderr piped; every line goes to `tx` and to the log file. `envs` are added to
|
||||
/// the child's environment (a miner passes IGNEUM_TUNING_FILE on to its GPU worker).
|
||||
pub fn spawn(src: Source, exe: &Path, args: &[String], cwd: Option<&Path>, log_path: &Path, tx: &Sender<Line>, envs: &[(String, String)]) -> std::io::Result<Proc> {
|
||||
let mut cmd = Command::new(exe);
|
||||
cmd.args(args).stdin(Stdio::piped()).stdout(Stdio::piped()).stderr(Stdio::piped());
|
||||
for (k, v) in envs {
|
||||
cmd.env(k, v);
|
||||
}
|
||||
if let Some(d) = cwd {
|
||||
cmd.current_dir(d);
|
||||
}
|
||||
|
|
|
|||
|
|
@ -70,6 +70,11 @@ pub struct CardState {
|
|||
pub temp_gpu: f64,
|
||||
pub temp_mem: f64,
|
||||
pub telemetry_at: f64,
|
||||
// the kernel variant race (docs/design/miner-tuning.md): what the worker's last race chose
|
||||
pub variant: String,
|
||||
pub race_mhs: f64,
|
||||
pub race_gain_pct: f64,
|
||||
pub race_variants: u32,
|
||||
}
|
||||
|
||||
#[derive(Clone, Serialize, Default)]
|
||||
|
|
|
|||
122
docs/plans/miner-perf.md
Normal file
122
docs/plans/miner-perf.md
Normal file
|
|
@ -0,0 +1,122 @@
|
|||
# Miner performance: the variant race on PC 1's RTX 5090
|
||||
|
||||
4 October 2026, evening. The race is built and measured on the Mac's Metal worker (`docs/bench-log.md`, "miner
|
||||
performance: variant racing"); the design is `docs/design/miner-tuning.md`. This plan is the same race on the RTX
|
||||
5090 with the NVRTC worker, as a signed job for PC 1 (machine `ae432dc7`), ready for the main session to publish.
|
||||
Nothing here was published by the agent: PC 1 mines, PC 2 runs a shard job.
|
||||
|
||||
## What the job does
|
||||
|
||||
Two jobs in the signed file, in this order (the app runs them one at a time, in file order):
|
||||
|
||||
1. `fetch-race-worker-20261004` (kind `fetch`, `--dir jobs --extract`): the race build of
|
||||
`igneum-worker-cuda.exe` (cross-compiled on the Mac with mingw from the `miner-perf` branch, the same
|
||||
`build-windows.sh` the package uses) lands in `%LOCALAPPDATA%\igneum\app\jobs\fetch-race-worker-20261004\`.
|
||||
2. `run-race-5090-20261004` (kind `run`, `--stop-miners`, cap 20 min): `relay/playbooks/race-5090.ps1`. The
|
||||
miners are stopped first, so the 5090 is the race's alone. The script copies the NVRTC DLLs from the installed
|
||||
app next to the fetched exe, takes this hour's prepared pack (`packs\prepare\<epoch16>-<day>`, else
|
||||
`packs\devnet`), prints the card's state from `nvidia-smi`, and runs the race twice
|
||||
(`--race --pack <dir> --race-rounds 3 --race-bench-ms 2000 --race-budget-s 400`): 17 variants, each
|
||||
self-tested against the pack's vectors and timed for 2 s, three interleaved rounds, best per variant. Every
|
||||
line starts with `RESULT`, so the dashboard's job strip and `tools/jobs.mjs` show them. The miners restart
|
||||
when the job ends (the engine's job path).
|
||||
|
||||
Expected wall time: a 1 GiB dataset build (about 10 s on the 5090 from the 4 October numbers), 17 NVRTC
|
||||
compiles in four threads, then 17 x 3 x about 2.3 s of timing, twice: about 5 min with the miners stopped.
|
||||
|
||||
## The script
|
||||
|
||||
`relay/playbooks/race-5090.ps1` (in the repo, parse-checked by `windows.yml` with the other playbooks):
|
||||
|
||||
```powershell
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$app = $env:IGNEUM_APP_DIR
|
||||
$jobs = Split-Path $env:IGNEUM_JOB_DIR
|
||||
$fetched = Join-Path $jobs 'fetch-race-worker-20261004'
|
||||
$exe = Join-Path $fetched 'igneum-worker-cuda.exe'
|
||||
if (-not (Test-Path $exe)) { Write-Output "RESULT race worker missing at $exe (the fetch job runs first)"; exit 2 }
|
||||
$inst = @("$env:LOCALAPPDATA\Programs\Igneum Miner", "$env:ProgramFiles\Igneum Miner") | Where-Object { Test-Path (Join-Path $_ 'igneum-worker-cuda.exe') } | Select-Object -First 1
|
||||
if (-not $inst) { Write-Output "RESULT no installed igneum-worker-cuda.exe found (the NVRTC DLLs come from there)"; exit 2 }
|
||||
Get-ChildItem $inst -Filter 'nvrtc*.dll' | Copy-Item -Destination $fetched -Force
|
||||
Write-Output "RESULT worker $exe with $((Get-ChildItem $fetched -Filter 'nvrtc*.dll').Count) NVRTC DLL(s) from $inst"
|
||||
$pack = Get-ChildItem "$app\packs\prepare" -Directory -ErrorAction SilentlyContinue | Sort-Object LastWriteTime -Descending | Select-Object -First 1
|
||||
if ($pack) { $pack = $pack.FullName } else { $pack = "$app\packs\devnet" }
|
||||
if (-not (Test-Path "$pack\seeds.txt")) { Write-Output "RESULT no pack with seeds.txt under $app\packs"; exit 2 }
|
||||
Write-Output "RESULT pack $pack"
|
||||
Get-Content "$pack\seeds.txt" | ForEach-Object { "RESULT seeds $_" }
|
||||
& nvidia-smi --query-gpu=name,driver_version,power.limit,power.default_limit,clocks.sm,clocks.mem,temperature.gpu --format=csv,noheader | ForEach-Object { "RESULT gpu $_" }
|
||||
foreach ($run in 1..2) {
|
||||
Write-Output "RESULT run $run start $(Get-Date -Format HH:mm:ss)"
|
||||
& $exe --race --pack $pack --race-rounds 3 --race-bench-ms 2000 --race-budget-s 400 2>&1 | ForEach-Object { "RESULT $_" }
|
||||
Write-Output "RESULT run $run exit $LASTEXITCODE"
|
||||
}
|
||||
& nvidia-smi --query-gpu=power.draw,clocks.sm,temperature.gpu --format=csv,noheader | ForEach-Object { "RESULT gpu-after $_" }
|
||||
exit 0
|
||||
```
|
||||
|
||||
Quoting, as in the jobs that worked on PC 2 (`run-20261004-173115`): a plain PowerShell file, no nested
|
||||
`bash -c` here because the NVRTC worker is a native Windows exe (no WSL); `$env:IGNEUM_*` come from the app
|
||||
(`jobrun.rs` `job_env`); the script runs with its job folder as the working directory. `& $exe ... 2>&1 |
|
||||
ForEach-Object { "RESULT $_" }` keeps stderr in the report.
|
||||
|
||||
## The publish commands (main session)
|
||||
|
||||
The zip: `igneum-worker-cuda-race.zip` (one file, `igneum-worker-cuda.exe`, the `miner-perf` build), produced
|
||||
by the agent in its scratchpad; copy it somewhere stable first (for example `~/Desktop/`). Its sha256 and size
|
||||
are in the agent's report and in `packaging/ota/publish-jobs.sh add` output (the script hashes the copy it puts in
|
||||
the downloads folder; the sha below is the agent's build and must match):
|
||||
|
||||
```bash
|
||||
cd ~/Projects/igneum # master or the miner-perf worktree: the script is the same
|
||||
# 1. the race worker (the fetch copies the zip into dl/<token>/ and hashes it)
|
||||
packaging/ota/publish-jobs.sh add --kind fetch --target ae432dc7 --platform windows \
|
||||
--file ~/Desktop/igneum-worker-cuda-race.zip --dir jobs --extract \
|
||||
--id fetch-race-worker-20261004 --title "Race build of the NVRTC worker for PC 1 (variant racing)" --expires-hours 48
|
||||
# 2. the race, miners stopped for its duration (about 5 min)
|
||||
packaging/ota/publish-jobs.sh add --kind run --target ae432dc7 --platform windows \
|
||||
--script relay/playbooks/race-5090.ps1 --stop-miners --timeout-minutes 20 \
|
||||
--id run-race-5090-20261004 --title "Variant race on the RTX 5090 (17 NVRTC variants, 3 rounds, twice)" --expires-hours 48 \
|
||||
--deploy
|
||||
# read back (the app polls every 10 minutes; Settings > remote jobs > Check now at once)
|
||||
node tools/jobs.mjs watch run-race-5090-20261004
|
||||
node tools/jobs.mjs run-race-5090-20261004 | grep '^RESULT'
|
||||
```
|
||||
|
||||
`--deploy` on the second `add` ships both (the first one leaves the file written, not deployed). The
|
||||
`--target` is PC 1 only. If `relay/playbooks/race-5090.ps1` is not on master yet, give its path in the
|
||||
`miner-perf` worktree (`~/Projects/igneum-wt-perf/relay/playbooks/race-5090.ps1`).
|
||||
|
||||
## What to read in the result
|
||||
|
||||
- `RESULT race <epoch16> device NVIDIA_GeForce_RTX_5090 driver ... variants 17 base=<MH/s>/<regs>r/1w w2=... winner <name> <MH/s> base <MH/s> gain <pct>% ...`,
|
||||
twice (run 1, run 2). The two winners should agree; the gain is the number for the bench log.
|
||||
- `| <variant>: <why>` after the line names any variant that was discarded (a compile error on sm_120, a vector
|
||||
mismatch) or not timed (budget).
|
||||
- `RESULT winner <name>: <regs> registers, <n> blocks/SM at <w> warp(s)/block`.
|
||||
- `RESULT gpu ...`: power limit (80% cap by the app's default) and clocks before; `gpu-after` after.
|
||||
|
||||
The gain on the 5090 is unknown until this runs. The Mac's Metal race (bench log) is a different compiler and a
|
||||
different memory system; its table says what the method finds there and nothing about NVIDIA. The range to
|
||||
measure on the 5090: from no gain (base stays, the compiler was already right for a 128-load random kernel) to
|
||||
whatever the load path and block variants give on a 1 GiB random-read kernel; the memory-hard kernel is
|
||||
bandwidth-bound by design, so a double-digit gain would be a surprise to check, not a claim.
|
||||
|
||||
## After the result
|
||||
|
||||
1. Add the two race lines to `docs/bench-log.md` under "miner performance: variant racing" (the PC 1 rows).
|
||||
2. If a variant wins twice: the race build of the worker goes into the next app version (it is the same
|
||||
`worker.cpp`; `build-windows.sh` then `packaging/windows/push-inputs.sh` as for any worker change), so every
|
||||
machine races at every prepare and logs the `TUNING` record; `node tools/tuning.mjs` shows the fleet table
|
||||
after a day; `node tools/tuning.mjs --write tuning.json` and `packaging/ota/publish-manifest.sh --version
|
||||
<current> --tuning tuning.json --deploy` send the per-card defaults back.
|
||||
3. If base wins both runs: the race costs the 5090 about 40 s of paused mining per hour for nothing on this
|
||||
program class; keep `--race on` for a day of records before deciding, since programs differ hour to hour.
|
||||
|
||||
## Untested
|
||||
|
||||
- The NVRTC worker's race has never run on a GPU (the Mac has none): the emulation suite covers the protocol and
|
||||
the base path only; the rewrites (`#pragma unroll`, `__ldg`/`__ldcg`/`__ldcs`, `__launch_bounds__`,
|
||||
`--maxrregcount`) are first compiled by the real NVRTC in this job. A variant NVRTC refuses is discarded with
|
||||
its reason in the line; the race still ends with a winner (base at worst).
|
||||
- The job's folder assumptions (`jobs\fetch-race-worker-20261004`, the per-user install folder for the DLLs) follow
|
||||
`jobrun.rs` and the 0.3.3 installer; PC 1 has not run a `fetch --dir jobs --extract` job before.
|
||||
|
|
@ -7,6 +7,10 @@
|
|||
# packaging/ota/publish-manifest.sh --version 0.3.1 --mac packaging/mac/dist/Igneum-Miner-0.3.1.dmg \
|
||||
# [--win packaging/windows/dist/Igneum-Miner-Setup-0.3.1.exe] --notes "one line of what changed" \
|
||||
# [--activation-height 120000 --deadline-note "difficulty v2"] [--min-supported 0.3.0] [--channel devnet] [--deploy]
|
||||
# [--override '{"difficulty_v2_activation_daa":33000}'] consensus.override, the node's --override-params-file object
|
||||
# [--tuning tuning.json | --no-tuning] the fleet's per-card kernel tuning (tools/tuning.mjs writes it;
|
||||
# docs/design/miner-tuning.md); carried over from the current
|
||||
# manifest when not given, as is consensus.override
|
||||
#
|
||||
# A platform you do not pass is carried over from the manifest already in the folder when that one has the same
|
||||
# version (the Windows build lands later than the Mac one: publish the Mac entry first, add the Windows entry when
|
||||
|
|
@ -29,6 +33,7 @@ TOKEN_FILE="$HOME/.config/igneum/dl-token"
|
|||
SIGNER="$ROOT/app/igneum-app/target/release/igneum-ota-sign"
|
||||
|
||||
VERSION="" MAC="" WIN="" NOTES="" ACTIVATION="" DEADLINE="" MIN_SUPPORTED="" CHANNEL="devnet" BASE="" DEST="" DEPLOY=0
|
||||
OVERRIDE="" TUNING_FILE="" NO_TUNING=0
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--version) VERSION="$2"; shift 2 ;;
|
||||
|
|
@ -39,6 +44,9 @@ while [ $# -gt 0 ]; do
|
|||
--deadline-note) DEADLINE="$2"; shift 2 ;;
|
||||
--min-supported) MIN_SUPPORTED="$2"; shift 2 ;;
|
||||
--channel) CHANNEL="$2"; shift 2 ;;
|
||||
--override) OVERRIDE="$2"; shift 2 ;;
|
||||
--tuning) TUNING_FILE="$2"; shift 2 ;;
|
||||
--no-tuning) NO_TUNING=1; shift ;;
|
||||
--base-url) BASE="$2"; shift 2 ;;
|
||||
--dest) DEST="$2"; shift 2 ;;
|
||||
--deploy) DEPLOY=1; shift ;;
|
||||
|
|
@ -112,12 +120,25 @@ fi
|
|||
if [ -z "$MIN_SUPPORTED" ] && [ -f "$OLD" ]; then
|
||||
MIN_SUPPORTED="$(python3 -c 'import json,sys; print(json.load(open(sys.argv[1])).get("min_supported_version",""))' "$OLD" 2>/dev/null || true)"
|
||||
fi
|
||||
# consensus.override and tuning: given here, else carried over from the current manifest (whatever its version)
|
||||
if [ -z "$OVERRIDE" ] && [ -f "$OLD" ]; then
|
||||
OVERRIDE="$(python3 -c 'import json,sys; o=json.load(open(sys.argv[1])).get("consensus",{}).get("override"); print(json.dumps(o, sort_keys=True, separators=(",",":")) if isinstance(o, dict) and o else "")' "$OLD" 2>/dev/null || true)"
|
||||
[ -n "$OVERRIDE" ] && echo "consensus.override: carried over from the current manifest: $OVERRIDE"
|
||||
fi
|
||||
TUNING=""
|
||||
if [ -n "$TUNING_FILE" ]; then
|
||||
[ -f "$TUNING_FILE" ] || { echo "missing: $TUNING_FILE" >&2; exit 1; }
|
||||
TUNING="$(python3 -c 'import json,sys; t=json.load(open(sys.argv[1])); assert isinstance(t.get("cards"), dict), "tuning needs a cards object"; print(json.dumps(t, sort_keys=True, separators=(",",":")))' "$TUNING_FILE")"
|
||||
elif [ "$NO_TUNING" = 0 ] && [ -f "$OLD" ]; then
|
||||
TUNING="$(python3 -c 'import json,sys; t=json.load(open(sys.argv[1])).get("tuning"); print(json.dumps(t, sort_keys=True, separators=(",",":")) if isinstance(t, dict) and isinstance(t.get("cards"), dict) else "")' "$OLD" 2>/dev/null || true)"
|
||||
[ -n "$TUNING" ] && echo "tuning: carried over from the current manifest ($(python3 -c 'import json,sys; print(len(json.loads(sys.argv[1])["cards"]))' "$TUNING") card model(s))"
|
||||
fi
|
||||
|
||||
# canonical JSON: sorted keys, no whitespace; the signature is over these exact bytes
|
||||
NEW="$DEST/igneum-app-latest.json.new"
|
||||
python3 - "$NEW" "$VERSION" "$CHANNEL" "$NOTES" "$MIN_SUPPORTED" "$ACTIVATION" "$DEADLINE" "$MAC_ENTRY" "$WIN_ENTRY" <<'PY'
|
||||
python3 - "$NEW" "$VERSION" "$CHANNEL" "$NOTES" "$MIN_SUPPORTED" "$ACTIVATION" "$DEADLINE" "$MAC_ENTRY" "$WIN_ENTRY" "$OVERRIDE" "$TUNING" <<'PY'
|
||||
import json, sys, datetime
|
||||
out, version, channel, notes, min_supported, activation, deadline, mac, win = sys.argv[1:10]
|
||||
out, version, channel, notes, min_supported, activation, deadline, mac, win, override, tuning = sys.argv[1:12]
|
||||
def entry(s):
|
||||
if not s: return None
|
||||
url, sha, size, kind = s.split()
|
||||
|
|
@ -131,6 +152,12 @@ m = {
|
|||
"notes": notes,
|
||||
"consensus": {"activation_height": int(activation) if activation else None, "deadline_note": deadline},
|
||||
}
|
||||
if override:
|
||||
o = json.loads(override)
|
||||
if not isinstance(o, dict) or not o: sys.exit("--override must be a non-empty JSON object")
|
||||
m["consensus"]["override"] = o
|
||||
if tuning:
|
||||
m["tuning"] = json.loads(tuning)
|
||||
open(out, "w").write(json.dumps(m, sort_keys=True, separators=(",", ":"), ensure_ascii=False))
|
||||
PY
|
||||
"$SIGNER" sign "$KEY" "$NEW" > "$NEW.sig"
|
||||
|
|
|
|||
31
relay/playbooks/race-5090.ps1
Normal file
31
relay/playbooks/race-5090.ps1
Normal file
|
|
@ -0,0 +1,31 @@
|
|||
# Variant race on PC 1's RTX 5090 with the NVRTC worker (docs/plans/miner-perf.md). A `run` job: the miners are
|
||||
# stopped first (--stop-miners), so the card is the race's alone. Needs the race worker fetched by the job
|
||||
# fetch-race-worker-20261004 (packaging/ota/publish-jobs.sh add --kind fetch ... --dir jobs --extract --id ...),
|
||||
# which lands next to this job's folder under <app data>\app\jobs\. Every line that matters starts with RESULT so
|
||||
# the dashboard and tools/jobs.mjs show it; the race lines are the same format the workers log in --serve.
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$app = $env:IGNEUM_APP_DIR
|
||||
$jobs = Split-Path $env:IGNEUM_JOB_DIR
|
||||
$fetched = Join-Path $jobs 'fetch-race-worker-20261004'
|
||||
$exe = Join-Path $fetched 'igneum-worker-cuda.exe'
|
||||
if (-not (Test-Path $exe)) { Write-Output "RESULT race worker missing at $exe (the fetch job runs first)"; exit 2 }
|
||||
# NVIDIA's NVRTC DLLs sit next to the installed worker (per-user install since 0.3.3, else Program Files)
|
||||
$inst = @("$env:LOCALAPPDATA\Programs\Igneum Miner", "$env:ProgramFiles\Igneum Miner") | Where-Object { Test-Path (Join-Path $_ 'igneum-worker-cuda.exe') } | Select-Object -First 1
|
||||
if (-not $inst) { Write-Output "RESULT no installed igneum-worker-cuda.exe found (the NVRTC DLLs come from there)"; exit 2 }
|
||||
Get-ChildItem $inst -Filter 'nvrtc*.dll' | Copy-Item -Destination $fetched -Force
|
||||
Write-Output "RESULT worker $exe with $((Get-ChildItem $fetched -Filter 'nvrtc*.dll').Count) NVRTC DLL(s) from $inst"
|
||||
# The pack: this hour's prepared pack (the miner writes one per pair under packs\prepare), else the one exported at the start
|
||||
$pack = Get-ChildItem "$app\packs\prepare" -Directory -ErrorAction SilentlyContinue | Sort-Object LastWriteTime -Descending | Select-Object -First 1
|
||||
if ($pack) { $pack = $pack.FullName } else { $pack = "$app\packs\devnet" }
|
||||
if (-not (Test-Path "$pack\seeds.txt")) { Write-Output "RESULT no pack with seeds.txt under $app\packs"; exit 2 }
|
||||
Write-Output "RESULT pack $pack"
|
||||
Get-Content "$pack\seeds.txt" | ForEach-Object { "RESULT seeds $_" }
|
||||
& nvidia-smi --query-gpu=name,driver_version,power.limit,power.default_limit,clocks.sm,clocks.mem,temperature.gpu --format=csv,noheader | ForEach-Object { "RESULT gpu $_" }
|
||||
# Two independent races of three interleaved rounds each (best of three per variant), 2 s per timed window
|
||||
foreach ($run in 1..2) {
|
||||
Write-Output "RESULT run $run start $(Get-Date -Format HH:mm:ss)"
|
||||
& $exe --race --pack $pack --race-rounds 3 --race-bench-ms 2000 --race-budget-s 400 2>&1 | ForEach-Object { "RESULT $_" }
|
||||
Write-Output "RESULT run $run exit $LASTEXITCODE"
|
||||
}
|
||||
& nvidia-smi --query-gpu=power.draw,clocks.sm,temperature.gpu --format=csv,noheader | ForEach-Object { "RESULT gpu-after $_" }
|
||||
exit 0
|
||||
133
tools/tuning.mjs
Normal file
133
tools/tuning.mjs
Normal file
|
|
@ -0,0 +1,133 @@
|
|||
#!/usr/bin/env node
|
||||
// Fleet learning for the GPU workers' kernel variants (docs/design/miner-tuning.md). Runs on the Mac.
|
||||
//
|
||||
// Every app logs one `TUNING {json}` line per hourly race (src/engine.rs race_line: card model, driver, program
|
||||
// class, every variant's MH/s, the winner, power cap, MH per watt). The app log reaches the intake (Neon table
|
||||
// miner_logs, the same one tools/logs.mjs reads). This script aggregates those records per card model and writes
|
||||
// the tuning object the over-the-air manifest carries back to every machine (publish-manifest.sh --tuning):
|
||||
//
|
||||
// node tools/tuning.mjs the table: per card model, per variant, samples, median MH/s, MH per watt
|
||||
// node tools/tuning.mjs --write tuning.json [--days 7] [--min-samples 3] [--by mhs|mhw] [--pin]
|
||||
// writes {"updated": ..., "cards": {"<card>": {"variant", "race", "candidates", "samples", "mhs", "base_mhs",
|
||||
// "gain_pct", "mh_per_w"}}}. The winner is the variant with the best median over the window; "candidates" are
|
||||
// the top three, which the workers race first at every prepare. --pin sets race false (the worker uses the
|
||||
// variant without a race; the record then carries only the self-test), the default keeps racing (the fleet keeps
|
||||
// learning while the card starts from the known best).
|
||||
// node tools/tuning.mjs --records [--days 7] [--card <model>] the raw records, newest first
|
||||
//
|
||||
// Reads DATABASE_URL from ~/.config/igneum/env. No dependencies: Neon HTTP SQL over fetch.
|
||||
import { readFileSync, writeFileSync } from 'node:fs';
|
||||
import { homedir } from 'node:os';
|
||||
|
||||
process.stdout.on('error', e => { if (e.code === 'EPIPE') process.exit(0); throw e; });
|
||||
|
||||
const env = readFileSync(`${homedir()}/.config/igneum/env`, 'utf8');
|
||||
const m = /^DATABASE_URL=(.*)$/m.exec(env);
|
||||
if (!m) { console.error('DATABASE_URL not found in ~/.config/igneum/env'); process.exit(1); }
|
||||
const url = m[1].trim().replace(/^['"]|['"]$/g, '');
|
||||
const host = new URL(url).hostname.replace('-pooler', '');
|
||||
|
||||
async function sql(query, params = []) {
|
||||
const r = await fetch(`https://${host}/sql`, {
|
||||
method: 'POST',
|
||||
headers: { 'Neon-Connection-String': url, 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({ query, params }),
|
||||
});
|
||||
const j = await r.json();
|
||||
if (!r.ok) throw new Error(j.message || JSON.stringify(j));
|
||||
return j.rows;
|
||||
}
|
||||
|
||||
const args = process.argv.slice(2);
|
||||
const opt = (name, def) => { const i = args.indexOf(name); return i >= 0 && i + 1 < args.length ? args[i + 1] : def; };
|
||||
const flag = name => args.includes(name);
|
||||
const days = Number(opt('--days', 7));
|
||||
const minSamples = Number(opt('--min-samples', 3));
|
||||
const by = opt('--by', 'mhs');
|
||||
const outFile = opt('--write', '');
|
||||
const onlyCard = opt('--card', '');
|
||||
|
||||
// Every app-log upload of the window; the TUNING lines out of them. The same race is uploaded many times (the log
|
||||
// is re-sent every minute), so records are de-duplicated on (machine, card, epoch).
|
||||
const rows = await sql(
|
||||
`SELECT machine, run_id, lines FROM miner_logs WHERE received_at > now() - ($1 || ' days')::interval AND lines LIKE '%TUNING {%' ORDER BY received_at DESC`,
|
||||
[String(days)]);
|
||||
const seen = new Set();
|
||||
const records = [];
|
||||
for (const r of rows) {
|
||||
for (const line of r.lines.split('\n')) {
|
||||
const i = line.indexOf('TUNING {');
|
||||
if (i < 0) continue;
|
||||
let rec;
|
||||
try { rec = JSON.parse(line.slice(i + 7)); } catch { continue; }
|
||||
if (!rec.card || !rec.winner) continue;
|
||||
const key = `${rec.machine}|${rec.card}|${rec.epoch}`;
|
||||
if (seen.has(key)) continue;
|
||||
seen.add(key);
|
||||
rec.run_id = r.run_id;
|
||||
records.push(rec);
|
||||
}
|
||||
}
|
||||
records.sort((a, b) => (b.ts || 0) - (a.ts || 0));
|
||||
|
||||
if (flag('--records')) {
|
||||
for (const r of records) if (!onlyCard || r.card === onlyCard) console.log(JSON.stringify(r));
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
const median = xs => { const s = [...xs].sort((a, b) => a - b); return s.length ? (s.length % 2 ? s[(s.length - 1) / 2] : (s[s.length / 2 - 1] + s[s.length / 2]) / 2) : 0; };
|
||||
|
||||
// Per card model (the worker's device name), per variant: every timed MH/s of every race (a race that could not time a
|
||||
// variant leaves it null), plus MH per watt from the race's power draw when the card reported one.
|
||||
const cards = {};
|
||||
for (const r of records) {
|
||||
if (onlyCard && r.card !== onlyCard) continue;
|
||||
const c = cards[r.card] ??= { worker: r.worker, vendor: r.vendor, machines: new Set(), races: 0, variants: {}, base: [], power: [] };
|
||||
c.machines.add(r.machine);
|
||||
c.races += 1;
|
||||
if (r.base_mhs > 0) c.base.push(r.base_mhs);
|
||||
if (r.power_w > 0) c.power.push(r.power_w);
|
||||
for (const [name, mhs] of Object.entries(r.variants || {})) {
|
||||
if (typeof mhs !== 'number' || mhs <= 0) continue;
|
||||
const v = c.variants[name] ??= { mhs: [], mhw: [] };
|
||||
v.mhs.push(mhs);
|
||||
if (r.power_w > 0) v.mhw.push(mhs / r.power_w);
|
||||
}
|
||||
}
|
||||
|
||||
const tuning = { updated: new Date().toISOString().replace(/\.\d{3}Z$/, 'Z'), window_days: days, cards: {} };
|
||||
const tableRows = [];
|
||||
for (const [card, c] of Object.entries(cards)) {
|
||||
const stats = Object.entries(c.variants).map(([name, v]) => ({ name, samples: v.mhs.length, mhs: median(v.mhs), mhw: v.mhw.length ? median(v.mhw) : 0 }));
|
||||
const baseMhs = stats.find(s => s.name === 'base')?.mhs || median(c.base);
|
||||
const eligible = stats.filter(s => s.samples >= minSamples);
|
||||
const ranked = (eligible.length ? eligible : stats).sort((a, b) => (by === 'mhw' ? b.mhw - a.mhw : b.mhs - a.mhs));
|
||||
for (const s of stats.sort((a, b) => b.mhs - a.mhs)) {
|
||||
tableRows.push({ card, variant: s.name, samples: s.samples, 'median MH/s': Number(s.mhs.toFixed(3)), 'vs base %': baseMhs ? Number(((s.mhs / baseMhs - 1) * 100).toFixed(2)) : null, 'MH/W': s.mhw ? Number(s.mhw.toFixed(3)) : null });
|
||||
}
|
||||
if (!ranked.length) continue;
|
||||
const best = ranked[0];
|
||||
tuning.cards[card] = {
|
||||
variant: best.name,
|
||||
race: !flag('--pin'),
|
||||
candidates: ranked.slice(0, 3).map(s => s.name),
|
||||
samples: best.samples,
|
||||
races: c.races,
|
||||
machines: c.machines.size,
|
||||
mhs: Number(best.mhs.toFixed(3)),
|
||||
base_mhs: Number(baseMhs.toFixed(3)),
|
||||
gain_pct: baseMhs ? Number(((best.mhs / baseMhs - 1) * 100).toFixed(2)) : 0,
|
||||
mh_per_w: best.mhw ? Number(best.mhw.toFixed(3)) : 0,
|
||||
worker: c.worker,
|
||||
by,
|
||||
};
|
||||
}
|
||||
|
||||
if (!records.length) { console.log(`No TUNING records in the last ${days} day(s). The workers log one per hourly race once the app with the race is on the machines.`); process.exit(0); }
|
||||
console.log(`${records.length} race record(s) from ${new Set(records.map(r => r.machine)).size} machine(s), last ${days} day(s), winner by ${by === 'mhw' ? 'MH per watt' : 'median MH/s'}, at least ${minSamples} sample(s) to be eligible`);
|
||||
console.table(tableRows);
|
||||
for (const [card, t] of Object.entries(tuning.cards)) console.log(`${card}: ${t.variant} (${t.samples} samples, ${t.mhs} MH/s, ${t.gain_pct >= 0 ? '+' : ''}${t.gain_pct}% over base ${t.base_mhs}${t.mh_per_w ? `, ${t.mh_per_w} MH/W` : ''}); candidates ${t.candidates.join(', ')}; race ${t.race}`);
|
||||
if (outFile) {
|
||||
writeFileSync(outFile, JSON.stringify(tuning, null, 2) + '\n');
|
||||
console.log(`written ${outFile}; publish with: packaging/ota/publish-manifest.sh --version <current> --tuning ${outFile} [--deploy]`);
|
||||
}
|
||||
Loading…
Reference in a new issue