diff --git a/app/igneum-app/src/engine.rs b/app/igneum-app/src/engine.rs index d87e2a2f4..7c966e1bc 100644 --- a/app/igneum-app/src/engine.rs +++ b/app/igneum-app/src/engine.rs @@ -51,6 +51,8 @@ pub enum Cmd { WorkerBuilt(usize, Result), /// local minus the latest block's timestamp, seconds (from the node's EVM RPC) ClockSample(f64), + /// the node readiness probe answered (src/execrpc.rs probe): did the RPC answer, does the exec follower hold a record + ExecProbe { answered: bool, has_record: bool }, /// another node holds the app's ports: what it answered (src/extnode.rs) ExternalNode(crate::extnode::Check), /// the node's own description over RPC (igneum_getNodeInfo), every 30 s @@ -556,10 +558,12 @@ struct MinerSlot { prepared: bool, // the pack is exported (and the worker built) for the next start last_status: Option, error_at: Option, - /// the per-card watchdog (src/watchdog.rs): one restart, then faulted + /// the per-card watchdog (src/watchdog.rs): restarts on the ladder, never a permanent fault watch: crate::watchdog::CardWatch, /// pack refusals (src/watchdog.rs): the pack is exported again before the restart, capped per epoch pack_rebuilds: crate::watchdog::PackRebuilds, + /// the next pack export is forced (a worker refused the pack); otherwise an export under 60 s old is reused (MF-4) + pack_force: bool, } pub struct Engine { @@ -675,6 +679,23 @@ pub struct Engine { node_watch: crate::watchdog::NodeWatch, /// the watchdogs' clock origin (they take seconds) t0: Instant, + /// node readiness (the project lead, 7 October 2026): a worker never starts before the node is synced AND its execution layer + /// reports an executed tip (igneum_getExecStatus); probed every 5 s off the engine thread + exec_ready: bool, + exec_probe_busy: bool, + exec_probe_next: Instant, + /// the last time the node's RPC answered anything (a sign of life for the node watchdog) + exec_answered_at: Option, + /// the node has been read as synced at least once since its start: before that its catch-up never counts + node_settled: bool, + /// fault reports sent to the log intake in the last hour (the cap) + fault_reports: Vec, + /// MF-6: the job that took the miners hold, when, and its cap in seconds + job_hold_owner: Option, + job_hold_since: Option, + job_hold_cap_s: f64, + /// MF-7: the orphan sweep (igneum-miner processes this engine does not track) runs at this time next + orphan_sweep_next: Instant, } impl Engine { @@ -779,6 +800,16 @@ impl Engine { sweep_retry: std::collections::HashMap::new(), sweep_attempts: std::collections::HashMap::new(), node_watch: crate::watchdog::NodeWatch::new(), + exec_ready: false, + exec_probe_busy: false, + exec_probe_next: now, + exec_answered_at: None, + node_settled: false, + fault_reports: Vec::new(), + job_hold_owner: None, + job_hold_since: None, + job_hold_cap_s: 3600.0, + orphan_sweep_next: now, t0: now, } } @@ -914,13 +945,12 @@ impl Engine { // live worker is re-armed (a faulted one reset first), its pack is exported again before the start // (prepared = false: the hour may have turned while paused), and a check 90 s later names any // enabled card that is not mining (`resume_check`). - let views: Vec = self.miners.iter().map(|m| ResumeSlot { faulted: m.watch.faulted().is_some(), live: m.proc.is_some() }).collect(); + let views: Vec = self.miners.iter().map(|m| ResumeSlot { faulted: m.watch.watchdog_restarts() > 0, live: m.proc.is_some() }).collect(); let now = Instant::now(); for i in slots_to_rearm_on_resume(&views) { let m = &mut self.miners[i]; - if m.watch.faulted().is_some() { - m.watch.event(0.0, crate::watchdog::Event::Reset); - } + // a user action: the ladder starts over + m.watch.event(0.0, crate::watchdog::Event::Reset); m.restart_at = Some(now); m.prepared = false; } @@ -933,7 +963,7 @@ impl Engine { self.shared.event("info", &format!("settings changed ({why}); the miners restart")); self.stop_miners("settings changed"); for m in self.miners.iter_mut() { - // a user action: a faulted card tries again + // a user action: the ladder starts over m.watch.event(0.0, crate::watchdog::Event::Reset); m.restart_at = Some(Instant::now()); } @@ -1024,6 +1054,19 @@ impl Engine { } if let Some(v) = info.version { if !v.is_empty() && st.node.version.is_empty() { st.node.version = v; } } } + Cmd::ExecProbe { answered, has_record } => { + self.exec_probe_busy = false; + if answered { + self.exec_answered_at = Some(Instant::now()); + } + if has_record && !self.exec_ready { + self.shared.log("node readiness: the execution layer reports an executed tip; the workers may start once the node is synced"); + } + if !has_record && self.exec_ready { + self.shared.log("node readiness: the execution layer reports no executed tip any more (a reset or a restart); the workers wait for it"); + } + self.exec_ready = has_record; + } Cmd::ClockSample(d) => { self.clock_samples.push(d); if self.clock_samples.len() > 9 { @@ -1375,6 +1418,18 @@ impl Engine { self.shared.event("warn", &format!("{name}: not usable ({problem}); its worker stopped. {}", crate::detect::PROBLEM_HINT)); } for (i, fresh) in diff.moved.iter() { + let driver_changed = self.st().mining.cards.get(*i).map(|c| !fresh.platform.is_empty() && c.platform != fresh.platform).unwrap_or(false); + if driver_changed { + // MF-4: a card held after a self-test failure tries again the minute its driver changes + let now_i = Instant::now(); + for m in self.miners.iter_mut().filter(|m| m.card == *i) { + if m.watch.driver_hold() { + m.watch.event(0.0, crate::watchdog::Event::DriverChanged); + m.restart_at = Some(now_i); + self.shared.event("info", &format!("{}: driver changed; its worker tries again now", fresh.name)); + } + } + } if let Some(c) = self.st().mining.cards.get_mut(*i) { self.shared.log(&format!("{}: device {} is now {} (the running worker keeps its device; the next start uses the new one)", c.name, c.device, fresh.device)); c.device = fresh.device.clone(); @@ -1642,6 +1697,9 @@ impl Engine { self.sync_prev = None; self.sync_stable_since = None; self.node_last_reading = None; + self.exec_ready = false; + self.exec_answered_at = None; + self.node_settled = false; self.shared.event(if self.node_starts == 1 { "ok" } else { "info" }, if self.node_starts == 1 { "node started" } else { "node restarted" }); } Err(e) => { @@ -1728,6 +1786,7 @@ impl Engine { error_at: None, watch: crate::watchdog::CardWatch::new(), pack_rebuilds: crate::watchdog::PackRebuilds::new(), + pack_force: false, }); } if self.miners.is_empty() { @@ -1808,6 +1867,12 @@ impl Engine { fn start_miner(&mut self, i: usize) { let card_idx = self.miners[i].card; let Some(card) = self.st().mining.cards.get(card_idx).cloned() else { return }; + // MF-7: a restart kills the old process before the new one starts; never two miners on one card + if let Some(mut p) = self.miners[i].proc.take() { + self.shared.log(&format!("miner {}: the previous process (pid {}) is still alive at restart; stopping it first", self.miners[i].label, p.pid())); + p.write_stdin("quit\n"); + p.stop(5); + } if self.shared.settings.lock().unwrap().address.is_empty() { self.shared.event("error", "no payout address; open settings and set one"); self.miners[i].restart_at = Some(Instant::now() + Duration::from_secs(60)); @@ -1870,10 +1935,14 @@ impl Engine { m.restart_at = None; m.watch.event(t, crate::watchdog::Event::Stopped); } + if any { + // MF-7: nothing of this engine's keeps hashing after a stop + self.sweep_orphan_miners("after a stop"); + } let mut st = self.st(); let paused = st.mining.paused; for c in st.mining.cards.iter_mut() { - if c.enabled && c.present() && c.state != "faulted" { + if c.enabled && c.present() { c.state = if paused { "off".into() } else { "waiting".into() }; c.hash_now = 0.0; c.pid = 0; @@ -1898,8 +1967,9 @@ impl Engine { let bins = self.bins.clone(); let vendor = self.st().mining.cards.get(card_idx).map(|c| c.vendor.clone()).unwrap_or_default(); let worker = self.miners[i].worker.clone(); + let force = std::mem::take(&mut self.miners[i].pack_force); std::thread::spawn(move || { - let r = if build { build_worker_from_source(&shared, &bins, &vendor) } else { export_pack(&shared, &bins).map(|_| worker) }; + let r = if build { build_worker_from_source(&shared, &bins, &vendor) } else { export_pack(&shared, &bins, force).map(|_| worker) }; shared.send(Cmd::WorkerBuilt(card_idx, r)); }); } @@ -3173,6 +3243,11 @@ impl Engine { if self.running { self.tick_node(now); self.tick_watch(now); + self.tick_exec_probe(now); + if now >= self.orphan_sweep_next { + self.orphan_sweep_next = now + Duration::from_secs(60); + self.sweep_orphan_miners("the minute sweep"); + } self.tick_miners(now); if let Some(at) = self.resume_check_at { if now >= at { @@ -3294,8 +3369,20 @@ impl Engine { if let Some(a) = self.jobs.tick(&self.shared) { self.job_action(a); } - if self.job_hold && !self.jobs.holds_miners() { + let release = if self.job_hold { + // MF-6: the hold belongs to the job that took it, releases when that job is gone (whatever runs next) + // or at its own cap; the runner's own flag is read too + let held_s = self.job_hold_since.map(|t| t.elapsed().as_secs_f64()).unwrap_or(0.0); + let active = self.jobs.active_id(); + crate::jobrun::hold_release(self.job_hold_owner.as_deref(), active.as_deref(), held_s, self.job_hold_cap_s).or(if !self.jobs.holds_miners() { Some("the runner released the miners") } else { None }) + } else { + None + }; + if let Some(why) = release { self.job_hold = false; + self.job_hold_owner = None; + self.job_hold_since = None; + self.shared.log(&format!("job hold released: {why}")); for c in self.st().mining.cards.iter_mut().filter(|c| c.state == "held" || c.state == "tuning") { c.state = "off".into(); c.sweep_state = if c.sweep_state == "running" { "idle".into() } else { c.sweep_state.clone() }; @@ -3317,6 +3404,9 @@ impl Engine { match a { Action::StopMiners(why) => { self.job_hold = true; + self.job_hold_owner = self.jobs.active_id(); + self.job_hold_since = Some(Instant::now()); + self.job_hold_cap_s = (self.jobs.active_cap_minutes().max(1) * 60) as f64; self.stop_miners(&why); for c in self.st().mining.cards.iter_mut().filter(|c| c.enabled) { // never a bare "off" at 0 MH/s: the row says why (a job holds the card) @@ -3325,6 +3415,17 @@ impl Engine { } self.jobs.miners_stopped(&self.shared); } + Action::ApplyCards(choices) => { + // the signed `cards` kind (7 October 2026): through the app's own card path, persisted, read back + let names: Vec = choices.iter().map(|c| format!("{} enabled={} identities={}", c.key, c.enabled, c.identities)).collect(); + self.shared.event("info", &format!("remote job: card settings: {}", names.join("; "))); + let keys: Vec = choices.iter().map(|c| c.key.clone()).collect(); + self.apply_cards(choices); + let readback: Vec<(String, bool, u32, u32)> = self.st().mining.cards.iter().filter(|c| keys.contains(&c.key)).map(|c| (c.key.clone(), c.enabled, c.identities, c.power_pct)).collect(); + if let Some(next) = self.jobs.cards_applied(&self.shared, readback) { + self.job_action(next); + } + } Action::CardsOff(keys) => { // `--cards-off` (6 October 2026): the runner switches the job's cards through the app's own card path // and keeps the exact choices that put them back; the script never touches /api/cards @@ -3468,6 +3569,7 @@ impl Engine { self.node_restarts += 1; let delay = jitter_secs(crate::platform::unix_now()); self.node_restart_at = Some(now + Duration::from_secs(delay)); + self.fault_report(None, "node-exit", &format!("igneumd exited with code {code} after {ran} s: {}", tail.last().cloned().unwrap_or_default())); let mut st = self.st(); st.node.state = "restarting".into(); st.node.synced = false; @@ -3518,13 +3620,19 @@ impl Engine { st.node.message = "no answer from the RPC yet (still opening its database?)".into(); } } - // the node watchdog: our node silent for 120 s (no reading, nothing accepted) is restarted in-process + // the node watchdog: our node with no sign of life for 120 s is restarted in-process. A sign of life is the + // watch reading, the exec probe's answer or an accepted block; and the node's catch-up never counts: until it + // has been read as synced once since its start only 30 minutes of total silence restarts it (PC 1, 7 October + // 2026: a 40-second restart loop while the node replayed) if self.node.is_some() && !self.node_external && self.node_restart_at.is_none() { - let silent_s = self.node_last_reading.map(|t| now.duration_since(t)).unwrap_or_else(|| self.node_started_at.elapsed()).as_secs_f64(); - if let Some(delay) = self.node_watch.tick(self.secs(now), true, silent_s, accepted_recent) { + let last_life = [self.node_last_reading, self.exec_answered_at, self.last_accepted].into_iter().flatten().max(); + let silent_s = last_life.map(|t| now.duration_since(t)).unwrap_or_else(|| self.node_started_at.elapsed()).as_secs_f64(); + if let Some(delay) = self.node_watch.tick(self.secs(now), true, silent_s, accepted_recent, self.node_settled) { let n = self.node_watch.restarts_in_window(); - self.shared.event("error", &format!("no reading from the node for {} s; the watchdog restarts it in {delay} s (restart {n} in the last 10 minutes)", silent_s as u64)); - self.restart_node("no reading from the node for 120 s", Duration::from_secs(delay)); + let why = if self.node_settled { "no sign of life from the node for 120 s" } else { "no answer from the node for 30 minutes after its start" }; + self.shared.event("error", &format!("{why}; the watchdog restarts it in {delay} s (restart {n} in the last 10 minutes)")); + self.fault_report(None, "node-silent", &format!("{why} (silent {} s); restart in {delay} s", silent_s as u64)); + self.restart_node(why, Duration::from_secs(delay)); } } } @@ -3532,6 +3640,10 @@ impl Engine { fn tick_miners(&mut self, now: Instant) { // a clock over the consensus bound: the node cannot sync and a miner would only submit rejected blocks let synced = self.st().node.synced && self.st().clock.severity != "block"; + // a worker starts, and is judged, only while the node is READY: synced and with an executed tip (the exec + // probe); PC 1, 7 October 2026: workers started into a node that was still catching up and were faulted for + // the silence that followed + let node_ready = synced && self.exec_ready; let paused = self.st().mining.paused; for i in 0..self.miners.len() { let card_idx = self.miners[i].card; @@ -3561,6 +3673,7 @@ impl Engine { self.shared.event("build", "program pack out of date, rebuilding"); self.shared.log(&format!("{name}: program pack out of date, rebuilding (export {n} of {cap} for epoch {epoch}): {why}")); self.miners[i].prepared = false; // prepare_worker exports the pack again before the start + self.miners[i].pack_force = true; self.miners[i].restart_at = Some(now); if let Some(c) = self.st().mining.cards.get_mut(card_idx) { c.state = "restarting".into(); @@ -3569,19 +3682,21 @@ impl Engine { } } crate::watchdog::PackAction::GiveUp { n: _, reason } => { - self.shared.event("error", &format!("{name}: {reason}")); - self.shared.log(&format!("{name}: {reason}; next try in 10 minutes or at the next hour")); + // never permanent: the next try in 5 minutes, and at every hour boundary + self.shared.event("error", &format!("{name}: {reason}; next try in 5 minutes")); + self.shared.log(&format!("{name}: {reason}; next try in 5 minutes or at the next hour")); + self.fault_report(Some(&name), "pack", &reason); self.miners[i].prepared = false; - self.miners[i].restart_at = Some(now + Duration::from_secs(600)); + self.miners[i].restart_at = Some(now + Duration::from_secs(300)); if let Some(c) = self.st().mining.cards.get_mut(card_idx) { - c.state = "failed".into(); + c.state = "restarting".into(); c.hash_now = 0.0; - c.message = reason; + c.message = format!("{reason}; next try in 5 minutes"); } } } } else if verdict != crate::watchdog::Action::None { - // exit 43: the miner gave up on its worker; once more, then the card is faulted + // exit 43: the miner gave up on its worker; a restart on the ladder self.watchdog_verdict(i, verdict, &tail); } else if is_stall_exit(code, &tail) { // N4: the miner found no new template for its stall span and stopped hashing and voting. Once: @@ -3615,6 +3730,8 @@ impl Engine { for l in &tail { self.shared.log(&format!(" miner: {l}")); } + let name = self.st().mining.cards.get(card_idx).map(|c| c.name.clone()).unwrap_or_else(|| self.miners[i].label.clone()); + self.fault_report(Some(&name), "miner-exit", &format!("the miner exited with code {code}: {}", tail.last().cloned().unwrap_or_default())); if let Some(c) = self.st().mining.cards.get_mut(card_idx) { c.state = "restarting".into(); c.restarts = self.miners[i].restarts; @@ -3623,9 +3740,10 @@ impl Engine { } } } else { - // the watchdog: no status for 90 s, or a zero rate for 60 s while synced, is one restart, then faulted + // the watchdog: no status for 90 s, or a zero rate for 60 s while the node is ready, is a restart + // on the ladder (10 s, 30 s, 2 min, 5 min, then every 5 min; never permanent) let t = self.secs(now); - let verdict = self.miners[i].watch.tick(t, synced); + let verdict = self.miners[i].watch.tick(t, node_ready); if verdict != crate::watchdog::Action::None { if let Some(mut p) = self.miners[i].proc.take() { p.write_stdin("quit\n"); @@ -3646,19 +3764,15 @@ impl Engine { } continue; } - if self.miners[i].watch.faulted().is_some() { - // faulted by the watchdog: not restarted until the user changes the card's settings or resumes - continue; - } if self.job_hold { // a remote job has the GPU; the miners wait until it lets go (src/jobrun.rs) continue; } - if paused || !synced { + if paused || !node_ready { if let Some(c) = self.st().mining.cards.get_mut(card_idx) { if !paused && c.state != "failed" { c.state = "waiting".into(); - c.message = "waiting for the node to sync".into(); + c.message = if synced { "waiting for the node to execute the tip (it is catching up)".into() } else { "waiting for the node to sync".into() }; } } continue; @@ -3710,27 +3824,18 @@ impl Engine { self.shared.log(&format!(" miner: {l}")); } match verdict { - Action::Restart(reason) => { + Action::Restart { reason, delay_s, attempt } => { self.miners[i].restarts += 1; - self.miners[i].restart_at = Some(Instant::now() + Duration::from_secs(1)); - self.shared.event("error", &format!("{name}: {reason}; the watchdog restarts the miner (restart {} of 1 before the card is marked faulted)", self.miners[i].watch.watchdog_restarts())); + self.miners[i].restart_at = Some(Instant::now() + Duration::from_secs(delay_s)); + let again = if attempt > 1 { format!(" (restart {attempt} since the card was last healthy; it keeps trying)") } else { String::new() }; + self.shared.event("error", &format!("{name}: {reason}; the worker restarts in {delay_s} s{again}")); + self.fault_report(Some(&name), "watchdog", &format!("{reason}; restart {attempt} in {delay_s} s")); if let Some(c) = self.st().mining.cards.get_mut(card_idx) { c.state = "restarting".into(); c.restarts = self.miners[i].restarts; c.hash_now = 0.0; c.pid = 0; - c.message = format!("watchdog: {reason}"); - } - } - Action::Fault(reason) => { - self.miners[i].restart_at = None; - self.shared.event("error", &format!("{name}: {reason}; the card is marked faulted and its miner is not restarted again (the other cards keep mining; change the card's settings or resume mining to try again)")); - if let Some(c) = self.st().mining.cards.get_mut(card_idx) { - c.state = "faulted".into(); - c.hash_now = 0.0; - c.pid = 0; - c.restart_in_s = 0; - c.message = format!("faulted: {reason}"); + c.message = format!("{reason}; trying again"); } } Action::None => {} @@ -3752,6 +3857,60 @@ impl Engine { st.node.message = format!("{why}; restart in {} s", delay.as_secs()); } + /// MF-7 (PC 1, 7 October 2026: orphan igneum-miner processes the app no longer tracked kept hitting the node's + /// template RPC beside the tracked ones). The engine owns every miner it started: any igneum-miner process whose + /// command line carries THIS engine's node RPC (the fence: the port this app's node listens on) and whose pid is + /// not in the tracked set is killed, one log line and one fault report per kill. Never by name alone. + fn sweep_orphan_miners(&mut self, why: &str) { + let fence = self.shared.runtime.rpc_url(); + let tracked: Vec = self.miners.iter().filter_map(|m| m.proc.as_ref().map(|p| p.pid())).collect(); + let orphans = crate::platform::miner_processes().into_iter().filter(|(pid, cmd)| cmd.contains(&fence) && !tracked.contains(pid)).collect::>(); + for (pid, cmd) in orphans { + crate::platform::kill_pid(pid); + self.shared.log(&format!("orphan miner killed ({why}): pid {pid}, not started by this engine: {}", short(&cmd, 200))); + self.fault_report(None, "orphan-miner", &format!("igneum-miner pid {pid} not tracked by the engine was killed ({why})")); + } + } + + /// Node readiness probe (the project lead, 7 October 2026): igneum_getExecStatus every 5 s off the engine thread while the + /// node is up. Its answer sets `exec_ready` (the workers' start gate with `synced`) and counts as the node's sign + /// of life for the node watchdog. + fn tick_exec_probe(&mut self, now: Instant) { + let node_up = self.node_external || self.node.is_some(); + if !node_up || self.exec_probe_busy || now < self.exec_probe_next { + return; + } + self.exec_probe_busy = true; + self.exec_probe_next = now + Duration::from_secs(5); + let port = self.shared.runtime.evm_port(); + let shared = self.shared.clone(); + std::thread::spawn(move || { + let (answered, has_record) = crate::execrpc::probe(port); + shared.send(Cmd::ExecProbe { answered, has_record }); + }); + } + + /// One fault line to the log intake, the moment it happens (the project lead, 7 October 2026: the team sees it before the + /// user): the card, the class, the reason and the app and node versions. At most 60 an hour, off the engine + /// thread; the same line is in the app log either way. + fn fault_report(&mut self, card: Option<&str>, class: &str, reason: &str) { + let now = Instant::now(); + self.fault_reports.retain(|t| now.duration_since(*t) < Duration::from_secs(3600)); + let line = format!("FAULT class={class} card=\"{}\" app={} reason=\"{}\"", card.unwrap_or("node"), VERSION, crate::platform::redact(reason).replace('"', "'")); + self.shared.log(&line); + let p = &self.shared.packaged; + if p.log_intake_url.is_empty() || p.log_intake_key.is_empty() || self.fault_reports.len() >= 60 { + return; + } + self.fault_reports.push(now); + let (url, key, machine, run_id) = (p.log_intake_url.clone(), p.log_intake_key.clone(), format!("{}-{}", self.shared.runtime.host, self.shared.runtime.id8()), format!("{}-{}", self.label_base, self.stamp)); + let label = format!("fault-{}", self.label_base); + let text = format!("{}\n{:.3} {line}", self.shared.upload_header(), crate::platform::unix_now_f()); + std::thread::spawn(move || { + crate::update::upload_text(&url, &key, &label, &machine, &run_id, &text); + }); + } + fn bins_worker_missing(&self, card_idx: usize) -> bool { let st = self.st(); match st.mining.cards.get(card_idx).map(|c| c.worker.as_str()) { @@ -4039,14 +4198,15 @@ impl Engine { if self.shared.ladder.lock().unwrap().mark("synced", crate::platform::unix_now_f()) { self.shared.save_ladder(); } let t = self.secs(now); let names: Vec = self.st().mining.cards.iter().map(|c| c.name.clone()).collect(); + self.node_settled = true; for m in self.miners.iter_mut() { - // a card faulted for silence while the node was not ready starts again now (PC 1, 7 October 2026) - if m.watch.faulted_by_node() { + // a restart the node's readiness explained does not count against the card (PC 1, 7 October 2026) + if m.watch.restarted_for_node() { m.watch.event(t, crate::watchdog::Event::NodeSynced); let name = names.get(m.card).cloned().unwrap_or_else(|| m.label.clone()); - self.shared.event("info", &format!("{name}: starts again now the node is synced (it was faulted for silence while the node was syncing)")); + self.shared.event("info", &format!("{name}: starts again now the node is synced (its earlier silence was the node's catch-up, not the card's)")); } - if m.proc.is_none() && m.restart_at.is_none() && m.watch.faulted().is_none() { + if m.proc.is_none() && m.restart_at.is_none() { m.restart_at = Some(now); } } @@ -4073,7 +4233,11 @@ impl Engine { self.shared.event("info", &format!("{card_name}: the node is not answering block templates yet; the miner keeps asking (the node catches up after a restart)")); } if let Some(c) = self.st().mining.cards.get_mut(card) { - c.message = "waiting for the node to answer block templates".into(); + // never while the card hashes (PC 1, 7 October 2026 11:4x UK: the label sat on a card accepting shares); + // a timed-out fetch for one identity beside a healthy rate is the node's latency, said by the STATUS line + if c.hash_now <= 0.0 { + c.message = "waiting for the node to answer block templates".into(); + } } return; } @@ -4097,6 +4261,21 @@ impl Engine { // the worker refused its program pack; the miner rebuilds the pack and restarts the worker itself // (or exits 44 for us to export it): the strip and the card name the condition in plain words self.pack_notice(i, card, &card_name, &why); + } else if text.contains("worker error") && is_self_test_failure(text) { + // MF-4: a worker that fails its self-test is not restarted every few seconds; the card is held for + // 30 minutes with the reason on its row, or until its driver changes + let reason = short(text.split("worker error:").nth(1).unwrap_or(text).trim(), 160); + let t = self.secs(Instant::now()); + let verdict = self.miners[i].watch.event(t, crate::watchdog::Event::SelfTestFailed(&reason)); + if let Some(mut p) = self.miners[i].proc.take() { + p.write_stdin("quit\n"); + p.stop(3); + } + self.watchdog_verdict(i, verdict, &[]); + if let Some(c) = self.st().mining.cards.get_mut(card) { + c.message = format!("not usable on this driver: {reason}; next try in 30 minutes or after a driver change"); + } + return; } else if text.contains("WORKER MISMATCH") || text.contains("worker error") || text.contains("worker exited") || text.contains("worker killed by a guard") || text.contains("panicked") || text.contains("CUDA error") || text.contains("submit error") { let now = Instant::now(); if now.duration_since(self.last_error_event) >= Duration::from_secs(30) { @@ -4131,6 +4310,7 @@ impl Engine { let t = self.secs(Instant::now()); self.miners[i].watch.event(t, crate::watchdog::Event::WorkerRestart(&reason)); self.shared.event("error", &format!("{card_name}: worker fault: {}; the miner restarts the worker", short(&reason, 200))); + self.fault_report(Some(&card_name), "worker-fault", &reason); if let Some(c) = self.st().mining.cards.get_mut(card) { c.faults += 1; c.hash_now = 0.0; @@ -4206,9 +4386,16 @@ impl Engine { if c.state == "starting" || c.state == "ready" { c.state = "mining".into(); } - if c.state == "mining" && s.hash_now > 0.0 { + if s.hash_now > 0.0 && (c.state == "mining" || c.message.starts_with("waiting for the node to answer") || c.message.starts_with("node slow")) { + // the first STATUS line with a rate clears every node-wait label, whatever the row's state word c.message = if s.mismatched > 0 { format!("{} share(s) failed the CPU re-check this run", s.mismatched) } else { String::new() }; } + // MF-5: the node's latency on the card in plain words, the worker kept + if s.template_wait_s > 0.0 && s.hash_now <= 0.0 { + c.message = format!("node slow: waiting for a block template for {:.0} s; the worker is kept", s.template_wait_s); + } else if s.template_ms >= 2000.0 && c.state == "mining" { + c.message = format!("node slow: a template takes {:.1} s{}", s.template_ms / 1000.0, if s.identities_active > 0 && s.identities_active < c.identities as u64 { format!("; {} of {} identities active until it answers faster", s.identities_active, c.identities) } else { String::new() }); + } } // miner-ui-5: the first-hour timeline's "card mining" mark, once per install, on the first status with a rate if s.hash_now > 0.0 && self.shared.ladder.lock().unwrap().mark("mining", crate::platform::unix_now_f()) { self.shared.save_ladder(); } @@ -4915,6 +5102,13 @@ mod tests { } } +/// A worker line that names a self-test failure (the vectors, the cache check or the source check did not pass): +/// `error 0 self-test FAIL ...`, `vectors 3 of 96 FAIL`, `Source check FAIL`, `self-test failed`. +pub(crate) fn is_self_test_failure(text: &str) -> bool { + let t = text.to_ascii_lowercase(); + (t.contains("self-test") || t.contains("selftest") || t.contains("source check") || t.contains("vectors")) && (t.contains("fail") || t.contains("mismatch")) +} + fn short(s: &str, n: usize) -> String { if s.chars().count() <= n { s.to_string() } else { format!("{}...", s.chars().take(n).collect::()) } } @@ -4945,13 +5139,27 @@ fn civil_from_days(z: i64) -> (i64, u32, u32) { /// OpenCL worker start then refused ("the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT") until the next /// export. The second export of a pair rewrites the same pack, which is harmless. static EXPORT_LOCK: std::sync::Mutex<()> = std::sync::Mutex::new(()); +/// When the pack was last exported: an export under `EXPORT_REUSE_S` old is reused unless forced (MF-4, 7 October +/// 2026: a card whose worker failed every few seconds exported on every restart and held two healthy cards in +/// "loading the program" past the watchdog). +static LAST_EXPORT: std::sync::Mutex> = std::sync::Mutex::new(None); +const EXPORT_REUSE_S: u64 = 60; /// Exports this hour's program pack from the node to \packs\devnet (the prebuilt workers read it with --pack). +/// One export per minute serves every card; `force` (a refused pack) exports again now. #[allow(unused_variables)] -fn export_pack(shared: &Arc, bins: &Bins) -> Result<(), String> { +fn export_pack(shared: &Arc, bins: &Bins, force: bool) -> Result<(), String> { let _one_at_a_time = EXPORT_LOCK.lock().unwrap_or_else(|e| e.into_inner()); let pack = shared.runtime.app_dir.join("packs").join("devnet"); let _ = std::fs::create_dir_all(&pack); + if !force && pack.join("seeds.txt").exists() { + let fresh = LAST_EXPORT.lock().unwrap_or_else(|e| e.into_inner()).map(|t| t.elapsed() < Duration::from_secs(EXPORT_REUSE_S)).unwrap_or(false); + if fresh { + shared.log("export-pack: reusing the pack exported under a minute ago"); + return Ok(()); + } + } + *LAST_EXPORT.lock().unwrap_or_else(|e| e.into_inner()) = Some(Instant::now()); let out = crate::detect::run_timeout(std::process::Command::new(&bins.miner).args(["export-pack", &shared.runtime.rpc_url(), &pack.display().to_string()]), None, Duration::from_secs(120)).unwrap_or_default(); shared.log(&format!("export-pack: {}", out.lines().last().unwrap_or("no output"))); if pack.join("seeds.txt").exists() { Ok(()) } else { Err("export-pack wrote no seeds.txt (is the node reachable?)".into()) } diff --git a/app/igneum-app/src/execrpc.rs b/app/igneum-app/src/execrpc.rs index be4674c22..29cbe478f 100644 --- a/app/igneum-app/src/execrpc.rs +++ b/app/igneum-app/src/execrpc.rs @@ -61,6 +61,15 @@ pub fn status_has_record(result: &Value) -> bool { result.get("executedTipHash").and_then(|h| h.as_str()).map(|h| !h.is_empty()).unwrap_or(false) } +/// The engine's readiness probe (every 5 s): did the node's RPC answer at all, and does the follower hold a record. +/// The first is the node watchdog's sign of life; the second, with `synced`, is the workers' start gate. +pub fn probe(evm_port: u16) -> (bool, bool) { + match post(evm_port, "igneum_getExecStatus", json!([]), Duration::from_secs(5)) { + Ok(r) => (true, status_has_record(&r)), + Err(e) => (!e.contains("did not answer"), false), + } +} + /// The gate: a safe method goes out; a gated one waits for a record; an unclassified method is refused (add it to a list). pub fn call(evm_port: u16, method: &str, params: Value, timeout: Duration) -> Result { if SAFE_ON_EMPTY.contains(&method) { diff --git a/app/igneum-app/src/jobrun.rs b/app/igneum-app/src/jobrun.rs index a6a1d26f5..3974ee10a 100644 --- a/app/igneum-app/src/jobrun.rs +++ b/app/igneum-app/src/jobrun.rs @@ -87,6 +87,9 @@ pub enum Action { StopMiners(String), RestartMiners, RestartNode, + /// The signed `cards` kind: apply these per-card choices through the app's own card path (persisted), then + /// call `cards_applied` with the read-back. + ApplyCards(Vec), /// The relaunch helper was started; the engine quits now. RestartApp, UpdateNow, @@ -564,6 +567,23 @@ impl Jobs { let cards_off: Vec = if job.kind == "run" { job.list_param("cards_off") } else { vec![] }; self.active = Some(Active { job: job.clone(), run_id: run_id.clone(), started: Instant::now(), started_unix: now, waiting_for_miners: needs_miners_stopped, holds_miners: false, waiting_for_cards: !cards_off.is_empty(), cards: CardHold::default(), ctl: ctl.clone() }); match job.kind.as_str() { + "cards" => { + // engine-side: refused at once for a card this machine does not have; else applied through the + // app's own card path and reported with the read-back (cards_applied) + let live: Vec<(String, bool, u32)> = shared.state.lock().unwrap().mining.cards.iter().filter(|c| c.present()).map(|c| (c.key.clone(), c.enabled, c.identities)).collect(); + let (choices, missing) = cards_job_choices(&job, &live); + let sink = Sink::new(shared, &job, &self.dir); + if !missing.is_empty() { + let summary = format!("refused: this machine has no card {}", missing.join(", ")); + sink.line(&format!("cards: {summary}; present: {}", live.iter().map(|c| c.0.as_str()).collect::>().join(", "))); + let uploaded = report(shared, &job, &sink, "failed", 2, now, crate::platform::unix_now(), &summary, json!({ "present": live.iter().map(|c| c.0.clone()).collect::>() })); + let o = Outcome { status: "failed".into(), exit: 2, summary, results: vec![], uploaded }; + self.event(shared, Event::Finished { id: job.id.clone(), outcome: o }); + return None; + } + sink.line(&format!("cards: applying {}", choices.iter().map(|c| format!("{} enabled={} identities={}{}", c.key, c.enabled, c.identities, c.power_pct.map(|p| format!(" power_pct={p}")).unwrap_or_default())).collect::>().join("; "))); + return Some(Action::ApplyCards(choices)); + } "restart" | "update-now" => { // engine-side; the report says what was asked and the ledger closes at once let what = job.str_param("what"); @@ -597,6 +617,35 @@ impl Jobs { } } + /// The running job's id, for the hold rule (a hold belongs to the job that took it). + pub fn active_id(&self) -> Option { + self.active.as_ref().map(|a| a.job.id.clone()) + } + + /// The running job's cap in minutes (the hold's hard cap). + pub fn active_cap_minutes(&self) -> u64 { + self.active.as_ref().map(|a| a.job.timeout_minutes()).unwrap_or(60) + } + + /// Called by the engine once a `cards` job's choices are applied: the report carries the read-back of every + /// card named (what the engine holds now) and the job closes done. + pub fn cards_applied(&mut self, shared: &Arc, readback: Vec<(String, bool, u32, u32)>) -> Option { + let Some(a) = self.active.as_ref() else { return None }; + if a.job.kind != "cards" { + return None; + } + let (job, started) = (a.job.clone(), a.started_unix); + let sink = Sink::new(shared, &job, &self.dir); + let lines: Vec = readback.iter().map(|(k, e, i, p)| format!("{k} enabled={e} identities={i} power_pct={p}")).collect(); + for l in &lines { + sink.line(&format!("cards: read back {l}")); + } + let summary = format!("cards applied: {}", lines.join("; ")); + let uploaded = report(shared, &job, &sink, "done", 0, started, crate::platform::unix_now(), &summary, json!({ "cards": readback.iter().map(|(k, e, i, p)| json!({ "key": k, "enabled": e, "identities": i, "power_pct": p })).collect::>() })); + let o = Outcome { status: "done".into(), exit: 0, summary, results: lines, uploaded }; + self.event(shared, Event::Finished { id: job.id.clone(), outcome: o }) + } + /// Called by the engine once a job's `--cards-off` cards are switched off; `restore` puts them back exactly. /// Next: the miners, if the job asked for them too, else the script. pub fn cards_off_done(&mut self, shared: &Arc, restore: Vec) -> Option { @@ -1316,6 +1365,50 @@ fn elevated_wrapper(env_lines: &str, script: &str, out_file: &str) -> String { ) } +/// The choices a `cards` job asks for, matched to the machine's present cards (key exact, or the key with the device +/// index left out: `vendor::name`); missing keys are returned for the refusal. A field the job leaves out keeps the +/// card's current value. +pub fn cards_job_choices(job: &Job, live: &[(String, bool, u32)]) -> (Vec, Vec) { + let mut out = Vec::new(); + let mut missing = Vec::new(); + let list = job.params.get("cards").and_then(|v| v.as_array()).cloned().unwrap_or_default(); + for c in list { + let key = c.get("key").and_then(|v| v.as_str()).unwrap_or("").trim().to_string(); + let found = live.iter().find(|(k, _, _)| *k == key).or_else(|| { + let parts: Vec<&str> = key.splitn(3, ':').collect(); + live.iter().find(|(k, _, _)| { let lp: Vec<&str> = k.splitn(3, ':').collect(); parts.len() == 3 && lp.len() == 3 && lp[0] == parts[0] && lp[2] == parts[2] && (parts[1].is_empty() || parts[1] == lp[1]) }) + }); + match found { + Some((k, enabled, identities)) => out.push(crate::engine::CardChoice { + key: k.clone(), + enabled: c.get("enabled").and_then(|v| v.as_bool()).unwrap_or(*enabled), + identities: c.get("identities").and_then(|v| v.as_u64()).map(|n| n as u32).unwrap_or(*identities), + power_pct: c.get("power_pct").and_then(|v| v.as_u64()).map(|n| n as u32), + }), + None => missing.push(key), + } + } + (out, missing) +} + +/// The hold rule (MF-6, PC 1, 7 October 2026: a read-only job that followed a --stop-miners job kept the cards off +/// for its whole run): a hold belongs to the job that took it and releases the moment that job is no longer the +/// running one, whatever runs next, or when the owner's own cap has passed. Returns the reason to release, or None. +pub fn hold_release(owner: Option<&str>, active: Option<&str>, held_s: f64, cap_s: f64) -> Option<&'static str> { + match owner { + None => Some("no job owns the hold"), + Some(o) => { + if active != Some(o) { + Some("the job that took the hold is no longer running") + } else if held_s >= cap_s { + Some("the hold passed the job's own cap") + } else { + None + } + } + } +} + fn finish_ran(ran: Ran, what: &str) -> Result { match ran.code { Some(0) => Ok(Done { status: "done".into(), exit: 0, summary: format!("{what} finished, exit 0"), extra: json!({}) }), @@ -2058,3 +2151,23 @@ fn account_warning(ctx: &str) -> String { fn short(s: &str, n: usize) -> String { if s.chars().count() <= n { s.to_string() } else { format!("{}...", s.chars().take(n).collect::()) } } + +#[cfg(test)] +mod hold_tests { + use super::hold_release; + + /// MF-6 (PC 1, 7 October 2026): a --stop-miners job's hold outlived it into a read-only watch job for three minutes. + #[test] + fn a_hold_belongs_to_the_job_that_took_it() { + // the owner is still running, under its cap: the hold stays + assert_eq!(hold_release(Some("job-a"), Some("job-a"), 30.0, 3600.0), None); + // another job runs now: released at once, whatever that job is + assert_eq!(hold_release(Some("job-a"), Some("job-b"), 30.0, 3600.0), Some("the job that took the hold is no longer running")); + // no job runs: released + assert_eq!(hold_release(Some("job-a"), None, 30.0, 3600.0), Some("the job that took the hold is no longer running")); + // the owner's own cap passed: released and logged + assert_eq!(hold_release(Some("job-a"), Some("job-a"), 3601.0, 3600.0), Some("the hold passed the job's own cap")); + // a hold with no owner (an older engine state) never sticks + assert_eq!(hold_release(None, Some("job-b"), 1.0, 3600.0), Some("no job owns the hold")); + } +} diff --git a/app/igneum-app/src/jobs.rs b/app/igneum-app/src/jobs.rs index e54805bf9..c36c1d501 100644 --- a/app/igneum-app/src/jobs.rs +++ b/app/igneum-app/src/jobs.rs @@ -59,7 +59,7 @@ pub const JOBS_FILE: &str = "igneum-jobs.json"; /// the signature is over the bytes of `file`, so the same key and the same signer sign both forms. pub const JOBS_SIGNED_FILE: &str = "igneum-jobs.signed.json"; pub const JOBS_SIGNED_FORMAT: &str = "igneum-jobs-signed-1"; -pub const KINDS: &[&str] = &["run", "fetch", "collect", "restart", "update-now", "shard-benchmark", "build"]; +pub const KINDS: &[&str] = &["run", "fetch", "collect", "restart", "update-now", "shard-benchmark", "build", "cards"]; /// Requirements the engine knows how to probe (src/jobrun.rs). An unknown requirement is never satisfied. pub const KNOWN_REQUIRES: &[&str] = &["wsl", "wsl-prover", "nvidia"]; /// Named folders a `fetch` may write into, all under the app data root. @@ -418,6 +418,33 @@ pub fn validate_params(job: &Job) -> Result<(), String> { } } "update-now" => {} + "cards" => { + // the signed `cards` kind (7 October 2026): per-card enabled and identities, applied by the app through + // its own card path and persisted; it closes the exception of a one-off script POSTing /api/cards + let list = job.params.get("cards").and_then(|v| v.as_array()).cloned().unwrap_or_default(); + if list.is_empty() { + return Err("cards: params.cards is empty (a list of {key, enabled, identities})".into()); + } + for c in &list { + let key = c.get("key").and_then(|v| v.as_str()).unwrap_or(""); + if key.trim().is_empty() || !key.contains(':') { + return Err(format!("cards: key '{key}' is not a card key (vendor:device:name)")); + } + if c.get("enabled").map(|v| !v.is_boolean()).unwrap_or(false) { + return Err(format!("cards: {key}: enabled must be true or false")); + } + if let Some(n) = c.get("identities") { + if !n.as_u64().map(|n| (1..=64).contains(&n)).unwrap_or(false) { + return Err(format!("cards: {key}: identities must be 1 to 64")); + } + } + if let Some(n) = c.get("power_pct") { + if !n.as_u64().map(|n| (50..=100).contains(&n)).unwrap_or(false) { + return Err(format!("cards: {key}: power_pct must be 50 to 100")); + } + } + } + } "shard-benchmark" => { let url = job.str_param("zip_url"); if !url.is_empty() && !https_ok(&url) { @@ -824,6 +851,10 @@ mod tests { assert!(j("restart", r#"{"what":"everything"}"#).unwrap_err().contains("restart")); assert!(j("restart", r#"{"what":"miners"}"#).is_ok()); assert!(j("update-now", r#"{}"#).is_ok()); + assert!(j("cards", r#"{}"#).unwrap_err().contains("cards")); + assert!(j("cards", r#"{"cards":[{"key":"nvidia:0:NVIDIA GeForce RTX 5090","enabled":true,"identities":8}]}"#).is_ok()); + assert!(j("cards", r#"{"cards":[{"key":"5090","enabled":true}]}"#).unwrap_err().contains("card key")); + assert!(j("cards", r#"{"cards":[{"key":"nvidia:0:x","identities":65}]}"#).unwrap_err().contains("1 to 64")); assert!(j("shard-benchmark", r#"{}"#).unwrap_err().contains("sha256")); assert!(j("shard-benchmark", &format!(r#"{{"sha256":"{}","fixtures":["../x"]}}"#, "b".repeat(64))).unwrap_err().contains("fixture")); assert!(j("run", r#"{"script":"ls","shell":"zsh"}"#).unwrap_err().contains("shell")); diff --git a/app/igneum-app/src/platform.rs b/app/igneum-app/src/platform.rs index 42e67793d..87d4f7359 100644 --- a/app/igneum-app/src/platform.rs +++ b/app/igneum-app/src/platform.rs @@ -268,6 +268,49 @@ pub fn keep_awake_tick() { } /// Asks a child to stop. Unix: SIGTERM (the node closes its database cleanly). Windows: TerminateProcess through +/// Every `igneum-miner` process on this machine with its command line: (pid, command line). Windows reads +/// Win32_Process through PowerShell; unix reads `ps`. An empty list when the tool fails (the caller kills nothing). +pub fn miner_processes() -> Vec<(u32, String)> { + let out = if cfg!(windows) { + let mut c = std::process::Command::new(tool("powershell")); + c.args(["-NoProfile", "-Command", "Get-CimInstance Win32_Process -Filter \"Name='igneum-miner.exe'\" | ForEach-Object { \"$($_.ProcessId)|$($_.CommandLine)\" }"]); + crate::detect::run_timeout(&mut c, None, std::time::Duration::from_secs(20)) + } else { + let mut c = std::process::Command::new("ps"); + c.args(["-eo", "pid=,args="]); + crate::detect::run_timeout(&mut c, None, std::time::Duration::from_secs(10)) + }; + let Some(out) = out else { return vec![] }; + let mut v = Vec::new(); + for l in out.lines() { + let l = l.trim(); + let (pid, cmd) = if cfg!(windows) { + let Some((p, c)) = l.split_once('|') else { continue }; + (p.trim(), c.trim()) + } else { + let Some((p, c)) = l.split_once(' ') else { continue }; + (p.trim(), c.trim()) + }; + let Ok(pid) = pid.parse::() else { continue }; + if !cfg!(windows) && !(cmd.contains("igneum-miner ") || cmd.ends_with("igneum-miner")) { + continue; + } + if cmd.contains(" mine ") { + v.push((pid, cmd.to_string())); + } + } + v +} + +/// Ends one process by pid (Windows: taskkill /T /F; unix: SIGKILL). +pub fn kill_pid(pid: u32) { + if cfg!(windows) { + let _ = std::process::Command::new(tool("taskkill")).args(["/PID", &pid.to_string(), "/T", "/F"]).output(); + } else { + let _ = std::process::Command::new("kill").args(["-9", &pid.to_string()]).output(); + } +} + /// std (what today's launcher does with taskkill /F). pub fn terminate(child: &mut std::process::Child) { #[cfg(unix)] diff --git a/app/igneum-app/src/state.rs b/app/igneum-app/src/state.rs index e746ae3ef..5a926fd6c 100644 --- a/app/igneum-app/src/state.rs +++ b/app/igneum-app/src/state.rs @@ -77,7 +77,7 @@ pub struct CardState { pub detail: String, // memory, cores pub device: String, // the worker's --device value (Windows) pub enabled: bool, - pub state: String, // off | waiting | starting | ready | mining | restarting | failed | faulted (the watchdog gave up on it) | unusable (the OS reports a problem) | removed (unplugged) + pub state: String, // off | waiting | starting | ready | mining | restarting | failed (no "faulted": no fault is permanent, 7 October 2026) | unusable (the OS reports a problem) | removed (unplugged) pub hash_now: f64, // MH/s, the last interval pub hash_avg: f64, // MH/s since the start pub accepted: u64, diff --git a/app/igneum-app/src/update.rs b/app/igneum-app/src/update.rs index 2ac67c8f9..50b061368 100644 --- a/app/igneum-app/src/update.rs +++ b/app/igneum-app/src/update.rs @@ -94,6 +94,15 @@ pub fn upload_log(url: &str, key: &str, label: &str, machine: &str, run_id: &str let tail: String = String::from_utf8_lossy(&data).lines().filter(|l| !crate::platform::carries_token(l)).map(|l| crate::platform::redact(l)).collect::>().join("\n"); // the first line of every upload names the app, the machine and the node (the console parses it) let text = format!("{header}\n{tail}"); + upload_text(url, key, label, machine, run_id, &text) +} + +/// One upload of ready text (a fault report, a job's lines): the same intake, the same token guard. +pub fn upload_text(url: &str, key: &str, label: &str, machine: &str, run_id: &str, text: &str) -> bool { + if url.is_empty() || key.is_empty() || text.is_empty() { + return false; + } + let text: String = text.lines().filter(|l| !crate::platform::carries_token(l)).collect::>().join("\n"); let body = serde_json::json!({ "label": label, "machine": machine, "run_id": run_id, "lines": text }); let tmp = std::env::temp_dir().join(format!("igneum-upload-{}-{}.json", std::process::id(), label)); if std::fs::write(&tmp, body.to_string()).is_err() { diff --git a/app/igneum-app/src/watchdog.rs b/app/igneum-app/src/watchdog.rs index 6ae01f539..cec7068aa 100644 --- a/app/igneum-app/src/watchdog.rs +++ b/app/igneum-app/src/watchdog.rs @@ -2,15 +2,21 @@ //! rules can be unit-tested with recorded miner lines. //! //! Per card (`CardWatch`): a miner that prints no status line for 90 s, or reports a hash rate of 0 for 60 s while -//! the node is synced, is restarted once; when that recurs before five minutes of healthy status, the card is marked -//! faulted with the reason, its miner is not restarted again, and the other cards keep mining. The miner's own -//! worker restarts (`WORKER FAULT`, `worker exited`) suspend both rules until the worker is ready again, so the app -//! never restarts a miner that is already restarting its worker (no double restarts); if the worker is not back -//! within 180 s the app steps in. Exit code 43 (the miner gave up on its worker after three guard trips) counts -//! like a watchdog restart: once, then faulted. +//! the node is ready, is restarted; when that recurs before five minutes of healthy status the next restart waits +//! longer (10 s, 30 s, 2 min, 5 min, then every 5 min, for ever: the project lead, 7 October 2026, "it needs to be truly plug, +//! tune, play"; no fault is permanent and the old one-restart-then-faulted state no longer exists). The reason stays on +//! the card in plain words and the hash resumes on its own. A miner is judged only while the node is READY: synced +//! and with an executed tip (`igneum_getExecStatus`); with no template it cannot print status, so every silence clock +//! holds while the node catches up (PC 1, 7 October 2026: three workers faulted "no status line" during the node's +//! catch-up and stayed faulted until a cards-API bounce). The miner's own worker restarts (`WORKER FAULT`, +//! `worker exited`) suspend both rules until the worker is ready again, so the app never restarts a miner that is +//! already restarting its worker (no double restarts); if the worker is not back within 180 s the app steps in. +//! Exit code 43 (the miner gave up on its worker after three guard trips) is a watchdog restart on the same ladder. //! -//! Per node (`NodeWatch`): a node of ours that answers no `watch` reading for 120 s is restarted by the app, with a -//! growing delay when it repeats inside ten minutes. +//! Per node (`NodeWatch`): a node of ours that gives no sign of life for 120 s is restarted by the app, with a +//! growing delay when it repeats inside ten minutes. Its catch-up never counts: until the node has been read as +//! synced once since its start, silence is not held against it (a node that never answers at all is restarted after +//! 30 minutes), and any RPC answer (the watch reading, the exec status probe, an accepted block) is a sign of life. //! //! Review round 4, X21 (4 October 2026): the app now reads `mismatched=` and `faults=` from STATUS lines and the //! `WORKER FAULT` lines, and shows them on the card. @@ -20,8 +26,27 @@ pub const NO_STATUS_S: f64 = 90.0; /// A worker that gives no status within this many seconds of its start, with the node synced, is one restart then /// faulted (PC 1, 7 October 2026 04:51Z: the row asks for a fault line at 60 s). pub const START_S: f64 = 60.0; -/// The start of a fault reason the node's readiness, not the card, explains: released when the node syncs. +/// The start of a restart reason the node's readiness, not the card, explains: the ladder resets when the node syncs. const NODE_FAULTS: [&str; 2] = ["no status line from the miner", "the worker gave no status within"]; +/// Seconds before the n-th restart of a card for a repeated fault (the last value repeats for ever). +pub const RETRY_LADDER_S: [u64; 4] = [10, 30, 120, 300]; +/// A node that has never answered since its start is given this long before the watchdog restarts it. +pub const NODE_STARTUP_CAP_S: f64 = 1800.0; +/// A worker that has not reported its program loaded (`ready`) this long after its start is restarted on the ladder. +/// The status clocks start at `ready`, never before (MF-4, 7 October 2026: two cards sat in "loading the program" +/// behind a third card's export storm and were faulted at 90 s). +pub const LOAD_S: f64 = 300.0; +/// A miner that exits (any code but 0, 42, 43, 44) within this many seconds of its start is in a crash loop: its +/// restarts follow the ladder instead of the 5 to 60 s jitter. +pub const EARLY_EXIT_S: f64 = 120.0; +/// A card whose worker failed its self-test is held this long before the next try (or until its driver changes). +pub const SELF_TEST_HOLD_S: u64 = 1800; + +/// The delay before restart number `attempt` (1-based) of a card: the ladder, then its last step for ever. +pub fn retry_delay_s(attempt: u32) -> u64 { + let i = (attempt.max(1) as usize - 1).min(RETRY_LADDER_S.len() - 1); + RETRY_LADDER_S[i] +} /// Hash rate 0 while the node is synced and the worker is ready for this long: restart the miner. pub const ZERO_RATE_S: f64 = 60.0; /// The miner is restarting its own worker: the app waits this long for `ready` before it steps in. @@ -50,6 +75,12 @@ pub struct Status { pub faults: u64, pub restarts: u64, pub synced: bool, + /// seconds the miner has waited for a template with none known (`template_wait=`, 0.3.20 miners; 0 before) + pub template_wait_s: f64, + /// the node's last template time in ms (`template_ms=`; 0 when the line has none) + pub template_ms: f64, + /// identities the miner fetches for now (`identities_active=`; 0 when the line has none) + pub identities_active: u64, } /// Parses a miner STATUS line; None for any other line. @@ -64,6 +95,9 @@ pub fn parse_status(text: &str) -> Option { faults: kv_u64(text, "faults").unwrap_or(0), restarts: kv_u64(text, "restarts").unwrap_or(0), synced: kv(text, "synced") == Some("true"), + template_wait_s: kv_f64(text, "template_wait").unwrap_or(0.0), + template_ms: kv_f64(text, "template_ms").unwrap_or(0.0), + identities_active: kv_u64(text, "identities_active").unwrap_or(0), }) } @@ -94,6 +128,10 @@ pub enum Event<'a> { /// The node became synced: a card faulted for silence while the node was not ready tries again (PC 1, 7 October /// 2026 04:51Z: every card faulted "no status line" during the post-relaunch sync and never came back). NodeSynced, + /// The worker failed its self-test (MF-4): the card is held for `SELF_TEST_HOLD_S` with the reason on its row. + SelfTestFailed(&'a str), + /// The card's driver or platform changed (a re-enumeration): a held card tries again now. + DriverChanged, /// The miner printed "template fetch timed out": it is alive and the node is not answering templates (PC 1, /// 7 October 2026 04:51Z: the node reported synced from the first second while its finality replay blocked the /// template RPC for three minutes; with no template the miner prints no status line). Silence is not the card's @@ -104,10 +142,9 @@ pub enum Event<'a> { #[derive(Debug, Clone, PartialEq)] pub enum Action { None, - /// Stop the miner and start it again now, for this reason. - Restart(String), - /// Mark the card faulted with this reason; do not restart its miner. - Fault(String), + /// Stop the miner and start it again after `delay_s`, for this reason (`attempt` counts restarts since the card + /// was last healthy for five minutes). + Restart { reason: String, delay_s: u64, attempt: u32 }, } #[derive(Debug, Default)] @@ -115,12 +152,17 @@ pub struct CardWatch { started_s: Option, last_status_s: Option, ready: bool, + /// when the worker reported its program loaded: the status clocks start here + ready_s: Option, + /// held after a self-test failure until the driver changes or the hold passes + driver_hold: bool, zero_since: Option, healthy_since: Option, worker_restart_since: Option, - /// app-level restarts without five healthy minutes since + /// app-level restarts without five healthy minutes since (the ladder position) restarts: u32, - faulted: Option, + /// the reason of the last restart (a node-caused one resets the ladder when the node syncs) + last_reason: String, last_fault: String, /// the miner's last line was a template timeout (the node not answering), cleared by the next status line templates_blocked: bool, @@ -138,10 +180,7 @@ impl CardWatch { Self::default() } - pub fn faulted(&self) -> Option<&str> { - self.faulted.as_deref() - } - + /// Restarts since the card was last healthy for five minutes: the ladder position. pub fn watchdog_restarts(&self) -> u32 { self.restarts } @@ -150,6 +189,7 @@ impl CardWatch { self.started_s = Some(now_s); self.last_status_s = None; self.ready = false; + self.ready_s = None; self.zero_since = None; self.healthy_since = None; self.worker_restart_since = None; @@ -162,14 +202,9 @@ impl CardWatch { self.zero_since = None; self.healthy_since = None; self.worker_restart_since = None; - if self.restarts >= 1 { - let r = format!("{reason} (restarted once already)"); - self.faulted = Some(r.clone()); - Action::Fault(r) - } else { - self.restarts += 1; - Action::Restart(reason) - } + self.restarts = self.restarts.saturating_add(1); + self.last_reason = reason.clone(); + Action::Restart { reason, delay_s: retry_delay_s(self.restarts), attempt: self.restarts } } pub fn event(&mut self, now_s: f64, ev: Event<'_>) -> Action { @@ -180,12 +215,22 @@ impl CardWatch { } Event::Ready => { self.ready = true; + self.ready_s = Some(now_s); + self.driver_hold = false; self.worker_restart_since = None; Action::None } Event::Status(s) => { - self.templates_blocked = false; self.last_status_s = Some(now_s); + if s.template_wait_s > 0.0 && s.hash_now <= 0.0 { + // MF-5: the miner is alive and waiting on the node for a template; a zero rate here is the + // node's latency, never the card's fault + self.templates_blocked = true; + self.zero_since = None; + self.healthy_since = None; + return Action::None; + } + self.templates_blocked = false; if s.hash_now > 0.0 { self.zero_since = None; let since = *self.healthy_since.get_or_insert(now_s); @@ -209,15 +254,35 @@ impl CardWatch { Action::None } Event::Exited(code) => { - let running = self.started_s.is_some(); + let started = self.started_s; self.started_s = None; - if code == MINER_GAVE_UP_CODE && running { + if code == MINER_GAVE_UP_CODE && started.is_some() { let why = if self.last_fault.is_empty() { "its guards tripped three times in ten minutes".to_string() } else { self.last_fault.clone() }; self.escalate(format!("the miner gave up on its worker: {why}")) } else { - Action::None + match started { + // a crash loop: the ladder, not the 5 to 60 s jitter (MF-4) + Some(t) if !matches!(code, 0 | 42 | 43 | PACK_OUT_OF_DATE_CODE) && now_s - t < EARLY_EXIT_S => { + self.escalate(format!("the miner exited with code {code} {:.0} s after starting", now_s - t)) + } + _ => Action::None, + } } } + Event::SelfTestFailed(reason) => { + self.started_s = None; + self.ready = false; + self.driver_hold = true; + self.last_reason = format!("not usable on this driver: {reason}"); + Action::Restart { reason: self.last_reason.clone(), delay_s: SELF_TEST_HOLD_S, attempt: self.restarts.max(1) } + } + Event::DriverChanged => { + if self.driver_hold { + self.driver_hold = false; + self.restarts = 0; + } + Action::None + } Event::Stopped => { self.started_s = None; self.last_status_s = None; @@ -241,24 +306,28 @@ impl CardWatch { Action::None } Event::NodeSynced => { - if self.faulted.as_deref().map(|f| NODE_FAULTS.iter().any(|p| f.starts_with(p))).unwrap_or(false) { - *self = Self::default(); + // restarts the node's readiness explained do not count against the card + if self.restarted_for_node() { + self.restarts = 0; + self.last_reason.clear(); } Action::None } } } - /// True when the card is faulted for a reason the node's readiness explains (released by Event::NodeSynced). - pub fn faulted_by_node(&self) -> bool { - self.faulted.as_deref().map(|f| NODE_FAULTS.iter().any(|p| f.starts_with(p))).unwrap_or(false) + /// True while the card waits out a self-test failure (released by a driver change or the hold's end). + pub fn driver_hold(&self) -> bool { + self.driver_hold } - /// Called every engine tick while the miner process is alive. - pub fn tick(&mut self, now_s: f64, node_synced: bool) -> Action { - if self.faulted.is_some() { - return Action::None; - } + /// True when the last restart was for a reason the node's readiness explains (released by Event::NodeSynced). + pub fn restarted_for_node(&self) -> bool { + NODE_FAULTS.iter().any(|p| self.last_reason.starts_with(p)) + } + + /// Called every engine tick while the miner process is alive. `node_ready`: synced AND an executed tip. + pub fn tick(&mut self, now_s: f64, node_ready: bool) -> Action { let Some(started) = self.started_s else { return Action::None }; if let Some(t) = self.worker_restart_since { // the miner is restarting its worker: its own guards own the card until the worker is ready @@ -267,8 +336,9 @@ impl CardWatch { } return Action::None; } + let node_synced = node_ready; if !node_synced { - // a miner is judged only while the node is synced: with no template it cannot print status, so the silence + // a miner is judged only while the node is ready: with no template it cannot print status, so the silence // clocks (start, no status, zero rate) hold at now until the node is back (PC 1, 7 October 2026) self.started_s = Some(now_s); if self.last_status_s.is_some() { @@ -277,10 +347,18 @@ impl CardWatch { self.zero_since = None; return Action::None; } - if !self.ready && self.last_status_s.is_none() && now_s - started > START_S { - return self.escalate(format!("the worker gave no status within {} s of starting", START_S as u64)); + // the status clocks start when the worker reports its program loaded (MF-4), never before; loading itself + // is bounded by LOAD_S + let Some(ready_at) = self.ready_s else { + if now_s - started > LOAD_S { + return self.escalate(format!("the worker did not load its program within {} s of starting", LOAD_S as u64)); + } + return Action::None; + }; + if self.last_status_s.is_none() && now_s - ready_at > START_S { + return self.escalate(format!("the worker gave no status within {} s of loading its program", START_S as u64)); } - let last = self.last_status_s.unwrap_or(started); + let last = self.last_status_s.unwrap_or(ready_at); if now_s - last > NO_STATUS_S { return self.escalate(format!("no status line from the miner for {} s", NO_STATUS_S as u64)); } @@ -306,10 +384,15 @@ impl NodeWatch { Self::default() } - /// `silent_s`: seconds since the last reading (or since the node started, when it never answered). Returns the - /// delay in seconds before the restart when one is due. - pub fn tick(&mut self, now_s: f64, ours: bool, silent_s: f64, accepted_recent: bool) -> Option { - if !ours || accepted_recent || silent_s < NODE_SILENT_S { + /// `silent_s`: seconds since the node's last sign of life (a watch reading, an exec status answer, an accepted + /// block; or since the node started, when it never answered). `settled`: the node has been read as synced at + /// least once since its start; before that its catch-up never counts, only `NODE_STARTUP_CAP_S` of total silence + /// does. Returns the delay in seconds before the restart when one is due. + pub fn tick(&mut self, now_s: f64, ours: bool, silent_s: f64, accepted_recent: bool, settled: bool) -> Option { + if !ours || accepted_recent { + return None; + } + if silent_s < if settled { NODE_SILENT_S } else { NODE_STARTUP_CAP_S } { return None; } self.restarts_s.retain(|t| now_s - *t <= NODE_WINDOW_S); @@ -456,7 +539,10 @@ mod tests { #[test] fn parses_status_and_fault_lines() { let s = parse_status(STATUS_OK).unwrap(); - assert_eq!(s, Status { hash_now: 124.10, mismatched: 0, faults: 0, restarts: 0, synced: true }); + assert_eq!(s, Status { hash_now: 124.10, mismatched: 0, faults: 0, restarts: 0, synced: true, template_wait_s: 0.0, template_ms: 0.0, identities_active: 0 }); + // a 0.3.20 miner waiting on a slow node (MF-5) + let w = parse_status("1791151000.000 STATUS 'win-1' [worker]: 120s jobs=0 accepted=0 rejected=0 fee=0 mismatched=0 extra=0 rate=0.00 blocks/s hash=0.00 MH/s wall (0.00 MH/s inside jobs) now=0.00 MH/s wall (0.00 MH/s inside jobs, 0 jobs, seed walk 0 calls) template_age=0.00s synced=true idle=100.0% (last 10s: 100.0%) queued=0 restarts=0 faults=0 identities=24 accepted_by_identity=0 tip_age_s=0 template_wait=37s template_ms=8120 identities_active=1").unwrap(); + assert_eq!((w.template_wait_s, w.template_ms, w.identities_active), (37.0, 8120.0, 1)); let m = parse_status(STATUS_MISMATCH).unwrap(); assert_eq!((m.mismatched, m.faults, m.restarts), (3, 1, 1)); assert_eq!(parse_status(STATUS_ZERO).unwrap().hash_now, 0.0); @@ -475,6 +561,19 @@ mod tests { } } + fn restart(a: &Action) -> (String, u64, u32) { + match a { + Action::Restart { reason, delay_s, attempt } => (reason.clone(), *delay_s, *attempt), + Action::None => panic!("expected a restart, got None"), + } + } + + #[test] + fn the_ladder() { + assert_eq!((1..=6).map(retry_delay_s).collect::>(), vec![10, 30, 120, 300, 300, 300]); + assert_eq!(retry_delay_s(0), 10); + } + #[test] fn healthy_miner_is_left_alone() { let mut w = CardWatch::new(); @@ -484,125 +583,99 @@ mod tests { assert_eq!(w.watchdog_restarts(), 0); } + /// the project lead's rule (7 October 2026): no fault is permanent. A zero rate restarts the miner on the ladder 10, 30, + /// 120, 300, 300 ... s, the reason stays in plain words, and the hash resumes on its own when the worker is back. #[test] - fn zero_rate_restarts_once_then_faults() { + fn zero_rate_restarts_on_the_ladder_for_ever_and_the_hash_resumes() { let mut w = CardWatch::new(); w.event(0.0, Event::Started); w.event(2.0, Event::Ready); healthy(&mut w, 10.0, 100.0); let z = parse_status(STATUS_ZERO).unwrap(); - for t in [110.0, 120.0, 130.0, 140.0, 150.0, 160.0] { - w.event(t, Event::Status(&z)); - assert_eq!(w.tick(t, true), Action::None, "under 60 s at {t}"); + let mut t = 110.0; + let mut seen = Vec::new(); + for expect in [(10u64, 1u32), (30, 2), (120, 3), (300, 4), (300, 5), (300, 6)] { + // the app restarts it after the delay; still zero: the next rung + w.event(t, Event::Started); + w.event(t + 2.0, Event::Ready); + let mut a = Action::None; + let mut k = 0.0; + while a == Action::None && k < 200.0 { + w.event(t + 10.0 + k, Event::Status(&z)); + a = w.tick(t + 10.0 + k, true); + k += 10.0; + } + let (reason, delay, attempt) = restart(&a); + assert!(reason.starts_with("hash rate 0 for 60 s"), "{reason}"); + assert!(!reason.contains("once already"), "the permanent state no longer exists: {reason}"); + assert_eq!((delay, attempt), expect, "rung {}", expect.1); + seen.push(delay); + t += 10.0 + k + delay as f64; } - w.event(170.0, Event::Status(&z)); - let a = w.tick(170.0, true); - assert!(matches!(a, Action::Restart(ref r) if r.contains("hash rate 0 for 60 s")), "{a:?}"); - // the app restarted it; still zero: faulted, not restarted again - w.event(172.0, Event::Started); - w.event(174.0, Event::Ready); - for t in [180.0, 190.0, 200.0, 210.0, 220.0, 230.0] { - w.event(t, Event::Status(&z)); - assert_eq!(w.tick(t, true), Action::None); - } - w.event(240.0, Event::Status(&z)); - let a = w.tick(240.0, true); - assert!(matches!(a, Action::Fault(ref r) if r.contains("restarted once already")), "{a:?}"); - assert!(w.faulted().is_some()); - // faulted stays: no more actions - w.event(250.0, Event::Status(&z)); - assert_eq!(w.tick(260.0, true), Action::None); - // the user changed the card's settings: a fresh start - w.event(300.0, Event::Reset); - assert!(w.faulted().is_none()); + assert_eq!(seen, vec![10, 30, 120, 300, 300, 300]); + // the worker is healthy again: five healthy minutes and the ladder starts over at 10 s + w.event(t, Event::Started); + w.event(t + 2.0, Event::Ready); + healthy(&mut w, t + 10.0, t + 320.0); + assert_eq!(w.watchdog_restarts(), 0); + let a = { let mut a = Action::None; let mut k = 0.0; while a == Action::None { w.event(t + 330.0 + k, Event::Status(&z)); a = w.tick(t + 330.0 + k, true); k += 10.0; } a }; + assert_eq!(restart(&a).1, 10); } #[test] - fn zero_rate_only_counts_with_a_synced_node_and_a_ready_worker() { + fn zero_rate_only_counts_with_a_ready_node_and_a_ready_worker() { let mut w = CardWatch::new(); w.event(0.0, Event::Started); let z = parse_status(STATUS_ZERO).unwrap(); // not ready yet (program loading): no zero timer - for t in [10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 70.0, 80.0] { + for t in [10.0, 20.0, 30.0, 40.0, 50.0] { w.event(t, Event::Status(&z)); assert_eq!(w.tick(t, true), Action::None); } - w.event(82.0, Event::Ready); - // node not synced: no zero timer either - for t in [90.0, 100.0, 110.0, 120.0, 130.0, 140.0, 150.0, 160.0] { + w.event(52.0, Event::Ready); + // node not ready (syncing, or no executed tip yet): no zero timer either + for t in [60.0, 70.0, 80.0, 90.0, 100.0, 110.0, 120.0, 130.0] { w.event(t, Event::Status(&z)); assert_eq!(w.tick(t, false), Action::None); } - assert_eq!(w.tick(170.0, true), Action::None, "the timer starts at the first zero status after ready"); - for t in [180.0, 190.0, 200.0, 210.0, 220.0, 230.0] { + assert_eq!(w.tick(140.0, true), Action::None, "the timer starts at the first zero status after ready"); + for t in [150.0, 160.0, 170.0, 180.0, 190.0, 200.0] { w.event(t, Event::Status(&z)); w.tick(t, true); } - assert!(matches!(w.tick(240.0, true), Action::Restart(_))); + assert!(matches!(w.tick(210.0, true), Action::Restart { .. })); } - /// PC 1, 7 October 2026 04:51:43Z (docs/plans/release-0.3.17.md, the PC 1 row): the app relaunched after the OTA - /// apply, started its workers while the node was still syncing, and every card ended "faulted: no status line from - /// the miner for 90 s (restarted once already)", pid 0, 0.0 MH/s from then on. Today's watchdog clocks the silence - /// while the node is not synced and never releases the fault. + /// PC 1, 7 October 2026: three workers started while the node caught up, printed nothing (no template), were + /// restarted once and then marked faulted for good. Now: the silence clocks hold while the node is not ready, a + /// restart the node explains does not climb the ladder once the node syncs, and nothing is ever permanent. #[test] - fn a_worker_started_while_the_node_syncs_is_not_faulted_and_a_faulted_one_returns_at_sync() { - // the recorded sequence (relaunch-engine-and-workers.txt): node synced=true from the first second, the miner - // started at +14 s, subscribed, then every template fetch timed out at 5 s for 180 s; no status line was printed - let mut r = CardWatch::new(); - r.event(14.0, Event::Started); - let mut t = 19.0; - while t <= 200.0 { - r.event(t, Event::TemplateTimeout); - assert_eq!(r.tick(t + 1.0, true), Action::None, "at {t}: a miner whose template fetches time out is not faulted on silence"); - assert!(r.templates_blocked()); - t += 5.0; - } - // the node answers templates again: the miner prints status and is healthy - healthy(&mut r, 210.0, 400.0); - assert!(!r.templates_blocked()); - assert_eq!(r.faulted(), None); - // a miner that stops printing even timeouts is judged again from its last line - let mut q = CardWatch::new(); - q.event(0.0, Event::Started); - q.event(50.0, Event::TemplateTimeout); - assert_eq!(q.tick(109.0, true), Action::None); - assert!(matches!(q.tick(111.0, true), Action::Restart(_)), "60 s of nothing at all after the last timeout line"); - + fn a_worker_started_while_the_node_catches_up_is_never_faulted() { let mut w = CardWatch::new(); w.event(0.0, Event::Started); - // the node reads unsynced for 200 s (the relaunch's sync): no verdict at all - let mut t = 5.0; - while t <= 200.0 { - assert_eq!(w.tick(t, false), Action::None, "at {t}: a miner is judged only while the node is synced"); - t += 10.0; + // the node is not ready for ten minutes: no action, no climb + let mut t = 1.0; + while t < 600.0 { + assert_eq!(w.tick(t, false), Action::None); + t += 1.0; } - // the node syncs; the worker answers inside 60 s and is healthy - assert_eq!(w.tick(210.0, true), Action::None); - w.event(230.0, Event::Ready); - healthy(&mut w, 240.0, 400.0); - assert_eq!(w.faulted(), None); - // the recorded state: a card faulted for silence while the node was not ready - let mut f = CardWatch::new(); - f.event(0.0, Event::Started); - assert!(matches!(f.tick(100.0, true), Action::Restart(_))); - f.event(101.0, Event::Started); - let a = f.tick(200.0, true); - assert!(matches!(a, Action::Fault(ref r) if (r.starts_with("no status line from the miner") || r.starts_with("the worker gave no status within")) && r.contains("restarted once already")), "{a:?}"); - assert!(f.faulted_by_node()); - f.event(300.0, Event::NodeSynced); - assert_eq!(f.faulted(), None, "the node syncing releases a silence fault"); - assert_eq!(f.watchdog_restarts(), 0); - // a fault the card owns (zero rate while synced) stays through NodeSynced - let mut z = CardWatch::new(); - z.event(0.0, Event::Started); - z.event(1.0, Event::Ready); - let zero = parse_status(STATUS_OK.replace("now=124.10 MH/s", "now=0.00 MH/s").as_str()).unwrap(); - for t in [10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 70.0, 80.0] { z.event(t, Event::Status(&zero)); z.tick(t, true); } - assert!(z.faulted().is_none() || !z.faulted_by_node()); - let before = z.faulted().map(|s| s.to_string()); - z.event(500.0, Event::NodeSynced); - assert_eq!(z.faulted().map(|s| s.to_string()), before); + // ready now, the worker has loaded but prints nothing: a restart after START_S, on rung 1 + w.event(t, Event::Ready); + let mut a = Action::None; + while a == Action::None && t < 700.0 { + a = w.tick(t, true); + t += 1.0; + } + let (reason, delay, attempt) = restart(&a); + assert!(reason.starts_with("the worker gave no status within"), "{reason}"); + assert_eq!((delay, attempt), (10, 1)); + assert!(w.restarted_for_node()); + // the node drops back to catching up and syncs again: the ladder resets for a node-caused restart + w.event(t, Event::NodeSynced); + assert_eq!(w.watchdog_restarts(), 0); + w.event(t + 10.0, Event::Started); + w.event(t + 12.0, Event::Ready); + healthy(&mut w, t + 20.0, t + 60.0); } #[test] @@ -611,25 +684,64 @@ mod tests { w.event(0.0, Event::Started); w.event(2.0, Event::Ready); healthy(&mut w, 10.0, 60.0); - // the miner goes silent (a stopped process, a hung RPC) assert_eq!(w.tick(149.0, true), Action::None); - let a = w.tick(151.0, true); - assert!(matches!(a, Action::Restart(ref r) if r.contains("no status line")), "{a:?}"); + let (reason, delay, _) = restart(&w.tick(151.0, true)); + assert!(reason.contains("no status line"), "{reason}"); + assert_eq!(delay, 10); } #[test] fn no_status_from_the_start() { - // a worker silent from its start, with the node synced, is judged at START_S (60 s), not NO_STATUS_S let mut w = CardWatch::new(); w.event(0.0, Event::Started); - assert_eq!(w.tick(59.0, true), Action::None); - assert!(matches!(w.tick(61.0, true), Action::Restart(ref r) if r.contains("gave no status within 60 s"))); + // loading: the status clock has not started; a worker that never loads is restarted at LOAD_S + assert_eq!(w.tick(290.0, true), Action::None); + let (reason, delay, _) = restart(&w.tick(301.0, true)); + assert!(reason.contains("did not load its program within 300 s"), "{reason}"); + assert_eq!(delay, 10); + // loaded at 200 s (a slow self-test behind another card's export): the 60 s status clock starts there + let mut w = CardWatch::new(); + w.event(0.0, Event::Started); + w.event(200.0, Event::Ready); + assert_eq!(w.tick(259.0, true), Action::None); + let (reason, _, _) = restart(&w.tick(261.0, true)); + assert!(reason.contains("gave no status within 60 s of loading"), "{reason}"); + } + + /// MF-4 (PC 1, 7 October 2026): a worker that fails its self-test is held for 30 minutes with the reason on its + /// row, not restarted every few seconds; a driver change releases it; a miner in a crash loop climbs the ladder. + #[test] + fn self_test_failure_is_held_and_a_crash_loop_climbs_the_ladder() { + let mut w = CardWatch::new(); + w.event(0.0, Event::Started); + let (reason, delay, _) = restart(&w.event(3.0, Event::SelfTestFailed("3 of 96 vectors mismatched"))); + assert_eq!(reason, "not usable on this driver: 3 of 96 vectors mismatched"); + assert_eq!(delay, SELF_TEST_HOLD_S); + assert!(w.driver_hold()); + w.event(100.0, Event::DriverChanged); + assert!(!w.driver_hold()); + // a crash loop: exits 2 s after each start climb 10, 30, 120 s + let mut w = CardWatch::new(); + let mut t = 0.0; + let mut delays = Vec::new(); + for _ in 0..3 { + w.event(t, Event::Started); + let (reason, delay, _) = restart(&w.event(t + 2.0, Event::Exited(3))); + assert!(reason.contains("exited with code 3"), "{reason}"); + delays.push(delay); + t += 2.0 + delay as f64; + } + assert_eq!(delays, vec![10, 30, 120]); + // an exit after a long healthy run is the engine's own jittered restart + let mut w = CardWatch::new(); + w.event(0.0, Event::Started); + w.event(2.0, Event::Ready); + healthy(&mut w, 10.0, 600.0); + assert_eq!(w.event(700.0, Event::Exited(3)), Action::None); } #[test] fn the_miners_own_worker_restart_is_not_doubled() { - // The gfx1036 fault: the miner prints WORKER FAULT, kills the worker, restarts it 2 s later; STATUS lines in - // between say now=0. The app must not restart the miner on top of that. let mut w = CardWatch::new(); w.event(0.0, Event::Started); w.event(2.0, Event::Ready); @@ -649,40 +761,66 @@ mod tests { w.event(t as f64, Event::Status(&z)); assert_eq!(w.tick(t as f64, true), Action::None); } - let a = w.tick(1091.0, true); - assert!(matches!(a, Action::Restart(ref r) if r.contains("did not come back within 180 s")), "{a:?}"); + let (reason, delay, _) = restart(&w.tick(1091.0, true)); + assert!(reason.contains("did not come back within 180 s"), "{reason}"); + assert_eq!(delay, 10); } #[test] - fn exit_43_once_then_faulted() { + fn exit_43_is_a_restart_on_the_ladder() { let mut w = CardWatch::new(); w.event(0.0, Event::Started); w.event(2.0, Event::Ready); healthy(&mut w, 10.0, 60.0); w.event(70.0, Event::WorkerRestart("cpu re-check: 3 consecutive mismatches (mismatched=3 in this run): the worker computes a wrong program")); - let a = w.event(75.0, Event::Exited(43)); - assert!(matches!(a, Action::Restart(ref r) if r.contains("gave up") && r.contains("wrong program")), "{a:?}"); - w.event(80.0, Event::Started); - let a = w.event(300.0, Event::Exited(43)); - assert!(matches!(a, Action::Fault(_)), "{a:?}"); - // an ordinary crash is the engine's own jittered restart, not the watchdog's + let (reason, delay, attempt) = restart(&w.event(75.0, Event::Exited(43))); + assert!(reason.contains("gave up") && reason.contains("wrong program"), "{reason}"); + assert_eq!((delay, attempt), (10, 1)); + w.event(85.0, Event::Started); + let (_, delay, attempt) = restart(&w.event(300.0, Event::Exited(43))); + assert_eq!((delay, attempt), (30, 2)); + // an exit 5 s after the start is the crash-loop class (MF-4): the ladder, not the 5 to 60 s jitter let mut w = CardWatch::new(); w.event(0.0, Event::Started); - assert_eq!(w.event(5.0, Event::Exited(1)), Action::None); + assert!(matches!(w.event(5.0, Event::Exited(1)), Action::Restart { delay_s: 10, .. })); } #[test] - fn five_healthy_minutes_renew_the_budget() { + fn template_timeouts_are_the_miners_heartbeat() { let mut w = CardWatch::new(); w.event(0.0, Event::Started); w.event(2.0, Event::Ready); - assert!(matches!(w.tick(100.0, true), Action::Restart(_))); - w.event(101.0, Event::Started); - w.event(103.0, Event::Ready); - healthy(&mut w, 110.0, 420.0); + healthy(&mut w, 10.0, 60.0); + // the node stops answering templates for four minutes while the app still reads it as ready + let mut t = 70.0; + while t < 300.0 { + w.event(t, Event::TemplateTimeout); + assert_eq!(w.tick(t + 1.0, true), Action::None, "a heartbeat at {t} is not silence"); + t += 5.0; + } + assert!(w.templates_blocked()); + healthy(&mut w, 310.0, 400.0); + assert!(!w.templates_blocked()); + } + + /// MF-5 (PC 1, 7 October 2026): the node answered templates past 5 s; the miner now prints STATUS every interval + /// with template_wait= while it waits, and the watchdog measures the worker, never the node. + #[test] + fn a_slow_node_never_faults_the_card() { + let mut w = CardWatch::new(); + w.event(0.0, Event::Started); + w.event(2.0, Event::Ready); + healthy(&mut w, 10.0, 60.0); + let slow = Status { hash_now: 0.0, template_wait_s: 8.0, template_ms: 8120.0, identities_active: 1, synced: true, ..Default::default() }; + let mut t = 70.0; + while t < 700.0 { + w.event(t, Event::Status(&slow)); + assert_eq!(w.tick(t, true), Action::None, "waiting on the node at {t} is not a fault"); + t += 10.0; + } + assert!(w.templates_blocked()); + healthy(&mut w, 710.0, 800.0); assert_eq!(w.watchdog_restarts(), 0); - // a second incident later is again a restart, not a fault - assert!(matches!(w.tick(520.0, true), Action::Restart(_))); } #[test] @@ -693,18 +831,26 @@ mod tests { w.event(30.0, Event::Stopped); assert_eq!(w.tick(500.0, true), Action::None); assert_eq!(w.event(500.0, Event::Exited(0)), Action::None); + w.event(600.0, Event::Reset); + assert_eq!(w.watchdog_restarts(), 0); } + /// The node's catch-up never counts against its watchdog (PC 1, 7 October 2026: a 40-second restart loop). #[test] - fn node_watch_restarts_a_silent_node_with_growing_delay() { + fn node_watch_waits_out_a_catch_up_and_restarts_a_dead_node_with_growing_delay() { let mut n = NodeWatch::new(); - assert_eq!(n.tick(100.0, true, 119.0, false), None); - assert_eq!(n.tick(100.0, false, 500.0, false), None, "an external node is never restarted"); - assert_eq!(n.tick(100.0, true, 500.0, true), None, "our block was accepted in the last minute: the node is alive"); - assert_eq!(n.tick(100.0, true, 120.0, false), Some(3)); - assert_eq!(n.tick(300.0, true, 120.0, false), Some(12)); - assert_eq!(n.tick(500.0, true, 120.0, false), Some(48)); + // catching up (never read as synced since its start): 25 minutes of silence is not a restart + assert_eq!(n.tick(100.0, true, 1500.0, false, false), None); + // a node that never answers at all: restarted at the 30-minute cap + assert_eq!(n.tick(100.0, true, 1801.0, false, false), Some(3)); + let mut n = NodeWatch::new(); + assert_eq!(n.tick(100.0, true, 119.0, false, true), None); + assert_eq!(n.tick(100.0, false, 500.0, false, true), None, "an external node is never restarted"); + assert_eq!(n.tick(100.0, true, 500.0, true, true), None, "our block was accepted in the last minute: the node is alive"); + assert_eq!(n.tick(100.0, true, 120.0, false, true), Some(3)); + assert_eq!(n.tick(300.0, true, 120.0, false, true), Some(12)); + assert_eq!(n.tick(500.0, true, 120.0, false, true), Some(48)); assert_eq!(n.restarts_in_window(), 3); - assert_eq!(n.tick(2000.0, true, 120.0, false), Some(3), "the window passed"); + assert_eq!(n.tick(2000.0, true, 120.0, false, true), Some(3), "the window passed"); } } diff --git a/app/igneum-app/ui/app.js b/app/igneum-app/ui/app.js index a4bf6ded2..d1ff3c607 100644 --- a/app/igneum-app/ui/app.js +++ b/app/igneum-app/ui/app.js @@ -311,8 +311,8 @@ var View = (function () { var on = !!cd.enabled && !removed && !unusable, st = removed ? 'removed' : unusable ? 'unusable' : (cd.state || 'off'); // the rule (docs/plans/ember-tune.md, 6 October 2026): a card never shows 0 MH/s without a reason word var tuning = st === 'tuning', held = st === 'held'; - var word = removed ? 'removed' : unusable ? ('not usable (' + cd.problem + ')') : !on ? 'off' : st === 'mining' ? 'mining' : tuning ? tuneWord(cd) : held ? (cd.message || 'paused for a job from the team') : st === 'restarting' ? 'restart in ' + (cd.restart_in_s || 0) + ' s' : st === 'waiting' ? 'waiting for the node' : st === 'faulted' ? 'stopped after repeated errors' : st === 'failed' ? 'failed' : st === 'starting' ? 'starting' : st === 'ready' ? 'ready' : st; - var tone = removed ? 'off' : unusable ? 'bad' : !on ? 'off' : (st === 'mining' || tuning) ? 'on' : (st === 'failed' || st === 'faulted' || st === 'restarting') ? 'bad' : ''; + var word = removed ? 'removed' : unusable ? ('not usable (' + cd.problem + ')') : !on ? 'off' : st === 'mining' ? 'mining' : tuning ? tuneWord(cd) : held ? (cd.message || 'paused for a job from the team') : st === 'restarting' ? 'restart in ' + (cd.restart_in_s || 0) + ' s' : st === 'waiting' ? 'waiting for the node' : st === 'failed' ? 'failed' : st === 'starting' ? 'starting' : st === 'ready' ? 'ready' : st; + var tone = removed ? 'off' : unusable ? 'bad' : !on ? 'off' : (st === 'mining' || tuning) ? 'on' : (st === 'failed' || st === 'restarting') ? 'bad' : ''; var hash = on && (st === 'mining' || tuning) ? (cd.hash_now >= 100 ? cd.hash_now.toFixed(0) : (cd.hash_now || 0).toFixed(1)) : ''; var temp = cd.temp_gpu > 0 ? Math.round(cd.temp_gpu) + ' °C' : ''; var tempTone = cd.temp_mem > 95 || cd.temp_gpu > 92 ? 'hot' : cd.temp_mem > 90 || cd.temp_gpu > 86 ? 'warm' : ''; diff --git a/app/igneum-app/ui/view.test.mjs b/app/igneum-app/ui/view.test.mjs index db83cb0e2..4152a8225 100644 --- a/app/igneum-app/ui/view.test.mjs +++ b/app/igneum-app/ui/view.test.mjs @@ -62,7 +62,8 @@ test('an integrated GPU is shown as integrated and off, with its reason', () => assert.equal(r.sub, 'integrated GPU: slow and shares the machine memory'); assert.equal(V.cardRow(card({ kind: 'unknown' })).canToggle, false); assert.equal(V.cardRow(card({ state: 'restarting', restart_in_s: 7 })).word, 'restart in 7 s'); - assert.equal(V.cardRow(card({ state: 'faulted' })).tone, 'bad'); + // no fault is permanent (7 October 2026, docs/plans/miner-faults.md): a restarting card is the bad tone, there is no faulted state + assert.equal(V.cardRow(card({ state: 'restarting', restart_in_s: 120, message: 'hash rate 0 for 60 s while the node is synced; trying again' })).tone, 'bad'); assert.equal(V.cardRow(card({ state: 'waiting' })).word, 'waiting for the node'); }); diff --git a/docs/plans/miner-faults.md b/docs/plans/miner-faults.md new file mode 100644 index 000000000..93af4cd8d --- /dev/null +++ b/docs/plans/miner-faults.md @@ -0,0 +1,43 @@ +# Miner fault-class register + +Started 7 October 2026 (the project lead, 11:2x UK: "we cannot have issues like this with the miner, it needs to be truly plug, +tune, play"). Every fault class found in the field gets a row the same day: the rule that makes the class impossible, +the test that proves the rule, and the gate line a cut must show before it publishes. A class is closed when all three +exist and the gate has fired once on a known-bad case and once on a known-good case (the watcher rule of 4 October). + +Standing rules behind every row (branch `miner-reliability`, off `release-0.3.19` db6f0964, for the 0.3.20 app; the miner rows on the fork branch `miner-reliability-20` off `release-0.3.20-node`): + +| Rule | Where | +|---|---| +| A worker never starts before the node is synced AND `igneum_getExecStatus` reports an executed tip; the node's catch-up never counts against its watchdog | `app/igneum-app/src/engine.rs` (`tick_exec_probe`, `node_ready`, `node_settled`), `src/watchdog.rs` (`NodeWatch::tick` with `settled`, `NODE_STARTUP_CAP_S`) | +| No fault is permanent: a restart waits 10 s, 30 s, 2 min, 5 min, then every 5 min, for ever; the reason stays on the card row in plain words; the hash resumes on its own; there is no "restarted once already" state | `src/watchdog.rs` (`RETRY_LADDER_S`, `retry_delay_s`, `Action::Restart { delay_s, attempt }`), `engine.rs` (`watchdog_verdict`, the pack give-up) | +| A card swap, a driver install or a restart needs no tap: the hot-plug enumeration every 60 s (`src/hotplug.rs`) starts the worker of a card that appears, recovers from a problem code or revives | `engine.rs` `merge_detection` → `plan_miners` | +| Every fault line reports to the log intake the moment it happens, with the card, the class, the reason and the app version | `engine.rs` `fault_report` → `update::upload_text`; label `fault--`; read with `node tools/logs.mjs` | +| Nothing on a user's machine is changed by a one-off script: card settings travel as the signed `cards` job kind (per card enabled, identities, power_pct), applied through the app's own card path, persisted, read back in the report, refused for a card the machine does not have | `src/jobs.rs` (`KINDS`, `validate_params`), `src/jobrun.rs` (`cards_job_choices`, `cards_applied`), `engine.rs` (`Action::ApplyCards`), `packaging/ota/publish-jobs.sh add --kind cards --cards "key=on:8"` | +| The fresh-install claim (LG-4) is a job, not a runbook: `tools/fleet/first-share-gate.mjs` on rented Windows boxes, on every cut, its line read by the shipper's publish | `relay/playbooks/first-share.ps1`, `tools/fleet/first-share-gate.mjs`, `site/evidence/first-share-.json` | + +## The register + +| Id | Found | Class (what the user saw) | Root cause | Rule | Test | Gate line | +|---|---|---|---|---|---|---| +| MF-1 | 7 Oct 2026, PC 1, 0.3.17 and 0.3.18 | The node restarted every 40 s; the card said "waiting for the node" for an hour | The app's clock sample, then its prover loop, called records-indexing exec RPCs (`eth_getBlockByNumber`, `igneum_getAssignedShards`) while the 0.3.17 node's exec follower held no record; the node panicked (`rpc.rs:591`, `rpc.rs:808`) | Every exec RPC call the app makes goes through `execrpc::call`; a method is safe on an empty state or gated on an executed tip, and an unclassified method is refused (shipper, `2dfb2e0c`). The node's whole records-indexing class is bounds-checked in the 0.3.20 node | `execrpc` unit test: every method literal in the tree is classified, no other file builds an exec request; `tools/reliability/app-run.mjs` step `catch-up`: a node with no executed tip gets no gated call and no worker for 120 s, no node restart | `app tests: execrpc callers classified` in the cut's plan; `app-run catch-up PASS` | +| MF-2 | 7 Oct 2026, PC 1, 0.3.18 | After a restart all three cards said "no status line from the miner for 90 s (restarted once already)" and never mined again until a cards-API bounce | The workers started the moment `synced` read true, before the node's catch-up (finality replay, exec follower) let templates flow; the watchdog's one-restart budget then marked them faulted for good | Workers start only when the node is READY (synced and an executed tip); silence while the node is not ready never counts; a restart the node's readiness explained resets the ladder at sync; the ladder never ends; the faulted state is gone | `watchdog` tests `a_worker_started_while_the_node_catches_up_is_never_faulted`, `zero_rate_restarts_on_the_ladder_for_ever_and_the_hash_resumes`, `node_watch_waits_out_a_catch_up...`; `app-run.mjs` steps `zero-ladder` (rungs 10, 30, 120 s observed, then the hash back on its own) and `catch-up` | `app tests: watchdog 16 green`; `app-run zero-ladder PASS`; the words "restarted once already" absent from `src/` (`tools/ci/forbidden-strings.txt`) | +| MF-4 | 7 Oct 2026, PC 1, Arc B580 beside a 5090 and a 9070 XT | One card's worker failed its self-test and restarted every few seconds; the other two cards sat in "loading the program" past the 90-second watchdog and were faulted | Every restart of the failing card exported the pack again under the export lock; the healthy cards' exports queued behind it; the watchdog's status clock ran from the process start, not from "program loaded" | A card's worker failure never blocks another card: the pack is exported once per minute for every card (`EXPORT_REUSE_S`; a refused pack forces one); a worker that fails its self-test is held 30 minutes with "not usable on this driver: " on its row and tried again on a driver change (`SelfTestFailed`, `DriverChanged`); a miner that exits inside 120 s of its start restarts on the ladder (10 s, 30 s, 2 min, 5 min), never every few seconds; the status clock starts at `ready` (program loaded), and loading itself is bounded by 300 s | `watchdog` tests `self_test_failure_is_held_and_a_crash_loop_climbs_the_ladder`, `no_status_from_the_start` (the clock from `ready`); `app-run.mjs` step `one-card-fails` (three cards, one failing for ever: the two mine on time, the third is held, at most 6 exports) | `app-run one-card-fails PASS` | +| MF-5 | 7 Oct 2026, PC 1, 0.3.17 node, 24 identities (the real cause of the 11:2x faults; MF-4 withdrawn as the cause) | Every card faulted "no status line from the miner for 90 s (restarted once already)" after the restart | Evidence (PC 1, 7 Oct 2026 11:4x UK): with 8 identities a card (24 template fetches a round) the 0.3.17 node answered no template in 5 s; with 2 a card (4 fetches) both cards mined at full rate (5090 122.4 MH/s, 9070 XT 18.9) within two minutes. The node's getBlockTemplate answered past 5 s with 24 identities fetching; the miners waited for a template inside their job-fill loop and printed no STATUS at all; the watchdog read the silence as the worker's; one restart, then faulted for good | Miner: STATUS every interval whatever the template state (`template_wait=` while it waits, `template_ms=` the node's last template time, `identities_active=`); the feed fetches only as many identities as fit one pass inside 8 s at the node's measured template time (`identities_for`: all of them when the node answers under 1 s; 24 at 8 s per template becomes 1), raised again when it answers faster; a pass whose fetches all fail backs off 2, 4, 8, 10 s and retries for ever with a `NODE SLOW` line once per 30 s. App: a STATUS with `template_wait>0` is the miner's heartbeat and the node's latency, never the card's fault (no zero-rate clock, no restart); the card reads "node slow: waiting for a block template for N s; the worker is kept" or "node slow: a template takes N s; k of n identities active"; the mitigation of the day (a one-off script POSTing /api/cards) is closed by the signed `cards` job kind | `watchdog` tests `a_slow_node_never_faults_the_card`, `parses_status_and_fault_lines` (the 0.3.20 line); `jobs` test for the `cards` kind; injector step `slow-node` (a template stub answering in 8 s while three cards run: open, needs the stub) | `app-run slow-node PASS` (0.3.20) | +| MF-6 | 7 Oct 2026, PC 1, 0.3.19 | After a `--stop-miners` job, a following read-only job kept both cards "off, held for a remote job" for its whole three minutes | The engine released the hold only when no job held the miners; the next job's active state hid the release (engine.rs 3032 class) | A hold belongs to the job that took it (`job_hold_owner`) and releases the moment that job is no longer the running one, whatever runs next, or when its own cap passes (logged); a read-only job never holds (`jobrun::hold_release`) | `jobrun` test `a_hold_belongs_to_the_job_that_took_it` (owner running, another job, no job, cap passed, no owner) | `app tests: hold rule green` | +| MF-7 | 7 Oct 2026, PC 1, 0.3.19 | Orphan `igneum-miner.exe` processes the app no longer tracked (two alive under `--stop-miners` with their rows at pid 0, one after) hammered the node's template RPC beside the tracked miners | The engine lost track of miners it had started (a stop that timed out, a restart over a live process) and never looked for them again | The engine owns every miner it started: at start, after every stop and every minute it kills any `igneum-miner` whose command line carries THIS engine's node RPC (the fence) and whose pid it does not track, one log line and one fault report per kill, never by name alone (`sweep_orphan_miners`, `platform::miner_processes`, `kill_pid`); a restart kills the slot's old process before the new one starts | injector step `orphan-miner` (a stray miner on the engine's node is killed inside the minute, the engine's own miner left alone) | `app-run orphan-miner PASS` | +| MF-3 | 7 Oct 2026, PC 1, Intel Arc | The Intel driver's first install did not bind: the device sat in Code 12 at install time; the card never mined until a reboot | A driver installed while the device reports a problem code (12, 43, 31) does not bind; nothing re-scanned the device afterwards, and the app only re-enumerates | The app re-enumerates every 60 s and starts the worker the minute the OS drives the card (`hotplug::diff` recovered / revived, `settle_new`); the row says what to do while it does not ("reboot with the card attached; if it persists, reinstall the driver with the card attached"); a Windows host asks for a re-scan (`pnputil /scan-devices`) after a problem code is seen, every 5 minutes, at most 6 times (follow-up, host side) | `hotplug` test `a_driven_card_that_turns_faulty_is_errored_and_recovers_later`; `app-run.mjs` step `card-appears` (a card listed after 2 minutes starts without a tap) | `app-run card-appears PASS` | + +## How a row is added + +1. The day the class is seen: the row with Found, Class, Root cause. The rule, the test and the gate line the same day + where they exist; "open" where they do not, with the owner. +2. The test is a unit test on recorded lines (`src/watchdog.rs`, `src/hotplug.rs`, `src/execrpc.rs`) or a step of the + fault injector (`tools/reliability/app-run.mjs` against `fake-worker.mjs`), run on the box. +3. The gate line is what the cut's plan (`docs/plans/release-.md`) must carry before publish; the shipper reads it. + +## The fault injector (box) + +`tools/reliability/app-run.mjs` drives a scratch engine (its own private node, the fake worker in place of the GPU +worker, Linux or macOS) through one step per class and prints `PASS`/`FAIL` with the seconds. The box runs it with +the Linux engine and node from `target-remote/`: `tools/reliability/box-run.sh` (the lock is the box's run slot). diff --git a/igneum-pow/src/generator.rs b/igneum-pow/src/generator.rs index b55e2c954..8bbc72c9f 100644 --- a/igneum-pow/src/generator.rs +++ b/igneum-pow/src/generator.rs @@ -326,6 +326,11 @@ pub fn era_draw(era_bytes: &[u8], allowed: &[u8]) -> EraParams { let j = i + (r[i] % n as u64) as usize; c.swap(i, j); } + // Draws 8 and 9, consumed and not used (spec 01 section 1.13.1 for `epoch_len`; `docs/design/latency-ladder.md` + // section 2 for the latency ladder): both parameters are set by miner signal, and consuming their slots here means a + // later use of either changes no other draw. Nothing below reads them, so every value drawn above is what it was. + let _epoch_len_draw = s.next(); + let _latency_ladder_draw = s.next(); let mut chosen: Vec = c[..free].to_vec(); chosen.sort_unstable(); let mut pos = [0u8; 4]; @@ -804,7 +809,43 @@ pub const V3_CLASS: LoadClass = LoadClass { era: None, hot: None, ..LoadClass::M /// of 256 ALU instructions run 27 times per iteration ("mx8+sh256x27", 55,296 shadow instructions per hash). The /// base program, the 16 loads, the item construction, the cache growth rule and the era draw are class v3's, draw /// for draw, so a v4 epoch's day cache and dataset are the v3 day's. Composed with the era exactly as V3 is. -pub const V4_CLASS: LoadClass = LoadClass { shadow: Some(ShadowClass { instrs: 256, reps: 27 }), ..V3_CLASS }; +pub const V4_CLASS: LoadClass = LoadClass { shadow: Some(ShadowClass { instrs: V4_SHADOW_INSTRS, reps: V4_SHADOW_REPS }), ..V3_CLASS }; + +/// The shadow block size of class v4 at every rung of the latency ladder (`docs/design/latency-ladder.md`): 256 +/// instructions. The ladder moves the pass count alone. +pub const V4_SHADOW_INSTRS: u16 = 256; + +/// The shadow passes of class v4 at rung 0 of the latency ladder: 27 (`mx8+sh256x27`, about 102,100 counted ops). +pub const V4_SHADOW_REPS: u16 = 27; + +/// Class v4 at a rung of the latency ladder (`docs/design/latency-ladder.md` section 2): [`V4_CLASS`] with the +/// 256-instruction shadow block run `reps` times per iteration. `reps` 0 means the class's own count, so +/// `v4_class_at(0) == v4_class_at(27) == V4_CLASS` and a caller that knows no rung changes nothing. +pub fn v4_class_at(reps: u16) -> LoadClass { + if reps == 0 { + V4_CLASS + } else { + LoadClass { shadow: Some(ShadowClass { instrs: V4_SHADOW_INSTRS, reps }), ..V3_CLASS } + } +} + +/// The shadow passes of a class v4 load class at any rung of the ladder, the era draw set aside (`Some(27)` for +/// [`V4_CLASS`] itself); `None` for every other class, a measurement class with another block size included. +pub fn v4_rung_reps(class: &LoadClass) -> Option { + let base = LoadClass { era: None, ..*class }; + match base.shadow { + Some(ShadowClass { instrs: V4_SHADOW_INSTRS, reps }) if LoadClass { shadow: None, ..base } == V3_CLASS => Some(reps), + _ => None, + } +} + +/// Counted integer ops per hash of class v4 at `reps` shadow passes, the convention of +/// `docs/analysis/latency-shadow-2026-10-06.md` section 1: 930 for the base program and its loads, 1.83 ops per +/// shadow instruction (137 / 75 over the non-load weights), 8 iterations x 256 instructions x `reps` shadow +/// instructions. Approximate by construction; the label of a rung, never a consensus value. +pub fn v4_counted_ops(reps: u16) -> u64 { + 930 + (ITERATIONS as u64 * V4_SHADOW_INSTRS as u64 * reps as u64 * 183).div_ceil(100) +} /// The width set class v3's era draw chooses from: 4 bytes only (the read-width decision of 5 October 2026; the /// draw is consumed, so widening the set at genesis keeps the derivation). @@ -1009,7 +1050,12 @@ impl Program { /// || attempt_le32`. Written into every pack so a version 1 program, or another attempt of the same seed, /// can never be mistaken for this one. pub fn program_id(&self) -> u64 { - if self.class.is_v2() || self.generator == GENERATOR_VERSION_V3 || self.generator == GENERATOR_VERSION_V4 { + // Latency ladder (docs/design/latency-ladder.md section 7): a class v4 program above rung 0 carries its shadow + // size in the id (`program_id_class`, the "shadow/" bytes), so two rungs of one seed never share an id and a + // pack of another rung is refused as a pack of another class is. Rung 0 keeps `program_id(4, seed, attempt)` + // byte for byte, so every v4 id written before the ladder stands. + let v4_rung_0 = self.generator == GENERATOR_VERSION_V4 && LoadClass { era: None, ..self.class } == V4_CLASS; + if self.class.is_v2() || self.generator == GENERATOR_VERSION_V3 || v4_rung_0 { // Spec 01 section 1.4.6: a class v3 program's id is `program_id(3, seed, attempt)`, a class v4 program's // `program_id(4, seed, attempt)` (Counter ASIC 3.0); the generator version in the preimage separates // them from every version 2 program of the same seed @@ -1378,6 +1424,27 @@ pub fn generate_from_seed_bytes_program_class(seed_string: &str, seed_bytes: &[u p } +/// [`generate_from_seed_bytes_program_class`] at a rung of the latency ladder (`docs/design/latency-ladder.md` +/// section 2): `shadow_reps` is the shadow pass count the chain's step gives the epoch, 0 for the class's own. Class +/// v4 at a rung above 0 draws from [`v4_class_at`] with the era inside and generator 4 stamped; every other class, +/// and class v4 at rung 0, is [`generate_from_seed_bytes_program_class`] byte for byte. The base program, the 16 loads +/// and the era draw do not move with the rung: only the pass count of the shadow block does. +pub fn generate_from_seed_bytes_program_class_shadow(seed_string: &str, seed_bytes: &[u8], class: ProgramClass, era_bytes: Option<&[u8]>, shadow_reps: u16) -> Program { + if class != ProgramClass::V4 || shadow_reps == 0 || shadow_reps == V4_SHADOW_REPS { + return generate_from_seed_bytes_program_class(seed_string, seed_bytes, class, era_bytes); + } + let base = v4_class_at(shadow_reps); + match era_bytes { + Some(era) => generate_era_generator(seed_string, seed_bytes, base, era, &V3_ALLOWED, GENERATOR_VERSION_V4), + None => { + let mut p = generate_from_seed_bytes_class(seed_string, seed_bytes, base); + p.generator = GENERATOR_VERSION_V4; + p.era_bytes = None; + p + } + } +} + /// The program of a seed string (its UTF-8 bytes are the program seed). pub fn generate(seed_string: &str) -> Program { generate_from_seed_bytes(seed_string, seed_string.as_bytes()) @@ -2027,4 +2094,72 @@ mod tests { assert_eq!(again.shadow, sh.shadow); assert_eq!(crate::verify::hash_warp(&again, 0, &ds), h1); } + + /// Latency ladder (`docs/design/latency-ladder.md`), the known-failed case first: before the ladder a changed N + /// was a hard fork. Two nodes drawing class v4 at 27 and at 35 passes from the same seeds build the same base + /// program and the same shadow block, carry generator 4 on both and, under the id rule as it stood, the SAME id + /// (`program_id(4, seed, attempt)` reads no shadow size), yet their warps hash differently: every block of one is + /// invalid to the other and no pack line told them apart. After: the rung is a parameter of the chain's step, + /// rung 0 is `V4_CLASS` byte for byte, and a rung above carries its pass count in the id. + #[test] + fn latency_ladder_known_failed_a_changed_n_was_a_hard_fork_and_rungs_are_class_v4() { + let era = [7u8; 32]; + let today = generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V4, Some(&era)); + let r0 = generate_from_seed_bytes_program_class_shadow("igneum-genesis", b"igneum-genesis", ProgramClass::V4, Some(&era), 0); + let r27 = generate_from_seed_bytes_program_class_shadow("igneum-genesis", b"igneum-genesis", ProgramClass::V4, Some(&era), V4_SHADOW_REPS); + assert_eq!(r0, today, "rung 0 is class v4 byte for byte"); + assert_eq!(r27, today, "27 passes is rung 0"); + let r1 = generate_from_seed_bytes_program_class_shadow("igneum-genesis", b"igneum-genesis", ProgramClass::V4, Some(&era), 35); + // the failed case, as it stood: the same base, the same block, the same seed words, generator 4 on both, one id + assert_eq!(r1.instrs, today.instrs, "the base program is the class's, draw for draw"); + assert_eq!(r1.shadow, today.shadow, "the block is the same draw; only the pass count moves"); + assert_eq!((r1.seed, r1.attempt, r1.generator, r1.era_bytes.clone()), (today.seed, today.attempt, today.generator, today.era_bytes.clone())); + assert_eq!(program_id(GENERATOR_VERSION_V4, &r1.seed, r1.attempt), program_id(GENERATOR_VERSION_V4, &today.seed, today.attempt), "the old rule gave both nodes one id"); + assert_eq!(r1.shadow_reps(), 35); + assert_eq!(r1.shadow_instrs_per_hash(), ITERATIONS * 256 * 35); + let ds = crate::verify::DatasetSource::new("2026-10-06", crate::verify::DatasetMode::ClosedForm, 20); + let h0 = crate::verify::hash_warp(&today, 0, &ds); + let h1 = crate::verify::hash_warp(&r1, 0, &ds); + assert_ne!(h0, h1, "a changed N is another hash: before the ladder, a hard fork"); + // after: the rung is in the id above rung 0; rung 0 keeps the id written before the ladder + assert_ne!(r1.program_id(), today.program_id(), "the rung is in the id"); + assert_eq!(r1.program_id(), program_id_class(GENERATOR_VERSION_V4, &r1.seed, r1.attempt, &r1.class)); + assert_eq!(today.program_id(), program_id(GENERATOR_VERSION_V4, &today.seed, today.attempt), "rung 0 keeps the v4 id"); + assert_eq!(r1.program_class(), ProgramClass::V4, "a rung is class v4"); + assert_eq!(v4_rung_reps(&r1.class), Some(35)); + assert_eq!(v4_rung_reps(&today.class), Some(27)); + assert_eq!(v4_rung_reps(&V4_CLASS), Some(27)); + assert_eq!(v4_rung_reps(&V3_CLASS), None); + assert_eq!(v4_rung_reps(&LoadClass::V2), None); + assert_eq!(v4_rung_reps(&LoadClass::MX8.with_shadow(64, 52)), None, "another block size is a measurement class, not a rung"); + assert_eq!(v4_class_at(0), V4_CLASS); + assert_eq!(v4_class_at(27), V4_CLASS); + assert_eq!(v4_class_at(35), LoadClass::parse("mx8+sh256x35").unwrap()); + assert_eq!(LoadClass { era: None, ..r1.class }, v4_class_at(35), "the era rides inside the rung's class"); + assert_eq!(r1.class.era, today.class.era, "the same era draw at every rung"); + assert!(check(&r1).is_ok(), "the acceptance rule reads the base program, which did not move"); + // every rung of the designed ladder is another program with its own id + let rungs = [27u16, 35, 53, 88, 173, 267]; + let ids: Vec = rungs.iter().map(|&r| generate_from_seed_bytes_program_class_shadow("igneum-genesis", b"igneum-genesis", ProgramClass::V4, Some(&era), r).program_id()).collect(); + for i in 0..ids.len() { + for j in 0..i { + assert_ne!(ids[i], ids[j], "rungs {} and {} share an id", rungs[i], rungs[j]); + } + } + // the ops labels of the rungs, within 1 percent of the measured rungs of algorithm.md 5.3a + for (r, ops) in [(27u16, 102_100u64), (35, 132_100), (53, 199_600), (88, 330_700), (173, 649_400), (267, 1_001_600)] { + let got = v4_counted_ops(r); + assert!(got.abs_diff(ops) * 100 < ops, "reps {r}: {got} counted ops against the label {ops}"); + } + // other classes ignore the rung; class v4 without an era takes it + assert_eq!(generate_from_seed_bytes_program_class_shadow("igneum-genesis", b"igneum-genesis", ProgramClass::V3, Some(&era), 35), generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V3, Some(&era))); + assert_eq!(generate_from_seed_bytes_program_class_shadow("igneum-genesis", b"igneum-genesis", ProgramClass::V2, None, 35), generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V2, None)); + let bare = generate_from_seed_bytes_program_class_shadow("igneum-genesis", b"igneum-genesis", ProgramClass::V4, None, 35); + assert_eq!((bare.generator, bare.shadow_reps(), bare.era_bytes.is_none()), (GENERATOR_VERSION_V4, 35, true)); + // the era stream consumes draws 8 and 9 after the seven it uses, so the seven are what they were: the pinned era + // packs of tests/packs.rs hold the values; here, the draw is a function of the bytes and the set alone + let e = era_draw(&era, &V3_ALLOWED); + assert_eq!(e, era_draw(&era, &V3_ALLOWED)); + assert_eq!(e.width_words, 1); + } } diff --git a/igneum-pow/src/lib.rs b/igneum-pow/src/lib.rs index 1b398fab5..244c4ab17 100644 --- a/igneum-pow/src/lib.rs +++ b/igneum-pow/src/lib.rs @@ -34,7 +34,7 @@ pub mod verify; pub use bind::{block_init_words, day_bytes, pow256_from_lane, target64_from_le256}; pub use accept::{check as accept_program, AcceptReport, Reject}; -pub use generator::{generate, generate_from_seed_bytes, generate_from_seed_bytes_program_class, Instr, LoadClass, Op, Program, ProgramClass, GENERATOR_VERSION, GENERATOR_VERSION_V3, GENERATOR_VERSION_V4, V3_CLASS, V4_CLASS}; +pub use generator::{generate, generate_from_seed_bytes, generate_from_seed_bytes_program_class, generate_from_seed_bytes_program_class_shadow, v4_class_at, v4_counted_ops, v4_rung_reps, Instr, LoadClass, Op, Program, ProgramClass, GENERATOR_VERSION, GENERATOR_VERSION_V3, GENERATOR_VERSION_V4, V3_CLASS, V4_CLASS, V4_SHADOW_INSTRS, V4_SHADOW_REPS}; pub use memhard::{cache_log2_words, dataset_log2_words, days_since_genesis, growth_doublings, Cache, MemhardCpu, MixParams, Shape}; pub use seed::{fnv1a64, seed_words, SplitMix64}; pub use verify::{hash_warp, interpret_warp_init, verify_block, DatasetMode, DatasetSource, Epoch}; diff --git a/igneum-pow/src/main.rs b/igneum-pow/src/main.rs index 3d32504ee..29bb77d80 100644 --- a/igneum-pow/src/main.rs +++ b/igneum-pow/src/main.rs @@ -44,6 +44,9 @@ struct Args { days: u64, /// The program class (Counter ASIC 2.0 seam): v2 (default), v3 (V3_CLASS, generator 3) or v4 (V4_CLASS, generator 4). program_class: Option, + /// The shadow pass count of a class v4 program at a rung of the latency ladder (`--shadow-reps`, 0 = the class's + /// own 27; `docs/design/latency-ladder.md`); read under `--program-class v4` only. + shadow_reps: u16, /// The era seed bytes a class v3 chain program records (`--era-hex`). era_hex: Option, /// Era layout: `--era igneum-era-test/` or `--era :<64 hex>` composes the era class over `--class` with @@ -99,6 +102,7 @@ fn usage() -> ! { \x20 also: w4, w16, w64, w64x4, p4,p16,p64[xN], m[g], +shx (latency-shadow block of S ALU instructions x R passes per iteration, Counter ASIC 3.0 item 8)\n\ \x20 --days N days since genesis for the cache growth rule of a class with it (default 0: the 2^26-word cache)\n\ \x20 --program-class v2|v3|v4 the program class of the seam (v3 = generator 3 on V3_CLASS, v4 = generator 4 on V4_CLASS = mx8+sh256x27, the chain's own derivation; --era-hex records the era seed)\n\ + \x20 --shadow-reps N class v4 at a rung of the latency ladder: the shadow block's pass count (0 = the class's own 27; docs/design/latency-ladder.md), with --program-class v4\n\ \x20 --era E era layout over --class: igneum-era-test/ or :<64 hex> (the 32-byte era seed E_n)\n\ \x20 --era-widths 4[,16,64] the width set the era draws from, in bytes (default 4: pinned; more lets the era draw it)" ); @@ -123,6 +127,7 @@ fn parse() -> Args { days: 0, program_class: None, era_hex: None, + shadow_reps: 0, era: None, era_widths: vec![1], }; @@ -146,6 +151,7 @@ fn parse() -> Args { "--days" => a.days = val().parse().unwrap_or_else(|_| usage()), "--program-class" => a.program_class = Some(ProgramClass::parse(&val()).unwrap_or_else(|| usage())), "--era-hex" => a.era_hex = Some(val()), + "--shadow-reps" => a.shadow_reps = val().parse().unwrap_or_else(|_| usage()), "--era" => a.era = Some(parse_era(&val()).unwrap_or_else(|| usage())), "--era-widths" => a.era_widths = parse_widths(&val()).unwrap_or_else(|| usage()), _ => usage(), @@ -222,7 +228,7 @@ fn epoch_of_class(a: &Args, mode: DatasetMode) -> (Epoch, String) { let label = format!("igneum-epoch/{eh}/day/{dh}"); let e = match a.program_class { Some(pc) => Epoch { - program: Epoch::chain_program(&eb, era.as_deref(), pc, &label), + program: Epoch::chain_program_shadow(&eb, era.as_deref(), pc, a.shadow_reps, &label), dataset: Epoch::chain_dataset_day(&db, pc, a.days, a.dataset_log2), }, None => Epoch::from_seed_bytes_day(&eb, &db, &label, a.class, a.days, a.dataset_log2), @@ -232,7 +238,7 @@ fn epoch_of_class(a: &Args, mode: DatasetMode) -> (Epoch, String) { _ => { let e = match a.program_class { Some(pc) => { - let program = igneum_pow::generator::generate_from_seed_bytes_program_class(&a.seed, a.seed.as_bytes(), pc, era.as_deref()); + let program = igneum_pow::generator::generate_from_seed_bytes_program_class_shadow(&a.seed, a.seed.as_bytes(), pc, era.as_deref(), a.shadow_reps); let lc = pc.load_class(); let shape = Shape::for_class_day(&lc, a.days); let log2 = if lc.growth { igneum_pow::memhard::dataset_log2_words(a.dataset_log2, a.days) } else { a.dataset_log2 }; @@ -410,7 +416,7 @@ fn show(a: &Args) { // the load class of --class, the era of --era composed over it let era_hex_bytes = a.era_hex.as_deref().map(|h| igneum_pow::bind::unhex(h).unwrap_or_else(|| usage())); let mut p = match a.program_class { - Some(pc) => igneum_pow::generator::generate_from_seed_bytes_program_class(&label, &bytes, pc, era_hex_bytes.as_deref()), + Some(pc) => igneum_pow::generator::generate_from_seed_bytes_program_class_shadow(&label, &bytes, pc, era_hex_bytes.as_deref(), a.shadow_reps), None => igneum_pow::generator::generate_from_seed_bytes_class(&label, &bytes, a.class), }; if let (None, Some((_, eb, _))) = (a.program_class, &a.era) { diff --git a/igneum-pow/src/verify.rs b/igneum-pow/src/verify.rs index 4b22630c4..34692b684 100644 --- a/igneum-pow/src/verify.rs +++ b/igneum-pow/src/verify.rs @@ -626,6 +626,13 @@ impl Epoch { crate::generator::generate_from_seed_bytes_program_class(label, epoch_seed, class, era_bytes) } + /// [`Epoch::chain_program`] at a rung of the latency ladder (`docs/design/latency-ladder.md`): `shadow_reps` is + /// the shadow pass count the chain's step gives the epoch (0 = the class's own, which is [`Epoch::chain_program`] + /// byte for byte). The node's engine and the miner's pack export call this with the step the template names. + pub fn chain_program_shadow(epoch_seed: &[u8], era_bytes: Option<&[u8]>, class: ProgramClass, shadow_reps: u16, label: &str) -> Program { + crate::generator::generate_from_seed_bytes_program_class_shadow(label, epoch_seed, class, era_bytes, shadow_reps) + } + /// The day's cache and dataset of [`Epoch::from_chain_seeds`], the one entry the node's engine builds a day /// cache through. The class is an argument because the Counter ASIC 2.0 integration gives class v3 its own item /// construction (the mixer multiplier) and cache size schedule (ca2-mixer); today both classes build the day of diff --git a/infra/build-server/README.md b/infra/build-server/README.md new file mode 100644 index 000000000..49f7ba106 --- /dev/null +++ b/infra/build-server/README.md @@ -0,0 +1,37 @@ +# The build boxes (infra/build-server) + +Three Hetzner dedicated servers in Falkenstein run everything the Mac must not: builds, test suites, benchmarks, CPU proving, the +devnet hands and the observer. The Mac keeps macOS binaries, the DMG and Metal tests (CLAUDE.md, "Running agents on this Mac"). + +## No mining on any Hetzner box, ever + +the project lead's rule through main, 7 October 2026: Hetzner's policies forbid crypto mining. The boxes run nodes, builds, tests, benchmarks and +CPU proving only. The pool's fast-time network runs its miners on rented GPU pods (tools/fleet), never on a box; a box may run the +network's nodes. The capacity layer refuses a job that would start `igneum-miner mine` or a GPU worker (capacity/run.sh), and no +hands unit carries a miner. A node started with `--enable-unsynced-mining` is a node flag, not a miner; nothing feeds it blocks here. + +## The boxes and the kind map + +| Box | Host file on the Mac | Takes | Never | +|---|---|---|---| +| igneum-build-1 (188.40.146.49, AX162-1-LTD) | `~/.config/igneum/build-server` | release gates (`--priority gate`), builds and cross-builds, checks, the GPU workers' host side, the devnet hands (node 1, the observer node, the observer), the Devnet 2 seed, the CI runner, the dashboard feed | suites and benches once box 2 exists | +| igneum-build-2 (AX162-1, on order) | `~/.config/igneum/build-server-2` | suites (`cargo test`), benches (`cargo bench`), the attack rows (`--box 2`) | gates, hands | +| igneum-build-3 (AX102-1, on order) | `~/.config/igneum/build-server-3` | proving and aggregation CPU work (proving/igneum-prove builds and suites), the second prover's shadow runner, the pool's fast-time NETWORK (nodes only, `--box 3`) | miners of any kind | + +`tools/build-remote.sh` routes by class (lib.sh `bs_route`): suite and bench to box 2, the proving crate to box 3, everything else +to box 1; `--box N` overrides; a class whose box has no host file yet falls back to box 1 and says so. `--priority gate` always runs +on box 1. Each box has its own mirrors, slots, locks and JSONL log under /srv; `run-from-mac.sh --box N ` provisions a box and +writes its host file; the dashboard collector reads every box it is told about. + +## Files + +| File | What | +|---|---| +| `provision.sh` | the box itself: install mode (rescue system, Ubuntu 24.04, RAID 1, no swap) and provision mode (user build, toolchains, the pin, sccache, zig, CUDA headers, docker, Caddy, mirrors, slots, sshd, ufw) | +| `run-from-mac.sh` | ships provision.sh, writes the host file, wires the `build` remotes and pushes every branch | +| `lib.sh`, `remote-run.sh` | the Mac and box halves of a remote run: sync, checkout, slots, scheduling classes, the JSONL line | +| `hands/` | the devnet hands' units, the mover and the restart read-backs | +| `capacity/` | the capacity layer (the box-work lane's): background jobs under the build slots, never a miner | +| `repro/`, `night/`, `prover/`, `runner/`, `workers/` | other lanes' pieces that live on the boxes | + +Plan, numbers and the gotchas: docs/plans/build-server.md; the hands: docs/plans/hands-on-build-1.md. diff --git a/infra/build-server/lib.sh b/infra/build-server/lib.sh index a2c758876..bb754b796 100755 --- a/infra/build-server/lib.sh +++ b/infra/build-server/lib.sh @@ -16,7 +16,19 @@ # shellcheck disable=SC2034 # shared with the scripts that source lib.sh BS_KEY="${IGNEUM_BUILD_KEY:-$HOME/.ssh/igneum_ed25519}" -BS_HOST_FILE="${IGNEUM_BUILD_HOST_FILE:-$HOME/.config/igneum/build-server}" # one line: build@ +BS_HOST_FILE="${IGNEUM_BUILD_HOST_FILE:-$HOME/.config/igneum/build-server}" # one line: build@ (box 1; box N is build-server-N) +# Several boxes (main, 7 October 2026: igneum-build-2 and igneum-build-3 on order). One host file per box: ~/.config/igneum/build-server +# is box 1, build-server-2 is box 2, build-server-3 is box 3 (run-from-mac.sh --box N writes it). The route by class, unless the +# caller passes --box: gates, builds, checks, cross-builds, the workers, the hands and the observer stay on box 1; suites, benches and +# the attack rows go to box 2; proving and aggregation CPU work, the second prover's shadow runner and the pool's fast-time NETWORK go +# to box 3. A class whose box has no host file yet falls back to box 1, and the log line says so. No box mines, ever (README.md). +bs_box_file() { case "${1:-1}" in 1) echo "$BS_HOST_FILE" ;; *) echo "${BS_HOST_FILE}-$1" ;; esac; } +bs_route() { # -> the box number, falling back to 1 + local want=1 + case "$1" in suite|bench|attack) want=2 ;; prove|shadow|fasttime) want=3 ;; esac + if [ "$want" != 1 ] && [ ! -s "$(bs_box_file "$want")" ]; then bs_log "class $1 routes to box $want, which has no host file yet ($(bs_box_file "$want")): box 1"; want=1; fi + echo "$want" +} BS_ROOT_REMOTE=/srv/builds BS_MIRROR_REPO=/srv/igneum.git BS_MIRROR_NODE=/srv/igneum-node.git @@ -24,11 +36,13 @@ BS_MIRROR_NODE=/srv/igneum-node.git bs_log() { printf '%s %s: %s\n' "$(date -u +%H:%M:%S)" "${BS_TOOL:-build-server}" "$*" >&2; } bs_die() { bs_log "ERROR: $*"; exit 1; } -bs_host() { +bs_host() { # [box number, default BS_BOX or 1] + BS_BOX="${1:-${BS_BOX:-1}}" BS_HOST="${BUILD_HOST:-}" if [ -z "$BS_HOST" ]; then - [ -s "$BS_HOST_FILE" ] || bs_die "no build server: write build@ to $BS_HOST_FILE (infra/build-server/run-from-mac.sh does) or set BUILD_HOST" - BS_HOST="$(head -1 "$BS_HOST_FILE" | tr -d '[:space:]')" + local f; f=$(bs_box_file "$BS_BOX") + [ -s "$f" ] || bs_die "no build server for box $BS_BOX: write build@ to $f (infra/build-server/run-from-mac.sh --box $BS_BOX does) or set BUILD_HOST" + BS_HOST="$(head -1 "$f" | tr -d '[:space:]')" fi case "$BS_HOST" in *@*) ;; *) bs_die "BUILD_HOST must be user@host, got '$BS_HOST'" ;; esac [ -r "$BS_KEY" ] || bs_die "no ssh key at $BS_KEY" @@ -43,15 +57,22 @@ bs_rsync() { rsync -e "$BS_SSH_CMD" "$@"; } # the two sides' rustc must agree (the box is pinned by provision.sh RUST_TOOLCHAIN; the Mac runs rustup's stable): # a different compiler gives different bytes and, across a minor version, different lints and errors +# the pin (main, 7 October 2026): rust-toolchain.toml at the worktree root (and the fork's own copy) names the channel; the Mac's +# rustc AS RESOLVED IN THE CRATE DIR (rustup reads the file), the box's rustc and the pin must agree, else the build is refused +# a tree without the file (an older release branch) gives an empty pin, never a failed pipeline: under pipefail the sed status would +# become the assignment's status and set -e would end the caller silently (7 Oct 2026, 03:36 UTC: the b3c228fa builds under the +# 0.3.16 app tree 5d118f58 died right after the pairing line) +bs_pin() { local f="${1:-$BS_WT_ROOT}/rust-toolchain.toml"; [ -f "$f" ] || { echo ""; return 0; }; { sed -n 's/^channel *= *"\([^"]*\)".*/\1/p' "$f" 2>/dev/null || true; } | head -1; } bs_toolchain_check() { - local mac box - mac=$("${CARGO_HOME:-$HOME/.cargo}/bin/rustc" --version 2>/dev/null | awk '{ print $2 }') + local mac box pin + pin=$(bs_pin "$BS_WT_ROOT"); [ -n "$pin" ] || pin=$(bs_pin "$BS_TOP") + mac=$(cd "${BS_CRATE:-.}" && "${CARGO_HOME:-$HOME/.cargo}/bin/rustc" --version 2>/dev/null | awk '{ print $2 }') box=$(bs_ssh '. /etc/profile.d/igneum-build.sh; rustc --version' 2>/dev/null | awk '{ print $2 }') [ -n "$box" ] || bs_die "cannot read rustc on $BS_HOST (is it provisioned? infra/build-server/run-from-mac.sh)" - if [ "$mac" != "$box" ]; then - if [ "${IGNEUM_TOOLCHAIN_MISMATCH:-}" = ok ]; then bs_log "WARNING: rustc $mac on the Mac, $box on the box (IGNEUM_TOOLCHAIN_MISMATCH=ok)" - else bs_die "rustc $mac on the Mac, $box on the box; re-provision with RUST_TOOLCHAIN=$mac or set IGNEUM_TOOLCHAIN_MISMATCH=ok"; fi - else bs_log "rustc $box on both sides"; fi + if [ -n "$pin" ] && { [ "$mac" != "$pin" ] || [ "$box" != "$pin" ]; } || [ "$mac" != "$box" ]; then + if [ "${IGNEUM_TOOLCHAIN_MISMATCH:-}" = ok ]; then bs_log "WARNING: pin ${pin:-none}, rustc $mac on the Mac, $box on the box (IGNEUM_TOOLCHAIN_MISMATCH=ok)" + else bs_die "toolchain mismatch: rust-toolchain.toml pins ${pin:-nothing}, the Mac resolves rustc $mac in $BS_CRATE, the box runs $box; align the pin (one commit), 'rustup toolchain install $pin' on the Mac, 'RUST_TOOLCHAIN=$pin infra/build-server/run-from-mac.sh ' for the box, or set IGNEUM_TOOLCHAIN_MISMATCH=ok"; fi + else bs_log "rustc $box on both sides${pin:+ (pinned $pin by rust-toolchain.toml)}"; fi } # Where am I? Sets BS_KIND (node = a worktree of the fork under vendor/; repo = a crate of the igneum repo), BS_TOP (the git @@ -59,6 +80,7 @@ bs_toolchain_check() { # BS_CRATE (the crate dir, = $PWD), BS_CRATE_REL (relative to BS_WT_ROOT), BS_MIRROR, BS_BRANCH, BS_SHA, BS_REMOTE_WT, # BS_REMOTE_CRATE, and BS_LOCAL_DIRS (every directory of a path dependency, relative to BS_WT_ROOT, from cargo metadata). bs_context() { + local d BS_CRATE="$PWD" [ -f "$BS_CRATE/Cargo.toml" ] || bs_die "no Cargo.toml in $BS_CRATE: run from the crate directory (the fork worktree, igneum-pow, app/igneum-app, proving/igneum-prove)" BS_TOP=$(git -C "$BS_CRATE" rev-parse --show-toplevel 2>/dev/null) || bs_die "$BS_CRATE is not inside a git worktree" @@ -115,6 +137,39 @@ print("\n".join(rel(m) for m in sorted(dirs))) print("VENDOR_REPOS " + " ".join(rel(t) for t in sorted(repos)))' "$BS_WT_ROOT" "$BS_TOP" "$BS_KIND") || bs_die "cargo metadata failed in $BS_CRATE" BS_VENDOR_REPOS=$(printf '%s\n' "$BS_LOCAL_DIRS" | sed -n 's/^VENDOR_REPOS //p') BS_LOCAL_DIRS=$(printf '%s\n' "$BS_LOCAL_DIRS" | grep -v '^VENDOR_REPOS ' || true) + # Files a crate reaches by include_bytes!/include_str! with a relative path that LEAVES its repository (the shipper, 7 Oct 2026: + # the fork's igneum-exec embeds proving/igneum-prove/elf/igneum-prove-program.vk five levels up, and the box build failed with + # "couldn't read ...: No such file or directory" because such a file is no path dependency and the overlay never carried it). + # Every .rs under the trees that travel is scanned; a path resolving outside BS_TOP but inside the worktree root adds its + # directory to the overlay. proving/igneum-prove/elf is added for a fork build whatever the scan finds (the fixed fallback). + local extra + extra=$(python3 - "$BS_WT_ROOT" "$BS_TOP" $BS_LOCAL_DIRS $BS_VENDOR_REPOS <<'PY2' +import os, re, sys +root, top, rels = sys.argv[1], sys.argv[2], sys.argv[3:] +rx = re.compile(r'include_(?:bytes|str)!\(\s*"((?:\.\./)+[^"]+)"') +out = set() +for rel in rels: + base = os.path.join(root, rel) + for dp, dn, fn in os.walk(base): + dn[:] = [d for d in dn if d not in ("target", ".git") and not d.startswith("target-")] + for f in fn: + if not f.endswith(".rs"): continue + p = os.path.join(dp, f) + try: text = open(p, encoding="utf-8", errors="ignore").read() + except OSError: continue + for m in rx.finditer(text): + a = os.path.normpath(os.path.join(dp, m.group(1))) + if (a == top or a.startswith(top + "/")): continue + if not a.startswith(root + "/"): continue + out.add(os.path.relpath(os.path.dirname(a), root)) +print("\n".join(sorted(out))) +PY2 +) || extra="" + [ "$BS_KIND" = node ] && [ -d "$BS_WT_ROOT/proving/igneum-prove/elf" ] && extra=$(printf '%s\n%s\n' "$extra" "proving/igneum-prove/elf" | grep . | sort -u) + for d in $extra; do + case " $BS_LOCAL_DIRS " in *" $d "*) ;; *) BS_LOCAL_DIRS="$BS_LOCAL_DIRS +$d"; bs_log "included by include_bytes!/include_str! outside the crate's repository: $d" ;; esac + done [ -n "$BS_LOCAL_DIRS$BS_VENDOR_REPOS" ] || bs_die "cargo metadata listed no local packages in $BS_CRATE" } @@ -239,9 +294,10 @@ bs_remote_run() { label="$label; agent=$agent" BR_DIR="$dir" BR_LABEL="$label" BR_CMD="$cmd" BR_TOOL="${BS_TOOL:-build-remote}" BR_WT="$BS_WT" BR_CRATE="$BS_CRATE_REL" \ BR_BRANCH="$BS_BRANCH" BR_SHA="$BS_SHA" BR_AGENT="$agent" BR_KIND="${BR_KIND:-other}" BR_COMMAND="${BR_COMMAND:-}" \ - BR_TARGET="${BR_TARGET:-}" BR_ARTEFACTS="${BR_ARTEFACTS:-}" BR_SDE="${BR_SDE:-}" \ + BR_TARGET="${BR_TARGET:-}" BR_ARTEFACTS="${BR_ARTEFACTS:-}" BR_SDE="${BR_SDE:-}" BR_PAIRS_WITH="${BR_PAIRS_WITH:-}" \ + BR_NICE="${BR_NICE:-0}" BR_CORES="${BR_CORES:-0}" BR_JOBS_CAP="${BR_JOBS_CAP:-0}" BR_PRIORITY="${BR_PRIORITY:-normal}" \ bash -c ' - for v in BR_DIR BR_LABEL BR_CMD BR_TOOL BR_WT BR_CRATE BR_BRANCH BR_SHA BR_AGENT BR_KIND BR_COMMAND BR_TARGET BR_ARTEFACTS BR_SDE; do + for v in BR_DIR BR_LABEL BR_CMD BR_TOOL BR_WT BR_CRATE BR_BRANCH BR_SHA BR_AGENT BR_KIND BR_COMMAND BR_TARGET BR_ARTEFACTS BR_SDE BR_PAIRS_WITH BR_NICE BR_CORES BR_JOBS_CAP BR_PRIORITY; do printf "export %s=%q\n" "$v" "${!v}" done cat "$0"' "$(dirname "${BASH_SOURCE[0]}")/remote-run.sh" | bs_ssh 'bash -s' diff --git a/infra/build-server/provision.sh b/infra/build-server/provision.sh index 945b128dc..6fdeb90ea 100755 --- a/infra/build-server/provision.sh +++ b/infra/build-server/provision.sh @@ -18,9 +18,9 @@ # directory per agent worktree, sshd key-only, ufw with 22 and the seed p2p ports. # # Settings (environment, all optional): -# RUST_TOOLCHAIN 1.99.0 the Mac's `rustc --version` on 6 October 2026. Neither the repo nor the fork carries a -# rust-toolchain file (checked 6 October 2026), so the pin lives here; the fork's Cargo.toml says -# rust-version 1.91.0. tools/build-remote.sh compares the two sides and refuses a mismatch. +# RUST_TOOLCHAIN 1.99.0 the channel of rust-toolchain.toml at the repo root (run-from-mac.sh passes it; one file pins +# the Mac, the box, the PCs and CI since 7 October 2026). tools/build-remote.sh refuses a build +# when the pin, the Mac's rustc or the box's rustc differ. # SCCACHE_GB 100 the local disk cache at /srv/sccache # SCCACHE_VERSION (unset) a `cargo install sccache --version` pin; unset = the newest on crates.io # NODE_MAJOR 22 @@ -139,7 +139,7 @@ step_os_check() { APT_PACKAGES=( build-essential clang lld llvm libclang-dev pkg-config libssl-dev cmake protobuf-compiler libprotobuf-dev gcc-mingw-w64-x86-64 g++-mingw-w64-x86-64 binutils-mingw-w64-x86-64 mingw-w64-x86-64-dev mingw-w64-tools - git tmux curl ca-certificates xz-utils zstd unzip rsync jq python3 ufw htop file caddy + git tmux curl ca-certificates xz-utils zstd unzip rsync jq python3 ufw htop file caddy docker.io # the GitHub Actions runner's .NET runtime needs libicu (bin/installdependencies.sh would install it); the two Python # simulators (sim/finality_v2.py, sim/difficulty/sim.py) need numpy, which ci.yml pip-installs on GitHub's runners libicu74 python3-numpy @@ -218,6 +218,14 @@ step_user() { worktree_count() { find /srv/builds -mindepth 1 -maxdepth 1 -type d -not -name '_*' | wc -l | tr -d ' '; } +# docker (7 October 2026): the glibc proof runs a shipped binary inside ubuntu:20.04 and ubuntu:22.04 on the box (tools/build-remote.sh +# --ship's proof); user build may run containers. Nothing else of the box runs in docker. +step_docker() { + command -v docker >/dev/null 2>&1 || die "docker is not installed (apt docker.io)" + systemctl is-active --quiet docker || systemctl enable --now docker >/dev/null 2>&1 + if id -nG "$BUILD_USER" | tr ' ' '\n' | grep -qx docker; then ok docker "$(docker --version | cut -d, -f1), $BUILD_USER in the docker group"; else usermod -aG docker "$BUILD_USER"; changed docker "$(docker --version | cut -d, -f1), $BUILD_USER added to the docker group (new ssh sessions see it)"; fi +} + step_dirs() { local d any=0 for d in /srv/builds /srv/builds/_locks /srv/sccache /srv/artefacts; do @@ -332,6 +340,22 @@ step_node() { changed node "$(/usr/local/bin/node --version) from nodejs.org (sha256 checked)" } +# zig (main, 7 October 2026): the glibc 2.36 Linux artefacts for Debian 12 seeds and HiveOS rigs come from cargo-zigbuild with zig as +# the C/C++ toolchain, as infra/cross/build-linux.sh does on the Mac (tonight's seed took 14 restarts and three minutes down on a +# glibc 2.39 native build). The release the Mac uses (0.17.0), from ziglang.org with the sha256 of the official download index. +ZIG_VERSION="${ZIG_VERSION:-0.17.0}" +ZIG_SHA256="${ZIG_SHA256:-1cbe9df9f27e6b78d14ccbca43b6703a404ef79ef1c463de901d7f088d4e2026}" # zig-x86_64-linux-0.17.0.tar.xz, index.json 7 Oct 2026 +step_zig() { + local have dir tar + have=$(/usr/local/bin/zig version 2>/dev/null || true) + if [ "$have" = "$ZIG_VERSION" ]; then ok zig "$have"; return; fi + dir=/usr/local/lib/zig; tar="zig-x86_64-linux-$ZIG_VERSION.tar.xz" + install -d "$dir" + ( cd "$dir" && curl -fsSLO "https://ziglang.org/download/$ZIG_VERSION/$tar" && echo "$ZIG_SHA256 $tar" | sha256sum -c --quiet - && tar -xJf "$tar" && rm -f "$tar" ) + ln -sfn "$dir/zig-x86_64-linux-$ZIG_VERSION/zig" /usr/local/bin/zig + changed zig "$(/usr/local/bin/zig version) from ziglang.org (sha256 checked) at /usr/local/bin/zig" +} + step_sshd() { local f=/etc/ssh/sshd_config.d/10-igneum-build.conf tmp tmp=$(mktemp) @@ -543,7 +567,9 @@ step_runner() { step_cargo_tools() { local cargo="$BUILD_HOME/.cargo/bin/cargo" any=0 if [ ! -x "$BUILD_HOME/.cargo/bin/cargo-audit" ]; then as_build "$cargo install cargo-audit --locked" >/dev/null 2>&1 || as_build "$cargo install cargo-audit" >/dev/null; any=1; fi - [ "$any" = 1 ] && changed cargo-tools "$(as_build "$BUILD_HOME/.cargo/bin/cargo-audit --version")" || ok cargo-tools "$(as_build "$BUILD_HOME/.cargo/bin/cargo-audit --version")" + # cargo-zigbuild (main, 7 October 2026): the glibc 2.36 Linux artefacts for Debian 12 seeds and HiveOS rigs (tools/build-remote.sh --ship) + if [ ! -x "$BUILD_HOME/.cargo/bin/cargo-zigbuild" ]; then as_build "$cargo install cargo-zigbuild --locked" >/dev/null 2>&1 || as_build "$cargo install cargo-zigbuild" >/dev/null; any=1; fi + [ "$any" = 1 ] && changed cargo-tools "$(as_build "$BUILD_HOME/.cargo/bin/cargo-audit --version"), $(as_build "$BUILD_HOME/.cargo/bin/cargo-zigbuild --version")" || ok cargo-tools "$(as_build "$BUILD_HOME/.cargo/bin/cargo-audit --version"), $(as_build "$BUILD_HOME/.cargo/bin/cargo-zigbuild --version")" } # the night battery (infra/build-server/night): the script and remote-run.sh into /srv/builds/_bin, the two units, the timer @@ -579,6 +605,7 @@ step_summary() { printf 'raid: %s\n' "$(grep -E '^md' /proc/mdstat 2>/dev/null | tr '\n' ';' || echo none)" printf 'rust: %s | %s | targets %s\n' "$(as_build "$BUILD_HOME/.cargo/bin/rustc --version")" "$(as_build "$BUILD_HOME/.cargo/bin/cargo --version")" "$(as_build "$BUILD_HOME/.cargo/bin/rustup target list --installed" | tr '\n' ' ')" printf 'sccache: %s, %s\n' "$(as_build "$BUILD_HOME/.cargo/bin/sccache --version")" "$(cat "$BUILD_HOME/.config/sccache/config" | tr '\n' ' ')" + printf 'zig: %s, cargo-zigbuild %s (glibc 2.36 Linux artefacts: tools/build-remote.sh --ship, tools/workers-remote.sh)\n' "$(/usr/local/bin/zig version 2>/dev/null || echo missing)" "$(as_build "$BUILD_HOME/.cargo/bin/cargo-zigbuild --version 2>/dev/null | awk '{ print \$NF }'" || echo missing)" printf 'mingw: %s\n' "$(x86_64-w64-mingw32-gcc-posix --version | head -1)" printf 'clang: %s | lld: %s\n' "$(clang --version | head -1)" "$(ld.lld --version | head -1)" printf 'node: %s | git: %s | tmux: %s\n' "$(/usr/local/bin/node --version)" "$(git --version)" "$(tmux -V)" @@ -603,6 +630,7 @@ do_provision() { step_sysctl step_limits step_user + step_docker step_dirs step_mirrors step_rustup @@ -616,6 +644,7 @@ do_provision() { step_ufw step_runner step_cargo_tools + step_zig step_night step_summary log "done" diff --git a/infra/build-server/remote-run.sh b/infra/build-server/remote-run.sh index b78b4b655..15facd9f4 100755 --- a/infra/build-server/remote-run.sh +++ b/infra/build-server/remote-run.sh @@ -51,6 +51,26 @@ _slots_env="${IGNEUM_BUILD_SLOTS_DIR:-}"; _log_env="${IGNEUM_BUILD_LOG_DIR:-}" [ -f /etc/profile.d/igneum-build.sh ] && . /etc/profile.d/igneum-build.sh [ -n "$_slots_env" ] && IGNEUM_BUILD_SLOTS_DIR="$_slots_env"; [ -n "$_log_env" ] && IGNEUM_BUILD_LOG_DIR="$_log_env" +# What the clean spares beyond target dirs and stamps: a lane's scratch. The fixed prefixes, plus every glob in the mirror-local +# file `.igneum-scratch-spare` (one per line, # comments; the file itself is spared). spare_args fills SPARE_ARGS with +# the `-e ` arguments; it is never empty (the fixed list), so bash 3.2's unbound-empty-array rule cannot bite the self-test. +SPARE_FIXED=('attack-*' 'scratch-*' 'target-attack-*' '.build-remote.log' '.igneum-scratch-spare') +spare_args() { + local p + SPARE_ARGS=() + for p in "${SPARE_FIXED[@]}"; do SPARE_ARGS+=(-e "$p"); done + [ -f "$1/.igneum-scratch-spare" ] || return 0 + while IFS= read -r p || [ -n "$p" ]; do + p="${p%%#*}"; p="${p#"${p%%[![:space:]]*}"}"; p="${p%"${p##*[![:space:]]}"}" + [ -n "$p" ] && SPARE_ARGS+=(-e "$p") + done < "$1/.igneum-scratch-spare" + return 0 +} +# the untracked paths the clean would still take, up to three (empty = the tree is clean); the same excludes as the clean line +tree_left() { + git -C "$1" clean -nd -e target -e 'target-*' -e '.build-remote-sha-*' -e '.cross-remote-sha-*' -e sccache "${SPARE_ARGS[@]}" | head -3 +} + # discard the previous overlay (tracked edits and untracked files; target dirs, the sha stamps and anything ignored are kept), # fetch, then the branch at the commit. Runs in BR_CO_DIR, clones it from BR_CO_MIRROR when it has no .git. checkout_tree() { @@ -68,7 +88,12 @@ checkout_tree() { if [ "$lock_age" -gt 30 ]; then rm -f .git/index.lock; echo "checkout: removed a stale .git/index.lock (${lock_age}s old) in $dir" >&2; fi fi git checkout -q -- . 2>/dev/null || true - git clean -qfd -e target -e 'target-*' -e '.build-remote-sha-*' -e '.cross-remote-sha-*' -e sccache + # 7 October 2026: the attack rows lost their scratch dirs (attack-f3/, attack-f1-venv/, tools/attack/*/target) to each + # other's builds, because this clean ran on the shared mirror before every build from any agent and took every untracked + # directory. A lane's scratch is now spared: the fixed prefixes and the mirror-local .igneum-scratch-spare (SPARE_ARGS, + # see spare_args above). Never -x here: .git/info/exclude still applies. tools/ci/scratch-spare-check.sh guards this line. + spare_args "$dir" + git clean -qfd -e target -e 'target-*' -e '.build-remote-sha-*' -e '.cross-remote-sha-*' -e sccache "${SPARE_ARGS[@]}" git fetch -q origin '+refs/heads/*:refs/remotes/origin/*' git checkout -q -B "$branch" "$sha" git reset -q --hard "$sha" @@ -78,11 +103,28 @@ checkout_tree() { # ... at any depth: a repo-kind crate (pool/, igneum-pow/, app/igneum-app/) writes its stamp in its own directory, and the # root-anchored pattern of the first version failed the second build of every such crate (6 October 2026, 18:51 UTC, the # pool build: "tree not clean after reset: ?? pool/.build-remote-sha-target"; the self-test has the subdirectory case now) - local left; left=$(git status --porcelain --untracked-files=all | grep -vE '^\?\? (.*/)?(target|target-|sccache|\.build-remote-sha-|\.cross-remote-sha-)' | head -3 || true) + # ... and since 7 October 2026 a lane's spared scratch, so the test asks the clean itself what it would still remove + # (tree_left: `git clean -nd` with the same excludes; a spared directory is no longer "not clean") + local left; left=$(tree_left "$dir" || true) [ -z "$left" ] || { echo "checkout: tree not clean after reset at $dir: $left" >&2; return 1; } set +e } +if [ "${1:-}" = --self-test-keeper ]; then + # the keeper restores a truncated holder line within its interval, and stops at release + t=$(mktemp -d); trap 'rm -rf "$t"' EXIT + BR_PID=$$; BR_LABEL="keeper self-test"; got=0; waited=3; SLOTS_DIR="$t"; BR_KEEP_S=1 + holder_line() { printf 'pid %s since %sZ waited %s s: %s\n' "$BR_PID" "$(date -u +%H:%M:%S)" "$1" "$BR_LABEL"; } + holder_line "$waited" > "$t/build-0" + keeper_pid=""; keep_line() { local f="$1" w="$2"; ( while kill -0 "$BR_PID" 2>/dev/null; do [ -s "$f" ] || holder_line "$w" > "$f" 2>/dev/null; sleep "${BR_KEEP_S:-20}"; done ) & keeper_pid=$!; } + keep_line "$t/build-0" "$waited" + : > "$t/build-0"; sleep 2.5 + grep -q "^pid $$ since .* waited 3 s: keeper self-test$" "$t/build-0" || { echo "keeper self-test: the truncated holder line was NOT restored"; kill "$keeper_pid" 2>/dev/null; exit 1; } + kill "$keeper_pid" 2>/dev/null; wait "$keeper_pid" 2>/dev/null; : > "$t/build-0"; sleep 2.5 + [ ! -s "$t/build-0" ] || { echo "keeper self-test: the keeper kept writing after release"; exit 1; } + echo "keeper self-test: a truncated holder line comes back within the interval; nothing is written after release"; exit 0 +fi + if [ "${1:-}" = --self-test ]; then t=$(mktemp -d); trap 'rm -rf "$t"' EXIT git init -q --bare -b master "$t/mirror.git" @@ -95,6 +137,11 @@ if [ "${1:-}" = --self-test ]; then echo edited > "$t/box/a.txt"; echo new > "$t/box/b.txt"; mkdir -p "$t/box/target/release"; echo bin > "$t/box/target/release/x"; echo "$sha1" > "$t/box/.build-remote-sha-target" # a crate in a subdirectory: its stamp and target dir must survive, its untracked overlay file must not mkdir -p "$t/box/sub/target/release"; echo bin > "$t/box/sub/target/release/y"; echo "$sha1" > "$t/box/sub/.build-remote-sha-target"; echo new > "$t/box/sub/c.txt" + # a lane's scratch (7 October 2026): a fixed-prefix dir at the root and nested, a dir declared in .igneum-scratch-spare + # (comments and blank lines in the file), and an undeclared dir that the clean must still take + mkdir -p "$t/box/attack-f3" "$t/box/sub/attack-f1-venv" "$t/box/bs-lane-1" "$t/box/undeclared-1" + echo x > "$t/box/attack-f3/x"; echo x > "$t/box/sub/attack-f1-venv/x"; echo x > "$t/box/bs-lane-1/x"; echo x > "$t/box/undeclared-1/x" + printf '# the build-server lane\n\n bs-lane-* # trailing comment\n' > "$t/box/.igneum-scratch-spare" : > "$t/box/.git/index.lock" # a run killed mid-git leaves this; the checkout mode removes it once it is older than 30 s touch -t "$(date -d '-2 min' +%Y%m%d%H%M.%S 2>/dev/null || date -v-2M +%Y%m%d%H%M.%S)" "$t/box/.git/index.lock" # backdated two minutes (GNU date, then BSD date) [ $(( $(date +%s) - $(stat -c %Y "$t/box/.git/index.lock" 2>/dev/null || stat -f %m "$t/box/.git/index.lock") )) -gt 30 ] || { echo "self-test: could not backdate the fixture lock"; exit 1; } @@ -111,12 +158,21 @@ if [ "${1:-}" = --self-test ]; then [ -f "$t/box/sub/.build-remote-sha-target" ] || { echo "self-test: a subdirectory crate's sha stamp was cleaned"; exit 1; } [ -f "$t/box/sub/target/release/y" ] || { echo "self-test: a subdirectory crate's target dir was cleaned"; exit 1; } [ ! -e "$t/box/sub/c.txt" ] || { echo "self-test: a subdirectory's untracked overlay file survived"; exit 1; } + # a lane's scratch (7 October 2026): fixed prefixes at any depth and a declared glob survive, the spare file survives, an + # undeclared dir is removed + [ -f "$t/box/attack-f3/x" ] || { echo "self-test: a fixed-prefix scratch dir (attack-*) was cleaned"; exit 1; } + [ -f "$t/box/sub/attack-f1-venv/x" ] || { echo "self-test: a nested fixed-prefix scratch dir was cleaned"; exit 1; } + [ -f "$t/box/bs-lane-1/x" ] || { echo "self-test: a scratch dir declared in .igneum-scratch-spare was cleaned"; exit 1; } + [ -f "$t/box/.igneum-scratch-spare" ] || { echo "self-test: the .igneum-scratch-spare file itself was cleaned"; exit 1; } + [ ! -e "$t/box/undeclared-1" ] || { echo "self-test: an undeclared scratch dir survived the clean"; exit 1; } # the known-failed case: an untracked file the overlay left that no rule keeps must fail the check echo stray > "$t/box/sub/stray.txt" - if (cd "$t/box" && left=$(git status --porcelain --untracked-files=all | grep -vE '^\?\? (.*/)?(target|target-|sccache|\.build-remote-sha-|\.cross-remote-sha-)' | head -3); [ -n "$left" ]); then :; else echo "self-test: the clean-tree check did NOT fire on a stray untracked file"; exit 1; fi + spare_args "$t/box"; left=$(tree_left "$t/box"); [ -n "$left" ] || { echo "self-test: the clean-tree check did NOT fire on a stray untracked file"; exit 1; } rm -f "$t/box/sub/stray.txt" + # ... and a spared directory alone must NOT fire it (the check asks the clean, not the status list) + left=$(tree_left "$t/box"); [ -z "$left" ] || { echo "self-test: the clean-tree check fired on spared scratch: $left"; exit 1; } [ ! -f "$t/box/.git/index.lock" ] || { echo "self-test: the stale index.lock survived"; exit 1; } - echo "self-test: checkout mode lands on the new commit with a clean tree, target dirs and sha stamps kept at any depth, stale index.lock removed, and fires on a stray file"; exit 0 + echo "self-test: checkout mode lands on the new commit with a clean tree, target dirs and sha stamps kept at any depth, declared and fixed-prefix scratch kept, an undeclared dir removed, stale index.lock removed, and fires on a stray file"; exit 0 fi if [ "${1:-}" = --self-test-slots ]; then @@ -196,7 +252,8 @@ d = { "agent": e.get('BR_AGENT'), "slot": num(e['BR_SLOT']), "wait_s": num(e['BR_WAIT']), "queued_at": iso(e['BR_T0']), "start": iso(e['BR_START']), "end": iso(e['BR_END']), "secs": num(e['BR_SECS']), "exit": num(e['BR_EXIT']), "compiles": num(e['BR_COMPILES']), "jobs": num(e.get('BR_JOBS')), "measure": (e.get('BR_MEASURE') == '1') or None, - "source_date_epoch": num(e.get('BR_SDE')), + "source_date_epoch": num(e.get('BR_SDE')), "pairs_with": e.get('BR_PAIRS_WITH') or None, + "nice": num(e.get('BR_NICE')) or 0, "cores": (num(e.get('BR_CORES')) or 0) or os.cpu_count(), "priority": e.get('BR_PRIORITY') or "normal", "class": e.get('BR_CLASS') or None, "run_log": e.get('BR_RUN_LOG') or None, } sc = {k: num(e[v]) for k, v in (("hits", "BR_HITS"), ("misses", "BR_MISSES"), ("hits_total", "BR_HITS_T"), ("misses_total", "BR_MISSES_T"))} @@ -280,6 +337,18 @@ else flock -s -w 7200 "$mfd" || give_up "the measurement to end" rm -f "$waitfile"; trap - EXIT fi + # scheduling (main, 7 Oct 2026): a gate announces itself (gate-pending-) and takes the next free slot; a suite, bench or + # other non-gate run that has not taken a slot yet yields while any gate-pending marker younger than 15 min exists + gatefile="" + if [ "${BR_PRIORITY:-normal}" = gate ]; then gatefile="$SLOTS_DIR/gate-pending-$BR_PID"; holder_line 0 > "$gatefile"; trap 'rm -f "$gatefile"' EXIT + else + yt0=$(date +%s) + while pending=$(find "$SLOTS_DIR" -maxdepth 1 -name 'gate-pending-*' -mmin -15 2>/dev/null | head -1) && [ -n "$pending" ]; do + [ -f "$waitfile" ] || { echo "build-remote: a gate is queued ($(head -c 120 "$pending")), this ${BR_KIND:-run} yields" >&2; holder_line 0 > "$waitfile"; trap 'rm -f "$waitfile"' EXIT; } + [ $(( $(date +%s) - yt0 )) -lt 7200 ] || give_up "the queued gate" + sleep 5 + done + fi got="" for k in $(seq 0 $((slots - 1))); do exec {fd}>>"$SLOTS_DIR/build-$k" @@ -308,11 +377,24 @@ else exec {tfd}>&- done if [ "$held" -gt 1 ]; then BR_JOBS=$JOBS_SHARED; else BR_JOBS=$JOBS_ALONE; fi + # the bounded classes cap the jobs (suite and bench: 32 unless the caller passed --priority gate) + if [ "${BR_JOBS_CAP:-0}" -gt 0 ] && [ "$BR_JOBS" -gt "$BR_JOBS_CAP" ]; then BR_JOBS=$BR_JOBS_CAP; fi export CARGO_BUILD_JOBS="$BR_JOBS" BR_JOBS - echo "build-remote: holding build-$got on $BR_HOST (waited $waited s; $held of $slots slots held, CARGO_BUILD_JOBS=$BR_JOBS)" >&2 + [ -n "$gatefile" ] && { rm -f "$gatefile"; trap - EXIT; } + echo "build-remote: holding build-$got on $BR_HOST (waited $waited s; $held of $slots slots held, CARGO_BUILD_JOBS=$BR_JOBS, kind ${BR_KIND:-other}, nice ${BR_NICE:-0}, cores $( [ "${BR_CORES:-0}" = 0 ] && nproc || echo "$BR_CORES"))" >&2 fi -release_slot() { if [ "$got" = measure ]; then : > "$SLOTS_DIR/measure"; else : > "$SLOTS_DIR/build-$got"; fi; } +# The holder keeps its own line (the watcher-trust rule, 7 October 2026 10:39Z: two held slots read EMPTY while two suites ran, +# because a run from a worktree without last night's append-mode fix still opens a busy sibling's slot file with `>` on every +# probe). A keeper re-writes the holder line whenever it finds the file empty, every BR_KEEP_S seconds, until release_slot. +keeper_pid="" +keep_line() { # + local f="$1" w="$2" + ( while kill -0 "$BR_PID" 2>/dev/null; do [ -s "$f" ] || holder_line "$w" > "$f" 2>/dev/null; sleep "${BR_KEEP_S:-20}"; done ) & + keeper_pid=$! +} +if [ "$got" = measure ]; then keep_line "$SLOTS_DIR/measure" "$waited"; else keep_line "$SLOTS_DIR/build-$got" "$waited"; fi +release_slot() { [ -n "$keeper_pid" ] && { kill "$keeper_pid" 2>/dev/null; wait "$keeper_pid" 2>/dev/null; keeper_pid=""; }; if [ "$got" = measure ]; then : > "$SLOTS_DIR/measure"; else : > "$SLOTS_DIR/build-$got"; fi; } cd "$BR_DIR" || { BR_CLASS=no-dir jsonlog 2 "$got" "$waited" "" "$(date +%s)" "" "" "" "" "" ""; redlog 2 0 no-dir; release_slot; exit 2; } # PRE-FLIGHT for a cargo command (the instant-death class, 6 October 2026): the manifest parses and every -p package exists, @@ -362,8 +444,15 @@ t1=$(date +%s) # streams stay where they were for the Mac (stdout to stdout, stderr to stderr), each teed into the run log RUN_LOG_DIR="$LOG_DIR/runs"; mkdir -p "$RUN_LOG_DIR" BR_RUN_LOG="$RUN_LOG_DIR/$BR_HOST-$BR_T0-$BR_PID.log"; export BR_RUN_LOG -( eval "$BR_CMD" ) > >(tee -a "$BR_RUN_LOG") 2> >(tee -a "$BR_RUN_LOG" >&2) +# the class's nice and core set apply to the command's subshell and everything it starts (renice and taskset on the subshell's own +# pid, BASHPID; cores are the LAST N of the box's set, so gates and builds keep the first ones to themselves) +ncpu=$(nproc); cores_str="0-$((ncpu - 1))" +if [ "${BR_CORES:-0}" -gt 0 ] && [ "${BR_CORES}" -lt "$ncpu" ]; then cores_str="$((ncpu - BR_CORES))-$((ncpu - 1))"; fi +( [ "${BR_NICE:-0}" -gt 0 ] && renice -n "$BR_NICE" -p $BASHPID >/dev/null 2>&1; [ "$cores_str" != "0-$((ncpu - 1))" ] && taskset -cp "$cores_str" $BASHPID >/dev/null 2>&1; eval "$BR_CMD" ) > >(tee -a "$BR_RUN_LOG") 2> >(tee -a "$BR_RUN_LOG" >&2) rc=$? +# the keeper stops BEFORE the bare `wait` (which flushes the two tees): a bare wait also waits for the keeper, and the keeper waits +# for this script, a deadlock that held build-2's first run 15 minutes after its test had passed (7 Oct 2026, 11:04 to 11:19Z) +if [ -n "$keeper_pid" ]; then kill "$keeper_pid" 2>/dev/null; wait "$keeper_pid" 2>/dev/null; keeper_pid=""; fi wait t2=$(date +%s); secs=$(( t2 - t1 )) exec_after=$(stat_field "Compile requests executed"); hits_after=$(stat_field "Cache hits "); misses_after=$(stat_field "Cache misses ") @@ -385,8 +474,8 @@ elif [[ "$BR_CMD" == cargo\ test* ]] && { [[ "$BR_CMD" == *" -- "* ]] || grep -q echo "build-remote: REFUSED after the run: the test filter matched no test in any binary (every 'running 0 tests'); check the name" >&2 fi export BR_CLASS -printf 'build-remote: RESULT rc=%s secs=%s compiles=%s sccache_hits=%s sccache_misses=%s sccache_hits_total=%s sccache_misses_total=%s jobs=%s load=%s class=%s\n' \ - "$rc" "$secs" "$compiles" "$hits" "$misses" "${hits_after:-?}" "${misses_after:-?}" "${BR_JOBS:-measure}" "$(cut -d' ' -f1-3 /proc/loadavg)" "${BR_CLASS:-ok}" +printf 'build-remote: RESULT rc=%s secs=%s compiles=%s sccache_hits=%s sccache_misses=%s sccache_hits_total=%s sccache_misses_total=%s jobs=%s nice=%s cores=%s load=%s class=%s\n' \ + "$rc" "$secs" "$compiles" "$hits" "$misses" "${hits_after:-?}" "${misses_after:-?}" "${BR_JOBS:-measure}" "${BR_NICE:-0}" "$( [ "${BR_CORES:-0}" = 0 ] && nproc || echo "$BR_CORES")" "$(cut -d' ' -f1-3 /proc/loadavg)" "${BR_CLASS:-ok}" jsonlog "$rc" "$([ "$got" = measure ] && echo "" || echo "$got")" "$waited" "$t1" "$t2" "$secs" "$compiles" "$hits" "$misses" "${hits_after:-}" "${misses_after:-}" [ "$rc" = 0 ] || redlog "$rc" "$secs" "$BR_CLASS" release_slot diff --git a/infra/build-server/run-from-mac.sh b/infra/build-server/run-from-mac.sh index f36fe94fc..a5a275a8c 100755 --- a/infra/build-server/run-from-mac.sh +++ b/infra/build-server/run-from-mac.sh @@ -20,11 +20,20 @@ BS_TOOL=run-from-mac # shellcheck source=lib.sh . "$HERE/lib.sh" +BOX=1; while [ "${1:-}" = --box ]; do BOX="$2"; shift 2; done # --box N: igneum-build-N, host file build-server-N (7 Oct 2026) IP="${1:-}"; shift || true -[ -n "$IP" ] || bs_die "usage: run-from-mac.sh [--wire-only]" +[ -n "$IP" ] || bs_die "usage: run-from-mac.sh [--box N] [--wire-only]" WIRE_ONLY=0; [ "${1:-}" = --wire-only ] && WIRE_ONLY=1 +# box N is igneum-build-N with its own dashboard feed name build-N.igneum.network (the dashboard lane's collector reads one server +# section per box; main adds the A record with the IP) +[ "$BOX" = 1 ] || { BOX_HOSTNAME="${BOX_HOSTNAME:-igneum-build-$BOX}"; WORKERS_HOST="${WORKERS_HOST:-build-$BOX.igneum.network}"; export BOX_HOSTNAME WORKERS_HOST; } +export BS_BOX="$BOX" # lib.sh's bs_host derives the host file from BS_BOX (box 1: build-server; box N: build-server-N); never set BS_HOST_FILE here (7 Oct 2026: a second suffix, build-server-2-2) +HOST_FILE=$(bs_box_file "$BOX") +REMOTE="build"; [ "$BOX" = 1 ] || REMOTE="build-$BOX" # one git remote per box in the two repositories +# the toolchain pin travels from rust-toolchain.toml unless RUST_TOOLCHAIN is set by hand (one file pins every side, 7 Oct 2026) +[ -n "${RUST_TOOLCHAIN:-}" ] || RUST_TOOLCHAIN=$(sed -n 's/^channel *= *"\([^"]*\)".*/\1/p' "$REPO/rust-toolchain.toml" 2>/dev/null | head -1); export RUST_TOOLCHAIN PASS="" # a string, not an array: bash 3.2 (the Mac) treats an empty array as unbound under set -u -for v in MODE RUST_TOOLCHAIN SCCACHE_GB SCCACHE_VERSION NODE_MAJOR SLOTS P2P_PORTS BOX_HOSTNAME SSH_PUBKEY; do +for v in MODE RUST_TOOLCHAIN SCCACHE_GB SCCACHE_VERSION NODE_MAJOR SLOTS P2P_PORTS BOX_HOSTNAME WORKERS_HOST SSH_PUBKEY; do [ -n "${!v:-}" ] && PASS="$PASS $v=$(printf '%q' "${!v}")" done @@ -41,19 +50,19 @@ if [ "$WIRE_ONLY" = 0 ]; then if "${ROOT_SSH[@]}" 'hostname' 2>/dev/null | grep -q '^rescue'; then bs_die "still in the rescue system after provision.sh"; fi fi -mkdir -p "$(dirname "$BS_HOST_FILE")" -printf 'build@%s\n' "$IP" > "$BS_HOST_FILE" -bs_log "wrote $BS_HOST_FILE" +mkdir -p "$(dirname "$HOST_FILE")" +printf 'build@%s\n' "$IP" > "$HOST_FILE" +bs_log "wrote $HOST_FILE" bs_host bs_ssh 'hostname; nproc' >/dev/null || bs_die "build@$IP does not answer with $BS_KEY" wire_remote() { # repo-dir mirror label local dir="$1" mirror="$2" label="$3" url="$BS_HOST:$2" cur - cur=$(git -C "$dir" remote get-url build 2>/dev/null || true) - if [ -z "$cur" ]; then git -C "$dir" remote add build "$url"; bs_log "$label: remote build = $url" - elif [ "$cur" != "$url" ]; then git -C "$dir" remote set-url build "$url"; bs_log "$label: remote build -> $url" - else bs_log "$label: remote build ok"; fi - GIT_SSH_COMMAND="$BS_SSH_CMD" git -C "$dir" push -q --force build --all && bs_log "$label: every branch pushed to $mirror ($(git -C "$dir" branch --list | wc -l | tr -d ' ') branches)" || bs_die "$label: push to $mirror failed" + cur=$(git -C "$dir" remote get-url "$REMOTE" 2>/dev/null || true) + if [ -z "$cur" ]; then git -C "$dir" remote add "$REMOTE" "$url"; bs_log "$label: remote $REMOTE = $url" + elif [ "$cur" != "$url" ]; then git -C "$dir" remote set-url "$REMOTE" "$url"; bs_log "$label: remote $REMOTE -> $url" + else bs_log "$label: remote $REMOTE ok"; fi + GIT_SSH_COMMAND="$BS_SSH_CMD" git -C "$dir" push -q --force "$REMOTE" --all && bs_log "$label: every branch pushed to $mirror ($(git -C "$dir" branch --list | wc -l | tr -d ' ') branches)" || bs_die "$label: push to $mirror failed" } wire_remote "$MAIN_REPO" "$BS_MIRROR_REPO" "igneum" wire_remote "$MAIN_REPO/vendor/igneum-node" "$BS_MIRROR_NODE" "igneum-node (the fork)" diff --git a/infra/build-server/workers/install.sh b/infra/build-server/workers/install.sh index 6842091e3..91bce76d4 100755 --- a/infra/build-server/workers/install.sh +++ b/infra/build-server/workers/install.sh @@ -1,11 +1,18 @@ #!/usr/bin/env bash # Install or refresh the worker dashboard collector on igneum-build-1 from this Mac. -# infra/build-server/workers/install.sh copies tools/workers/{lib,collect}.mjs and the two units, enables the timer -# Needs root over ssh (root@ with ~/.ssh/igneum_ed25519); the host ip comes from ~/.config/igneum/build-server (build@). +# infra/build-server/workers/install.sh build-1: copies tools/workers/{lib,collect}.mjs and the two units, enables the timer +# infra/build-server/workers/install.sh 2 igneum-build-2 (host line in ~/.config/igneum/build-server-2), 3 for build-3 +# infra/build-server/workers/install.sh build@ any box by its host line +# Needs root over ssh (root@ with ~/.ssh/igneum_ed25519); the host ip comes from ~/.config/igneum/build-server[-N] (build@). set -euo pipefail HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"; ROOT="$(cd "$HERE/../../.." && pwd)" KEY="${IGNEUM_BUILD_KEY:-$HOME/.ssh/igneum_ed25519}" -HOST_LINE="$(head -1 "${IGNEUM_BUILD_HOST_FILE:-$HOME/.config/igneum/build-server}" | tr -d '[:space:]')" +case "${1:-}" in + "") HOST_LINE="$(head -1 "${IGNEUM_BUILD_HOST_FILE:-$HOME/.config/igneum/build-server}" | tr -d '[:space:]')" ;; + [2-9]) HOST_LINE="$(head -1 "$HOME/.config/igneum/build-server-$1" | tr -d '[:space:]')" ;; + *@*) HOST_LINE="$1" ;; + *) echo "usage: install.sh [N | build@]" >&2; exit 2 ;; +esac IP="${HOST_LINE#*@}"; [ -n "$IP" ] || { echo "no build server in ~/.config/igneum/build-server" >&2; exit 1; } SSH=(ssh -i "$KEY" -o BatchMode=yes -o StrictHostKeyChecking=accept-new -o ConnectTimeout=10) "${SSH[@]}" "root@$IP" 'install -d -o build -g build /srv/workers /srv/workers/bin /srv/workers/sources /srv/builds/_log; chown build:build /srv/builds/_log' diff --git a/packaging/ota/publish-jobs.sh b/packaging/ota/publish-jobs.sh index 82f1f126d..e2c019f7e 100755 --- a/packaging/ota/publish-jobs.sh +++ b/packaging/ota/publish-jobs.sh @@ -18,6 +18,10 @@ # [--to name] [--extract] [--extract-dir sub] [--fresh] (or --url https://... --sha256 ... [--size N]) # packaging/ota/publish-jobs.sh add --kind collect --target all --glob "logs/app-*.log" [--glob ...] [--command "nvidia-smi"] # packaging/ota/publish-jobs.sh add --kind restart --target 1ccfe586 --what miners|node|app +# packaging/ota/publish-jobs.sh add --kind cards --target 1ccfe586 --cards "nvidia:0:NVIDIA GeForce RTX 5090=on:8,amd::Radeon=off" +# (the signed card settings kind, 7 October 2026: per card, on|off and the identities after the colon, power_pct +# after a second colon; a card the machine does not have is refused; applied through the app's own card path +# and persisted; the report reads every named card back) # packaging/ota/publish-jobs.sh add --kind update-now --target all # packaging/ota/publish-jobs.sh add --kind shard-benchmark --target 1ccfe586 [--zip ~/Desktop/igneum-prove-wsl2.zip] \ # [--fixtures "block-338-shard1 block-341-shards2 block-344-shards4"] [--cap-minutes 90] [--distro Ubuntu-24.04] [--wsl-user [user]] @@ -53,7 +57,7 @@ CMD="${1:-}"; [ $# -gt 0 ] && shift KIND="" TARGET="" PLATFORM="" REQUIRES="" REQUIRES_SET=0 ID="" TITLE="" EXPIRES_H="48" DEPLOY=0 BASE="" DEST="" TRIES=12 SCRIPT="" SHELL_KIND="" ELEVATED=0 STOP_MINERS=0 TIMEOUT_MIN="" CARDS_OFF="" FILE="" URL="" SHA="" SIZE="" DIR="" TO="" EXTRACT=0 EXTRACT_DIR="" FRESH=0 -GLOBS=() COMMAND="" WHAT="" +GLOBS=() COMMAND="" WHAT="" CARDS="" ZIP="" FIXTURES="" CAP_MIN="" DISTRO="" WSL_USER="" REMOVE_ID="" TARGETS="" BUDGET_MIN="" STAGE_MIN="" MIN_FREE_GB="" TESTS=1 RELAY_URL="" NICE="" CARGO_JOBS="" case "$CMD" in @@ -86,6 +90,7 @@ while [ $# -gt 0 ]; do --glob) GLOBS+=("$2"); shift 2 ;; --command) COMMAND="$2"; shift 2 ;; --what) WHAT="$2"; shift 2 ;; + --cards) CARDS="$2"; shift 2 ;; --zip) ZIP="$2"; shift 2 ;; --fixtures) FIXTURES="$2"; shift 2 ;; --cap-minutes) CAP_MIN="$2"; shift 2 ;; @@ -310,6 +315,22 @@ print(json.dumps(d))' "$COMMAND" "${GLOBS[@]:-}")" update-now) [ -n "$TITLE" ] || TITLE="update now" ;; + cards) + [ -n "$CARDS" ] || { echo "cards: --cards \"key=on:8,key=off\"" >&2; exit 2; } + PARAMS="$(python3 -c 'import json,sys +out=[] +for item in [x for x in sys.argv[1].split(",") if x.strip()]: + key, _, v = item.partition("=") + parts = v.split(":") + d = {"key": key.strip()} + if parts[0] in ("on", "off"): d["enabled"] = parts[0] == "on" + elif parts[0]: raise SystemExit("cards: %s: on|off expected, got %s" % (key, parts[0])) + if len(parts) > 1 and parts[1]: d["identities"] = int(parts[1]) + if len(parts) > 2 and parts[2]: d["power_pct"] = int(parts[2]) + out.append(d) +print(json.dumps({"cards": out}))' "$CARDS")" + [ -n "$TITLE" ] || TITLE="card settings" + ;; shard-benchmark) [ -n "$ZIP" ] || ZIP="$HOME/Desktop/igneum-prove-wsl2.zip" read -r ZURL ZSHA ZSIZE < <(hosted_file "$ZIP") diff --git a/relay/playbooks/first-share.ps1 b/relay/playbooks/first-share.ps1 new file mode 100644 index 000000000..ac0ce4028 --- /dev/null +++ b/relay/playbooks/first-share.ps1 @@ -0,0 +1,87 @@ +# Igneum playbook: one fresh Windows install timed to its first share or block (launch gate LG-4, +# docs/plans/miner-ui-5-first-share-runbook.md; the job form of it: docs/plans/miner-faults.md). +# +# powershell -ExecutionPolicy Bypass -File first-share.ps1 -Url [-Address 0x...] [-BudgetSeconds 1800] +# [-AllowInstalled] +# +# Runs on a FRESH Windows box only: a machine with the Igneum Miner installed is refused unless -AllowInstalled is given +# (a fresh-install measurement replaces the installed app and wipes %LOCALAPPDATA%\igneum; on a machine that mines for +# someone that is their call, never the job's: the rule of 5 October 2026). Prints the six RESULT lines of the runbook, +# then uninstalls and wipes so the next run on the same box is fresh again. Reads only the engine it installed (its own +# app.url); never pauses, resumes or quits another engine (tools/ci/playbook-quit-check.sh). +param( + [Parameter(Mandatory = $true)][string]$Url, + [string]$Address = '0x4242424242424242424242424242424242424242', + [int]$BudgetSeconds = 1800, + [switch]$AllowInstalled +) +$ErrorActionPreference = 'Stop' +function Now { [int64]([DateTimeOffset]::UtcNow.ToUnixTimeMilliseconds()) } +function Result([string]$line) { Write-Host ("RESULT " + $line) } +$appData = Join-Path $env:LOCALAPPDATA 'igneum' +$programs = Join-Path $env:LOCALAPPDATA 'Programs\Igneum Miner' +if ((Test-Path $appData) -or (Test-Path $programs)) { + if (-not $AllowInstalled) { Result "total_s=0 pass=false reason=installed_app_present (this is not a fresh box; -AllowInstalled overrides, with the owner's word)"; exit 2 } + Get-Process -Name 'igneum-app', 'Igneum Miner', 'igneumd', 'igneum-miner' -ErrorAction SilentlyContinue | Stop-Process -Force -ErrorAction SilentlyContinue + Start-Sleep -Seconds 3 + Remove-Item -Recurse -Force $appData -ErrorAction SilentlyContinue +} +# 1. download: the public installer through the dl host, timed from the first byte +$setup = Join-Path $env:TEMP 'Igneum-Miner-Setup-first-share.exe' +Remove-Item $setup -ErrorAction SilentlyContinue +$tDl0 = Now +Invoke-WebRequest -Uri $Url -OutFile $setup -UseBasicParsing +$tDl1 = Now +$bytes = (Get-Item $setup).Length +$sha = (Get-FileHash $setup -Algorithm SHA256).Hash.ToLower() +Result "step=download start=$tDl0 end=$tDl1 bytes=$bytes sha256=$sha" +# 2. install (Inno Setup, per user, silent), then the engine's first screen = app.url present and api/state answering +$p = Start-Process -FilePath $setup -ArgumentList '/VERYSILENT', '/SUPPRESSMSGBOXES', '/NORESTART', '/SP-' -Wait -PassThru +if ($p.ExitCode -ne 0) { Result "step=install end=$(Now) exit=$($p.ExitCode)"; Result "total_s=$([int](((Now) - $tDl0) / 1000)) pass=false reason=installer_exit_$($p.ExitCode)"; exit 3 } +$exe = Get-ChildItem -Path $programs -Filter 'Igneum Miner.exe' -Recurse -ErrorAction SilentlyContinue | Select-Object -First 1 +if (-not $exe) { $exe = Get-ChildItem -Path $programs -Filter 'igneum-app.exe' -Recurse -ErrorAction SilentlyContinue | Select-Object -First 1 } +if (-not $exe) { Result "total_s=$([int](((Now) - $tDl0) / 1000)) pass=false reason=no_app_exe_after_install"; exit 3 } +Start-Process -FilePath $exe.FullName | Out-Null +$urlFile = Join-Path $appData 'app\app.url' +$url = $null +$deadline = (Now) + 180000 +while ((Now) -lt $deadline) { + if (Test-Path $urlFile) { $url = (Get-Content $urlFile -Raw).Trim(); if ($url) { try { $null = Invoke-RestMethod -Uri ($url + 'api/state') -TimeoutSec 5; break } catch {} } } + Start-Sleep -Milliseconds 500 +} +if (-not $url) { Result "total_s=$([int](((Now) - $tDl0) / 1000)) pass=false reason=engine_not_up_in_180s"; exit 4 } +$tInstall = Now +Result "step=install end=$tInstall" +# 3. setup: the address the owner pasted, then start (what the welcome screens post) +try { $null = Invoke-RestMethod -Uri ($url + 'api/setup') -Method Post -ContentType 'application/json' -Body (@{ mode = 'paste'; address = $Address } | ConvertTo-Json) -TimeoutSec 10 } catch { Write-Host "api/setup: $_" } +try { $null = Invoke-RestMethod -Uri ($url + 'api/key/saved') -Method Post -TimeoutSec 10 } catch {} +try { $null = Invoke-RestMethod -Uri ($url + 'api/start') -Method Post -TimeoutSec 10 } catch { Write-Host "api/start: $_" } +$tSetup = Now +Result "step=setup end=$tSetup" +# 4. node synced, 5. the first accepted block or share (ladder.first_block_at), within the budget +$synced = $null; $first = $null; $stall = 'node'; $mining = $null; $hash = '' +$deadline = $tDl0 + ($BudgetSeconds * 1000) +while ((Now) -lt $deadline) { + try { $st = Invoke-RestMethod -Uri ($url + 'api/state') -TimeoutSec 5 } catch { Start-Sleep -Seconds 2; continue } + if (-not $synced -and $st.node.synced) { $synced = Now; $stall = 'card'; Result "step=node end=$synced synced_at=$($st.ladder.first_synced_at)" } + if ($st.ladder -and $st.ladder.first_mining_at -and -not $mining) { $mining = $st.ladder.first_mining_at } + if ($st.ladder -and $st.ladder.first_block_at) { $first = Now; $hash = $st.ladder.first_block_hash; break } + if ($st.ladder -and $st.ladder.first_share_at) { $first = Now; $hash = 'share'; break } + Start-Sleep -Seconds 2 +} +$total = [int](((Now) - $tDl0) / 1000) +if ($first) { + Result "step=card end=$first mining_at=$mining first_block_at=$first hash=$hash" + $pass = ($total -le 600) + Result "total_s=$total pass=$($pass.ToString().ToLower())" +} else { + Result "total_s=$total pass=false stalled_in=$stall" +} +# 6. fresh again: quit the engine this job installed (its own URL), uninstall, wipe +try { $null = Invoke-RestMethod -Uri ($url + 'api/quit') -Method Post -TimeoutSec 10 } catch {} +Start-Sleep -Seconds 5 +$unins = Get-ChildItem -Path $programs -Filter 'unins*.exe' -Recurse -ErrorAction SilentlyContinue | Select-Object -First 1 +if ($unins) { Start-Process -FilePath $unins.FullName -ArgumentList '/VERYSILENT', '/SUPPRESSMSGBOXES', '/NORESTART' -Wait | Out-Null } +Remove-Item -Recurse -Force $appData -ErrorAction SilentlyContinue +Remove-Item $setup -ErrorAction SilentlyContinue +if ($first -and $total -le 600) { exit 0 } else { exit 1 } diff --git a/tools/build-remote.sh b/tools/build-remote.sh index 30cffa0a7..9618ec3d5 100755 --- a/tools/build-remote.sh +++ b/tools/build-remote.sh @@ -12,6 +12,30 @@ # tools/build-remote.sh --jobs 48 -- check # tools/build-remote.sh --target-dir target-exp -- build --release another persistent target dir on the box # tools/build-remote.sh --no-fetch -- clippy --all-targets nothing comes back (tests, check, clippy) +# tools/build-remote.sh --ship [hive|rig|seed|linux] [--glibc X.Y] anything that SHIPS: cargo zigbuild for +# x86_64-unknown-linux-gnu. where the glibc comes +# from the class (hive/rig 2.31: HiveOS is Ubuntu 20.04 +# based; seed/linux 2.35: Debian 12 and Ubuntu 22.04; +# default seed; --glibc overrides) +# (zig as the C/C++ toolchain, as the Mac's +# infra/cross/build-linux.sh), artefacts from +# target/x86_64-unknown-linux-gnu/release, each checked by +# tools/ci/glibc-ceiling-check.sh (need at most ). +# A plain build is native glibc 2.39: the box, the fleet's +# Ubuntu 24.04 hosts, never a seed or a rig (7 Oct 2026: +# a seed took 14 restarts on a 2.39 binary). +# tools/build-remote.sh --priority gate -- test ... a RELEASE GATE (the app gate, the canary cut, the pre-push +# self-tests): nice 0, the full core set, a slot ahead of +# queued suites and benches +# tools/build-remote.sh --plan [--priority gate] -- print the resolved class (kind, nice, cores, jobs) and stop; +# no box, no crate needed (the CI check uses it) +# +# Scheduling (main, 7 October 2026, after a load of 190 on 96 threads: a release join bench and the 0.3.19 app gate starving each +# other and "builds" of 16 minutes): every SUITE (cargo test) and BENCH (cargo bench) runs under nice 10 on the last 32 cores with +# -j 32 unless the caller passes --priority gate; builds keep the box's own jobs rule (90 alone, 45 beside another slot holder); a +# gate runs at nice 0 on the full set and takes a slot ahead of queued suites and benches (remote-run.sh gate-pending marker). The +# class travels in the slot label ("; kind=suite nice=10 cores=32") and the JSONL line, so the dashboard's job card shows why a +# job is slow. A call without a priority flag defaults to the bounded class for suites and benches: tools/ci/build-kind-default-check.sh. # tools/build-remote.sh --self-test-repro [--full] from a fork worktree: igneum-miner built twice a minute # apart without sccache into one target dir must give one # sha256, and a per-run target path must not (prost's @@ -47,8 +71,11 @@ BS_TOOL=build-remote # shellcheck source=../infra/build-server/lib.sh . "$HERE/../infra/build-server/lib.sh" +{ # whole-body: bash parses this block entirely before running a line of it, so an edit to this file while a run is in + # flight cannot reach the running copy (7 Oct 2026: build-remote.sh was edited mid-run and died on shifted bytes after a 4-min build) + # JOBS empty = the box decides: 90 alone, 45 beside another slot holder (remote-run.sh, main's ruling 6 Oct 2026) -JOBS="${JOBS:-}"; OUT=""; ARTEFACTS=""; TARGET_DIR="target"; FETCH=1; CARGO_ARGS=(); SELFTEST=0; FULL=0 +JOBS="${JOBS:-}"; OUT=""; ARTEFACTS=""; TARGET_DIR="target"; FETCH=1; CARGO_ARGS=(); SELFTEST=0; FULL=0; SHIP=0; SHIP_CLASS="${SHIP_CLASS:-seed}"; PRIORITY="${PRIORITY:-normal}"; PLAN=0; BOX="${BOX:-}"; GLIBC="${GLIBC:-}" while [ $# -gt 0 ]; do case "$1" in --jobs) JOBS="$2"; shift 2 ;; @@ -57,6 +84,12 @@ while [ $# -gt 0 ]; do --target-dir) TARGET_DIR="$2"; shift 2 ;; --no-fetch) FETCH=0; shift ;; --self-test-repro) SELFTEST=1; shift ;; + --ship) SHIP=1; case "${2:-}" in hive|rig|seed|linux|native) SHIP_CLASS="$2"; shift 2 ;; *) shift ;; esac ;; + --priority) PRIORITY="$2"; shift 2 ;; + --box) BOX="$2"; BOX_GIVEN=1; shift 2 ;; + --gate) PRIORITY=gate; shift ;; + --plan) PLAN=1; shift ;; + --glibc) GLIBC="$2"; shift 2 ;; --full) FULL=1; shift ;; --) shift; CARGO_ARGS=("$@"); break ;; -h|--help) sed -n '2,32p' "$0"; exit 0 ;; @@ -66,8 +99,37 @@ done [ "${CARGO_ARGS[0]:-}" = cargo ] && CARGO_ARGS=("${CARGO_ARGS[@]:1}") CARGO_ARGS_GIVEN=""; [ -n "${CARGO_ARGS[*]:-}" ] && CARGO_ARGS_GIVEN=1 -bs_host +# the scheduling class, from the cargo subcommand and the priority flag alone (no box, no crate): BR_NICE, BR_CORES (the last N of the +# box's 96; 0 = the full set), BR_JOBS_CAP (0 = the box's rule), BR_PRIORITY; the kind for the log is refined after bs_context +BR_NICE=0; BR_CORES=0; BR_JOBS_CAP=0; BR_PRIORITY="$PRIORITY"; SCHED_CLASS=build +case "$PRIORITY" in gate|normal) ;; *) bs_die "--priority takes gate or normal, not '$PRIORITY'" ;; esac +case "${CARGO_ARGS[0]:-build}" in + test) SCHED_CLASS=suite ;; + bench) SCHED_CLASS=bench ;; + check|clippy) SCHED_CLASS=check ;; + *) SCHED_CLASS=build ;; +esac +if [ "$PRIORITY" = gate ]; then BR_NICE=0; BR_CORES=0; BR_JOBS_CAP=0 +elif [ "$SCHED_CLASS" = suite ] || [ "$SCHED_CLASS" = bench ]; then BR_NICE=10; BR_CORES=32; BR_JOBS_CAP=32; fi +# an explicit --jobs above the bounded class's cap is clamped (main's rule: bounded unless a priority flag; 7 Oct 2026: two suites +# ran at -j 90 on a box at load 190 because their callers passed --jobs 90) +if [ "$BR_JOBS_CAP" -gt 0 ] && [ -n "$JOBS" ] && [ "$JOBS" -gt "$BR_JOBS_CAP" ]; then bs_log "--jobs $JOBS clamped to $BR_JOBS_CAP for a $SCHED_CLASS (pass --priority gate for the full set)"; JOBS=$BR_JOBS_CAP; fi +export BR_NICE BR_CORES BR_JOBS_CAP BR_PRIORITY +if [ "$PLAN" = 1 ]; then + k="$SCHED_CLASS"; [ "$PRIORITY" = gate ] && k=gate + printf 'kind=%s nice=%s cores=%s jobs=%s priority=%s\n' "$k" "$BR_NICE" "$( [ "$BR_CORES" = 0 ] && echo 96 || echo "$BR_CORES")" "$( [ "$BR_JOBS_CAP" = 0 ] && echo box || echo "$BR_JOBS_CAP")" "$PRIORITY" + exit 0 +fi + +# the box: --box N, else the route by class (suite and bench to box 2, prove to box 3, the rest to box 1; lib.sh bs_route) +if [ -z "$BOX" ]; then + case "$SCHED_CLASS:$PRIORITY" in *:gate) BOX=1 ;; suite:*|bench:*) BOX=$(bs_route "$SCHED_CLASS") ;; *) BOX=1 ;; esac +fi +bs_host "$BOX" bs_context +# a proving crate routes to box 3 unless the caller chose (its builds and suites alike) +if [ "$BS_CRATE_REL" = proving/igneum-prove ] && [ -z "${BOX_GIVEN:-}" ] && [ "$PRIORITY" != gate ]; then b=$(bs_route prove); [ "$b" != "$BOX" ] && { BOX=$b; bs_host "$BOX"; }; fi +bs_log "box $BOX ($BS_HOST) for class $SCHED_CLASS, priority $PRIORITY" if [ "$SELFTEST" = 1 ]; then [ "$BS_KIND" = node ] || bs_die "--self-test-repro runs from a fork worktree (igneum-miner and kaspad live there)" @@ -103,24 +165,46 @@ fi exit 0 fi +# --ship: the zig path and the target triple dir for the artefacts +SHIP_TARGET=x86_64-unknown-linux-gnu +if [ "$SHIP" = 1 ]; then + [ -n "$GLIBC" ] || GLIBC=$("$HERE/ci/glibc-ceiling-check.sh" --ceiling-of "$SHIP_CLASS") || bs_die "unknown ship class $SHIP_CLASS" + [ "$GLIBC" != native ] || bs_die "--ship native is a plain build: drop --ship" + TARGET_SUB="$SHIP_TARGET/release" +else TARGET_SUB="release"; fi # defaults per crate case "$BS_KIND:$BS_CRATE_REL" in node:*) [ -n "${CARGO_ARGS[*]:-}" ] || CARGO_ARGS=(build --release -p kaspad -p igneum-miner --features kaspad/igneum-pow) - [ -n "$ARTEFACTS" ] || ARTEFACTS="$TARGET_DIR/release/igneumd $TARGET_DIR/release/igneum-miner" ;; + [ -n "$ARTEFACTS" ] || ARTEFACTS="$TARGET_DIR/$TARGET_SUB/igneumd $TARGET_DIR/$TARGET_SUB/igneum-miner" ;; repo:app/igneum-app) [ -n "${CARGO_ARGS[*]:-}" ] || CARGO_ARGS=(build --release) - [ -n "$ARTEFACTS" ] || ARTEFACTS="$TARGET_DIR/release/igneum-app $TARGET_DIR/release/igneum-ota-sign $TARGET_DIR/release/igneum-prove-verify" ;; + [ -n "$ARTEFACTS" ] || ARTEFACTS="$TARGET_DIR/$TARGET_SUB/igneum-app $TARGET_DIR/$TARGET_SUB/igneum-ota-sign $TARGET_DIR/$TARGET_SUB/igneum-prove-verify" ;; repo:proving/igneum-prove) [ -n "${CARGO_ARGS[*]:-}" ] || CARGO_ARGS=(build --release) - [ -n "$ARTEFACTS" ] || ARTEFACTS="$TARGET_DIR/release/igneum-prove-host $TARGET_DIR/release/igneum-prove-export" ;; + [ -n "$ARTEFACTS" ] || ARTEFACTS="$TARGET_DIR/$TARGET_SUB/igneum-prove-host $TARGET_DIR/$TARGET_SUB/igneum-prove-export" ;; *) [ -n "${CARGO_ARGS[*]:-}" ] || CARGO_ARGS=(build --release) ;; esac case "${CARGO_ARGS[0]}" in build) ;; *) [ -n "${ARTEFACTS_SET:-}" ] || { FETCH=0; ARTEFACTS=""; } ;; esac # test, check, clippy: nothing to fetch +if [ "$SHIP" = 1 ]; then + [ "${CARGO_ARGS[0]}" = build ] || bs_die "--ship is for cargo build" + CARGO_ARGS=(zigbuild "${CARGO_ARGS[@]:1}" --target "$SHIP_TARGET.$GLIBC") + BR_TARGET_SHIP="$SHIP_TARGET.$GLIBC" +fi [ -n "$OUT" ] || OUT="$BS_CRATE/target-remote" bs_log "$BS_KIND crate $BS_WT/$BS_CRATE_REL at $BS_SHA ($BS_BRANCH) -> $BS_HOST:$BS_REMOTE_CRATE; cargo ${CARGO_ARGS[*]} -j ${JOBS:-auto}; target dir $TARGET_DIR" +# a fork build pairs with the igneum-pow of the igneum worktree it sits in (the fork's path dependency ../../../../igneum-pow), so the +# pairing is said on the first line and recorded (7 Oct 2026, 0.3.17: a fork at 12153428 under a master worktree failed in kaspa-pow +# four minutes in, "no associated function chain_program_shadow", because master's igneum-pow predates the release branch's) +if [ "$BS_KIND" = node ]; then + pair_branch=$(git -C "$BS_WT_ROOT" branch --show-current 2>/dev/null || true); pair_dirty=$(git -C "$BS_WT_ROOT" status --porcelain -- igneum-pow 2>/dev/null || true) + # no `[ ... ] && echo` inside an assignment's $( ): its false status is the assignment's status and set -e ends the script (7 Oct 2026, 03:0x UTC) + PAIR="igneum $(git -C "$BS_WT_ROOT" rev-parse --short HEAD) (${pair_branch:-detached})${pair_dirty:+ with uncommitted igneum-pow changes}" + bs_log "pairs with $PAIR: igneum-pow $(grep -m1 '^version' "$BS_WT_ROOT/igneum-pow/Cargo.toml" | sed 's/.*"\(.*\)".*/\1/')" + export BR_PAIRS_WITH="$PAIR" +fi bs_toolchain_check t_sync0=$(date +%s) bs_sync_sources @@ -130,12 +214,15 @@ bs_log "sources in place after $(( $(date +%s) - t_sync0 )) s (every changed fil # rerun-if-changed and is never run again by cargo (release-0.3.11 plan: `cargo clean -p kaspa-build-info` first); so when # the commit the box builds differs from the last one built in this target dir, that one crate is cleaned (a relink, seconds) pre="" -if [ "$BS_KIND" = node ] && [ "${CARGO_ARGS[0]}" = build ]; then +if [ "$BS_KIND" = node ] && { [ "${CARGO_ARGS[0]}" = build ] || [ "${CARGO_ARGS[0]}" = zigbuild ]; }; then pre="[ \"\$(cat '.build-remote-sha-$TARGET_DIR' 2>/dev/null)\" = '$BS_SHA' ] || CARGO_TARGET_DIR='$TARGET_DIR' cargo clean -q --release -p kaspa-build-info 2>/dev/null; " fi cmd="$(bs_repro_env)${pre}CARGO_TARGET_DIR='$TARGET_DIR' cargo $(printf '%q ' "${CARGO_ARGS[@]}")${JOBS:+-j $JOBS} 2>&1 | tee -a '$BS_REMOTE_WT/.build-remote.log'; rc=\${PIPESTATUS[0]}; [ \$rc = 0 ] && echo '$BS_SHA' > '.build-remote-sha-$TARGET_DIR'; ( exit \$rc )" # a subshell exit: the runner reads \$? and still prints its RESULT line label="$BS_WT/$BS_CRATE_REL cargo ${CARGO_ARGS[*]}" -BR_KIND=$(bs_kind build-remote "${CARGO_ARGS[0]}"); BR_COMMAND="cargo ${CARGO_ARGS[*]}"; BR_TARGET=x86_64-unknown-linux-gnu +BR_KIND=$(bs_kind build-remote "$( [ "${CARGO_ARGS[0]}" = zigbuild ] && echo build || echo "${CARGO_ARGS[0]}")"); BR_COMMAND="cargo ${CARGO_ARGS[*]}"; BR_TARGET="${BR_TARGET_SHIP:-x86_64-unknown-linux-gnu}" +[ "$SCHED_CLASS" = bench ] && BR_KIND=bench +[ "$PRIORITY" = gate ] && BR_KIND=gate +label="$label; kind=$BR_KIND nice=$BR_NICE cores=$( [ "$BR_CORES" = 0 ] && echo 96 || echo "$BR_CORES")" for ((i = 0; i < ${#CARGO_ARGS[@]}; i++)); do [ "${CARGO_ARGS[$i]}" = --target ] && BR_TARGET="${CARGO_ARGS[$((i + 1))]:-}"; done BR_ARTEFACTS=""; [ "$FETCH" = 1 ] && BR_ARTEFACTS="$ARTEFACTS" export BR_KIND BR_COMMAND BR_TARGET BR_ARTEFACTS @@ -162,6 +249,10 @@ if [ "$FETCH" = 1 ] && [ -n "$ARTEFACTS" ]; then bs_log "artefact $dest: $(bs_size "$dest") bytes, sha256 $(bs_sha256 "$dest"), $(file -b "$dest" | cut -c1-60)" # the commit-string gate (rule of 6 October 2026): a node binary without its commit in its strings fails the run case "$BS_KIND:$(basename "$dest")" in node:igneumd) "$HERE/ci/commit-string-check.sh" "$dest" "$BS_SHA" || bs_die "commit-string gate failed for $a" ;; esac # only kaspad depends on kaspa-build-info + # the glibc ceiling of anything that ships (main, 7 Oct 2026): a seed or a rig refuses a binary needing more than 2.36 + if [ "$SHIP" = 1 ]; then "$HERE/ci/glibc-ceiling-check.sh" "$dest" "$GLIBC" || bs_die "glibc ceiling gate failed for $a"; fi done fi bs_wt_unlock +exit 0 +} diff --git a/tools/ci/permanent-fault-check.sh b/tools/ci/permanent-fault-check.sh new file mode 100755 index 000000000..fd304c8c9 --- /dev/null +++ b/tools/ci/permanent-fault-check.sh @@ -0,0 +1,28 @@ +#!/usr/bin/env bash +# No fault is permanent (the project lead, 7 October 2026, docs/plans/miner-faults.md MF-2): the app's engine may never mark a card +# "faulted" or say "restarted once already"; every fault retries on the ladder. This check fails when either word comes +# back into the engine's sources or its UI. +# tools/ci/permanent-fault-check.sh the tree +# tools/ci/permanent-fault-check.sh --self-test fires on a fixture with the words, passes on one without +set -uo pipefail +cd "$(git rev-parse --show-toplevel)" || exit 1 +check() { + # $1: directory; a hit is a line in a .rs or .js file that sets or tests the faulted state or carries the words + grep -rnE --include='*.rs' --include='*.js' -e 'restarted once already' -e '"faulted"\.into\(\)' -e "state === 'faulted'" -e 'c\.state = "faulted"' "$1" 2>/dev/null +} +if [ "${1:-}" = "--self-test" ]; then + d="$(mktemp -d)"; trap 'rm -rf "$d"' EXIT + mkdir -p "$d/bad" "$d/good" + printf 'fn x() { c.state = "faulted".into(); }\n' > "$d/bad/a.rs" + printf 'fn x() { c.state = "restarting".into(); }\n' > "$d/good/a.rs" + if ! check "$d/bad" >/dev/null; then echo "self-test failed: the fixture with the faulted state passed"; exit 1; fi + if check "$d/good" >/dev/null; then echo "self-test failed: the clean fixture was flagged"; exit 1; fi + echo "self-test ok"; exit 0 +fi +hits="$(check app/igneum-app/src; check app/igneum-app/ui)" +if [ -n "$hits" ]; then + echo "permanent fault words in the engine (docs/plans/miner-faults.md MF-2: no fault is permanent):" + echo "$hits" + exit 1 +fi +exit 0 diff --git a/tools/ci/pre-push.sh b/tools/ci/pre-push.sh index c6364234d..f52318a92 100755 --- a/tools/ci/pre-push.sh +++ b/tools/ci/pre-push.sh @@ -68,6 +68,7 @@ tree_checks() { run "run jobs test their fetched kit before use" bash -c 'bash tools/ci/kit-path-check.sh --self-test && bash tools/ci/kit-path-check.sh' run "every Windows spawn of the app runs with a hidden console" bash -c 'node tools/ci/windows-spawn-check.mjs --self-test && node tools/ci/windows-spawn-check.mjs' run "pinned guest programs match their manifest" bash tools/ci/pinned-guests-check.sh + run "no permanent miner fault in the engine (miner-faults.md MF-2)" bash -c 'bash tools/ci/permanent-fault-check.sh --self-test && bash tools/ci/permanent-fault-check.sh' run "root prover playbooks kill the GPU server and unlink its socket" bash tools/ci/prover-socket-check.sh run "commit-string gate self-test" bash tools/ci/commit-string-check.sh --self-test run "build server remote checkout self-test" bash infra/build-server/remote-run.sh --self-test diff --git a/tools/cross-remote.sh b/tools/cross-remote.sh index 431137abe..d7a58e973 100755 --- a/tools/cross-remote.sh +++ b/tools/cross-remote.sh @@ -36,6 +36,9 @@ BS_TOOL=cross-remote # shellcheck source=../infra/build-server/lib.sh . "$HERE/../infra/build-server/lib.sh" +{ # whole-body: bash parses this block entirely before running a line of it, so an edit to this file while a run is in + # flight cannot reach the running copy (7 Oct 2026: build-remote.sh was edited mid-run and died on shifted bytes after a 4-min build) + TARGET=x86_64-pc-windows-gnu # JOBS empty = the box decides: 90 alone, 45 beside another slot holder (remote-run.sh, main's ruling 6 Oct 2026). One line, no # trailing comment: a `#` on this line once swallowed every assignment after it, CARGO_ARGS was never declared and the default @@ -126,3 +129,5 @@ if printf '%s\n' "$dlls" | grep -q 'libstdc++-6.dll'; then fi bs_log "exes in $OUT/$TARGET/release" bs_wt_unlock +exit 0 +} diff --git a/tools/fleet/first-share-gate.mjs b/tools/fleet/first-share-gate.mjs new file mode 100755 index 000000000..57f16c8d4 --- /dev/null +++ b/tools/fleet/first-share-gate.mjs @@ -0,0 +1,92 @@ +#!/usr/bin/env node +// Launch gate LG-4 as a job: "first share within 10 minutes of download in 9 of 10 fresh Windows installs" +// (docs/plans/miner-ui-5-first-share-runbook.md, docs/plans/miner-faults.md). Runs relay/playbooks/first-share.ps1 on +// Windows boxes over SSH, ten installs in all, and writes the evidence table the shipper's publish reads. +// +// node tools/fleet/first-share-gate.mjs run --version 0.3.20 --url https://dl.igneum.network//Igneum-Miner-Setup-0.3.20.exe +// [--boxes ~/igneum-fleet/windows-boxes.json] [--installs 10] [--allow-installed] +// node tools/fleet/first-share-gate.mjs check --version 0.3.20 exit 0 on a PASS line for that version, 1 otherwise +// +// Boxes file: a JSON array of the fleet registry's Windows rows ({kind:"windows", label, host, port, user, key, transport:"ssh"}); +// with no row the gate writes "LG-4 not run: no Windows box" and `check` fails. Each box runs installs in turn; boxes run in +// parallel. Output: site/evidence/first-share-.json (the runbook's keys per row, the medians, the pass count, the +// installer's sha256) and the gate line on stdout and in the file: "LG-4 PASS k/10" or "LG-4 FAIL k/10". +// Never touches PC 1 (the project lead's desk) and refuses a box whose installed app would be replaced unless --allow-installed is given +// and the box's row says "job_box": true (PC 2's case needs the project lead's word, recorded in the row by the fleet lane). + +import { readFileSync, writeFileSync, existsSync, mkdirSync } from 'node:fs'; +import { spawnSync } from 'node:child_process'; +import { homedir } from 'node:os'; +import { join, dirname, resolve } from 'node:path'; +import { fileURLToPath } from 'node:url'; + +const ROOT = resolve(dirname(fileURLToPath(import.meta.url)), '..', '..'); +const args = process.argv.slice(2); +const cmd = args[0]; +const opt = (n, d) => { const i = args.indexOf(n); return i >= 0 ? args[i + 1] : d; }; +const flag = (n) => args.includes(n); +const version = opt('--version'); +if (!version) { console.error('need --version'); process.exit(2); } +const evidence = join(ROOT, 'site', 'evidence', `first-share-${version}.json`); + +function gateLine(rows, installs) { + const passed = rows.filter(r => r.pass).length; + return `LG-4 ${passed >= Math.ceil(installs * 0.9) ? 'PASS' : 'FAIL'} ${passed}/${installs}`; +} + +if (cmd === 'check') { + if (!existsSync(evidence)) { console.log(`LG-4 not run for ${version}: no ${evidence}`); process.exit(1); } + const e = JSON.parse(readFileSync(evidence, 'utf8')); + console.log(e.gate); + process.exit(/^LG-4 PASS/.test(e.gate) ? 0 : 1); +} +if (cmd !== 'run') { console.log(readFileSync(fileURLToPath(import.meta.url), 'utf8').split('\n').slice(1, 16).map(l => l.replace(/^\/\/ ?/, '')).join('\n')); process.exit(2); } + +const url = opt('--url'); +if (!url) { console.error('need --url (the public installer)'); process.exit(2); } +const installs = Number(opt('--installs', '10')); +const boxesFile = opt('--boxes', join(homedir(), 'igneum-fleet', 'windows-boxes.json')).replace(/^~/, homedir()); +const boxes = existsSync(boxesFile) ? JSON.parse(readFileSync(boxesFile, 'utf8')).filter(b => b.kind === 'windows' && b.transport === 'ssh' && b.host) : []; +mkdirSync(dirname(evidence), { recursive: true }); +if (!boxes.length) { + const out = { version, url, date: new Date().toISOString(), installs, rows: [], gate: 'LG-4 not run: no Windows box', note: 'the fleet lane has no rented Windows GPU box; PC 2 over the relay needs the owner\'s word (docs/plans/miner-faults.md)' }; + writeFileSync(evidence, JSON.stringify(out, null, 2) + '\n'); + console.log(out.gate); + process.exit(1); +} +for (const b of boxes) { + if (/ae432dc7|pc1|pc-1/i.test(`${b.label} ${b.machine || ''}`)) { console.error(`refusing ${b.label}: PC 1 is never a job box`); process.exit(2); } +} +const allowInstalled = flag('--allow-installed'); +const parse = (text) => { + const r = { pass: false }; + for (const l of text.split('\n')) { + const m = /^RESULT (.*)$/.exec(l.trim()); if (!m) continue; + for (const kv of m[1].split(/\s+/)) { const [k, v] = kv.split('='); if (k === 'step') r._step = v; else if (r._step && k !== 'step') r[`${r._step}_${k}`] = v; else r[k] = v; } + } + r.pass = r.pass === 'true'; + r.total_s = Number(r.total_s || 0); + return r; +}; +const rows = []; +const perBox = Math.ceil(installs / boxes.length); +const started = Date.now(); +// boxes in parallel (one ssh each, installs in turn inside it), the Mac waits +const procs = boxes.map((b) => { + const n = Math.min(perBox, installs - rows.length); + const remote = `$ErrorActionPreference='Continue'; $f=\"$env:TEMP\\first-share.ps1\"; [IO.File]::WriteAllText($f, [Text.Encoding]::UTF8.GetString([Convert]::FromBase64String('${Buffer.from(readFileSync(join(ROOT, 'relay', 'playbooks', 'first-share.ps1'))).toString('base64')}'))); 1..${n} | ForEach-Object { Write-Host \"INSTALL $_\"; powershell -ExecutionPolicy Bypass -File $f -Url '${url}' ${allowInstalled && b.job_box ? '-AllowInstalled' : ''} }`; + const key = (b.key || '~/.ssh/igneum-fleet').replace(/^~/, homedir()); + const p = spawnSync('ssh', ['-o', 'ConnectTimeout=20', '-o', 'StrictHostKeyChecking=accept-new', '-i', key, '-p', String(b.port || 22), `${b.user}@${b.host}`, 'powershell', '-NoProfile', '-Command', remote], { encoding: 'utf8', timeout: (n * 1900 + 120) * 1000 }); + return { box: b, out: (p.stdout || '') + (p.stderr || ''), status: p.status }; +}); +for (const { box, out } of procs) { + for (const chunk of out.split(/^INSTALL \d+\r?$/m).slice(1)) { + const r = parse(chunk); r.box = box.label; r.gpu = box.gpu || ''; rows.push(r); + } + if (!out.includes('RESULT')) rows.push({ box: box.label, gpu: box.gpu || '', pass: false, total_s: 0, error: out.trim().split('\n').slice(-3).join(' | ') }); +} +const med = (k) => { const v = rows.map(r => Number(r[k])).filter(x => x > 0).sort((a, b) => a - b); return v.length ? v[Math.floor(v.length / 2)] : null; }; +const out = { version, url, date: new Date().toISOString(), installs, boxes: boxes.map(b => b.label), rows, median_total_s: med('total_s'), pass_count: rows.filter(r => r.pass).length, sha256: rows.find(r => r.download_sha256)?.download_sha256 || '', gate: gateLine(rows, installs), took_s: Math.round((Date.now() - started) / 1000) }; +writeFileSync(evidence, JSON.stringify(out, null, 2) + '\n'); +console.log(out.gate); +process.exit(/^LG-4 PASS/.test(out.gate) ? 0 : 1); diff --git a/tools/reliability/app-run.mjs b/tools/reliability/app-run.mjs index 2b9ef375f..48dab7bc3 100755 --- a/tools/reliability/app-run.mjs +++ b/tools/reliability/app-run.mjs @@ -1,24 +1,34 @@ #!/usr/bin/env node -// The app engine's watchdog measured end to end on a Mac: a scratch Igneum Miner engine with its own private node -// (ports 29960 and 29961, devnet suffix 9960, no peers, unsynced mining allowed, so the node counts as synced) and -// fake-worker.mjs standing in for the Metal worker, driven through the control file and with signals. +// The app engine's fault injector (docs/plans/miner-faults.md): a scratch Igneum Miner engine with its own private node +// (ports 29960 and 29961, devnet suffix 9960, no peers, unsynced mining allowed) and fake-worker.mjs standing in for +// the GPU worker (the Metal worker on macOS, the OpenCL worker on Linux and Windows), driven through the control file +// and with signals. One step per fault class, each stating what must happen and what must not. Never touches the live +// devnet. Runs on the box (Linux engine and node from target-remote/) or on a Mac. // -// node tools/reliability/app-run.mjs --app --miner [--node ] +// node tools/reliability/app-run.mjs --app --miner --node [--only a,b] // -// Steps, in one engine run (each states what must happen and what must not): +// Steps, in one engine run: +// catch-up MF-1, MF-2: the node is synced (private) but its execution layer holds no record: no worker starts +// for 120 s, the card says it waits for the executed tip, the node is NOT restarted by the watchdog; +// a CPU block producer then makes blocks, the record appears, the worker starts on its own // own-restart the worker reports jobs done in 0.3 ms: the miner's guard restarts the worker; the app shows the // fault and does NOT restart the miner (same pid, no app restart) -// zero-once the worker completes jobs with 0 hashes: hash rate 0 for 60 s while synced, the app restarts the -// miner once; the worker is healthy again: mining resumes -// zero-faulted the same again inside five minutes: the card is marked faulted with the reason, its miner is not -// restarted, the node keeps running; "resume" clears it +// zero-ladder MF-2: jobs complete with 0 hashes: the watchdog restarts the miner at 10 s, then 30 s, then 120 s +// (the ladder), the reason stays on the card in plain words, nothing says "restarted once already" +// and no card is ever "faulted"; the worker is healthy again: mining resumes with no tap // no-status the miner process is stopped with SIGSTOP: no status line for 90 s, the app restarts it -// node-silent the node is stopped with SIGSTOP: no reading for 120 s, the app restarts the node in-process and -// the miner comes back once it is synced -// Never touches the live devnet. The app's own log is in the scratch directory. +// card-appears MF-3: the card is absent from the enumeration at start (unplugged, or a problem code) and appears +// two minutes later: the hot-plug pass starts its worker with no tap (Linux and Windows only) +// node-silent the node is stopped with SIGSTOP: no sign of life for 120 s, the app restarts the node in-process +// and the miner comes back once it is ready +// orphan-miner MF-7: a stray igneum-miner on this engine's node, not started by it, is killed by the minute sweep; +// the engine's own miner is left alone +// one-card-fails MF-4: two more cards appear, one failing its self-test for ever: the healthy two mine, the failing +// one is held 30 minutes with the reason on its row, the pack is not exported per failure (Linux) +// The app's own log is in the scratch directory. Every FAULT line the engine wrote is counted at the end. -import { spawn } from 'node:child_process'; -import { mkdirSync, rmSync, writeFileSync, existsSync, symlinkSync, readFileSync, appendFileSync, chmodSync } from 'node:fs'; +import { spawn, spawnSync } from 'node:child_process'; +import { mkdirSync, rmSync, writeFileSync, existsSync, symlinkSync, readFileSync, appendFileSync, chmodSync, readdirSync } from 'node:fs'; import { fileURLToPath } from 'node:url'; import { dirname, join } from 'node:path'; @@ -27,10 +37,11 @@ const args = process.argv.slice(2); const opt = (n, d) => { const i = args.indexOf(n); return i >= 0 ? args[i + 1] : d; }; const APP = opt('--app'); const MINER = opt('--miner'); -const NODE = opt('--node', join(here, '../../vendor/igneum-node/target-integration/release/igneumd')); +const NODE = opt('--node'); const ONLY = opt('--only', '').split(',').filter(Boolean); const SCRATCH = process.env.SCRATCH || `/tmp/igneum-reliability-app-${process.pid}`; const RPC = 29960, P2P = 29961, SUFFIX = 9960; +const MAC = process.platform === 'darwin'; for (const [k, v] of Object.entries({ APP, MINER, NODE })) if (!v || !existsSync(v)) { console.error(`missing ${k} (${v})`); process.exit(2); } const t0 = Date.now(); @@ -44,46 +55,51 @@ const bin = join(SCRATCH, 'stage', 'bin'); mkdirSync(bin, { recursive: true }); symlinkSync(NODE, join(bin, 'igneumd')); symlinkSync(MINER, join(bin, 'igneum-miner')); chmodSync(join(here, 'fake-worker.mjs'), 0o755); -symlinkSync(join(here, 'fake-worker.mjs'), join(bin, 'igneum-bench')); +// the fake worker under the name the engine's detection looks for on this platform +symlinkSync(join(here, 'fake-worker.mjs'), join(bin, MAC ? 'igneum-bench' : 'igneum-worker-opencl')); const data = join(SCRATCH, 'data'); mkdirSync(join(data, 'app'), { recursive: true }); const logs = join(SCRATCH, 'logs'); mkdirSync(logs, { recursive: true }); const CTL = join(SCRATCH, 'ctl'); const setMode = (m) => { writeFileSync(CTL, m + '\n'); log(`fake worker mode -> ${m}`); }; +const cardKey = MAC ? 'apple::Fake GPU' : 'other:0:Fake GPU'; setMode('ok'); writeFileSync(join(data, 'app', 'settings.json'), JSON.stringify({ setup_done: true, address: '0x4242424242424242424242424242424242424242', address_source: 'pasted', key_saved: true, identities: 1, - cards: { 'apple::Fake GPU': { enabled: true, identities: 1, power_pct: 0 } }, vote: false, paused: false, accepted_total: 0, - auto_update: false, remote_jobs: false, prove: false, + cards: { [cardKey]: { enabled: true, identities: 1, power_pct: 0 } }, vote: false, paused: false, accepted_total: 0, + auto_update: false, remote_jobs: false, prove: false, dev_fee: false, }, null, 2)); const env = { ...process.env, IGNEUM_APP_DATA: data, IGNEUM_APP_LOGS: logs, IGNEUM_APP_BIN: bin, IGNEUM_APP_RPC_PORT: String(RPC), IGNEUM_APP_P2P_PORT: String(P2P), IGNEUM_APP_PEERS: '', IGNEUM_APP_UNSYNCED: '1', IGNEUM_APP_DEVNET_SUFFIX: String(SUFFIX), IGNEUM_APP_STATUS_SECS: '10', FAKE_WORKER_CTL: CTL, + // easy genesis bits so the CPU block producer of the catch-up step makes blocks on two threads + IGNEUM_DEVNET_GENESIS_BITS: '0x1f100000', }; const app = spawn(APP, ['--no-open'], { stdio: ['pipe', 'pipe', 'pipe'], env }); const appOut = join(SCRATCH, 'app.out'); app.stdout.on('data', d => appendFileSync(appOut, d)); app.stderr.on('data', d => appendFileSync(appOut, d)); let appExit = null; app.on('exit', c => { appExit = c; log(`app exited ${c}`); }); -process.on('exit', () => { try { app.kill('SIGKILL'); } catch {} }); +const started = [app]; +process.on('exit', () => { for (const p of started) { try { p.kill('SIGKILL'); } catch {} } }); process.on('SIGINT', () => process.exit(130)); -log(`app pid ${app.pid}, scratch ${SCRATCH}`); +log(`app pid ${app.pid}, scratch ${SCRATCH}, platform ${process.platform}`); let url = null; -for (let i = 0; i < 100 && !url; i++) { try { url = readFileSync(join(data, 'app', 'app.url'), 'utf8').trim(); } catch { await sleep(200); } } +for (let i = 0; i < 150 && !url; i++) { try { url = readFileSync(join(data, 'app', 'app.url'), 'utf8').trim(); } catch { await sleep(200); } } if (!url) { log('no app.url'); process.exit(1); } async function state() { try { const r = await fetch(url + 'api/state'); return await r.json(); } catch { return null; } } const card = (st) => (st && st.mining && st.mining.cards && st.mining.cards[0]) || {}; -const events = []; // (t, card state, message, hash, pid, restarts, faults, node state) transitions, for the record +const events = []; let last = ''; async function poll() { const st = await state(); if (!st) return null; const c = card(st); - const key = [c.state, c.message, c.pid, c.restarts, c.faults, st.node.state, c.hash_now > 0 ? '>0' : '0'].join('|'); - if (key !== last) { last = key; events.push({ t: Date.now(), key }); log(`card ${c.state} pid ${c.pid} restarts ${c.restarts} faults ${c.faults} hash ${Number(c.hash_now).toFixed(1)} node ${st.node.state} ${c.message ? '"' + c.message + '"' : ''}`); } + const key = [c.state, c.message, c.pid, c.restarts, c.faults, st.node.state, st.node.starts, c.hash_now > 0 ? '>0' : '0'].join('|'); + if (key !== last) { last = key; events.push({ t: Date.now(), key }); log(`card ${c.state || '(none)'} pid ${c.pid} restarts ${c.restarts} faults ${c.faults} hash ${Number(c.hash_now || 0).toFixed(1)} node ${st.node.state} starts ${st.node.starts} ${c.message ? '"' + c.message + '"' : ''}`); } return st; } async function until(pred, timeoutMs, what) { @@ -102,10 +118,50 @@ function verdict(name, checks, metrics) { for (const [k, v] of Object.entries(metrics)) log(` ${k}: ${v}`); } async function steady(ms) { const end = Date.now() + ms; while (Date.now() < end) { await poll(); await sleep(1000); } } +const engineLogLines = (re) => { const out = []; for (const f of readdirSync(logs).filter(f => f.startsWith('app-'))) { for (const l of readFileSync(join(logs, f), 'utf8').split('\n')) if (re.test(l)) out.push(l); } return out; }; const S = {}; +S['catch-up'] = async () => { + // the fake card is present for this step + setMode('ok'); + const tStart = Date.now(); + // the private node reads synced within seconds; its execution layer holds no record until a block executes + const synced = await until((st) => st.node.synced, 120000, 'node synced'); + const tSynced = Date.now(); + // 120 s: no worker starts, the card waits on the executed tip, the node is not restarted + let started = false, nodeRestarts = 0, waitedForTip = false; + const end = Date.now() + 120000; + while (Date.now() < end) { + const st = await poll(); if (!st) break; + const c = card(st); + if (c.pid > 0 || c.state === 'mining' || c.state === 'starting') started = true; + if (/execute the tip/.test(c.message || '')) waitedForTip = true; + nodeRestarts = Math.max(nodeRestarts, (st.node.starts || 1) - 1); + await sleep(1000); + } + // now blocks: a CPU block producer against the app's node makes the first executed record + // it stays up for the rest of the run: every later step needs an executed tip to exist + const cpu = spawn(MINER, ['mine', `grpc://127.0.0.1:${RPC}`, '2', '36000', 'cpu', '--engine', 'igneum-pow', '--no-vote', '--payout-label', 'cpu'], { stdio: ['ignore', 'ignore', 'ignore'] }); + started.push(cpu); + log(`cpu block producer pid ${cpu.pid}`); + const tBlocks = Date.now(); + const back = await until(mining, 300000, 'the worker to start once the execution layer holds a record'); + const tMining = Date.now(); + const readyLines = engineLogLines(/node readiness: the execution layer reports an executed tip/); + const gatedCalls = engineLogLines(/holds no record yet/); + verdict('catch-up', [ + { ok: !!synced, what: 'the private node read synced' }, + { ok: !started, what: 'no worker started while the execution layer held no record (120 s)' }, + { ok: waitedForTip, what: 'the card said it waits for the executed tip' }, + { ok: nodeRestarts === 0, what: `the node watchdog did not restart the node during the catch-up (restarts ${nodeRestarts})` }, + { ok: readyLines.length >= 1, what: 'the engine logged the executed tip when it appeared' }, + { ok: !!back, what: 'the worker started on its own once the record existed' }, + ], { synced_after: s(tSynced - tStart), blocks_to_mining: s(tMining - tBlocks), gated_exec_calls_refused: gatedCalls.length }); +}; + S['own-restart'] = async () => { - const st0 = await until(mining, 180000, 'first mining'); + setMode('ok'); + const st0 = await until(mining, 300000, 'first mining'); const pid0 = card(st0).pid, r0 = card(st0).restarts; const tInject = Date.now(); setMode('fast'); @@ -114,7 +170,7 @@ S['own-restart'] = async () => { setMode('ok'); const back = await until((st, c) => mining(st, c) && c.faults >= 1, 90000, 'mining again after the worker restart'); const tBack = Date.now(); - await steady(20000); + await steady(15000); const st1 = await poll(); verdict('own-restart', [ { ok: !!st0, what: 'the card mined with a rate above 0 on the fake worker' }, @@ -123,53 +179,50 @@ S['own-restart'] = async () => { { ok: st1 && card(st1).pid === pid0 && card(st1).restarts === r0, what: `the app did not restart the miner (pid ${pid0} -> ${card(st1 || {}).pid}, app restarts ${r0} -> ${card(st1 || {}).restarts})` }, ], { inject_to_fault_on_card: s(tFault - tInject), fault_to_mining_again: s(tBack - tFault) }); }; -S['zero-once'] = async () => { - const st0 = await until(mining, 60000, 'mining'); - const pid0 = card(st0).pid, r0 = card(st0).restarts; - const tInject = Date.now(); - setMode('zero'); - const rs = await until((st, c) => c.restarts > r0 || /watchdog/.test(c.message || ''), 150000, 'the watchdog restart'); - const tRestart = Date.now(); + +S['zero-ladder'] = async () => { setMode('ok'); - const back = await until((st, c) => mining(st, c) && c.pid !== pid0, 120000, 'mining on the restarted miner'); - const tBack = Date.now(); - await steady(15000); - const st1 = await poll(); - verdict('zero-once', [ - { ok: !!rs && /hash rate 0/.test(card(rs).message || ''), what: `the watchdog restarted the miner for a zero rate (${JSON.stringify(card(rs || {}).message)})` }, - { ok: rs && tRestart - tInject >= 55000 && tRestart - tInject <= 100000, what: `between 55 and 100 s after the rate went to 0 (${s(tRestart - tInject)})` }, - { ok: !!back, what: 'mining resumed on a new miner process' }, - { ok: st1 && card(st1).restarts === r0 + 1, what: `exactly one app restart (${r0} -> ${card(st1 || {}).restarts})` }, - ], { zero_to_restart: s(tRestart - tInject), restart_to_mining: s(tBack - tRestart), zero_to_mining: s(tBack - tInject) }); -}; -S['zero-faulted'] = async () => { - const st0 = await until(mining, 60000, 'mining'); + const st0 = await until(mining, 120000, 'mining'); const r0 = card(st0).restarts; const tInject = Date.now(); setMode('zero'); - const f = await until((st, c) => c.state === 'faulted', 150000, 'the card to be marked faulted'); - const tFault = Date.now(); - await steady(45000); - const st1 = await poll(); + // three rungs: the gaps between consecutive watchdog restarts must grow 10, 30, 120 s (plus the 60 s rule each time) + const marks = []; + let lastR = r0; + const end = Date.now() + 600000; + let faultedSeen = false, onceAlready = false; + while (Date.now() < end && marks.length < 3) { + const st = await poll(); if (!st) break; + const c = card(st); + if (c.state === 'faulted') faultedSeen = true; + if (/restarted once already/.test(c.message || '')) onceAlready = true; + if ((c.restarts || 0) > lastR) { lastR = c.restarts; marks.push({ t: Date.now(), wait: c.restart_in_s, msg: c.message }); log(`rung ${marks.length}: restart_in_s ${c.restart_in_s} "${c.message}"`); } + await sleep(500); + } setMode('ok'); - app.stdin.write('resume\n'); - const back = await until(mining, 90000, 'mining after resume'); - verdict('zero-faulted', [ - { ok: !!f && /restarted once already/.test(card(f).message || ''), what: `the card was marked faulted with the reason (${JSON.stringify(card(f || {}).message)})` }, - { ok: st1 && card(st1).state === 'faulted' && card(st1).pid === 0 && card(st1).restarts === r0, what: `45 s later: still faulted, no miner process, no further restart (restarts ${r0} -> ${card(st1 || {}).restarts})` }, - { ok: st1 && st1.node.synced && appExit === null, what: 'the node kept running and the app stayed up' }, - { ok: !!back, what: 'resume cleared the fault and mining resumed' }, - ], { zero_to_faulted: s(tFault - tInject) }); + const back = await until(mining, 400000, 'mining again on its own after the worker is healthy'); + const tBack = Date.now(); + const waits = marks.map(m => m.wait); + const ladderOk = waits.length === 3 && waits[0] >= 8 && waits[0] <= 10 && waits[1] >= 28 && waits[1] <= 30 && waits[2] >= 118 && waits[2] <= 120; + verdict('zero-ladder', [ + { ok: marks.length === 3, what: `three watchdog restarts observed (${marks.length})` }, + { ok: ladderOk, what: `the restart delays follow the ladder 10, 30, 120 s (saw ${waits.join(', ')})` }, + { ok: marks.every(m => /hash rate 0 for 60 s/.test(m.msg || '')), what: 'the reason stayed on the card in plain words' }, + { ok: !faultedSeen && !onceAlready, what: 'no "faulted" state and no "restarted once already" words' }, + { ok: !!back, what: 'mining resumed on its own once the worker was healthy' }, + ], { zero_to_rung1: marks[0] ? s(marks[0].t - tInject) : 'n/a', rung_gaps: marks.slice(1).map((m, i) => s(m.t - marks[i].t)).join(', '), healthy_to_mining: s(tBack - (marks[2] ? marks[2].t : tInject)) }); }; + S['no-status'] = async () => { - const st0 = await until(mining, 60000, 'mining'); + setMode('ok'); + const st0 = await until(mining, 400000, 'mining'); const pid0 = card(st0).pid, r0 = card(st0).restarts; const tInject = Date.now(); process.kill(pid0, 'SIGSTOP'); log(`SIGSTOP miner pid ${pid0}`); const rs = await until((st, c) => c.restarts > r0 || /no status line/.test(c.message || ''), 150000, 'the watchdog restart'); const tRestart = Date.now(); - const back = await until((st, c) => mining(st, c) && c.pid !== pid0 && c.pid > 0, 120000, 'mining on the restarted miner'); + const back = await until((st, c) => mining(st, c) && c.pid !== pid0 && c.pid > 0, 400000, 'mining on the restarted miner'); const tBack = Date.now(); let gone = false; try { process.kill(pid0, 0); } catch { gone = true; } if (!gone) { try { process.kill(pid0, 'SIGKILL'); } catch {} } @@ -180,38 +233,111 @@ S['no-status'] = async () => { { ok: gone, what: 'the stopped miner process was killed' }, ], { quiet_to_restart: s(tRestart - tInject), restart_to_mining: s(tBack - tRestart) }); }; + +S['card-appears'] = async () => { + if (MAC) { verdict('card-appears', [{ ok: true, what: 'skipped on macOS (Apple silicon has no GPU hot-plug; the Metal path enumerates once)' }], {}); return; } + // the card leaves the enumeration (unplugged, or a problem code: the hot-plug pass marks it removed and stops its + // worker), then comes back: its worker must start with no tap + await until(mining, 300000, 'mining before the card leaves'); + setMode('absent'); + const tGone = Date.now(); + const gone = await until((st, c) => c.state === 'removed' || c.removed === true || /removed/.test(c.state || ''), 150000, 'the hot-plug pass to mark the card removed'); + const tMarked = Date.now(); + await steady(30000); + setMode('ok'); + const tAppear = Date.now(); + const back = await until(mining, 300000, 'the worker to start with no tap once the card is back'); + const tBack = Date.now(); + verdict('card-appears', [ + { ok: !!gone, what: `the hot-plug pass marked the card removed when it left (${s(tMarked - tGone)})` }, + { ok: !!back, what: 'its worker started with no tap once it was back' }, + ], { gone_to_marked: s(tMarked - tGone), back_to_mining: s(tBack - tAppear) }); +}; + S['node-silent'] = async () => { - const st0 = await until(mining, 60000, 'mining'); + setMode('ok'); + const st0 = await until(mining, 400000, 'mining'); const npid = st0.node.pid, starts0 = st0.node.starts; const tInject = Date.now(); process.kill(npid, 'SIGSTOP'); log(`SIGSTOP node pid ${npid}`); - const rs = await until((st) => st.node.state === 'restarting' || st.node.starts > starts0, 200000, 'the node restart'); + const rs = await until((st) => st.node.state === 'restarting' || st.node.starts > starts0, 240000, 'the node restart'); const tRestart = Date.now(); - const synced = await until((st) => st.node.starts > starts0 && st.node.synced, 120000, 'the restarted node to sync'); + const synced = await until((st) => st.node.starts > starts0 && st.node.synced, 180000, 'the restarted node to sync'); const tSynced = Date.now(); - const back = await until(mining, 120000, 'mining again'); + const back = await until(mining, 400000, 'mining again'); const tBack = Date.now(); let gone = false; try { process.kill(npid, 0); } catch { gone = true; } if (!gone) { try { process.kill(npid, 'SIGKILL'); } catch {} } verdict('node-silent', [ { ok: !!rs, what: 'the app restarted the node' }, - { ok: rs && tRestart - tInject >= 115000 && tRestart - tInject <= 170000, what: `between 115 and 170 s after the node went quiet (${s(tRestart - tInject)})` }, + { ok: rs && tRestart - tInject >= 115000 && tRestart - tInject <= 190000, what: `between 115 and 190 s after the node went quiet (${s(tRestart - tInject)})` }, { ok: !!synced, what: `the new node synced (starts ${starts0} -> ${synced ? synced.node.starts : '?'})` }, - { ok: !!back, what: 'mining resumed' }, + { ok: !!back, what: 'mining resumed once the node was ready' }, { ok: gone, what: 'the stopped node process was killed' }, ], { quiet_to_restart: s(tRestart - tInject), restart_to_synced: s(tSynced - tRestart), quiet_to_mining: s(tBack - tInject) }); }; -const names = ONLY.length ? ONLY : Object.keys(S); +S['one-card-fails'] = async () => { + if (MAC) { verdict('one-card-fails', [{ ok: true, what: 'skipped on macOS (one Metal device)' }], {}); return; } + // MF-4: two more cards appear, one of them failing its self-test for ever; the healthy two must mine on time, + // the failing one is held with the reason on its row, and the pack is not exported once per failure + const exportsBefore = engineLogLines(/export-pack: /).length; + writeFileSync(CTL, 'ok\ndevices 3\ndev 2 selftest\n'); log('fake worker: 3 devices, device 2 fails its self-test'); + const tInject = Date.now(); + const listed = await until((st) => st.mining.cards.length >= 3, 200000, 'three cards listed'); + const tListed = Date.now(); + const twoMine = await until((st) => st.mining.cards.filter(c => c.state === 'mining' && c.hash_now > 0).length >= 2, 300000, 'two healthy cards mining'); + const tTwo = Date.now(); + const held = await until((st) => st.mining.cards.some(c => /not usable on this driver/.test(c.message || '')), 120000, 'the failing card held with the reason'); + await steady(90000); + const st1 = await poll(); + const bad = st1 ? st1.mining.cards.find(c => /not usable on this driver/.test(c.message || '')) : null; + const good = st1 ? st1.mining.cards.filter(c => !/not usable/.test(c.message || '')) : []; + const exportsAfter = engineLogLines(/export-pack: /).length - exportsBefore; + const reused = engineLogLines(/export-pack: reusing/).length; + verdict('one-card-fails', [ + { ok: !!listed, what: 'the hot-plug pass listed the two new cards' }, + { ok: !!twoMine, what: 'the two healthy cards mined' }, + { ok: !!held && bad && bad.restart_in_s >= 1500, what: `the failing card is held 30 minutes with the reason on its row (restart_in_s ${bad ? bad.restart_in_s : '?'}, "${bad ? bad.message : ''}")` }, + { ok: good.length >= 2 && good.every(c => c.state === 'mining' && c.hash_now > 0), what: 'the healthy cards still mine 90 s later (no restart from the failing card)' }, + { ok: exportsAfter <= 6, what: `the pack was exported at most 6 times for three starts and the failures (${exportsAfter}, ${reused} reused)` }, + ], { listed_after: s(tListed - tInject), two_mining_after: s(tTwo - tInject), exports: exportsAfter, exports_reused: reused }); +}; + +S['orphan-miner'] = async () => { + // MF-7: an igneum-miner the engine did not start, on this engine's node (the fence), is killed by the minute sweep + const st0 = await until(mining, 300000, 'mining'); + const stray = spawn(MINER, ['mine', `grpc://127.0.0.1:${RPC}`, '1', '3600', 'stray', '--worker', join(here, 'fake-worker.mjs'), '--status-secs', '10', '--no-vote', '--payout-label', 'stray'], { stdio: ['ignore', 'ignore', 'ignore'], env: { ...process.env, FAKE_WORKER_CTL: CTL } }); + started.push(stray); + const tInject = Date.now(); + log(`stray miner pid ${stray.pid} on the engine's node`); + let killedAt = null; + const end = Date.now() + 150000; + while (Date.now() < end) { await poll(); try { process.kill(stray.pid, 0); } catch { killedAt = Date.now(); break; } await sleep(1000); } + const lines = engineLogLines(/orphan miner killed/); + const st1 = await poll(); + const own = st1 ? card(st1) : {}; + let ownAlive = false; try { process.kill(own.pid, 0); ownAlive = true; } catch {} + verdict('orphan-miner', [ + { ok: !!killedAt, what: `the stray miner was killed by the engine (${killedAt ? s(killedAt - tInject) : 'still alive after 150 s'})` }, + { ok: lines.length >= 1, what: `one log line per kill (${lines.length})` }, + { ok: ownAlive && own.pid > 0, what: `the engine's own miner (pid ${own.pid}) was left alone` }, + ], { inject_to_kill: killedAt ? s(killedAt - tInject) : 'n/a' }); +}; + +const order = ['catch-up', 'card-appears', 'own-restart', 'zero-ladder', 'no-status', 'node-silent', 'one-card-fails', 'orphan-miner']; +const names = ONLY.length ? order.filter(n => ONLY.includes(n)) : order; for (const n of names) { - if (!S[n]) { log(`unknown step ${n}`); continue; } log(`=== ${n}`); try { await S[n](); } catch (e) { log(`step ${n} threw: ${e.stack || e}`); results.push({ name: n, pass: false, checks: [{ ok: false, what: String(e) }], metrics: {} }); } } +const faults = engineLogLines(/ FAULT class=/); +log(`FAULT lines the engine wrote: ${faults.length}`); +for (const f of faults.slice(0, 12)) log(` ${f.slice(0, 200)}`); app.stdin.write('quit\n'); await sleep(8000); -const report = { date: new Date().toISOString(), app: APP, miner: MINER, node: NODE, scratch: SCRATCH, results, transitions: events.map(e => ({ t: new Date(e.t).toISOString(), key: e.key })) }; +const report = { date: new Date().toISOString(), app: APP, miner: MINER, node: NODE, platform: process.platform, scratch: SCRATCH, results, fault_lines: faults.length, transitions: events.map(e => ({ t: new Date(e.t).toISOString(), key: e.key })) }; writeFileSync(join(SCRATCH, 'report.json'), JSON.stringify(report, null, 2)); console.log(JSON.stringify({ ...report, transitions: undefined }, null, 2)); process.exit(results.every(r => r.pass) ? 0 : 1); diff --git a/tools/reliability/fake-worker.mjs b/tools/reliability/fake-worker.mjs index bd5fe0b04..c6527a851 100755 --- a/tools/reliability/fake-worker.mjs +++ b/tools/reliability/fake-worker.mjs @@ -13,25 +13,54 @@ // badfound a `found` with a wrong hash before every `done`: a worker on a wrong program // preparefail every `prepare` is answered `prepare-failed`; jobs run as in ok // exit the process exits with code 7 at the next job +// absent `--list` shows no device (the card is unplugged or in a problem code); jobs run as in ok +// selftest the worker prints a self-test failure and exits 3 before ready (MF-4) +// Extra lines in the control file: `devices N` (how many --list shows), `dev N ` (a mode for device N only) // The mode is read from argv `--mode ` when no control file is set. `--serve` is accepted and ignored. // Lines on stderr are prefixed `fake-worker:` and never match a miner pattern. import { readFileSync, existsSync, writeSync } from 'node:fs'; +import { createRequire } from 'node:module'; +const require = createRequire(import.meta.url); import { createInterface } from 'node:readline'; +// `--list` (the engine's OpenCL enumeration on Windows and Linux, src/detect.rs): one fake GPU, or none while the +// control file says `absent` (the hot-plug step: a card that appears later must start without a tap) +if (process.argv.includes('--list')) { + const m = (() => { try { return require('node:fs').readFileSync(process.env.FAKE_WORKER_CTL, 'utf8').trim().split(/\s+/)[0]; } catch { return 'ok'; } })(); + const n = m === 'absent' ? 0 : (() => { try { const l = require('node:fs').readFileSync(process.env.FAKE_WORKER_CTL, 'utf8').split('\n').map(x => x.trim().split(/\s+/)).find(f => f[0] === 'devices'); return l ? Number(l[1]) : 1; } catch { return 1; } })(); + for (let i = 0; i < n; i++) { + process.stdout.write(`[${i}] Fake GPU ${i + 1} | Fake Platform (OpenCL 1.2)\n GPU, vendor Fake Silicon, driver 1.0.0, OpenCL C 1.2, 8 compute units, 8192 MB\n`); + } + process.exit(0); +} const ctl = process.env.FAKE_WORKER_CTL; const argMode = (() => { const i = process.argv.indexOf('--mode'); return i >= 0 ? process.argv[i + 1] : 'ok'; })(); +// the device this process serves (`--device N` from the engine's OpenCL path); the control file's first word is the +// mode for every device, a line `dev N ` overrides it for device N, a line `devices N` sets how many --list shows +const device = (() => { const i = process.argv.indexOf('--device'); return i >= 0 ? process.argv[i + 1] : '0'; })(); function mode() { if (ctl && existsSync(ctl)) { - const m = readFileSync(ctl, 'utf8').trim().split(/\s+/)[0]; + const lines = readFileSync(ctl, 'utf8').trim().split('\n'); + const own = lines.map(l => l.trim().split(/\s+/)).find(f => f[0] === 'dev' && f[1] === device); + if (own && own[2]) return own[2]; + const m = (lines[0] || '').trim().split(/\s+/)[0]; if (m) return m; } return argMode; } +function deviceCount() { + try { const l = readFileSync(ctl, 'utf8').split('\n').map(x => x.trim().split(/\s+/)).find(f => f[0] === 'devices'); return l ? Number(l[1]) : 1; } catch { return 1; } +} // synchronous writes: a piped stdout is asynchronous in Node and process.exit would drop pending lines const out = (s) => { writeSync(1, s + '\n'); }; const sleep = (ms) => new Promise(r => setTimeout(r, ms)); +if (mode() === 'selftest') { + // MF-4: the worker fails its self-test and exits before it is ready (the Arc B580 on PC 1, 7 October 2026) + out('error 0 self-test FAIL: 3 of 96 vectors mismatched (fake worker, device ' + device + ')'); + process.exit(3); +} out('ready fake Fake_GPU dataset-log2 28 batch 16777216 prepare 1'); const queue = [];