From 089b078a3a607f1e2cead4a2f15318794e0ab395 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 11:27:55 +0000 Subject: [PATCH 01/18] Miner app: plug, tune, play (the project lead, 7 October 2026): node readiness gate, the retry ladder, no permanent fault, fault lines to the intake, the signed cards job, the fault-class register docs/plans/miner-faults.md: MF-1 to MF-7, each with its rule, test and gate line. - MF-1/MF-2: a worker starts and is judged only when the node is READY (synced and igneum_getExecStatus reports an executed tip; execrpc::probe every 5 s off the engine thread); the node watchdog never counts the catch-up (settled once read synced; 30 min cap before that; any RPC answer is a sign of life); the watchdog restarts on a ladder 10 s, 30 s, 2 min, 5 min, then every 5 min for ever (watchdog::RETRY_LADDER_S); the faulted state and the one-restart budget are gone (tools/ci/permanent-fault-check.sh in the gate); a node-caused restart resets the ladder at sync. - MF-3: the hot-plug pass starts a recovered or revived card's worker (unchanged rule, now in the register). - MF-4: the status clock starts at ready (program loaded), loading bounded by 300 s; a self-test failure holds the card 30 min with the reason on its row, released on a driver change; a crash loop climbs the ladder; the pack is exported once a minute for every card (a refused pack forces one). - MF-5: the app reads template_wait=, template_ms=, identities_active= from the 0.3.20 miner's STATUS; waiting on the node is never the card's fault; the row says node slow; every node-wait label clears on the first rate. - MF-6: a miners hold belongs to the job that took it and releases when that job is gone or at its own cap. - MF-7: the engine owns every igneum-miner it started: an untracked one on this engine's node RPC is killed at start, after every stop and every minute, one line and one fault report per kill; a restart kills the old process first. - Every fault line posts one FAULT line to the log intake (label fault-, app and node version, 60/h cap). - The signed cards job kind (per card enabled, identities, power_pct; refused for a card the machine lacks; applied through the app's own card path, persisted, read back): packaging/ota/publish-jobs.sh add --kind cards. - LG-4 as a job: relay/playbooks/first-share.ps1 and tools/fleet/first-share-gate.mjs (no Windows box yet). - tools/reliability: the fault injector with one step per class (catch-up, card-appears, own-restart, zero-ladder, no-status, node-silent, one-card-fails, orphan-miner); fake-worker.mjs lists devices and fails self-tests on command. - master's build tooling (97255a4e) and release-0.3.20's igneum-pow taken into the worktree for the box routes. Box: app 198 + 27 + 8 tests green on igneum-build-2; the tree gate green (33 checks). Co-Authored-By: Claude Fable 5.1 --- app/igneum-app/src/engine.rs | 308 ++++++++++++++++--- app/igneum-app/src/execrpc.rs | 158 ++++++++++ app/igneum-app/src/jobrun.rs | 113 +++++++ app/igneum-app/src/jobs.rs | 33 +- app/igneum-app/src/platform.rs | 43 +++ app/igneum-app/src/state.rs | 2 +- app/igneum-app/src/update.rs | 9 + app/igneum-app/src/watchdog.rs | 492 +++++++++++++++++++----------- app/igneum-app/ui/app.js | 4 +- app/igneum-app/ui/view.test.mjs | 3 +- docs/plans/miner-faults.md | 43 +++ packaging/ota/publish-jobs.sh | 23 +- relay/playbooks/first-share.ps1 | 87 ++++++ tools/ci/permanent-fault-check.sh | 28 ++ tools/ci/pre-push.sh | 1 + tools/fleet/first-share-gate.mjs | 92 ++++++ tools/reliability/app-run.mjs | 264 +++++++++++----- tools/reliability/fake-worker.mjs | 31 +- 18 files changed, 1435 insertions(+), 299 deletions(-) create mode 100644 app/igneum-app/src/execrpc.rs create mode 100644 docs/plans/miner-faults.md create mode 100644 relay/playbooks/first-share.ps1 create mode 100755 tools/ci/permanent-fault-check.sh create mode 100755 tools/fleet/first-share-gate.mjs diff --git a/app/igneum-app/src/engine.rs b/app/igneum-app/src/engine.rs index 2c19f5b64..1ef9fdd78 100644 --- a/app/igneum-app/src/engine.rs +++ b/app/igneum-app/src/engine.rs @@ -59,6 +59,8 @@ pub enum Cmd { WorkerBuilt(usize, Result), /// local minus the latest block's timestamp, seconds (from the node's EVM RPC) ClockSample(f64), + /// the node readiness probe answered (src/execrpc.rs probe): did the RPC answer, does the exec follower hold a record + ExecProbe { answered: bool, has_record: bool }, /// another node holds the app's ports: what it answered (src/extnode.rs) ExternalNode(crate::extnode::Check), /// the node's own description over RPC (igneum_getNodeInfo), every 30 s @@ -614,10 +616,12 @@ struct MinerSlot { prepared: bool, // the pack is exported (and the worker built) for the next start last_status: Option, error_at: Option, - /// the per-card watchdog (src/watchdog.rs): one restart, then faulted + /// the per-card watchdog (src/watchdog.rs): restarts on the ladder, never a permanent fault watch: crate::watchdog::CardWatch, /// pack refusals (src/watchdog.rs): the pack is exported again before the restart, capped per epoch pack_rebuilds: crate::watchdog::PackRebuilds, + /// the next pack export is forced (a worker refused the pack); otherwise an export under 60 s old is reused (MF-4) + pack_force: bool, } pub struct Engine { @@ -754,6 +758,23 @@ pub struct Engine { heat_room_seen: f64, /// the last half minute of (unix s, coolest card sensor) for the cooling-tail slope (heat::room) heat_card_samples: std::collections::VecDeque<(f64, f64)>, + /// node readiness (the project lead, 7 October 2026): a worker never starts before the node is synced AND its execution layer + /// reports an executed tip (igneum_getExecStatus); probed every 5 s off the engine thread + exec_ready: bool, + exec_probe_busy: bool, + exec_probe_next: Instant, + /// the last time the node's RPC answered anything (a sign of life for the node watchdog) + exec_answered_at: Option, + /// the node has been read as synced at least once since its start: before that its catch-up never counts + node_settled: bool, + /// fault reports sent to the log intake in the last hour (the cap) + fault_reports: Vec, + /// MF-6: the job that took the miners hold, when, and its cap in seconds + job_hold_owner: Option, + job_hold_since: Option, + job_hold_cap_s: f64, + /// MF-7: the orphan sweep (igneum-miner processes this engine does not track) runs at this time next + orphan_sweep_next: Instant, } impl Engine { @@ -865,6 +886,16 @@ impl Engine { sweep_retry: std::collections::HashMap::new(), sweep_attempts: std::collections::HashMap::new(), node_watch: crate::watchdog::NodeWatch::new(), + exec_ready: false, + exec_probe_busy: false, + exec_probe_next: now, + exec_answered_at: None, + node_settled: false, + fault_reports: Vec::new(), + job_hold_owner: None, + job_hold_since: None, + job_hold_cap_s: 3600.0, + orphan_sweep_next: now, t0: now, heat: crate::heat::Controller::new(), heat_rest: false, @@ -1016,13 +1047,12 @@ impl Engine { // live worker is re-armed (a faulted one reset first), its pack is exported again before the start // (prepared = false: the hour may have turned while paused), and a check 90 s later names any // enabled card that is not mining (`resume_check`). - let views: Vec = self.miners.iter().map(|m| ResumeSlot { faulted: m.watch.faulted().is_some(), live: m.proc.is_some() }).collect(); + let views: Vec = self.miners.iter().map(|m| ResumeSlot { faulted: m.watch.watchdog_restarts() > 0, live: m.proc.is_some() }).collect(); let now = Instant::now(); for i in slots_to_rearm_on_resume(&views) { let m = &mut self.miners[i]; - if m.watch.faulted().is_some() { - m.watch.event(0.0, crate::watchdog::Event::Reset); - } + // a user action: the ladder starts over + m.watch.event(0.0, crate::watchdog::Event::Reset); m.restart_at = Some(now); m.prepared = false; } @@ -1035,7 +1065,7 @@ impl Engine { self.shared.event("info", &format!("settings changed ({why}); the miners restart")); self.stop_miners("settings changed"); for m in self.miners.iter_mut() { - // a user action: a faulted card tries again + // a user action: the ladder starts over m.watch.event(0.0, crate::watchdog::Event::Reset); m.restart_at = Some(Instant::now()); } @@ -1130,6 +1160,19 @@ impl Engine { } if let Some(v) = info.version { if !v.is_empty() && st.node.version.is_empty() { st.node.version = v; } } } + Cmd::ExecProbe { answered, has_record } => { + self.exec_probe_busy = false; + if answered { + self.exec_answered_at = Some(Instant::now()); + } + if has_record && !self.exec_ready { + self.shared.log("node readiness: the execution layer reports an executed tip; the workers may start once the node is synced"); + } + if !has_record && self.exec_ready { + self.shared.log("node readiness: the execution layer reports no executed tip any more (a reset or a restart); the workers wait for it"); + } + self.exec_ready = has_record; + } Cmd::ClockSample(d) => { self.clock_samples.push(d); if self.clock_samples.len() > 9 { @@ -1692,6 +1735,18 @@ impl Engine { self.shared.event("warn", &format!("{name}: not usable ({problem}); its worker stopped. {}", crate::detect::PROBLEM_HINT)); } for (i, fresh) in diff.moved.iter() { + let driver_changed = self.st().mining.cards.get(*i).map(|c| !fresh.platform.is_empty() && c.platform != fresh.platform).unwrap_or(false); + if driver_changed { + // MF-4: a card held after a self-test failure tries again the minute its driver changes + let now_i = Instant::now(); + for m in self.miners.iter_mut().filter(|m| m.card == *i) { + if m.watch.driver_hold() { + m.watch.event(0.0, crate::watchdog::Event::DriverChanged); + m.restart_at = Some(now_i); + self.shared.event("info", &format!("{}: driver changed; its worker tries again now", fresh.name)); + } + } + } if let Some(c) = self.st().mining.cards.get_mut(*i) { self.shared.log(&format!("{}: device {} is now {} (the running worker keeps its device; the next start uses the new one)", c.name, c.device, fresh.device)); c.device = fresh.device.clone(); @@ -1959,6 +2014,9 @@ impl Engine { self.sync_prev = None; self.sync_stable_since = None; self.node_last_reading = None; + self.exec_ready = false; + self.exec_answered_at = None; + self.node_settled = false; self.shared.event(if self.node_starts == 1 { "ok" } else { "info" }, if self.node_starts == 1 { "node started" } else { "node restarted" }); } Err(e) => { @@ -2045,6 +2103,7 @@ impl Engine { error_at: None, watch: crate::watchdog::CardWatch::new(), pack_rebuilds: crate::watchdog::PackRebuilds::new(), + pack_force: false, }); } if self.miners.is_empty() { @@ -2125,6 +2184,12 @@ impl Engine { fn start_miner(&mut self, i: usize) { let card_idx = self.miners[i].card; let Some(card) = self.st().mining.cards.get(card_idx).cloned() else { return }; + // MF-7: a restart kills the old process before the new one starts; never two miners on one card + if let Some(mut p) = self.miners[i].proc.take() { + self.shared.log(&format!("miner {}: the previous process (pid {}) is still alive at restart; stopping it first", self.miners[i].label, p.pid())); + p.write_stdin("quit\n"); + p.stop(5); + } if self.shared.settings.lock().unwrap().address.is_empty() { self.shared.event("error", "no payout address; open settings and set one"); self.miners[i].restart_at = Some(Instant::now() + Duration::from_secs(60)); @@ -2187,10 +2252,14 @@ impl Engine { m.restart_at = None; m.watch.event(t, crate::watchdog::Event::Stopped); } + if any { + // MF-7: nothing of this engine's keeps hashing after a stop + self.sweep_orphan_miners("after a stop"); + } let mut st = self.st(); let paused = st.mining.paused; for c in st.mining.cards.iter_mut() { - if c.enabled && c.present() && c.state != "faulted" { + if c.enabled && c.present() { c.state = if paused { "off".into() } else { "waiting".into() }; c.hash_now = 0.0; c.pid = 0; @@ -2215,8 +2284,9 @@ impl Engine { let bins = self.bins.clone(); let vendor = self.st().mining.cards.get(card_idx).map(|c| c.vendor.clone()).unwrap_or_default(); let worker = self.miners[i].worker.clone(); + let force = std::mem::take(&mut self.miners[i].pack_force); std::thread::spawn(move || { - let r = if build { build_worker_from_source(&shared, &bins, &vendor) } else { export_pack(&shared, &bins).map(|_| worker) }; + let r = if build { build_worker_from_source(&shared, &bins, &vendor) } else { export_pack(&shared, &bins, force).map(|_| worker) }; shared.send(Cmd::WorkerBuilt(card_idx, r)); }); } @@ -3499,6 +3569,11 @@ impl Engine { if self.running { self.tick_node(now); self.tick_watch(now); + self.tick_exec_probe(now); + if now >= self.orphan_sweep_next { + self.orphan_sweep_next = now + Duration::from_secs(60); + self.sweep_orphan_miners("the minute sweep"); + } self.tick_miners(now); if let Some(at) = self.resume_check_at { if now >= at { @@ -3633,8 +3708,20 @@ impl Engine { if let Some(a) = self.jobs.tick(&self.shared) { self.job_action(a); } - if self.job_hold && !self.jobs.holds_miners() { + let release = if self.job_hold { + // MF-6: the hold belongs to the job that took it, releases when that job is gone (whatever runs next) + // or at its own cap; the runner's own flag is read too + let held_s = self.job_hold_since.map(|t| t.elapsed().as_secs_f64()).unwrap_or(0.0); + let active = self.jobs.active_id(); + crate::jobrun::hold_release(self.job_hold_owner.as_deref(), active.as_deref(), held_s, self.job_hold_cap_s).or(if !self.jobs.holds_miners() { Some("the runner released the miners") } else { None }) + } else { + None + }; + if let Some(why) = release { self.job_hold = false; + self.job_hold_owner = None; + self.job_hold_since = None; + self.shared.log(&format!("job hold released: {why}")); for c in self.st().mining.cards.iter_mut().filter(|c| c.state == "held" || c.state == "tuning") { c.state = "off".into(); c.sweep_state = if c.sweep_state == "running" { "idle".into() } else { c.sweep_state.clone() }; @@ -3785,6 +3872,9 @@ impl Engine { match a { Action::StopMiners(why) => { self.job_hold = true; + self.job_hold_owner = self.jobs.active_id(); + self.job_hold_since = Some(Instant::now()); + self.job_hold_cap_s = (self.jobs.active_cap_minutes().max(1) * 60) as f64; self.stop_miners(&why); for c in self.st().mining.cards.iter_mut().filter(|c| c.enabled) { // never a bare "off" at 0 MH/s: the row says why (a job holds the card) @@ -3793,6 +3883,17 @@ impl Engine { } self.jobs.miners_stopped(&self.shared); } + Action::ApplyCards(choices) => { + // the signed `cards` kind (7 October 2026): through the app's own card path, persisted, read back + let names: Vec = choices.iter().map(|c| format!("{} enabled={} identities={}", c.key, c.enabled, c.identities)).collect(); + self.shared.event("info", &format!("remote job: card settings: {}", names.join("; "))); + let keys: Vec = choices.iter().map(|c| c.key.clone()).collect(); + self.apply_cards(choices); + let readback: Vec<(String, bool, u32, u32)> = self.st().mining.cards.iter().filter(|c| keys.contains(&c.key)).map(|c| (c.key.clone(), c.enabled, c.identities, c.power_pct)).collect(); + if let Some(next) = self.jobs.cards_applied(&self.shared, readback) { + self.job_action(next); + } + } Action::CardsOff(keys) => { // `--cards-off` (6 October 2026): the runner switches the job's cards through the app's own card path // and keeps the exact choices that put them back; the script never touches /api/cards @@ -3936,6 +4037,7 @@ impl Engine { self.node_restarts += 1; let delay = jitter_secs(crate::platform::unix_now()); self.node_restart_at = Some(now + Duration::from_secs(delay)); + self.fault_report(None, "node-exit", &format!("igneumd exited with code {code} after {ran} s: {}", tail.last().cloned().unwrap_or_default())); let mut st = self.st(); st.node.state = "restarting".into(); st.node.synced = false; @@ -3986,13 +4088,19 @@ impl Engine { st.node.message = "no answer from the RPC yet (still opening its database?)".into(); } } - // the node watchdog: our node silent for 120 s (no reading, nothing accepted) is restarted in-process + // the node watchdog: our node with no sign of life for 120 s is restarted in-process. A sign of life is the + // watch reading, the exec probe's answer or an accepted block; and the node's catch-up never counts: until it + // has been read as synced once since its start only 30 minutes of total silence restarts it (PC 1, 7 October + // 2026: a 40-second restart loop while the node replayed) if self.node.is_some() && !self.node_external && self.node_restart_at.is_none() { - let silent_s = self.node_last_reading.map(|t| now.duration_since(t)).unwrap_or_else(|| self.node_started_at.elapsed()).as_secs_f64(); - if let Some(delay) = self.node_watch.tick(self.secs(now), true, silent_s, accepted_recent) { + let last_life = [self.node_last_reading, self.exec_answered_at, self.last_accepted].into_iter().flatten().max(); + let silent_s = last_life.map(|t| now.duration_since(t)).unwrap_or_else(|| self.node_started_at.elapsed()).as_secs_f64(); + if let Some(delay) = self.node_watch.tick(self.secs(now), true, silent_s, accepted_recent, self.node_settled) { let n = self.node_watch.restarts_in_window(); - self.shared.event("error", &format!("no reading from the node for {} s; the watchdog restarts it in {delay} s (restart {n} in the last 10 minutes)", silent_s as u64)); - self.restart_node("no reading from the node for 120 s", Duration::from_secs(delay)); + let why = if self.node_settled { "no sign of life from the node for 120 s" } else { "no answer from the node for 30 minutes after its start" }; + self.shared.event("error", &format!("{why}; the watchdog restarts it in {delay} s (restart {n} in the last 10 minutes)")); + self.fault_report(None, "node-silent", &format!("{why} (silent {} s); restart in {delay} s", silent_s as u64)); + self.restart_node(why, Duration::from_secs(delay)); } } } @@ -4000,6 +4108,10 @@ impl Engine { fn tick_miners(&mut self, now: Instant) { // a clock over the consensus bound: the node cannot sync and a miner would only submit rejected blocks let synced = self.st().node.synced && self.st().clock.severity != "block"; + // a worker starts, and is judged, only while the node is READY: synced and with an executed tip (the exec + // probe); PC 1, 7 October 2026: workers started into a node that was still catching up and were faulted for + // the silence that followed + let node_ready = synced && self.exec_ready; let paused = self.st().mining.paused; for i in 0..self.miners.len() { let card_idx = self.miners[i].card; @@ -4029,6 +4141,7 @@ impl Engine { self.shared.event("build", "program pack out of date, rebuilding"); self.shared.log(&format!("{name}: program pack out of date, rebuilding (export {n} of {cap} for epoch {epoch}): {why}")); self.miners[i].prepared = false; // prepare_worker exports the pack again before the start + self.miners[i].pack_force = true; self.miners[i].restart_at = Some(now); if let Some(c) = self.st().mining.cards.get_mut(card_idx) { c.state = "restarting".into(); @@ -4037,19 +4150,21 @@ impl Engine { } } crate::watchdog::PackAction::GiveUp { n: _, reason } => { - self.shared.event("error", &format!("{name}: {reason}")); - self.shared.log(&format!("{name}: {reason}; next try in 10 minutes or at the next hour")); + // never permanent: the next try in 5 minutes, and at every hour boundary + self.shared.event("error", &format!("{name}: {reason}; next try in 5 minutes")); + self.shared.log(&format!("{name}: {reason}; next try in 5 minutes or at the next hour")); + self.fault_report(Some(&name), "pack", &reason); self.miners[i].prepared = false; - self.miners[i].restart_at = Some(now + Duration::from_secs(600)); + self.miners[i].restart_at = Some(now + Duration::from_secs(300)); if let Some(c) = self.st().mining.cards.get_mut(card_idx) { - c.state = "failed".into(); + c.state = "restarting".into(); c.hash_now = 0.0; - c.message = reason; + c.message = format!("{reason}; next try in 5 minutes"); } } } } else if verdict != crate::watchdog::Action::None { - // exit 43: the miner gave up on its worker; once more, then the card is faulted + // exit 43: the miner gave up on its worker; a restart on the ladder self.watchdog_verdict(i, verdict, &tail); } else if is_stall_exit(code, &tail) { // N4: the miner found no new template for its stall span and stopped hashing and voting. Once: @@ -4083,6 +4198,8 @@ impl Engine { for l in &tail { self.shared.log(&format!(" miner: {l}")); } + let name = self.st().mining.cards.get(card_idx).map(|c| c.name.clone()).unwrap_or_else(|| self.miners[i].label.clone()); + self.fault_report(Some(&name), "miner-exit", &format!("the miner exited with code {code}: {}", tail.last().cloned().unwrap_or_default())); if let Some(c) = self.st().mining.cards.get_mut(card_idx) { c.state = "restarting".into(); c.restarts = self.miners[i].restarts; @@ -4091,9 +4208,10 @@ impl Engine { } } } else { - // the watchdog: no status for 90 s, or a zero rate for 60 s while synced, is one restart, then faulted + // the watchdog: no status for 90 s, or a zero rate for 60 s while the node is ready, is a restart + // on the ladder (10 s, 30 s, 2 min, 5 min, then every 5 min; never permanent) let t = self.secs(now); - let verdict = self.miners[i].watch.tick(t, synced); + let verdict = self.miners[i].watch.tick(t, node_ready); if verdict != crate::watchdog::Action::None { if let Some(mut p) = self.miners[i].proc.take() { p.write_stdin("quit\n"); @@ -4114,10 +4232,6 @@ impl Engine { } continue; } - if self.miners[i].watch.faulted().is_some() { - // faulted by the watchdog: not restarted until the user changes the card's settings or resumes - continue; - } if self.job_hold { // a remote job has the GPU; the miners wait until it lets go (src/jobrun.rs) continue; @@ -4126,11 +4240,11 @@ impl Engine { // heat mode rests the cards until the room wants heat again (src/heat.rs, tick_heat) continue; } - if paused || !synced { + if paused || !node_ready { if let Some(c) = self.st().mining.cards.get_mut(card_idx) { if !paused && c.state != "failed" { c.state = "waiting".into(); - c.message = "waiting for the node to sync".into(); + c.message = if synced { "waiting for the node to execute the tip (it is catching up)".into() } else { "waiting for the node to sync".into() }; } } continue; @@ -4182,27 +4296,18 @@ impl Engine { self.shared.log(&format!(" miner: {l}")); } match verdict { - Action::Restart(reason) => { + Action::Restart { reason, delay_s, attempt } => { self.miners[i].restarts += 1; - self.miners[i].restart_at = Some(Instant::now() + Duration::from_secs(1)); - self.shared.event("error", &format!("{name}: {reason}; the watchdog restarts the miner (restart {} of 1 before the card is marked faulted)", self.miners[i].watch.watchdog_restarts())); + self.miners[i].restart_at = Some(Instant::now() + Duration::from_secs(delay_s)); + let again = if attempt > 1 { format!(" (restart {attempt} since the card was last healthy; it keeps trying)") } else { String::new() }; + self.shared.event("error", &format!("{name}: {reason}; the worker restarts in {delay_s} s{again}")); + self.fault_report(Some(&name), "watchdog", &format!("{reason}; restart {attempt} in {delay_s} s")); if let Some(c) = self.st().mining.cards.get_mut(card_idx) { c.state = "restarting".into(); c.restarts = self.miners[i].restarts; c.hash_now = 0.0; c.pid = 0; - c.message = format!("watchdog: {reason}"); - } - } - Action::Fault(reason) => { - self.miners[i].restart_at = None; - self.shared.event("error", &format!("{name}: {reason}; the card is marked faulted and its miner is not restarted again (the other cards keep mining; change the card's settings or resume mining to try again)")); - if let Some(c) = self.st().mining.cards.get_mut(card_idx) { - c.state = "faulted".into(); - c.hash_now = 0.0; - c.pid = 0; - c.restart_in_s = 0; - c.message = format!("faulted: {reason}"); + c.message = format!("{reason}; trying again"); } } Action::None => {} @@ -4224,6 +4329,60 @@ impl Engine { st.node.message = format!("{why}; restart in {} s", delay.as_secs()); } + /// MF-7 (PC 1, 7 October 2026: orphan igneum-miner processes the app no longer tracked kept hitting the node's + /// template RPC beside the tracked ones). The engine owns every miner it started: any igneum-miner process whose + /// command line carries THIS engine's node RPC (the fence: the port this app's node listens on) and whose pid is + /// not in the tracked set is killed, one log line and one fault report per kill. Never by name alone. + fn sweep_orphan_miners(&mut self, why: &str) { + let fence = self.shared.runtime.rpc_url(); + let tracked: Vec = self.miners.iter().filter_map(|m| m.proc.as_ref().map(|p| p.pid())).collect(); + let orphans = crate::platform::miner_processes().into_iter().filter(|(pid, cmd)| cmd.contains(&fence) && !tracked.contains(pid)).collect::>(); + for (pid, cmd) in orphans { + crate::platform::kill_pid(pid); + self.shared.log(&format!("orphan miner killed ({why}): pid {pid}, not started by this engine: {}", short(&cmd, 200))); + self.fault_report(None, "orphan-miner", &format!("igneum-miner pid {pid} not tracked by the engine was killed ({why})")); + } + } + + /// Node readiness probe (the project lead, 7 October 2026): igneum_getExecStatus every 5 s off the engine thread while the + /// node is up. Its answer sets `exec_ready` (the workers' start gate with `synced`) and counts as the node's sign + /// of life for the node watchdog. + fn tick_exec_probe(&mut self, now: Instant) { + let node_up = self.node_external || self.node.is_some(); + if !node_up || self.exec_probe_busy || now < self.exec_probe_next { + return; + } + self.exec_probe_busy = true; + self.exec_probe_next = now + Duration::from_secs(5); + let port = self.shared.runtime.evm_port(); + let shared = self.shared.clone(); + std::thread::spawn(move || { + let (answered, has_record) = crate::execrpc::probe(port); + shared.send(Cmd::ExecProbe { answered, has_record }); + }); + } + + /// One fault line to the log intake, the moment it happens (the project lead, 7 October 2026: the team sees it before the + /// user): the card, the class, the reason and the app and node versions. At most 60 an hour, off the engine + /// thread; the same line is in the app log either way. + fn fault_report(&mut self, card: Option<&str>, class: &str, reason: &str) { + let now = Instant::now(); + self.fault_reports.retain(|t| now.duration_since(*t) < Duration::from_secs(3600)); + let line = format!("FAULT class={class} card=\"{}\" app={} reason=\"{}\"", card.unwrap_or("node"), VERSION, crate::platform::redact(reason).replace('"', "'")); + self.shared.log(&line); + let p = &self.shared.packaged; + if p.log_intake_url.is_empty() || p.log_intake_key.is_empty() || self.fault_reports.len() >= 60 { + return; + } + self.fault_reports.push(now); + let (url, key, machine, run_id) = (p.log_intake_url.clone(), p.log_intake_key.clone(), format!("{}-{}", self.shared.runtime.host, self.shared.runtime.id8()), format!("{}-{}", self.label_base, self.stamp)); + let label = format!("fault-{}", self.label_base); + let text = format!("{}\n{:.3} {line}", self.shared.upload_header(), crate::platform::unix_now_f()); + std::thread::spawn(move || { + crate::update::upload_text(&url, &key, &label, &machine, &run_id, &text); + }); + } + fn bins_worker_missing(&self, card_idx: usize) -> bool { let st = self.st(); match st.mining.cards.get(card_idx).map(|c| c.worker.as_str()) { @@ -4557,14 +4716,15 @@ impl Engine { if self.shared.ladder.lock().unwrap().mark("synced", crate::platform::unix_now_f()) { self.shared.save_ladder(); } let t = self.secs(now); let names: Vec = self.st().mining.cards.iter().map(|c| c.name.clone()).collect(); + self.node_settled = true; for m in self.miners.iter_mut() { - // a card faulted for silence while the node was not ready starts again now (PC 1, 7 October 2026) - if m.watch.faulted_by_node() { + // a restart the node's readiness explained does not count against the card (PC 1, 7 October 2026) + if m.watch.restarted_for_node() { m.watch.event(t, crate::watchdog::Event::NodeSynced); let name = names.get(m.card).cloned().unwrap_or_else(|| m.label.clone()); - self.shared.event("info", &format!("{name}: starts again now the node is synced (it was faulted for silence while the node was syncing)")); + self.shared.event("info", &format!("{name}: starts again now the node is synced (its earlier silence was the node's catch-up, not the card's)")); } - if m.proc.is_none() && m.restart_at.is_none() && m.watch.faulted().is_none() { + if m.proc.is_none() && m.restart_at.is_none() { m.restart_at = Some(now); } } @@ -4591,7 +4751,11 @@ impl Engine { self.shared.event("info", &format!("{card_name}: the node is not answering block templates yet; the miner keeps asking (the node catches up after a restart)")); } if let Some(c) = self.st().mining.cards.get_mut(card) { - c.message = "waiting for the node to answer block templates".into(); + // never while the card hashes (PC 1, 7 October 2026 11:4x UK: the label sat on a card accepting shares); + // a timed-out fetch for one identity beside a healthy rate is the node's latency, said by the STATUS line + if c.hash_now <= 0.0 { + c.message = "waiting for the node to answer block templates".into(); + } } return; } @@ -4615,6 +4779,21 @@ impl Engine { // the worker refused its program pack; the miner rebuilds the pack and restarts the worker itself // (or exits 44 for us to export it): the strip and the card name the condition in plain words self.pack_notice(i, card, &card_name, &why); + } else if text.contains("worker error") && is_self_test_failure(text) { + // MF-4: a worker that fails its self-test is not restarted every few seconds; the card is held for + // 30 minutes with the reason on its row, or until its driver changes + let reason = short(text.split("worker error:").nth(1).unwrap_or(text).trim(), 160); + let t = self.secs(Instant::now()); + let verdict = self.miners[i].watch.event(t, crate::watchdog::Event::SelfTestFailed(&reason)); + if let Some(mut p) = self.miners[i].proc.take() { + p.write_stdin("quit\n"); + p.stop(3); + } + self.watchdog_verdict(i, verdict, &[]); + if let Some(c) = self.st().mining.cards.get_mut(card) { + c.message = format!("not usable on this driver: {reason}; next try in 30 minutes or after a driver change"); + } + return; } else if text.contains("WORKER MISMATCH") || text.contains("worker error") || text.contains("worker exited") || text.contains("worker killed by a guard") || text.contains("panicked") || text.contains("CUDA error") || text.contains("submit error") { let now = Instant::now(); if now.duration_since(self.last_error_event) >= Duration::from_secs(30) { @@ -4649,6 +4828,7 @@ impl Engine { let t = self.secs(Instant::now()); self.miners[i].watch.event(t, crate::watchdog::Event::WorkerRestart(&reason)); self.shared.event("error", &format!("{card_name}: worker fault: {}; the miner restarts the worker", short(&reason, 200))); + self.fault_report(Some(&card_name), "worker-fault", &reason); if let Some(c) = self.st().mining.cards.get_mut(card) { c.faults += 1; c.hash_now = 0.0; @@ -4724,9 +4904,16 @@ impl Engine { if c.state == "starting" || c.state == "ready" { c.state = "mining".into(); } - if c.state == "mining" && s.hash_now > 0.0 { + if s.hash_now > 0.0 && (c.state == "mining" || c.message.starts_with("waiting for the node to answer") || c.message.starts_with("node slow")) { + // the first STATUS line with a rate clears every node-wait label, whatever the row's state word c.message = if s.mismatched > 0 { format!("{} share(s) failed the CPU re-check this run", s.mismatched) } else { String::new() }; } + // MF-5: the node's latency on the card in plain words, the worker kept + if s.template_wait_s > 0.0 && s.hash_now <= 0.0 { + c.message = format!("node slow: waiting for a block template for {:.0} s; the worker is kept", s.template_wait_s); + } else if s.template_ms >= 2000.0 && c.state == "mining" { + c.message = format!("node slow: a template takes {:.1} s{}", s.template_ms / 1000.0, if s.identities_active > 0 && s.identities_active < c.identities as u64 { format!("; {} of {} identities active until it answers faster", s.identities_active, c.identities) } else { String::new() }); + } } // miner-ui-5: the first-hour timeline's "card mining" mark, once per install, on the first status with a rate if s.hash_now > 0.0 && self.shared.ladder.lock().unwrap().mark("mining", crate::platform::unix_now_f()) { self.shared.save_ladder(); } @@ -5445,6 +5632,13 @@ mod tests { } } +/// A worker line that names a self-test failure (the vectors, the cache check or the source check did not pass): +/// `error 0 self-test FAIL ...`, `vectors 3 of 96 FAIL`, `Source check FAIL`, `self-test failed`. +pub(crate) fn is_self_test_failure(text: &str) -> bool { + let t = text.to_ascii_lowercase(); + (t.contains("self-test") || t.contains("selftest") || t.contains("source check") || t.contains("vectors")) && (t.contains("fail") || t.contains("mismatch")) +} + fn short(s: &str, n: usize) -> String { if s.chars().count() <= n { s.to_string() } else { format!("{}...", s.chars().take(n).collect::()) } } @@ -5475,13 +5669,27 @@ fn civil_from_days(z: i64) -> (i64, u32, u32) { /// OpenCL worker start then refused ("the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT") until the next /// export. The second export of a pair rewrites the same pack, which is harmless. static EXPORT_LOCK: std::sync::Mutex<()> = std::sync::Mutex::new(()); +/// When the pack was last exported: an export under `EXPORT_REUSE_S` old is reused unless forced (MF-4, 7 October +/// 2026: a card whose worker failed every few seconds exported on every restart and held two healthy cards in +/// "loading the program" past the watchdog). +static LAST_EXPORT: std::sync::Mutex> = std::sync::Mutex::new(None); +const EXPORT_REUSE_S: u64 = 60; /// Exports this hour's program pack from the node to \packs\devnet (the prebuilt workers read it with --pack). +/// One export per minute serves every card; `force` (a refused pack) exports again now. #[allow(unused_variables)] -fn export_pack(shared: &Arc, bins: &Bins) -> Result<(), String> { +fn export_pack(shared: &Arc, bins: &Bins, force: bool) -> Result<(), String> { let _one_at_a_time = EXPORT_LOCK.lock().unwrap_or_else(|e| e.into_inner()); let pack = shared.runtime.app_dir.join("packs").join("devnet"); let _ = std::fs::create_dir_all(&pack); + if !force && pack.join("seeds.txt").exists() { + let fresh = LAST_EXPORT.lock().unwrap_or_else(|e| e.into_inner()).map(|t| t.elapsed() < Duration::from_secs(EXPORT_REUSE_S)).unwrap_or(false); + if fresh { + shared.log("export-pack: reusing the pack exported under a minute ago"); + return Ok(()); + } + } + *LAST_EXPORT.lock().unwrap_or_else(|e| e.into_inner()) = Some(Instant::now()); let out = crate::detect::run_timeout(std::process::Command::new(&bins.miner).args(["export-pack", &shared.runtime.rpc_url(), &pack.display().to_string()]), None, Duration::from_secs(120)).unwrap_or_default(); shared.log(&format!("export-pack: {}", out.lines().last().unwrap_or("no output"))); if pack.join("seeds.txt").exists() { Ok(()) } else { Err("export-pack wrote no seeds.txt (is the node reachable?)".into()) } diff --git a/app/igneum-app/src/execrpc.rs b/app/igneum-app/src/execrpc.rs new file mode 100644 index 000000000..29cbe478f --- /dev/null +++ b/app/igneum-app/src/execrpc.rs @@ -0,0 +1,158 @@ +//! The one path to the node's execution-layer JSON-RPC (ledger N7, 7 October 2026, main's rule for 0.3.19). +//! +//! A node before the exec RPC bounds fix (every 0.3.17 node) dies when a method that resolves a block number or indexes +//! the record vector is asked while its exec follower holds no record: `rpc.rs` indexes `records[0]` (or slices +//! `records[1..=0]`) on an empty vector, the panic hook exits the process, and the app restarts it. PC 1 crash-looped on +//! two callers in one night (the clock sample's eth_getBlockByNumber, then the prover's igneum_getAssignedShards after the +//! node read synced seconds before a slow follower loaded). The node-side fix ships with the 0.3.20 node; a 0.3.19 app on a +//! 0.3.17 node must be safe by itself, so every exec RPC call the app makes goes through [`call`]: a method in +//! [`SAFE_ON_EMPTY`] goes out at once; any other waits until igneum_getExecStatus reports an executed tip. The unit test +//! below enumerates the callers: no other file may build an exec JSON-RPC request, and every method name in the tree must +//! be classified here, so a new caller or method cannot bypass the gate. +use serde_json::{json, Value}; +use std::process::Command; +use std::time::Duration; + +/// Methods that index nothing on an empty exec state (read from the 0.3.17 node's rpc.rs): safe at any time. +pub const SAFE_ON_EMPTY: &[&str] = &[ + "eth_chainId", "eth_blockNumber", "eth_syncing", "igneum_getExecStatus", "igneum_getProvingStatus", "igneum_getNodeInfo", +]; + +/// Methods the app sends that resolve a block number, index or slice the record vector, or simulate at a block: held until +/// the follower holds a record. Every method literal outside this module must be in one of the two lists. +pub const GATED: &[&str] = &[ + "eth_getBlockByNumber", "igneum_getAssignedShards", "igneum_getProofRecords", "igneum_getSegmentRecords", "igneum_getSegmentStatement", + "igneum_getProofBytes", "igneum_getSegmentProofBytes", "igneum_exportSegments", "igneum_submitProofRecord", "igneum_submitSegmentRecord", + "igneum_getFinalityWeights", +]; + +/// One JSON-RPC POST to 127.0.0.1: through curl (the engine carries no HTTP client); the body goes through a file +/// so a large export request is not an argument. The reply's `result` (null allowed), or the error's message. +fn post(evm_port: u16, method: &str, params: Value, timeout: Duration) -> Result { + let body = json!({ "jsonrpc": "2.0", "id": 1, "method": method, "params": params }).to_string(); + let tmp = std::env::temp_dir().join(format!("igneum-rpc-{}-{}-{}.json", std::process::id(), method, crate::platform::unix_now_f() as u64)); + std::fs::write(&tmp, body).map_err(|e| e.to_string())?; + let url = format!("http://127.0.0.1:{evm_port}"); + let out = crate::detect::run_timeout( + Command::new(crate::platform::tool("curl")).args(["-s", "--max-time", &timeout.as_secs().max(1).to_string(), "-X", "POST", &url, "-H", "Content-Type: application/json", "-d", &format!("@{}", tmp.display())]), + None, + timeout + Duration::from_secs(2), + ); + let _ = std::fs::remove_file(&tmp); + let out = out.ok_or_else(|| format!("{method}: the node's RPC did not answer"))?; + let v: Value = serde_json::from_str(&out).map_err(|e| format!("{method}: {e}"))?; + if let Some(err) = v.get("error") { + return Err(format!("{method}: {}", err.get("message").and_then(|m| m.as_str()).unwrap_or("error"))); + } + Ok(v.get("result").cloned().unwrap_or(Value::Null)) +} + +/// True once the node's exec follower holds a record (igneum_getExecStatus's executedTipHash is set). Any error or an +/// unreachable node reads false: the gated call waits rather than asks. +pub fn has_record(evm_port: u16) -> bool { + match post(evm_port, "igneum_getExecStatus", json!([]), Duration::from_secs(5)) { + Ok(r) => status_has_record(&r), + Err(_) => false, + } +} + +/// The reading of an igneum_getExecStatus result: a record is held when executedTipHash is a non-empty string. +pub fn status_has_record(result: &Value) -> bool { + result.get("executedTipHash").and_then(|h| h.as_str()).map(|h| !h.is_empty()).unwrap_or(false) +} + +/// The engine's readiness probe (every 5 s): did the node's RPC answer at all, and does the follower hold a record. +/// The first is the node watchdog's sign of life; the second, with `synced`, is the workers' start gate. +pub fn probe(evm_port: u16) -> (bool, bool) { + match post(evm_port, "igneum_getExecStatus", json!([]), Duration::from_secs(5)) { + Ok(r) => (true, status_has_record(&r)), + Err(e) => (!e.contains("did not answer"), false), + } +} + +/// The gate: a safe method goes out; a gated one waits for a record; an unclassified method is refused (add it to a list). +pub fn call(evm_port: u16, method: &str, params: Value, timeout: Duration) -> Result { + if SAFE_ON_EMPTY.contains(&method) { + return post(evm_port, method, params, timeout); + } + if !GATED.contains(&method) { + return Err(format!("{method}: not classified in execrpc (safe on an empty state, or gated); add it before calling")); + } + if !has_record(evm_port) { + return Err(format!("{method}: the node's execution layer holds no record yet")); + } + post(evm_port, method, params, timeout) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn status_reading() { + assert!(status_has_record(&json!({"executedTip":"0x38a5","executedTipHash":"0xa3ae37ec"}))); + assert!(!status_has_record(&json!({"executedTip":"0x0","executedTipHash":null}))); + assert!(!status_has_record(&json!({}))); + assert!(!status_has_record(&json!({"executedTipHash":""}))); + } + + #[test] + fn lists_are_disjoint_and_sorted_enough() { + for m in GATED { + assert!(!SAFE_ON_EMPTY.contains(m), "{m} in both lists"); + } + } + + /// Ledger N7: every exec JSON-RPC request the app builds goes through this module, and every exec method name in the + /// tree is classified here. A new direct caller (a file that builds a "jsonrpc" POST) or an unclassified method fails. + #[test] + fn every_caller_goes_through_the_gate() { + let src = std::path::Path::new(env!("CARGO_MANIFEST_DIR")).join("src"); + // every eth_* or igneum_* identifier on a line (a hand scanner: no regex crate in this binary) + fn methods_in(line: &str) -> Vec { + // ASCII-only scan over char boundaries: a token of [A-Za-z0-9_] that starts with eth_ or igneum_ + let mut out = Vec::new(); + let mut token = String::new(); + let mut flush = |t: &mut String| { + if t.starts_with("eth_") || t.starts_with("igneum_") { out.push(t.clone()); } + t.clear(); + }; + for ch in line.chars() { + if ch.is_ascii_alphanumeric() || ch == '_' { token.push(ch); } else { flush(&mut token); } + } + flush(&mut token); + out + } + let mut offenders = Vec::new(); + let mut unclassified = Vec::new(); + for entry in std::fs::read_dir(&src).unwrap() { + let path = entry.unwrap().path(); + if path.extension().and_then(|e| e.to_str()) != Some("rs") { continue; } + let name = path.file_name().unwrap().to_string_lossy().to_string(); + let text = std::fs::read_to_string(&path).unwrap(); + // a JSON-RPC REQUEST names a method next to "jsonrpc" (replies and test fixtures carry "result" or "error"); + // only this module may build one + if name != "execrpc.rs" { + for (i, line) in text.lines().enumerate() { + let l = line.replace('\\', ""); + if l.contains("\"jsonrpc\"") && l.contains("\"method\"") && !l.contains("\"result\"") && !l.contains("\"error\"") { + offenders.push(format!("{name}:{}", i + 1)); + } + } + } + for (i, line) in text.lines().enumerate() { + if line.trim_start().starts_with("//") { continue; } + for m in methods_in(line) { + let m = m.as_str(); + // identifiers that are not RPC methods: crate and file names (igneum_app, igneum_miner, ...) carry no camel-case method part + let looks_like_method = m.contains("_get") || m.contains("_submit") || m.contains("_export") || m.contains("_estimate") || m.contains("_send") || m == "eth_chainId" || m == "eth_blockNumber" || m == "eth_syncing"; + if looks_like_method && !SAFE_ON_EMPTY.contains(&m) && !GATED.contains(&m) { + unclassified.push(format!("{name}:{}: {m}", i + 1)); + } + } + } + } + assert!(offenders.is_empty(), "exec JSON-RPC built outside execrpc.rs: {offenders:?}"); + assert!(unclassified.is_empty(), "exec methods not classified in execrpc.rs: {unclassified:?}"); + } +} diff --git a/app/igneum-app/src/jobrun.rs b/app/igneum-app/src/jobrun.rs index c25e29eb3..e9229d78f 100644 --- a/app/igneum-app/src/jobrun.rs +++ b/app/igneum-app/src/jobrun.rs @@ -87,6 +87,9 @@ pub enum Action { StopMiners(String), RestartMiners, RestartNode, + /// The signed `cards` kind: apply these per-card choices through the app's own card path (persisted), then + /// call `cards_applied` with the read-back. + ApplyCards(Vec), /// The relaunch helper was started; the engine quits now. RestartApp, UpdateNow, @@ -570,6 +573,23 @@ impl Jobs { let cards_off: Vec = if job.kind == "run" { job.list_param("cards_off") } else { vec![] }; self.active = Some(Active { job: job.clone(), run_id: run_id.clone(), started: Instant::now(), started_unix: now, waiting_for_miners: needs_miners_stopped, holds_miners: false, waiting_for_cards: !cards_off.is_empty(), cards: CardHold::default(), ctl: ctl.clone() }); match job.kind.as_str() { + "cards" => { + // engine-side: refused at once for a card this machine does not have; else applied through the + // app's own card path and reported with the read-back (cards_applied) + let live: Vec<(String, bool, u32)> = shared.state.lock().unwrap().mining.cards.iter().filter(|c| c.present()).map(|c| (c.key.clone(), c.enabled, c.identities)).collect(); + let (choices, missing) = cards_job_choices(&job, &live); + let sink = Sink::new(shared, &job, &self.dir); + if !missing.is_empty() { + let summary = format!("refused: this machine has no card {}", missing.join(", ")); + sink.line(&format!("cards: {summary}; present: {}", live.iter().map(|c| c.0.as_str()).collect::>().join(", "))); + let uploaded = report(shared, &job, &sink, "failed", 2, now, crate::platform::unix_now(), &summary, json!({ "present": live.iter().map(|c| c.0.clone()).collect::>() })); + let o = Outcome { status: "failed".into(), exit: 2, summary, results: vec![], uploaded }; + self.event(shared, Event::Finished { id: job.id.clone(), outcome: o }); + return None; + } + sink.line(&format!("cards: applying {}", choices.iter().map(|c| format!("{} enabled={} identities={}{}", c.key, c.enabled, c.identities, c.power_pct.map(|p| format!(" power_pct={p}")).unwrap_or_default())).collect::>().join("; "))); + return Some(Action::ApplyCards(choices)); + } "restart" | "update-now" => { // engine-side; the report says what was asked and the ledger closes at once let what = job.str_param("what"); @@ -603,6 +623,35 @@ impl Jobs { } } + /// The running job's id, for the hold rule (a hold belongs to the job that took it). + pub fn active_id(&self) -> Option { + self.active.as_ref().map(|a| a.job.id.clone()) + } + + /// The running job's cap in minutes (the hold's hard cap). + pub fn active_cap_minutes(&self) -> u64 { + self.active.as_ref().map(|a| a.job.timeout_minutes()).unwrap_or(60) + } + + /// Called by the engine once a `cards` job's choices are applied: the report carries the read-back of every + /// card named (what the engine holds now) and the job closes done. + pub fn cards_applied(&mut self, shared: &Arc, readback: Vec<(String, bool, u32, u32)>) -> Option { + let Some(a) = self.active.as_ref() else { return None }; + if a.job.kind != "cards" { + return None; + } + let (job, started) = (a.job.clone(), a.started_unix); + let sink = Sink::new(shared, &job, &self.dir); + let lines: Vec = readback.iter().map(|(k, e, i, p)| format!("{k} enabled={e} identities={i} power_pct={p}")).collect(); + for l in &lines { + sink.line(&format!("cards: read back {l}")); + } + let summary = format!("cards applied: {}", lines.join("; ")); + let uploaded = report(shared, &job, &sink, "done", 0, started, crate::platform::unix_now(), &summary, json!({ "cards": readback.iter().map(|(k, e, i, p)| json!({ "key": k, "enabled": e, "identities": i, "power_pct": p })).collect::>() })); + let o = Outcome { status: "done".into(), exit: 0, summary, results: lines, uploaded }; + self.event(shared, Event::Finished { id: job.id.clone(), outcome: o }) + } + /// Called by the engine once a job's `--cards-off` cards are switched off; `restore` puts them back exactly. /// Next: the miners, if the job asked for them too, else the script. pub fn cards_off_done(&mut self, shared: &Arc, restore: Vec) -> Option { @@ -1330,6 +1379,50 @@ fn elevated_wrapper(env_lines: &str, script: &str, out_file: &str) -> String { ) } +/// The choices a `cards` job asks for, matched to the machine's present cards (key exact, or the key with the device +/// index left out: `vendor::name`); missing keys are returned for the refusal. A field the job leaves out keeps the +/// card's current value. +pub fn cards_job_choices(job: &Job, live: &[(String, bool, u32)]) -> (Vec, Vec) { + let mut out = Vec::new(); + let mut missing = Vec::new(); + let list = job.params.get("cards").and_then(|v| v.as_array()).cloned().unwrap_or_default(); + for c in list { + let key = c.get("key").and_then(|v| v.as_str()).unwrap_or("").trim().to_string(); + let found = live.iter().find(|(k, _, _)| *k == key).or_else(|| { + let parts: Vec<&str> = key.splitn(3, ':').collect(); + live.iter().find(|(k, _, _)| { let lp: Vec<&str> = k.splitn(3, ':').collect(); parts.len() == 3 && lp.len() == 3 && lp[0] == parts[0] && lp[2] == parts[2] && (parts[1].is_empty() || parts[1] == lp[1]) }) + }); + match found { + Some((k, enabled, identities)) => out.push(crate::engine::CardChoice { + key: k.clone(), + enabled: c.get("enabled").and_then(|v| v.as_bool()).unwrap_or(*enabled), + identities: c.get("identities").and_then(|v| v.as_u64()).map(|n| n as u32).unwrap_or(*identities), + power_pct: c.get("power_pct").and_then(|v| v.as_u64()).map(|n| n as u32), + }), + None => missing.push(key), + } + } + (out, missing) +} + +/// The hold rule (MF-6, PC 1, 7 October 2026: a read-only job that followed a --stop-miners job kept the cards off +/// for its whole run): a hold belongs to the job that took it and releases the moment that job is no longer the +/// running one, whatever runs next, or when the owner's own cap has passed. Returns the reason to release, or None. +pub fn hold_release(owner: Option<&str>, active: Option<&str>, held_s: f64, cap_s: f64) -> Option<&'static str> { + match owner { + None => Some("no job owns the hold"), + Some(o) => { + if active != Some(o) { + Some("the job that took the hold is no longer running") + } else if held_s >= cap_s { + Some("the hold passed the job's own cap") + } else { + None + } + } + } +} + fn finish_ran(ran: Ran, what: &str) -> Result { match ran.code { Some(0) => Ok(Done { status: "done".into(), exit: 0, summary: format!("{what} finished, exit 0"), extra: json!({}) }), @@ -2089,3 +2182,23 @@ fn account_warning(ctx: &str) -> String { fn short(s: &str, n: usize) -> String { if s.chars().count() <= n { s.to_string() } else { format!("{}...", s.chars().take(n).collect::()) } } + +#[cfg(test)] +mod hold_tests { + use super::hold_release; + + /// MF-6 (PC 1, 7 October 2026): a --stop-miners job's hold outlived it into a read-only watch job for three minutes. + #[test] + fn a_hold_belongs_to_the_job_that_took_it() { + // the owner is still running, under its cap: the hold stays + assert_eq!(hold_release(Some("job-a"), Some("job-a"), 30.0, 3600.0), None); + // another job runs now: released at once, whatever that job is + assert_eq!(hold_release(Some("job-a"), Some("job-b"), 30.0, 3600.0), Some("the job that took the hold is no longer running")); + // no job runs: released + assert_eq!(hold_release(Some("job-a"), None, 30.0, 3600.0), Some("the job that took the hold is no longer running")); + // the owner's own cap passed: released and logged + assert_eq!(hold_release(Some("job-a"), Some("job-a"), 3601.0, 3600.0), Some("the hold passed the job's own cap")); + // a hold with no owner (an older engine state) never sticks + assert_eq!(hold_release(None, Some("job-b"), 1.0, 3600.0), Some("no job owns the hold")); + } +} diff --git a/app/igneum-app/src/jobs.rs b/app/igneum-app/src/jobs.rs index 20662e2ad..94d68b669 100644 --- a/app/igneum-app/src/jobs.rs +++ b/app/igneum-app/src/jobs.rs @@ -61,7 +61,7 @@ pub const JOBS_FILE: &str = "igneum-jobs.json"; /// the signature is over the bytes of `file`, so the same key and the same signer sign both forms. pub const JOBS_SIGNED_FILE: &str = "igneum-jobs.signed.json"; pub const JOBS_SIGNED_FORMAT: &str = "igneum-jobs-signed-1"; -pub const KINDS: &[&str] = &["run", "fetch", "collect", "restart", "update-now", "shard-benchmark", "build"]; +pub const KINDS: &[&str] = &["run", "fetch", "collect", "restart", "update-now", "shard-benchmark", "build", "cards"]; /// Requirements the engine knows how to probe (src/jobrun.rs). An unknown requirement is never satisfied. pub const KNOWN_REQUIRES: &[&str] = &["wsl", "wsl-prover", "nvidia"]; /// Named folders a `fetch` may write into, all under the app data root. @@ -420,6 +420,33 @@ pub fn validate_params(job: &Job) -> Result<(), String> { } } "update-now" => {} + "cards" => { + // the signed `cards` kind (7 October 2026): per-card enabled and identities, applied by the app through + // its own card path and persisted; it closes the exception of a one-off script POSTing /api/cards + let list = job.params.get("cards").and_then(|v| v.as_array()).cloned().unwrap_or_default(); + if list.is_empty() { + return Err("cards: params.cards is empty (a list of {key, enabled, identities})".into()); + } + for c in &list { + let key = c.get("key").and_then(|v| v.as_str()).unwrap_or(""); + if key.trim().is_empty() || !key.contains(':') { + return Err(format!("cards: key '{key}' is not a card key (vendor:device:name)")); + } + if c.get("enabled").map(|v| !v.is_boolean()).unwrap_or(false) { + return Err(format!("cards: {key}: enabled must be true or false")); + } + if let Some(n) = c.get("identities") { + if !n.as_u64().map(|n| (1..=64).contains(&n)).unwrap_or(false) { + return Err(format!("cards: {key}: identities must be 1 to 64")); + } + } + if let Some(n) = c.get("power_pct") { + if !n.as_u64().map(|n| (50..=100).contains(&n)).unwrap_or(false) { + return Err(format!("cards: {key}: power_pct must be 50 to 100")); + } + } + } + } "shard-benchmark" => { let url = job.str_param("zip_url"); if !url.is_empty() && !https_ok(&url) { @@ -826,6 +853,10 @@ mod tests { assert!(j("restart", r#"{"what":"everything"}"#).unwrap_err().contains("restart")); assert!(j("restart", r#"{"what":"miners"}"#).is_ok()); assert!(j("update-now", r#"{}"#).is_ok()); + assert!(j("cards", r#"{}"#).unwrap_err().contains("cards")); + assert!(j("cards", r#"{"cards":[{"key":"nvidia:0:NVIDIA GeForce RTX 5090","enabled":true,"identities":8}]}"#).is_ok()); + assert!(j("cards", r#"{"cards":[{"key":"5090","enabled":true}]}"#).unwrap_err().contains("card key")); + assert!(j("cards", r#"{"cards":[{"key":"nvidia:0:x","identities":65}]}"#).unwrap_err().contains("1 to 64")); assert!(j("shard-benchmark", r#"{}"#).unwrap_err().contains("sha256")); assert!(j("shard-benchmark", &format!(r#"{{"sha256":"{}","fixtures":["../x"]}}"#, "b".repeat(64))).unwrap_err().contains("fixture")); assert!(j("run", r#"{"script":"ls","shell":"zsh"}"#).unwrap_err().contains("shell")); diff --git a/app/igneum-app/src/platform.rs b/app/igneum-app/src/platform.rs index 42e67793d..87d4f7359 100644 --- a/app/igneum-app/src/platform.rs +++ b/app/igneum-app/src/platform.rs @@ -268,6 +268,49 @@ pub fn keep_awake_tick() { } /// Asks a child to stop. Unix: SIGTERM (the node closes its database cleanly). Windows: TerminateProcess through +/// Every `igneum-miner` process on this machine with its command line: (pid, command line). Windows reads +/// Win32_Process through PowerShell; unix reads `ps`. An empty list when the tool fails (the caller kills nothing). +pub fn miner_processes() -> Vec<(u32, String)> { + let out = if cfg!(windows) { + let mut c = std::process::Command::new(tool("powershell")); + c.args(["-NoProfile", "-Command", "Get-CimInstance Win32_Process -Filter \"Name='igneum-miner.exe'\" | ForEach-Object { \"$($_.ProcessId)|$($_.CommandLine)\" }"]); + crate::detect::run_timeout(&mut c, None, std::time::Duration::from_secs(20)) + } else { + let mut c = std::process::Command::new("ps"); + c.args(["-eo", "pid=,args="]); + crate::detect::run_timeout(&mut c, None, std::time::Duration::from_secs(10)) + }; + let Some(out) = out else { return vec![] }; + let mut v = Vec::new(); + for l in out.lines() { + let l = l.trim(); + let (pid, cmd) = if cfg!(windows) { + let Some((p, c)) = l.split_once('|') else { continue }; + (p.trim(), c.trim()) + } else { + let Some((p, c)) = l.split_once(' ') else { continue }; + (p.trim(), c.trim()) + }; + let Ok(pid) = pid.parse::() else { continue }; + if !cfg!(windows) && !(cmd.contains("igneum-miner ") || cmd.ends_with("igneum-miner")) { + continue; + } + if cmd.contains(" mine ") { + v.push((pid, cmd.to_string())); + } + } + v +} + +/// Ends one process by pid (Windows: taskkill /T /F; unix: SIGKILL). +pub fn kill_pid(pid: u32) { + if cfg!(windows) { + let _ = std::process::Command::new(tool("taskkill")).args(["/PID", &pid.to_string(), "/T", "/F"]).output(); + } else { + let _ = std::process::Command::new("kill").args(["-9", &pid.to_string()]).output(); + } +} + /// std (what today's launcher does with taskkill /F). pub fn terminate(child: &mut std::process::Child) { #[cfg(unix)] diff --git a/app/igneum-app/src/state.rs b/app/igneum-app/src/state.rs index 511085d2b..8414b8909 100644 --- a/app/igneum-app/src/state.rs +++ b/app/igneum-app/src/state.rs @@ -77,7 +77,7 @@ pub struct CardState { pub detail: String, // memory, cores pub device: String, // the worker's --device value (Windows) pub enabled: bool, - pub state: String, // off | waiting | starting | ready | mining | restarting | failed | faulted (the watchdog gave up on it) | unusable (the OS reports a problem) | removed (unplugged) + pub state: String, // off | waiting | starting | ready | mining | restarting | failed (no "faulted": no fault is permanent, 7 October 2026) | unusable (the OS reports a problem) | removed (unplugged) pub hash_now: f64, // MH/s, the last interval pub hash_avg: f64, // MH/s since the start pub accepted: u64, diff --git a/app/igneum-app/src/update.rs b/app/igneum-app/src/update.rs index 382823536..2978d8804 100644 --- a/app/igneum-app/src/update.rs +++ b/app/igneum-app/src/update.rs @@ -115,6 +115,15 @@ pub fn upload_log(url: &str, key: &str, label: &str, machine: &str, run_id: &str let tail: String = String::from_utf8_lossy(&data).lines().filter(|l| !crate::platform::carries_token(l)).map(|l| crate::platform::redact(l)).collect::>().join("\n"); // the first line of every upload names the app, the machine and the node (the console parses it) let text = format!("{header}\n{tail}"); + upload_text(url, key, label, machine, run_id, &text) +} + +/// One upload of ready text (a fault report, a job's lines): the same intake, the same token guard. +pub fn upload_text(url: &str, key: &str, label: &str, machine: &str, run_id: &str, text: &str) -> bool { + if url.is_empty() || key.is_empty() || text.is_empty() { + return false; + } + let text: String = text.lines().filter(|l| !crate::platform::carries_token(l)).collect::>().join("\n"); let body = serde_json::json!({ "label": label, "machine": machine, "run_id": run_id, "lines": text }); let tmp = std::env::temp_dir().join(format!("igneum-upload-{}-{}.json", std::process::id(), label)); if std::fs::write(&tmp, body.to_string()).is_err() { diff --git a/app/igneum-app/src/watchdog.rs b/app/igneum-app/src/watchdog.rs index 6ae01f539..cec7068aa 100644 --- a/app/igneum-app/src/watchdog.rs +++ b/app/igneum-app/src/watchdog.rs @@ -2,15 +2,21 @@ //! rules can be unit-tested with recorded miner lines. //! //! Per card (`CardWatch`): a miner that prints no status line for 90 s, or reports a hash rate of 0 for 60 s while -//! the node is synced, is restarted once; when that recurs before five minutes of healthy status, the card is marked -//! faulted with the reason, its miner is not restarted again, and the other cards keep mining. The miner's own -//! worker restarts (`WORKER FAULT`, `worker exited`) suspend both rules until the worker is ready again, so the app -//! never restarts a miner that is already restarting its worker (no double restarts); if the worker is not back -//! within 180 s the app steps in. Exit code 43 (the miner gave up on its worker after three guard trips) counts -//! like a watchdog restart: once, then faulted. +//! the node is ready, is restarted; when that recurs before five minutes of healthy status the next restart waits +//! longer (10 s, 30 s, 2 min, 5 min, then every 5 min, for ever: the project lead, 7 October 2026, "it needs to be truly plug, +//! tune, play"; no fault is permanent and the old one-restart-then-faulted state no longer exists). The reason stays on +//! the card in plain words and the hash resumes on its own. A miner is judged only while the node is READY: synced +//! and with an executed tip (`igneum_getExecStatus`); with no template it cannot print status, so every silence clock +//! holds while the node catches up (PC 1, 7 October 2026: three workers faulted "no status line" during the node's +//! catch-up and stayed faulted until a cards-API bounce). The miner's own worker restarts (`WORKER FAULT`, +//! `worker exited`) suspend both rules until the worker is ready again, so the app never restarts a miner that is +//! already restarting its worker (no double restarts); if the worker is not back within 180 s the app steps in. +//! Exit code 43 (the miner gave up on its worker after three guard trips) is a watchdog restart on the same ladder. //! -//! Per node (`NodeWatch`): a node of ours that answers no `watch` reading for 120 s is restarted by the app, with a -//! growing delay when it repeats inside ten minutes. +//! Per node (`NodeWatch`): a node of ours that gives no sign of life for 120 s is restarted by the app, with a +//! growing delay when it repeats inside ten minutes. Its catch-up never counts: until the node has been read as +//! synced once since its start, silence is not held against it (a node that never answers at all is restarted after +//! 30 minutes), and any RPC answer (the watch reading, the exec status probe, an accepted block) is a sign of life. //! //! Review round 4, X21 (4 October 2026): the app now reads `mismatched=` and `faults=` from STATUS lines and the //! `WORKER FAULT` lines, and shows them on the card. @@ -20,8 +26,27 @@ pub const NO_STATUS_S: f64 = 90.0; /// A worker that gives no status within this many seconds of its start, with the node synced, is one restart then /// faulted (PC 1, 7 October 2026 04:51Z: the row asks for a fault line at 60 s). pub const START_S: f64 = 60.0; -/// The start of a fault reason the node's readiness, not the card, explains: released when the node syncs. +/// The start of a restart reason the node's readiness, not the card, explains: the ladder resets when the node syncs. const NODE_FAULTS: [&str; 2] = ["no status line from the miner", "the worker gave no status within"]; +/// Seconds before the n-th restart of a card for a repeated fault (the last value repeats for ever). +pub const RETRY_LADDER_S: [u64; 4] = [10, 30, 120, 300]; +/// A node that has never answered since its start is given this long before the watchdog restarts it. +pub const NODE_STARTUP_CAP_S: f64 = 1800.0; +/// A worker that has not reported its program loaded (`ready`) this long after its start is restarted on the ladder. +/// The status clocks start at `ready`, never before (MF-4, 7 October 2026: two cards sat in "loading the program" +/// behind a third card's export storm and were faulted at 90 s). +pub const LOAD_S: f64 = 300.0; +/// A miner that exits (any code but 0, 42, 43, 44) within this many seconds of its start is in a crash loop: its +/// restarts follow the ladder instead of the 5 to 60 s jitter. +pub const EARLY_EXIT_S: f64 = 120.0; +/// A card whose worker failed its self-test is held this long before the next try (or until its driver changes). +pub const SELF_TEST_HOLD_S: u64 = 1800; + +/// The delay before restart number `attempt` (1-based) of a card: the ladder, then its last step for ever. +pub fn retry_delay_s(attempt: u32) -> u64 { + let i = (attempt.max(1) as usize - 1).min(RETRY_LADDER_S.len() - 1); + RETRY_LADDER_S[i] +} /// Hash rate 0 while the node is synced and the worker is ready for this long: restart the miner. pub const ZERO_RATE_S: f64 = 60.0; /// The miner is restarting its own worker: the app waits this long for `ready` before it steps in. @@ -50,6 +75,12 @@ pub struct Status { pub faults: u64, pub restarts: u64, pub synced: bool, + /// seconds the miner has waited for a template with none known (`template_wait=`, 0.3.20 miners; 0 before) + pub template_wait_s: f64, + /// the node's last template time in ms (`template_ms=`; 0 when the line has none) + pub template_ms: f64, + /// identities the miner fetches for now (`identities_active=`; 0 when the line has none) + pub identities_active: u64, } /// Parses a miner STATUS line; None for any other line. @@ -64,6 +95,9 @@ pub fn parse_status(text: &str) -> Option { faults: kv_u64(text, "faults").unwrap_or(0), restarts: kv_u64(text, "restarts").unwrap_or(0), synced: kv(text, "synced") == Some("true"), + template_wait_s: kv_f64(text, "template_wait").unwrap_or(0.0), + template_ms: kv_f64(text, "template_ms").unwrap_or(0.0), + identities_active: kv_u64(text, "identities_active").unwrap_or(0), }) } @@ -94,6 +128,10 @@ pub enum Event<'a> { /// The node became synced: a card faulted for silence while the node was not ready tries again (PC 1, 7 October /// 2026 04:51Z: every card faulted "no status line" during the post-relaunch sync and never came back). NodeSynced, + /// The worker failed its self-test (MF-4): the card is held for `SELF_TEST_HOLD_S` with the reason on its row. + SelfTestFailed(&'a str), + /// The card's driver or platform changed (a re-enumeration): a held card tries again now. + DriverChanged, /// The miner printed "template fetch timed out": it is alive and the node is not answering templates (PC 1, /// 7 October 2026 04:51Z: the node reported synced from the first second while its finality replay blocked the /// template RPC for three minutes; with no template the miner prints no status line). Silence is not the card's @@ -104,10 +142,9 @@ pub enum Event<'a> { #[derive(Debug, Clone, PartialEq)] pub enum Action { None, - /// Stop the miner and start it again now, for this reason. - Restart(String), - /// Mark the card faulted with this reason; do not restart its miner. - Fault(String), + /// Stop the miner and start it again after `delay_s`, for this reason (`attempt` counts restarts since the card + /// was last healthy for five minutes). + Restart { reason: String, delay_s: u64, attempt: u32 }, } #[derive(Debug, Default)] @@ -115,12 +152,17 @@ pub struct CardWatch { started_s: Option, last_status_s: Option, ready: bool, + /// when the worker reported its program loaded: the status clocks start here + ready_s: Option, + /// held after a self-test failure until the driver changes or the hold passes + driver_hold: bool, zero_since: Option, healthy_since: Option, worker_restart_since: Option, - /// app-level restarts without five healthy minutes since + /// app-level restarts without five healthy minutes since (the ladder position) restarts: u32, - faulted: Option, + /// the reason of the last restart (a node-caused one resets the ladder when the node syncs) + last_reason: String, last_fault: String, /// the miner's last line was a template timeout (the node not answering), cleared by the next status line templates_blocked: bool, @@ -138,10 +180,7 @@ impl CardWatch { Self::default() } - pub fn faulted(&self) -> Option<&str> { - self.faulted.as_deref() - } - + /// Restarts since the card was last healthy for five minutes: the ladder position. pub fn watchdog_restarts(&self) -> u32 { self.restarts } @@ -150,6 +189,7 @@ impl CardWatch { self.started_s = Some(now_s); self.last_status_s = None; self.ready = false; + self.ready_s = None; self.zero_since = None; self.healthy_since = None; self.worker_restart_since = None; @@ -162,14 +202,9 @@ impl CardWatch { self.zero_since = None; self.healthy_since = None; self.worker_restart_since = None; - if self.restarts >= 1 { - let r = format!("{reason} (restarted once already)"); - self.faulted = Some(r.clone()); - Action::Fault(r) - } else { - self.restarts += 1; - Action::Restart(reason) - } + self.restarts = self.restarts.saturating_add(1); + self.last_reason = reason.clone(); + Action::Restart { reason, delay_s: retry_delay_s(self.restarts), attempt: self.restarts } } pub fn event(&mut self, now_s: f64, ev: Event<'_>) -> Action { @@ -180,12 +215,22 @@ impl CardWatch { } Event::Ready => { self.ready = true; + self.ready_s = Some(now_s); + self.driver_hold = false; self.worker_restart_since = None; Action::None } Event::Status(s) => { - self.templates_blocked = false; self.last_status_s = Some(now_s); + if s.template_wait_s > 0.0 && s.hash_now <= 0.0 { + // MF-5: the miner is alive and waiting on the node for a template; a zero rate here is the + // node's latency, never the card's fault + self.templates_blocked = true; + self.zero_since = None; + self.healthy_since = None; + return Action::None; + } + self.templates_blocked = false; if s.hash_now > 0.0 { self.zero_since = None; let since = *self.healthy_since.get_or_insert(now_s); @@ -209,15 +254,35 @@ impl CardWatch { Action::None } Event::Exited(code) => { - let running = self.started_s.is_some(); + let started = self.started_s; self.started_s = None; - if code == MINER_GAVE_UP_CODE && running { + if code == MINER_GAVE_UP_CODE && started.is_some() { let why = if self.last_fault.is_empty() { "its guards tripped three times in ten minutes".to_string() } else { self.last_fault.clone() }; self.escalate(format!("the miner gave up on its worker: {why}")) } else { - Action::None + match started { + // a crash loop: the ladder, not the 5 to 60 s jitter (MF-4) + Some(t) if !matches!(code, 0 | 42 | 43 | PACK_OUT_OF_DATE_CODE) && now_s - t < EARLY_EXIT_S => { + self.escalate(format!("the miner exited with code {code} {:.0} s after starting", now_s - t)) + } + _ => Action::None, + } } } + Event::SelfTestFailed(reason) => { + self.started_s = None; + self.ready = false; + self.driver_hold = true; + self.last_reason = format!("not usable on this driver: {reason}"); + Action::Restart { reason: self.last_reason.clone(), delay_s: SELF_TEST_HOLD_S, attempt: self.restarts.max(1) } + } + Event::DriverChanged => { + if self.driver_hold { + self.driver_hold = false; + self.restarts = 0; + } + Action::None + } Event::Stopped => { self.started_s = None; self.last_status_s = None; @@ -241,24 +306,28 @@ impl CardWatch { Action::None } Event::NodeSynced => { - if self.faulted.as_deref().map(|f| NODE_FAULTS.iter().any(|p| f.starts_with(p))).unwrap_or(false) { - *self = Self::default(); + // restarts the node's readiness explained do not count against the card + if self.restarted_for_node() { + self.restarts = 0; + self.last_reason.clear(); } Action::None } } } - /// True when the card is faulted for a reason the node's readiness explains (released by Event::NodeSynced). - pub fn faulted_by_node(&self) -> bool { - self.faulted.as_deref().map(|f| NODE_FAULTS.iter().any(|p| f.starts_with(p))).unwrap_or(false) + /// True while the card waits out a self-test failure (released by a driver change or the hold's end). + pub fn driver_hold(&self) -> bool { + self.driver_hold } - /// Called every engine tick while the miner process is alive. - pub fn tick(&mut self, now_s: f64, node_synced: bool) -> Action { - if self.faulted.is_some() { - return Action::None; - } + /// True when the last restart was for a reason the node's readiness explains (released by Event::NodeSynced). + pub fn restarted_for_node(&self) -> bool { + NODE_FAULTS.iter().any(|p| self.last_reason.starts_with(p)) + } + + /// Called every engine tick while the miner process is alive. `node_ready`: synced AND an executed tip. + pub fn tick(&mut self, now_s: f64, node_ready: bool) -> Action { let Some(started) = self.started_s else { return Action::None }; if let Some(t) = self.worker_restart_since { // the miner is restarting its worker: its own guards own the card until the worker is ready @@ -267,8 +336,9 @@ impl CardWatch { } return Action::None; } + let node_synced = node_ready; if !node_synced { - // a miner is judged only while the node is synced: with no template it cannot print status, so the silence + // a miner is judged only while the node is ready: with no template it cannot print status, so the silence // clocks (start, no status, zero rate) hold at now until the node is back (PC 1, 7 October 2026) self.started_s = Some(now_s); if self.last_status_s.is_some() { @@ -277,10 +347,18 @@ impl CardWatch { self.zero_since = None; return Action::None; } - if !self.ready && self.last_status_s.is_none() && now_s - started > START_S { - return self.escalate(format!("the worker gave no status within {} s of starting", START_S as u64)); + // the status clocks start when the worker reports its program loaded (MF-4), never before; loading itself + // is bounded by LOAD_S + let Some(ready_at) = self.ready_s else { + if now_s - started > LOAD_S { + return self.escalate(format!("the worker did not load its program within {} s of starting", LOAD_S as u64)); + } + return Action::None; + }; + if self.last_status_s.is_none() && now_s - ready_at > START_S { + return self.escalate(format!("the worker gave no status within {} s of loading its program", START_S as u64)); } - let last = self.last_status_s.unwrap_or(started); + let last = self.last_status_s.unwrap_or(ready_at); if now_s - last > NO_STATUS_S { return self.escalate(format!("no status line from the miner for {} s", NO_STATUS_S as u64)); } @@ -306,10 +384,15 @@ impl NodeWatch { Self::default() } - /// `silent_s`: seconds since the last reading (or since the node started, when it never answered). Returns the - /// delay in seconds before the restart when one is due. - pub fn tick(&mut self, now_s: f64, ours: bool, silent_s: f64, accepted_recent: bool) -> Option { - if !ours || accepted_recent || silent_s < NODE_SILENT_S { + /// `silent_s`: seconds since the node's last sign of life (a watch reading, an exec status answer, an accepted + /// block; or since the node started, when it never answered). `settled`: the node has been read as synced at + /// least once since its start; before that its catch-up never counts, only `NODE_STARTUP_CAP_S` of total silence + /// does. Returns the delay in seconds before the restart when one is due. + pub fn tick(&mut self, now_s: f64, ours: bool, silent_s: f64, accepted_recent: bool, settled: bool) -> Option { + if !ours || accepted_recent { + return None; + } + if silent_s < if settled { NODE_SILENT_S } else { NODE_STARTUP_CAP_S } { return None; } self.restarts_s.retain(|t| now_s - *t <= NODE_WINDOW_S); @@ -456,7 +539,10 @@ mod tests { #[test] fn parses_status_and_fault_lines() { let s = parse_status(STATUS_OK).unwrap(); - assert_eq!(s, Status { hash_now: 124.10, mismatched: 0, faults: 0, restarts: 0, synced: true }); + assert_eq!(s, Status { hash_now: 124.10, mismatched: 0, faults: 0, restarts: 0, synced: true, template_wait_s: 0.0, template_ms: 0.0, identities_active: 0 }); + // a 0.3.20 miner waiting on a slow node (MF-5) + let w = parse_status("1791151000.000 STATUS 'win-1' [worker]: 120s jobs=0 accepted=0 rejected=0 fee=0 mismatched=0 extra=0 rate=0.00 blocks/s hash=0.00 MH/s wall (0.00 MH/s inside jobs) now=0.00 MH/s wall (0.00 MH/s inside jobs, 0 jobs, seed walk 0 calls) template_age=0.00s synced=true idle=100.0% (last 10s: 100.0%) queued=0 restarts=0 faults=0 identities=24 accepted_by_identity=0 tip_age_s=0 template_wait=37s template_ms=8120 identities_active=1").unwrap(); + assert_eq!((w.template_wait_s, w.template_ms, w.identities_active), (37.0, 8120.0, 1)); let m = parse_status(STATUS_MISMATCH).unwrap(); assert_eq!((m.mismatched, m.faults, m.restarts), (3, 1, 1)); assert_eq!(parse_status(STATUS_ZERO).unwrap().hash_now, 0.0); @@ -475,6 +561,19 @@ mod tests { } } + fn restart(a: &Action) -> (String, u64, u32) { + match a { + Action::Restart { reason, delay_s, attempt } => (reason.clone(), *delay_s, *attempt), + Action::None => panic!("expected a restart, got None"), + } + } + + #[test] + fn the_ladder() { + assert_eq!((1..=6).map(retry_delay_s).collect::>(), vec![10, 30, 120, 300, 300, 300]); + assert_eq!(retry_delay_s(0), 10); + } + #[test] fn healthy_miner_is_left_alone() { let mut w = CardWatch::new(); @@ -484,125 +583,99 @@ mod tests { assert_eq!(w.watchdog_restarts(), 0); } + /// the project lead's rule (7 October 2026): no fault is permanent. A zero rate restarts the miner on the ladder 10, 30, + /// 120, 300, 300 ... s, the reason stays in plain words, and the hash resumes on its own when the worker is back. #[test] - fn zero_rate_restarts_once_then_faults() { + fn zero_rate_restarts_on_the_ladder_for_ever_and_the_hash_resumes() { let mut w = CardWatch::new(); w.event(0.0, Event::Started); w.event(2.0, Event::Ready); healthy(&mut w, 10.0, 100.0); let z = parse_status(STATUS_ZERO).unwrap(); - for t in [110.0, 120.0, 130.0, 140.0, 150.0, 160.0] { - w.event(t, Event::Status(&z)); - assert_eq!(w.tick(t, true), Action::None, "under 60 s at {t}"); + let mut t = 110.0; + let mut seen = Vec::new(); + for expect in [(10u64, 1u32), (30, 2), (120, 3), (300, 4), (300, 5), (300, 6)] { + // the app restarts it after the delay; still zero: the next rung + w.event(t, Event::Started); + w.event(t + 2.0, Event::Ready); + let mut a = Action::None; + let mut k = 0.0; + while a == Action::None && k < 200.0 { + w.event(t + 10.0 + k, Event::Status(&z)); + a = w.tick(t + 10.0 + k, true); + k += 10.0; + } + let (reason, delay, attempt) = restart(&a); + assert!(reason.starts_with("hash rate 0 for 60 s"), "{reason}"); + assert!(!reason.contains("once already"), "the permanent state no longer exists: {reason}"); + assert_eq!((delay, attempt), expect, "rung {}", expect.1); + seen.push(delay); + t += 10.0 + k + delay as f64; } - w.event(170.0, Event::Status(&z)); - let a = w.tick(170.0, true); - assert!(matches!(a, Action::Restart(ref r) if r.contains("hash rate 0 for 60 s")), "{a:?}"); - // the app restarted it; still zero: faulted, not restarted again - w.event(172.0, Event::Started); - w.event(174.0, Event::Ready); - for t in [180.0, 190.0, 200.0, 210.0, 220.0, 230.0] { - w.event(t, Event::Status(&z)); - assert_eq!(w.tick(t, true), Action::None); - } - w.event(240.0, Event::Status(&z)); - let a = w.tick(240.0, true); - assert!(matches!(a, Action::Fault(ref r) if r.contains("restarted once already")), "{a:?}"); - assert!(w.faulted().is_some()); - // faulted stays: no more actions - w.event(250.0, Event::Status(&z)); - assert_eq!(w.tick(260.0, true), Action::None); - // the user changed the card's settings: a fresh start - w.event(300.0, Event::Reset); - assert!(w.faulted().is_none()); + assert_eq!(seen, vec![10, 30, 120, 300, 300, 300]); + // the worker is healthy again: five healthy minutes and the ladder starts over at 10 s + w.event(t, Event::Started); + w.event(t + 2.0, Event::Ready); + healthy(&mut w, t + 10.0, t + 320.0); + assert_eq!(w.watchdog_restarts(), 0); + let a = { let mut a = Action::None; let mut k = 0.0; while a == Action::None { w.event(t + 330.0 + k, Event::Status(&z)); a = w.tick(t + 330.0 + k, true); k += 10.0; } a }; + assert_eq!(restart(&a).1, 10); } #[test] - fn zero_rate_only_counts_with_a_synced_node_and_a_ready_worker() { + fn zero_rate_only_counts_with_a_ready_node_and_a_ready_worker() { let mut w = CardWatch::new(); w.event(0.0, Event::Started); let z = parse_status(STATUS_ZERO).unwrap(); // not ready yet (program loading): no zero timer - for t in [10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 70.0, 80.0] { + for t in [10.0, 20.0, 30.0, 40.0, 50.0] { w.event(t, Event::Status(&z)); assert_eq!(w.tick(t, true), Action::None); } - w.event(82.0, Event::Ready); - // node not synced: no zero timer either - for t in [90.0, 100.0, 110.0, 120.0, 130.0, 140.0, 150.0, 160.0] { + w.event(52.0, Event::Ready); + // node not ready (syncing, or no executed tip yet): no zero timer either + for t in [60.0, 70.0, 80.0, 90.0, 100.0, 110.0, 120.0, 130.0] { w.event(t, Event::Status(&z)); assert_eq!(w.tick(t, false), Action::None); } - assert_eq!(w.tick(170.0, true), Action::None, "the timer starts at the first zero status after ready"); - for t in [180.0, 190.0, 200.0, 210.0, 220.0, 230.0] { + assert_eq!(w.tick(140.0, true), Action::None, "the timer starts at the first zero status after ready"); + for t in [150.0, 160.0, 170.0, 180.0, 190.0, 200.0] { w.event(t, Event::Status(&z)); w.tick(t, true); } - assert!(matches!(w.tick(240.0, true), Action::Restart(_))); + assert!(matches!(w.tick(210.0, true), Action::Restart { .. })); } - /// PC 1, 7 October 2026 04:51:43Z (docs/plans/release-0.3.17.md, the PC 1 row): the app relaunched after the OTA - /// apply, started its workers while the node was still syncing, and every card ended "faulted: no status line from - /// the miner for 90 s (restarted once already)", pid 0, 0.0 MH/s from then on. Today's watchdog clocks the silence - /// while the node is not synced and never releases the fault. + /// PC 1, 7 October 2026: three workers started while the node caught up, printed nothing (no template), were + /// restarted once and then marked faulted for good. Now: the silence clocks hold while the node is not ready, a + /// restart the node explains does not climb the ladder once the node syncs, and nothing is ever permanent. #[test] - fn a_worker_started_while_the_node_syncs_is_not_faulted_and_a_faulted_one_returns_at_sync() { - // the recorded sequence (relaunch-engine-and-workers.txt): node synced=true from the first second, the miner - // started at +14 s, subscribed, then every template fetch timed out at 5 s for 180 s; no status line was printed - let mut r = CardWatch::new(); - r.event(14.0, Event::Started); - let mut t = 19.0; - while t <= 200.0 { - r.event(t, Event::TemplateTimeout); - assert_eq!(r.tick(t + 1.0, true), Action::None, "at {t}: a miner whose template fetches time out is not faulted on silence"); - assert!(r.templates_blocked()); - t += 5.0; - } - // the node answers templates again: the miner prints status and is healthy - healthy(&mut r, 210.0, 400.0); - assert!(!r.templates_blocked()); - assert_eq!(r.faulted(), None); - // a miner that stops printing even timeouts is judged again from its last line - let mut q = CardWatch::new(); - q.event(0.0, Event::Started); - q.event(50.0, Event::TemplateTimeout); - assert_eq!(q.tick(109.0, true), Action::None); - assert!(matches!(q.tick(111.0, true), Action::Restart(_)), "60 s of nothing at all after the last timeout line"); - + fn a_worker_started_while_the_node_catches_up_is_never_faulted() { let mut w = CardWatch::new(); w.event(0.0, Event::Started); - // the node reads unsynced for 200 s (the relaunch's sync): no verdict at all - let mut t = 5.0; - while t <= 200.0 { - assert_eq!(w.tick(t, false), Action::None, "at {t}: a miner is judged only while the node is synced"); - t += 10.0; + // the node is not ready for ten minutes: no action, no climb + let mut t = 1.0; + while t < 600.0 { + assert_eq!(w.tick(t, false), Action::None); + t += 1.0; } - // the node syncs; the worker answers inside 60 s and is healthy - assert_eq!(w.tick(210.0, true), Action::None); - w.event(230.0, Event::Ready); - healthy(&mut w, 240.0, 400.0); - assert_eq!(w.faulted(), None); - // the recorded state: a card faulted for silence while the node was not ready - let mut f = CardWatch::new(); - f.event(0.0, Event::Started); - assert!(matches!(f.tick(100.0, true), Action::Restart(_))); - f.event(101.0, Event::Started); - let a = f.tick(200.0, true); - assert!(matches!(a, Action::Fault(ref r) if (r.starts_with("no status line from the miner") || r.starts_with("the worker gave no status within")) && r.contains("restarted once already")), "{a:?}"); - assert!(f.faulted_by_node()); - f.event(300.0, Event::NodeSynced); - assert_eq!(f.faulted(), None, "the node syncing releases a silence fault"); - assert_eq!(f.watchdog_restarts(), 0); - // a fault the card owns (zero rate while synced) stays through NodeSynced - let mut z = CardWatch::new(); - z.event(0.0, Event::Started); - z.event(1.0, Event::Ready); - let zero = parse_status(STATUS_OK.replace("now=124.10 MH/s", "now=0.00 MH/s").as_str()).unwrap(); - for t in [10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 70.0, 80.0] { z.event(t, Event::Status(&zero)); z.tick(t, true); } - assert!(z.faulted().is_none() || !z.faulted_by_node()); - let before = z.faulted().map(|s| s.to_string()); - z.event(500.0, Event::NodeSynced); - assert_eq!(z.faulted().map(|s| s.to_string()), before); + // ready now, the worker has loaded but prints nothing: a restart after START_S, on rung 1 + w.event(t, Event::Ready); + let mut a = Action::None; + while a == Action::None && t < 700.0 { + a = w.tick(t, true); + t += 1.0; + } + let (reason, delay, attempt) = restart(&a); + assert!(reason.starts_with("the worker gave no status within"), "{reason}"); + assert_eq!((delay, attempt), (10, 1)); + assert!(w.restarted_for_node()); + // the node drops back to catching up and syncs again: the ladder resets for a node-caused restart + w.event(t, Event::NodeSynced); + assert_eq!(w.watchdog_restarts(), 0); + w.event(t + 10.0, Event::Started); + w.event(t + 12.0, Event::Ready); + healthy(&mut w, t + 20.0, t + 60.0); } #[test] @@ -611,25 +684,64 @@ mod tests { w.event(0.0, Event::Started); w.event(2.0, Event::Ready); healthy(&mut w, 10.0, 60.0); - // the miner goes silent (a stopped process, a hung RPC) assert_eq!(w.tick(149.0, true), Action::None); - let a = w.tick(151.0, true); - assert!(matches!(a, Action::Restart(ref r) if r.contains("no status line")), "{a:?}"); + let (reason, delay, _) = restart(&w.tick(151.0, true)); + assert!(reason.contains("no status line"), "{reason}"); + assert_eq!(delay, 10); } #[test] fn no_status_from_the_start() { - // a worker silent from its start, with the node synced, is judged at START_S (60 s), not NO_STATUS_S let mut w = CardWatch::new(); w.event(0.0, Event::Started); - assert_eq!(w.tick(59.0, true), Action::None); - assert!(matches!(w.tick(61.0, true), Action::Restart(ref r) if r.contains("gave no status within 60 s"))); + // loading: the status clock has not started; a worker that never loads is restarted at LOAD_S + assert_eq!(w.tick(290.0, true), Action::None); + let (reason, delay, _) = restart(&w.tick(301.0, true)); + assert!(reason.contains("did not load its program within 300 s"), "{reason}"); + assert_eq!(delay, 10); + // loaded at 200 s (a slow self-test behind another card's export): the 60 s status clock starts there + let mut w = CardWatch::new(); + w.event(0.0, Event::Started); + w.event(200.0, Event::Ready); + assert_eq!(w.tick(259.0, true), Action::None); + let (reason, _, _) = restart(&w.tick(261.0, true)); + assert!(reason.contains("gave no status within 60 s of loading"), "{reason}"); + } + + /// MF-4 (PC 1, 7 October 2026): a worker that fails its self-test is held for 30 minutes with the reason on its + /// row, not restarted every few seconds; a driver change releases it; a miner in a crash loop climbs the ladder. + #[test] + fn self_test_failure_is_held_and_a_crash_loop_climbs_the_ladder() { + let mut w = CardWatch::new(); + w.event(0.0, Event::Started); + let (reason, delay, _) = restart(&w.event(3.0, Event::SelfTestFailed("3 of 96 vectors mismatched"))); + assert_eq!(reason, "not usable on this driver: 3 of 96 vectors mismatched"); + assert_eq!(delay, SELF_TEST_HOLD_S); + assert!(w.driver_hold()); + w.event(100.0, Event::DriverChanged); + assert!(!w.driver_hold()); + // a crash loop: exits 2 s after each start climb 10, 30, 120 s + let mut w = CardWatch::new(); + let mut t = 0.0; + let mut delays = Vec::new(); + for _ in 0..3 { + w.event(t, Event::Started); + let (reason, delay, _) = restart(&w.event(t + 2.0, Event::Exited(3))); + assert!(reason.contains("exited with code 3"), "{reason}"); + delays.push(delay); + t += 2.0 + delay as f64; + } + assert_eq!(delays, vec![10, 30, 120]); + // an exit after a long healthy run is the engine's own jittered restart + let mut w = CardWatch::new(); + w.event(0.0, Event::Started); + w.event(2.0, Event::Ready); + healthy(&mut w, 10.0, 600.0); + assert_eq!(w.event(700.0, Event::Exited(3)), Action::None); } #[test] fn the_miners_own_worker_restart_is_not_doubled() { - // The gfx1036 fault: the miner prints WORKER FAULT, kills the worker, restarts it 2 s later; STATUS lines in - // between say now=0. The app must not restart the miner on top of that. let mut w = CardWatch::new(); w.event(0.0, Event::Started); w.event(2.0, Event::Ready); @@ -649,40 +761,66 @@ mod tests { w.event(t as f64, Event::Status(&z)); assert_eq!(w.tick(t as f64, true), Action::None); } - let a = w.tick(1091.0, true); - assert!(matches!(a, Action::Restart(ref r) if r.contains("did not come back within 180 s")), "{a:?}"); + let (reason, delay, _) = restart(&w.tick(1091.0, true)); + assert!(reason.contains("did not come back within 180 s"), "{reason}"); + assert_eq!(delay, 10); } #[test] - fn exit_43_once_then_faulted() { + fn exit_43_is_a_restart_on_the_ladder() { let mut w = CardWatch::new(); w.event(0.0, Event::Started); w.event(2.0, Event::Ready); healthy(&mut w, 10.0, 60.0); w.event(70.0, Event::WorkerRestart("cpu re-check: 3 consecutive mismatches (mismatched=3 in this run): the worker computes a wrong program")); - let a = w.event(75.0, Event::Exited(43)); - assert!(matches!(a, Action::Restart(ref r) if r.contains("gave up") && r.contains("wrong program")), "{a:?}"); - w.event(80.0, Event::Started); - let a = w.event(300.0, Event::Exited(43)); - assert!(matches!(a, Action::Fault(_)), "{a:?}"); - // an ordinary crash is the engine's own jittered restart, not the watchdog's + let (reason, delay, attempt) = restart(&w.event(75.0, Event::Exited(43))); + assert!(reason.contains("gave up") && reason.contains("wrong program"), "{reason}"); + assert_eq!((delay, attempt), (10, 1)); + w.event(85.0, Event::Started); + let (_, delay, attempt) = restart(&w.event(300.0, Event::Exited(43))); + assert_eq!((delay, attempt), (30, 2)); + // an exit 5 s after the start is the crash-loop class (MF-4): the ladder, not the 5 to 60 s jitter let mut w = CardWatch::new(); w.event(0.0, Event::Started); - assert_eq!(w.event(5.0, Event::Exited(1)), Action::None); + assert!(matches!(w.event(5.0, Event::Exited(1)), Action::Restart { delay_s: 10, .. })); } #[test] - fn five_healthy_minutes_renew_the_budget() { + fn template_timeouts_are_the_miners_heartbeat() { let mut w = CardWatch::new(); w.event(0.0, Event::Started); w.event(2.0, Event::Ready); - assert!(matches!(w.tick(100.0, true), Action::Restart(_))); - w.event(101.0, Event::Started); - w.event(103.0, Event::Ready); - healthy(&mut w, 110.0, 420.0); + healthy(&mut w, 10.0, 60.0); + // the node stops answering templates for four minutes while the app still reads it as ready + let mut t = 70.0; + while t < 300.0 { + w.event(t, Event::TemplateTimeout); + assert_eq!(w.tick(t + 1.0, true), Action::None, "a heartbeat at {t} is not silence"); + t += 5.0; + } + assert!(w.templates_blocked()); + healthy(&mut w, 310.0, 400.0); + assert!(!w.templates_blocked()); + } + + /// MF-5 (PC 1, 7 October 2026): the node answered templates past 5 s; the miner now prints STATUS every interval + /// with template_wait= while it waits, and the watchdog measures the worker, never the node. + #[test] + fn a_slow_node_never_faults_the_card() { + let mut w = CardWatch::new(); + w.event(0.0, Event::Started); + w.event(2.0, Event::Ready); + healthy(&mut w, 10.0, 60.0); + let slow = Status { hash_now: 0.0, template_wait_s: 8.0, template_ms: 8120.0, identities_active: 1, synced: true, ..Default::default() }; + let mut t = 70.0; + while t < 700.0 { + w.event(t, Event::Status(&slow)); + assert_eq!(w.tick(t, true), Action::None, "waiting on the node at {t} is not a fault"); + t += 10.0; + } + assert!(w.templates_blocked()); + healthy(&mut w, 710.0, 800.0); assert_eq!(w.watchdog_restarts(), 0); - // a second incident later is again a restart, not a fault - assert!(matches!(w.tick(520.0, true), Action::Restart(_))); } #[test] @@ -693,18 +831,26 @@ mod tests { w.event(30.0, Event::Stopped); assert_eq!(w.tick(500.0, true), Action::None); assert_eq!(w.event(500.0, Event::Exited(0)), Action::None); + w.event(600.0, Event::Reset); + assert_eq!(w.watchdog_restarts(), 0); } + /// The node's catch-up never counts against its watchdog (PC 1, 7 October 2026: a 40-second restart loop). #[test] - fn node_watch_restarts_a_silent_node_with_growing_delay() { + fn node_watch_waits_out_a_catch_up_and_restarts_a_dead_node_with_growing_delay() { let mut n = NodeWatch::new(); - assert_eq!(n.tick(100.0, true, 119.0, false), None); - assert_eq!(n.tick(100.0, false, 500.0, false), None, "an external node is never restarted"); - assert_eq!(n.tick(100.0, true, 500.0, true), None, "our block was accepted in the last minute: the node is alive"); - assert_eq!(n.tick(100.0, true, 120.0, false), Some(3)); - assert_eq!(n.tick(300.0, true, 120.0, false), Some(12)); - assert_eq!(n.tick(500.0, true, 120.0, false), Some(48)); + // catching up (never read as synced since its start): 25 minutes of silence is not a restart + assert_eq!(n.tick(100.0, true, 1500.0, false, false), None); + // a node that never answers at all: restarted at the 30-minute cap + assert_eq!(n.tick(100.0, true, 1801.0, false, false), Some(3)); + let mut n = NodeWatch::new(); + assert_eq!(n.tick(100.0, true, 119.0, false, true), None); + assert_eq!(n.tick(100.0, false, 500.0, false, true), None, "an external node is never restarted"); + assert_eq!(n.tick(100.0, true, 500.0, true, true), None, "our block was accepted in the last minute: the node is alive"); + assert_eq!(n.tick(100.0, true, 120.0, false, true), Some(3)); + assert_eq!(n.tick(300.0, true, 120.0, false, true), Some(12)); + assert_eq!(n.tick(500.0, true, 120.0, false, true), Some(48)); assert_eq!(n.restarts_in_window(), 3); - assert_eq!(n.tick(2000.0, true, 120.0, false), Some(3), "the window passed"); + assert_eq!(n.tick(2000.0, true, 120.0, false, true), Some(3), "the window passed"); } } diff --git a/app/igneum-app/ui/app.js b/app/igneum-app/ui/app.js index f9d34a2ca..f314f738b 100644 --- a/app/igneum-app/ui/app.js +++ b/app/igneum-app/ui/app.js @@ -311,8 +311,8 @@ var View = (function () { var on = !!cd.enabled && !removed && !unusable, st = removed ? 'removed' : unusable ? 'unusable' : (cd.state || 'off'); // the rule (docs/plans/ember-tune.md, 6 October 2026): a card never shows 0 MH/s without a reason word var tuning = st === 'tuning', held = st === 'held'; - var word = removed ? 'removed' : unusable ? ('not usable (' + cd.problem + ')') : !on ? 'off' : st === 'mining' ? 'mining' : tuning ? tuneWord(cd) : held ? (cd.message || 'paused for a job from the team') : st === 'resting' ? 'resting (heat mode)' : st === 'restarting' ? 'restart in ' + (cd.restart_in_s || 0) + ' s' : st === 'waiting' ? 'waiting for the node' : st === 'faulted' ? 'stopped after repeated errors' : st === 'failed' ? 'failed' : st === 'starting' ? 'starting' : st === 'ready' ? 'ready' : st; - var tone = removed ? 'off' : unusable ? 'bad' : !on ? 'off' : (st === 'mining' || tuning) ? 'on' : (st === 'failed' || st === 'faulted' || st === 'restarting') ? 'bad' : ''; + var word = removed ? 'removed' : unusable ? ('not usable (' + cd.problem + ')') : !on ? 'off' : st === 'mining' ? 'mining' : tuning ? tuneWord(cd) : held ? (cd.message || 'paused for a job from the team') : st === 'resting' ? 'resting (heat mode)' : st === 'restarting' ? 'restart in ' + (cd.restart_in_s || 0) + ' s' : st === 'waiting' ? 'waiting for the node' : st === 'failed' ? 'failed' : st === 'starting' ? 'starting' : st === 'ready' ? 'ready' : st; + var tone = removed ? 'off' : unusable ? 'bad' : !on ? 'off' : (st === 'mining' || tuning) ? 'on' : (st === 'failed' || st === 'restarting') ? 'bad' : ''; var hash = on && (st === 'mining' || tuning) ? (cd.hash_now >= 100 ? cd.hash_now.toFixed(0) : (cd.hash_now || 0).toFixed(1)) : ''; var temp = cd.temp_gpu > 0 ? Math.round(cd.temp_gpu) + ' °C' : ''; var tempTone = cd.temp_mem > 95 || cd.temp_gpu > 92 ? 'hot' : cd.temp_mem > 90 || cd.temp_gpu > 86 ? 'warm' : ''; diff --git a/app/igneum-app/ui/view.test.mjs b/app/igneum-app/ui/view.test.mjs index 636eff103..3feddfe3b 100644 --- a/app/igneum-app/ui/view.test.mjs +++ b/app/igneum-app/ui/view.test.mjs @@ -123,7 +123,8 @@ test('an integrated GPU is shown as integrated and off, with its reason', () => assert.equal(r.sub, 'integrated GPU: slow and shares the machine memory'); assert.equal(V.cardRow(card({ kind: 'unknown' })).canToggle, false); assert.equal(V.cardRow(card({ state: 'restarting', restart_in_s: 7 })).word, 'restart in 7 s'); - assert.equal(V.cardRow(card({ state: 'faulted' })).tone, 'bad'); + // no fault is permanent (7 October 2026, docs/plans/miner-faults.md): a restarting card is the bad tone, there is no faulted state + assert.equal(V.cardRow(card({ state: 'restarting', restart_in_s: 120, message: 'hash rate 0 for 60 s while the node is synced; trying again' })).tone, 'bad'); assert.equal(V.cardRow(card({ state: 'waiting' })).word, 'waiting for the node'); }); diff --git a/docs/plans/miner-faults.md b/docs/plans/miner-faults.md new file mode 100644 index 000000000..93af4cd8d --- /dev/null +++ b/docs/plans/miner-faults.md @@ -0,0 +1,43 @@ +# Miner fault-class register + +Started 7 October 2026 (the project lead, 11:2x UK: "we cannot have issues like this with the miner, it needs to be truly plug, +tune, play"). Every fault class found in the field gets a row the same day: the rule that makes the class impossible, +the test that proves the rule, and the gate line a cut must show before it publishes. A class is closed when all three +exist and the gate has fired once on a known-bad case and once on a known-good case (the watcher rule of 4 October). + +Standing rules behind every row (branch `miner-reliability`, off `release-0.3.19` db6f0964, for the 0.3.20 app; the miner rows on the fork branch `miner-reliability-20` off `release-0.3.20-node`): + +| Rule | Where | +|---|---| +| A worker never starts before the node is synced AND `igneum_getExecStatus` reports an executed tip; the node's catch-up never counts against its watchdog | `app/igneum-app/src/engine.rs` (`tick_exec_probe`, `node_ready`, `node_settled`), `src/watchdog.rs` (`NodeWatch::tick` with `settled`, `NODE_STARTUP_CAP_S`) | +| No fault is permanent: a restart waits 10 s, 30 s, 2 min, 5 min, then every 5 min, for ever; the reason stays on the card row in plain words; the hash resumes on its own; there is no "restarted once already" state | `src/watchdog.rs` (`RETRY_LADDER_S`, `retry_delay_s`, `Action::Restart { delay_s, attempt }`), `engine.rs` (`watchdog_verdict`, the pack give-up) | +| A card swap, a driver install or a restart needs no tap: the hot-plug enumeration every 60 s (`src/hotplug.rs`) starts the worker of a card that appears, recovers from a problem code or revives | `engine.rs` `merge_detection` → `plan_miners` | +| Every fault line reports to the log intake the moment it happens, with the card, the class, the reason and the app version | `engine.rs` `fault_report` → `update::upload_text`; label `fault--`; read with `node tools/logs.mjs` | +| Nothing on a user's machine is changed by a one-off script: card settings travel as the signed `cards` job kind (per card enabled, identities, power_pct), applied through the app's own card path, persisted, read back in the report, refused for a card the machine does not have | `src/jobs.rs` (`KINDS`, `validate_params`), `src/jobrun.rs` (`cards_job_choices`, `cards_applied`), `engine.rs` (`Action::ApplyCards`), `packaging/ota/publish-jobs.sh add --kind cards --cards "key=on:8"` | +| The fresh-install claim (LG-4) is a job, not a runbook: `tools/fleet/first-share-gate.mjs` on rented Windows boxes, on every cut, its line read by the shipper's publish | `relay/playbooks/first-share.ps1`, `tools/fleet/first-share-gate.mjs`, `site/evidence/first-share-.json` | + +## The register + +| Id | Found | Class (what the user saw) | Root cause | Rule | Test | Gate line | +|---|---|---|---|---|---|---| +| MF-1 | 7 Oct 2026, PC 1, 0.3.17 and 0.3.18 | The node restarted every 40 s; the card said "waiting for the node" for an hour | The app's clock sample, then its prover loop, called records-indexing exec RPCs (`eth_getBlockByNumber`, `igneum_getAssignedShards`) while the 0.3.17 node's exec follower held no record; the node panicked (`rpc.rs:591`, `rpc.rs:808`) | Every exec RPC call the app makes goes through `execrpc::call`; a method is safe on an empty state or gated on an executed tip, and an unclassified method is refused (shipper, `2dfb2e0c`). The node's whole records-indexing class is bounds-checked in the 0.3.20 node | `execrpc` unit test: every method literal in the tree is classified, no other file builds an exec request; `tools/reliability/app-run.mjs` step `catch-up`: a node with no executed tip gets no gated call and no worker for 120 s, no node restart | `app tests: execrpc callers classified` in the cut's plan; `app-run catch-up PASS` | +| MF-2 | 7 Oct 2026, PC 1, 0.3.18 | After a restart all three cards said "no status line from the miner for 90 s (restarted once already)" and never mined again until a cards-API bounce | The workers started the moment `synced` read true, before the node's catch-up (finality replay, exec follower) let templates flow; the watchdog's one-restart budget then marked them faulted for good | Workers start only when the node is READY (synced and an executed tip); silence while the node is not ready never counts; a restart the node's readiness explained resets the ladder at sync; the ladder never ends; the faulted state is gone | `watchdog` tests `a_worker_started_while_the_node_catches_up_is_never_faulted`, `zero_rate_restarts_on_the_ladder_for_ever_and_the_hash_resumes`, `node_watch_waits_out_a_catch_up...`; `app-run.mjs` steps `zero-ladder` (rungs 10, 30, 120 s observed, then the hash back on its own) and `catch-up` | `app tests: watchdog 16 green`; `app-run zero-ladder PASS`; the words "restarted once already" absent from `src/` (`tools/ci/forbidden-strings.txt`) | +| MF-4 | 7 Oct 2026, PC 1, Arc B580 beside a 5090 and a 9070 XT | One card's worker failed its self-test and restarted every few seconds; the other two cards sat in "loading the program" past the 90-second watchdog and were faulted | Every restart of the failing card exported the pack again under the export lock; the healthy cards' exports queued behind it; the watchdog's status clock ran from the process start, not from "program loaded" | A card's worker failure never blocks another card: the pack is exported once per minute for every card (`EXPORT_REUSE_S`; a refused pack forces one); a worker that fails its self-test is held 30 minutes with "not usable on this driver: " on its row and tried again on a driver change (`SelfTestFailed`, `DriverChanged`); a miner that exits inside 120 s of its start restarts on the ladder (10 s, 30 s, 2 min, 5 min), never every few seconds; the status clock starts at `ready` (program loaded), and loading itself is bounded by 300 s | `watchdog` tests `self_test_failure_is_held_and_a_crash_loop_climbs_the_ladder`, `no_status_from_the_start` (the clock from `ready`); `app-run.mjs` step `one-card-fails` (three cards, one failing for ever: the two mine on time, the third is held, at most 6 exports) | `app-run one-card-fails PASS` | +| MF-5 | 7 Oct 2026, PC 1, 0.3.17 node, 24 identities (the real cause of the 11:2x faults; MF-4 withdrawn as the cause) | Every card faulted "no status line from the miner for 90 s (restarted once already)" after the restart | Evidence (PC 1, 7 Oct 2026 11:4x UK): with 8 identities a card (24 template fetches a round) the 0.3.17 node answered no template in 5 s; with 2 a card (4 fetches) both cards mined at full rate (5090 122.4 MH/s, 9070 XT 18.9) within two minutes. The node's getBlockTemplate answered past 5 s with 24 identities fetching; the miners waited for a template inside their job-fill loop and printed no STATUS at all; the watchdog read the silence as the worker's; one restart, then faulted for good | Miner: STATUS every interval whatever the template state (`template_wait=` while it waits, `template_ms=` the node's last template time, `identities_active=`); the feed fetches only as many identities as fit one pass inside 8 s at the node's measured template time (`identities_for`: all of them when the node answers under 1 s; 24 at 8 s per template becomes 1), raised again when it answers faster; a pass whose fetches all fail backs off 2, 4, 8, 10 s and retries for ever with a `NODE SLOW` line once per 30 s. App: a STATUS with `template_wait>0` is the miner's heartbeat and the node's latency, never the card's fault (no zero-rate clock, no restart); the card reads "node slow: waiting for a block template for N s; the worker is kept" or "node slow: a template takes N s; k of n identities active"; the mitigation of the day (a one-off script POSTing /api/cards) is closed by the signed `cards` job kind | `watchdog` tests `a_slow_node_never_faults_the_card`, `parses_status_and_fault_lines` (the 0.3.20 line); `jobs` test for the `cards` kind; injector step `slow-node` (a template stub answering in 8 s while three cards run: open, needs the stub) | `app-run slow-node PASS` (0.3.20) | +| MF-6 | 7 Oct 2026, PC 1, 0.3.19 | After a `--stop-miners` job, a following read-only job kept both cards "off, held for a remote job" for its whole three minutes | The engine released the hold only when no job held the miners; the next job's active state hid the release (engine.rs 3032 class) | A hold belongs to the job that took it (`job_hold_owner`) and releases the moment that job is no longer the running one, whatever runs next, or when its own cap passes (logged); a read-only job never holds (`jobrun::hold_release`) | `jobrun` test `a_hold_belongs_to_the_job_that_took_it` (owner running, another job, no job, cap passed, no owner) | `app tests: hold rule green` | +| MF-7 | 7 Oct 2026, PC 1, 0.3.19 | Orphan `igneum-miner.exe` processes the app no longer tracked (two alive under `--stop-miners` with their rows at pid 0, one after) hammered the node's template RPC beside the tracked miners | The engine lost track of miners it had started (a stop that timed out, a restart over a live process) and never looked for them again | The engine owns every miner it started: at start, after every stop and every minute it kills any `igneum-miner` whose command line carries THIS engine's node RPC (the fence) and whose pid it does not track, one log line and one fault report per kill, never by name alone (`sweep_orphan_miners`, `platform::miner_processes`, `kill_pid`); a restart kills the slot's old process before the new one starts | injector step `orphan-miner` (a stray miner on the engine's node is killed inside the minute, the engine's own miner left alone) | `app-run orphan-miner PASS` | +| MF-3 | 7 Oct 2026, PC 1, Intel Arc | The Intel driver's first install did not bind: the device sat in Code 12 at install time; the card never mined until a reboot | A driver installed while the device reports a problem code (12, 43, 31) does not bind; nothing re-scanned the device afterwards, and the app only re-enumerates | The app re-enumerates every 60 s and starts the worker the minute the OS drives the card (`hotplug::diff` recovered / revived, `settle_new`); the row says what to do while it does not ("reboot with the card attached; if it persists, reinstall the driver with the card attached"); a Windows host asks for a re-scan (`pnputil /scan-devices`) after a problem code is seen, every 5 minutes, at most 6 times (follow-up, host side) | `hotplug` test `a_driven_card_that_turns_faulty_is_errored_and_recovers_later`; `app-run.mjs` step `card-appears` (a card listed after 2 minutes starts without a tap) | `app-run card-appears PASS` | + +## How a row is added + +1. The day the class is seen: the row with Found, Class, Root cause. The rule, the test and the gate line the same day + where they exist; "open" where they do not, with the owner. +2. The test is a unit test on recorded lines (`src/watchdog.rs`, `src/hotplug.rs`, `src/execrpc.rs`) or a step of the + fault injector (`tools/reliability/app-run.mjs` against `fake-worker.mjs`), run on the box. +3. The gate line is what the cut's plan (`docs/plans/release-.md`) must carry before publish; the shipper reads it. + +## The fault injector (box) + +`tools/reliability/app-run.mjs` drives a scratch engine (its own private node, the fake worker in place of the GPU +worker, Linux or macOS) through one step per class and prints `PASS`/`FAIL` with the seconds. The box runs it with +the Linux engine and node from `target-remote/`: `tools/reliability/box-run.sh` (the lock is the box's run slot). diff --git a/packaging/ota/publish-jobs.sh b/packaging/ota/publish-jobs.sh index 3d8146e2d..d659f53b5 100755 --- a/packaging/ota/publish-jobs.sh +++ b/packaging/ota/publish-jobs.sh @@ -20,6 +20,10 @@ # [--to name] [--extract] [--extract-dir sub] [--fresh] (or --url https://... --sha256 ... [--size N]) # packaging/ota/publish-jobs.sh add --kind collect --target all --glob "logs/app-*.log" [--glob ...] [--command "nvidia-smi"] # packaging/ota/publish-jobs.sh add --kind restart --target 1ccfe586 --what miners|node|app +# packaging/ota/publish-jobs.sh add --kind cards --target 1ccfe586 --cards "nvidia:0:NVIDIA GeForce RTX 5090=on:8,amd::Radeon=off" +# (the signed card settings kind, 7 October 2026: per card, on|off and the identities after the colon, power_pct +# after a second colon; a card the machine does not have is refused; applied through the app's own card path +# and persisted; the report reads every named card back) # packaging/ota/publish-jobs.sh add --kind update-now --target all # packaging/ota/publish-jobs.sh add --kind shard-benchmark --target 1ccfe586 [--zip ~/Desktop/igneum-prove-wsl2.zip] \ # [--fixtures "block-338-shard1 block-341-shards2 block-344-shards4"] [--cap-minutes 90] [--distro Ubuntu-24.04] [--wsl-user [user]] @@ -56,7 +60,7 @@ CMD="${1:-}"; [ $# -gt 0 ] && shift KIND="" TARGET="" PLATFORM="" REQUIRES="" REQUIRES_SET=0 ID="" TITLE="" EXPIRES_H="48" DEPLOY=0 BASE="" DEST="" TRIES=12 SCRIPT="" SHELL_KIND="" ELEVATED=0 STOP_MINERS=0 TIMEOUT_MIN="" CARDS_OFF="" CARDS_LEAVE_OFF=0 FILE="" URL="" SHA="" SIZE="" DIR="" TO="" EXTRACT=0 EXTRACT_DIR="" FRESH=0 -GLOBS=() COMMAND="" WHAT="" +GLOBS=() COMMAND="" WHAT="" CARDS="" ZIP="" FIXTURES="" CAP_MIN="" DISTRO="" WSL_USER="" REMOVE_ID="" TARGETS="" BUDGET_MIN="" STAGE_MIN="" MIN_FREE_GB="" TESTS=1 RELAY_URL="" NICE="" CARGO_JOBS="" case "$CMD" in @@ -90,6 +94,7 @@ while [ $# -gt 0 ]; do --glob) GLOBS+=("$2"); shift 2 ;; --command) COMMAND="$2"; shift 2 ;; --what) WHAT="$2"; shift 2 ;; + --cards) CARDS="$2"; shift 2 ;; --zip) ZIP="$2"; shift 2 ;; --fixtures) FIXTURES="$2"; shift 2 ;; --cap-minutes) CAP_MIN="$2"; shift 2 ;; @@ -314,6 +319,22 @@ print(json.dumps(d))' "$COMMAND" "${GLOBS[@]:-}")" update-now) [ -n "$TITLE" ] || TITLE="update now" ;; + cards) + [ -n "$CARDS" ] || { echo "cards: --cards \"key=on:8,key=off\"" >&2; exit 2; } + PARAMS="$(python3 -c 'import json,sys +out=[] +for item in [x for x in sys.argv[1].split(",") if x.strip()]: + key, _, v = item.partition("=") + parts = v.split(":") + d = {"key": key.strip()} + if parts[0] in ("on", "off"): d["enabled"] = parts[0] == "on" + elif parts[0]: raise SystemExit("cards: %s: on|off expected, got %s" % (key, parts[0])) + if len(parts) > 1 and parts[1]: d["identities"] = int(parts[1]) + if len(parts) > 2 and parts[2]: d["power_pct"] = int(parts[2]) + out.append(d) +print(json.dumps({"cards": out}))' "$CARDS")" + [ -n "$TITLE" ] || TITLE="card settings" + ;; shard-benchmark) [ -n "$ZIP" ] || ZIP="$HOME/Desktop/igneum-prove-wsl2.zip" read -r ZURL ZSHA ZSIZE < <(hosted_file "$ZIP") diff --git a/relay/playbooks/first-share.ps1 b/relay/playbooks/first-share.ps1 new file mode 100644 index 000000000..ac0ce4028 --- /dev/null +++ b/relay/playbooks/first-share.ps1 @@ -0,0 +1,87 @@ +# Igneum playbook: one fresh Windows install timed to its first share or block (launch gate LG-4, +# docs/plans/miner-ui-5-first-share-runbook.md; the job form of it: docs/plans/miner-faults.md). +# +# powershell -ExecutionPolicy Bypass -File first-share.ps1 -Url [-Address 0x...] [-BudgetSeconds 1800] +# [-AllowInstalled] +# +# Runs on a FRESH Windows box only: a machine with the Igneum Miner installed is refused unless -AllowInstalled is given +# (a fresh-install measurement replaces the installed app and wipes %LOCALAPPDATA%\igneum; on a machine that mines for +# someone that is their call, never the job's: the rule of 5 October 2026). Prints the six RESULT lines of the runbook, +# then uninstalls and wipes so the next run on the same box is fresh again. Reads only the engine it installed (its own +# app.url); never pauses, resumes or quits another engine (tools/ci/playbook-quit-check.sh). +param( + [Parameter(Mandatory = $true)][string]$Url, + [string]$Address = '0x4242424242424242424242424242424242424242', + [int]$BudgetSeconds = 1800, + [switch]$AllowInstalled +) +$ErrorActionPreference = 'Stop' +function Now { [int64]([DateTimeOffset]::UtcNow.ToUnixTimeMilliseconds()) } +function Result([string]$line) { Write-Host ("RESULT " + $line) } +$appData = Join-Path $env:LOCALAPPDATA 'igneum' +$programs = Join-Path $env:LOCALAPPDATA 'Programs\Igneum Miner' +if ((Test-Path $appData) -or (Test-Path $programs)) { + if (-not $AllowInstalled) { Result "total_s=0 pass=false reason=installed_app_present (this is not a fresh box; -AllowInstalled overrides, with the owner's word)"; exit 2 } + Get-Process -Name 'igneum-app', 'Igneum Miner', 'igneumd', 'igneum-miner' -ErrorAction SilentlyContinue | Stop-Process -Force -ErrorAction SilentlyContinue + Start-Sleep -Seconds 3 + Remove-Item -Recurse -Force $appData -ErrorAction SilentlyContinue +} +# 1. download: the public installer through the dl host, timed from the first byte +$setup = Join-Path $env:TEMP 'Igneum-Miner-Setup-first-share.exe' +Remove-Item $setup -ErrorAction SilentlyContinue +$tDl0 = Now +Invoke-WebRequest -Uri $Url -OutFile $setup -UseBasicParsing +$tDl1 = Now +$bytes = (Get-Item $setup).Length +$sha = (Get-FileHash $setup -Algorithm SHA256).Hash.ToLower() +Result "step=download start=$tDl0 end=$tDl1 bytes=$bytes sha256=$sha" +# 2. install (Inno Setup, per user, silent), then the engine's first screen = app.url present and api/state answering +$p = Start-Process -FilePath $setup -ArgumentList '/VERYSILENT', '/SUPPRESSMSGBOXES', '/NORESTART', '/SP-' -Wait -PassThru +if ($p.ExitCode -ne 0) { Result "step=install end=$(Now) exit=$($p.ExitCode)"; Result "total_s=$([int](((Now) - $tDl0) / 1000)) pass=false reason=installer_exit_$($p.ExitCode)"; exit 3 } +$exe = Get-ChildItem -Path $programs -Filter 'Igneum Miner.exe' -Recurse -ErrorAction SilentlyContinue | Select-Object -First 1 +if (-not $exe) { $exe = Get-ChildItem -Path $programs -Filter 'igneum-app.exe' -Recurse -ErrorAction SilentlyContinue | Select-Object -First 1 } +if (-not $exe) { Result "total_s=$([int](((Now) - $tDl0) / 1000)) pass=false reason=no_app_exe_after_install"; exit 3 } +Start-Process -FilePath $exe.FullName | Out-Null +$urlFile = Join-Path $appData 'app\app.url' +$url = $null +$deadline = (Now) + 180000 +while ((Now) -lt $deadline) { + if (Test-Path $urlFile) { $url = (Get-Content $urlFile -Raw).Trim(); if ($url) { try { $null = Invoke-RestMethod -Uri ($url + 'api/state') -TimeoutSec 5; break } catch {} } } + Start-Sleep -Milliseconds 500 +} +if (-not $url) { Result "total_s=$([int](((Now) - $tDl0) / 1000)) pass=false reason=engine_not_up_in_180s"; exit 4 } +$tInstall = Now +Result "step=install end=$tInstall" +# 3. setup: the address the owner pasted, then start (what the welcome screens post) +try { $null = Invoke-RestMethod -Uri ($url + 'api/setup') -Method Post -ContentType 'application/json' -Body (@{ mode = 'paste'; address = $Address } | ConvertTo-Json) -TimeoutSec 10 } catch { Write-Host "api/setup: $_" } +try { $null = Invoke-RestMethod -Uri ($url + 'api/key/saved') -Method Post -TimeoutSec 10 } catch {} +try { $null = Invoke-RestMethod -Uri ($url + 'api/start') -Method Post -TimeoutSec 10 } catch { Write-Host "api/start: $_" } +$tSetup = Now +Result "step=setup end=$tSetup" +# 4. node synced, 5. the first accepted block or share (ladder.first_block_at), within the budget +$synced = $null; $first = $null; $stall = 'node'; $mining = $null; $hash = '' +$deadline = $tDl0 + ($BudgetSeconds * 1000) +while ((Now) -lt $deadline) { + try { $st = Invoke-RestMethod -Uri ($url + 'api/state') -TimeoutSec 5 } catch { Start-Sleep -Seconds 2; continue } + if (-not $synced -and $st.node.synced) { $synced = Now; $stall = 'card'; Result "step=node end=$synced synced_at=$($st.ladder.first_synced_at)" } + if ($st.ladder -and $st.ladder.first_mining_at -and -not $mining) { $mining = $st.ladder.first_mining_at } + if ($st.ladder -and $st.ladder.first_block_at) { $first = Now; $hash = $st.ladder.first_block_hash; break } + if ($st.ladder -and $st.ladder.first_share_at) { $first = Now; $hash = 'share'; break } + Start-Sleep -Seconds 2 +} +$total = [int](((Now) - $tDl0) / 1000) +if ($first) { + Result "step=card end=$first mining_at=$mining first_block_at=$first hash=$hash" + $pass = ($total -le 600) + Result "total_s=$total pass=$($pass.ToString().ToLower())" +} else { + Result "total_s=$total pass=false stalled_in=$stall" +} +# 6. fresh again: quit the engine this job installed (its own URL), uninstall, wipe +try { $null = Invoke-RestMethod -Uri ($url + 'api/quit') -Method Post -TimeoutSec 10 } catch {} +Start-Sleep -Seconds 5 +$unins = Get-ChildItem -Path $programs -Filter 'unins*.exe' -Recurse -ErrorAction SilentlyContinue | Select-Object -First 1 +if ($unins) { Start-Process -FilePath $unins.FullName -ArgumentList '/VERYSILENT', '/SUPPRESSMSGBOXES', '/NORESTART' -Wait | Out-Null } +Remove-Item -Recurse -Force $appData -ErrorAction SilentlyContinue +Remove-Item $setup -ErrorAction SilentlyContinue +if ($first -and $total -le 600) { exit 0 } else { exit 1 } diff --git a/tools/ci/permanent-fault-check.sh b/tools/ci/permanent-fault-check.sh new file mode 100755 index 000000000..fd304c8c9 --- /dev/null +++ b/tools/ci/permanent-fault-check.sh @@ -0,0 +1,28 @@ +#!/usr/bin/env bash +# No fault is permanent (the project lead, 7 October 2026, docs/plans/miner-faults.md MF-2): the app's engine may never mark a card +# "faulted" or say "restarted once already"; every fault retries on the ladder. This check fails when either word comes +# back into the engine's sources or its UI. +# tools/ci/permanent-fault-check.sh the tree +# tools/ci/permanent-fault-check.sh --self-test fires on a fixture with the words, passes on one without +set -uo pipefail +cd "$(git rev-parse --show-toplevel)" || exit 1 +check() { + # $1: directory; a hit is a line in a .rs or .js file that sets or tests the faulted state or carries the words + grep -rnE --include='*.rs' --include='*.js' -e 'restarted once already' -e '"faulted"\.into\(\)' -e "state === 'faulted'" -e 'c\.state = "faulted"' "$1" 2>/dev/null +} +if [ "${1:-}" = "--self-test" ]; then + d="$(mktemp -d)"; trap 'rm -rf "$d"' EXIT + mkdir -p "$d/bad" "$d/good" + printf 'fn x() { c.state = "faulted".into(); }\n' > "$d/bad/a.rs" + printf 'fn x() { c.state = "restarting".into(); }\n' > "$d/good/a.rs" + if ! check "$d/bad" >/dev/null; then echo "self-test failed: the fixture with the faulted state passed"; exit 1; fi + if check "$d/good" >/dev/null; then echo "self-test failed: the clean fixture was flagged"; exit 1; fi + echo "self-test ok"; exit 0 +fi +hits="$(check app/igneum-app/src; check app/igneum-app/ui)" +if [ -n "$hits" ]; then + echo "permanent fault words in the engine (docs/plans/miner-faults.md MF-2: no fault is permanent):" + echo "$hits" + exit 1 +fi +exit 0 diff --git a/tools/ci/pre-push.sh b/tools/ci/pre-push.sh index b7b8d8afb..efc61d988 100755 --- a/tools/ci/pre-push.sh +++ b/tools/ci/pre-push.sh @@ -69,6 +69,7 @@ tree_checks() { run "run jobs test their fetched kit before use" bash -c 'bash tools/ci/kit-path-check.sh --self-test && bash tools/ci/kit-path-check.sh' run "every Windows spawn of the app runs with a hidden console" bash -c 'node tools/ci/windows-spawn-check.mjs --self-test && node tools/ci/windows-spawn-check.mjs' run "pinned guest programs match their manifest" bash tools/ci/pinned-guests-check.sh + run "no permanent miner fault in the engine (miner-faults.md MF-2)" bash -c 'bash tools/ci/permanent-fault-check.sh --self-test && bash tools/ci/permanent-fault-check.sh' run "root prover playbooks kill the GPU server and unlink its socket" bash tools/ci/prover-socket-check.sh run "commit-string gate self-test" bash tools/ci/commit-string-check.sh --self-test run "build server remote checkout self-test" bash infra/build-server/remote-run.sh --self-test diff --git a/tools/fleet/first-share-gate.mjs b/tools/fleet/first-share-gate.mjs new file mode 100755 index 000000000..57f16c8d4 --- /dev/null +++ b/tools/fleet/first-share-gate.mjs @@ -0,0 +1,92 @@ +#!/usr/bin/env node +// Launch gate LG-4 as a job: "first share within 10 minutes of download in 9 of 10 fresh Windows installs" +// (docs/plans/miner-ui-5-first-share-runbook.md, docs/plans/miner-faults.md). Runs relay/playbooks/first-share.ps1 on +// Windows boxes over SSH, ten installs in all, and writes the evidence table the shipper's publish reads. +// +// node tools/fleet/first-share-gate.mjs run --version 0.3.20 --url https://dl.igneum.network//Igneum-Miner-Setup-0.3.20.exe +// [--boxes ~/igneum-fleet/windows-boxes.json] [--installs 10] [--allow-installed] +// node tools/fleet/first-share-gate.mjs check --version 0.3.20 exit 0 on a PASS line for that version, 1 otherwise +// +// Boxes file: a JSON array of the fleet registry's Windows rows ({kind:"windows", label, host, port, user, key, transport:"ssh"}); +// with no row the gate writes "LG-4 not run: no Windows box" and `check` fails. Each box runs installs in turn; boxes run in +// parallel. Output: site/evidence/first-share-.json (the runbook's keys per row, the medians, the pass count, the +// installer's sha256) and the gate line on stdout and in the file: "LG-4 PASS k/10" or "LG-4 FAIL k/10". +// Never touches PC 1 (the project lead's desk) and refuses a box whose installed app would be replaced unless --allow-installed is given +// and the box's row says "job_box": true (PC 2's case needs the project lead's word, recorded in the row by the fleet lane). + +import { readFileSync, writeFileSync, existsSync, mkdirSync } from 'node:fs'; +import { spawnSync } from 'node:child_process'; +import { homedir } from 'node:os'; +import { join, dirname, resolve } from 'node:path'; +import { fileURLToPath } from 'node:url'; + +const ROOT = resolve(dirname(fileURLToPath(import.meta.url)), '..', '..'); +const args = process.argv.slice(2); +const cmd = args[0]; +const opt = (n, d) => { const i = args.indexOf(n); return i >= 0 ? args[i + 1] : d; }; +const flag = (n) => args.includes(n); +const version = opt('--version'); +if (!version) { console.error('need --version'); process.exit(2); } +const evidence = join(ROOT, 'site', 'evidence', `first-share-${version}.json`); + +function gateLine(rows, installs) { + const passed = rows.filter(r => r.pass).length; + return `LG-4 ${passed >= Math.ceil(installs * 0.9) ? 'PASS' : 'FAIL'} ${passed}/${installs}`; +} + +if (cmd === 'check') { + if (!existsSync(evidence)) { console.log(`LG-4 not run for ${version}: no ${evidence}`); process.exit(1); } + const e = JSON.parse(readFileSync(evidence, 'utf8')); + console.log(e.gate); + process.exit(/^LG-4 PASS/.test(e.gate) ? 0 : 1); +} +if (cmd !== 'run') { console.log(readFileSync(fileURLToPath(import.meta.url), 'utf8').split('\n').slice(1, 16).map(l => l.replace(/^\/\/ ?/, '')).join('\n')); process.exit(2); } + +const url = opt('--url'); +if (!url) { console.error('need --url (the public installer)'); process.exit(2); } +const installs = Number(opt('--installs', '10')); +const boxesFile = opt('--boxes', join(homedir(), 'igneum-fleet', 'windows-boxes.json')).replace(/^~/, homedir()); +const boxes = existsSync(boxesFile) ? JSON.parse(readFileSync(boxesFile, 'utf8')).filter(b => b.kind === 'windows' && b.transport === 'ssh' && b.host) : []; +mkdirSync(dirname(evidence), { recursive: true }); +if (!boxes.length) { + const out = { version, url, date: new Date().toISOString(), installs, rows: [], gate: 'LG-4 not run: no Windows box', note: 'the fleet lane has no rented Windows GPU box; PC 2 over the relay needs the owner\'s word (docs/plans/miner-faults.md)' }; + writeFileSync(evidence, JSON.stringify(out, null, 2) + '\n'); + console.log(out.gate); + process.exit(1); +} +for (const b of boxes) { + if (/ae432dc7|pc1|pc-1/i.test(`${b.label} ${b.machine || ''}`)) { console.error(`refusing ${b.label}: PC 1 is never a job box`); process.exit(2); } +} +const allowInstalled = flag('--allow-installed'); +const parse = (text) => { + const r = { pass: false }; + for (const l of text.split('\n')) { + const m = /^RESULT (.*)$/.exec(l.trim()); if (!m) continue; + for (const kv of m[1].split(/\s+/)) { const [k, v] = kv.split('='); if (k === 'step') r._step = v; else if (r._step && k !== 'step') r[`${r._step}_${k}`] = v; else r[k] = v; } + } + r.pass = r.pass === 'true'; + r.total_s = Number(r.total_s || 0); + return r; +}; +const rows = []; +const perBox = Math.ceil(installs / boxes.length); +const started = Date.now(); +// boxes in parallel (one ssh each, installs in turn inside it), the Mac waits +const procs = boxes.map((b) => { + const n = Math.min(perBox, installs - rows.length); + const remote = `$ErrorActionPreference='Continue'; $f=\"$env:TEMP\\first-share.ps1\"; [IO.File]::WriteAllText($f, [Text.Encoding]::UTF8.GetString([Convert]::FromBase64String('${Buffer.from(readFileSync(join(ROOT, 'relay', 'playbooks', 'first-share.ps1'))).toString('base64')}'))); 1..${n} | ForEach-Object { Write-Host \"INSTALL $_\"; powershell -ExecutionPolicy Bypass -File $f -Url '${url}' ${allowInstalled && b.job_box ? '-AllowInstalled' : ''} }`; + const key = (b.key || '~/.ssh/igneum-fleet').replace(/^~/, homedir()); + const p = spawnSync('ssh', ['-o', 'ConnectTimeout=20', '-o', 'StrictHostKeyChecking=accept-new', '-i', key, '-p', String(b.port || 22), `${b.user}@${b.host}`, 'powershell', '-NoProfile', '-Command', remote], { encoding: 'utf8', timeout: (n * 1900 + 120) * 1000 }); + return { box: b, out: (p.stdout || '') + (p.stderr || ''), status: p.status }; +}); +for (const { box, out } of procs) { + for (const chunk of out.split(/^INSTALL \d+\r?$/m).slice(1)) { + const r = parse(chunk); r.box = box.label; r.gpu = box.gpu || ''; rows.push(r); + } + if (!out.includes('RESULT')) rows.push({ box: box.label, gpu: box.gpu || '', pass: false, total_s: 0, error: out.trim().split('\n').slice(-3).join(' | ') }); +} +const med = (k) => { const v = rows.map(r => Number(r[k])).filter(x => x > 0).sort((a, b) => a - b); return v.length ? v[Math.floor(v.length / 2)] : null; }; +const out = { version, url, date: new Date().toISOString(), installs, boxes: boxes.map(b => b.label), rows, median_total_s: med('total_s'), pass_count: rows.filter(r => r.pass).length, sha256: rows.find(r => r.download_sha256)?.download_sha256 || '', gate: gateLine(rows, installs), took_s: Math.round((Date.now() - started) / 1000) }; +writeFileSync(evidence, JSON.stringify(out, null, 2) + '\n'); +console.log(out.gate); +process.exit(/^LG-4 PASS/.test(out.gate) ? 0 : 1); diff --git a/tools/reliability/app-run.mjs b/tools/reliability/app-run.mjs index 2b9ef375f..48dab7bc3 100755 --- a/tools/reliability/app-run.mjs +++ b/tools/reliability/app-run.mjs @@ -1,24 +1,34 @@ #!/usr/bin/env node -// The app engine's watchdog measured end to end on a Mac: a scratch Igneum Miner engine with its own private node -// (ports 29960 and 29961, devnet suffix 9960, no peers, unsynced mining allowed, so the node counts as synced) and -// fake-worker.mjs standing in for the Metal worker, driven through the control file and with signals. +// The app engine's fault injector (docs/plans/miner-faults.md): a scratch Igneum Miner engine with its own private node +// (ports 29960 and 29961, devnet suffix 9960, no peers, unsynced mining allowed) and fake-worker.mjs standing in for +// the GPU worker (the Metal worker on macOS, the OpenCL worker on Linux and Windows), driven through the control file +// and with signals. One step per fault class, each stating what must happen and what must not. Never touches the live +// devnet. Runs on the box (Linux engine and node from target-remote/) or on a Mac. // -// node tools/reliability/app-run.mjs --app --miner [--node ] +// node tools/reliability/app-run.mjs --app --miner --node [--only a,b] // -// Steps, in one engine run (each states what must happen and what must not): +// Steps, in one engine run: +// catch-up MF-1, MF-2: the node is synced (private) but its execution layer holds no record: no worker starts +// for 120 s, the card says it waits for the executed tip, the node is NOT restarted by the watchdog; +// a CPU block producer then makes blocks, the record appears, the worker starts on its own // own-restart the worker reports jobs done in 0.3 ms: the miner's guard restarts the worker; the app shows the // fault and does NOT restart the miner (same pid, no app restart) -// zero-once the worker completes jobs with 0 hashes: hash rate 0 for 60 s while synced, the app restarts the -// miner once; the worker is healthy again: mining resumes -// zero-faulted the same again inside five minutes: the card is marked faulted with the reason, its miner is not -// restarted, the node keeps running; "resume" clears it +// zero-ladder MF-2: jobs complete with 0 hashes: the watchdog restarts the miner at 10 s, then 30 s, then 120 s +// (the ladder), the reason stays on the card in plain words, nothing says "restarted once already" +// and no card is ever "faulted"; the worker is healthy again: mining resumes with no tap // no-status the miner process is stopped with SIGSTOP: no status line for 90 s, the app restarts it -// node-silent the node is stopped with SIGSTOP: no reading for 120 s, the app restarts the node in-process and -// the miner comes back once it is synced -// Never touches the live devnet. The app's own log is in the scratch directory. +// card-appears MF-3: the card is absent from the enumeration at start (unplugged, or a problem code) and appears +// two minutes later: the hot-plug pass starts its worker with no tap (Linux and Windows only) +// node-silent the node is stopped with SIGSTOP: no sign of life for 120 s, the app restarts the node in-process +// and the miner comes back once it is ready +// orphan-miner MF-7: a stray igneum-miner on this engine's node, not started by it, is killed by the minute sweep; +// the engine's own miner is left alone +// one-card-fails MF-4: two more cards appear, one failing its self-test for ever: the healthy two mine, the failing +// one is held 30 minutes with the reason on its row, the pack is not exported per failure (Linux) +// The app's own log is in the scratch directory. Every FAULT line the engine wrote is counted at the end. -import { spawn } from 'node:child_process'; -import { mkdirSync, rmSync, writeFileSync, existsSync, symlinkSync, readFileSync, appendFileSync, chmodSync } from 'node:fs'; +import { spawn, spawnSync } from 'node:child_process'; +import { mkdirSync, rmSync, writeFileSync, existsSync, symlinkSync, readFileSync, appendFileSync, chmodSync, readdirSync } from 'node:fs'; import { fileURLToPath } from 'node:url'; import { dirname, join } from 'node:path'; @@ -27,10 +37,11 @@ const args = process.argv.slice(2); const opt = (n, d) => { const i = args.indexOf(n); return i >= 0 ? args[i + 1] : d; }; const APP = opt('--app'); const MINER = opt('--miner'); -const NODE = opt('--node', join(here, '../../vendor/igneum-node/target-integration/release/igneumd')); +const NODE = opt('--node'); const ONLY = opt('--only', '').split(',').filter(Boolean); const SCRATCH = process.env.SCRATCH || `/tmp/igneum-reliability-app-${process.pid}`; const RPC = 29960, P2P = 29961, SUFFIX = 9960; +const MAC = process.platform === 'darwin'; for (const [k, v] of Object.entries({ APP, MINER, NODE })) if (!v || !existsSync(v)) { console.error(`missing ${k} (${v})`); process.exit(2); } const t0 = Date.now(); @@ -44,46 +55,51 @@ const bin = join(SCRATCH, 'stage', 'bin'); mkdirSync(bin, { recursive: true }); symlinkSync(NODE, join(bin, 'igneumd')); symlinkSync(MINER, join(bin, 'igneum-miner')); chmodSync(join(here, 'fake-worker.mjs'), 0o755); -symlinkSync(join(here, 'fake-worker.mjs'), join(bin, 'igneum-bench')); +// the fake worker under the name the engine's detection looks for on this platform +symlinkSync(join(here, 'fake-worker.mjs'), join(bin, MAC ? 'igneum-bench' : 'igneum-worker-opencl')); const data = join(SCRATCH, 'data'); mkdirSync(join(data, 'app'), { recursive: true }); const logs = join(SCRATCH, 'logs'); mkdirSync(logs, { recursive: true }); const CTL = join(SCRATCH, 'ctl'); const setMode = (m) => { writeFileSync(CTL, m + '\n'); log(`fake worker mode -> ${m}`); }; +const cardKey = MAC ? 'apple::Fake GPU' : 'other:0:Fake GPU'; setMode('ok'); writeFileSync(join(data, 'app', 'settings.json'), JSON.stringify({ setup_done: true, address: '0x4242424242424242424242424242424242424242', address_source: 'pasted', key_saved: true, identities: 1, - cards: { 'apple::Fake GPU': { enabled: true, identities: 1, power_pct: 0 } }, vote: false, paused: false, accepted_total: 0, - auto_update: false, remote_jobs: false, prove: false, + cards: { [cardKey]: { enabled: true, identities: 1, power_pct: 0 } }, vote: false, paused: false, accepted_total: 0, + auto_update: false, remote_jobs: false, prove: false, dev_fee: false, }, null, 2)); const env = { ...process.env, IGNEUM_APP_DATA: data, IGNEUM_APP_LOGS: logs, IGNEUM_APP_BIN: bin, IGNEUM_APP_RPC_PORT: String(RPC), IGNEUM_APP_P2P_PORT: String(P2P), IGNEUM_APP_PEERS: '', IGNEUM_APP_UNSYNCED: '1', IGNEUM_APP_DEVNET_SUFFIX: String(SUFFIX), IGNEUM_APP_STATUS_SECS: '10', FAKE_WORKER_CTL: CTL, + // easy genesis bits so the CPU block producer of the catch-up step makes blocks on two threads + IGNEUM_DEVNET_GENESIS_BITS: '0x1f100000', }; const app = spawn(APP, ['--no-open'], { stdio: ['pipe', 'pipe', 'pipe'], env }); const appOut = join(SCRATCH, 'app.out'); app.stdout.on('data', d => appendFileSync(appOut, d)); app.stderr.on('data', d => appendFileSync(appOut, d)); let appExit = null; app.on('exit', c => { appExit = c; log(`app exited ${c}`); }); -process.on('exit', () => { try { app.kill('SIGKILL'); } catch {} }); +const started = [app]; +process.on('exit', () => { for (const p of started) { try { p.kill('SIGKILL'); } catch {} } }); process.on('SIGINT', () => process.exit(130)); -log(`app pid ${app.pid}, scratch ${SCRATCH}`); +log(`app pid ${app.pid}, scratch ${SCRATCH}, platform ${process.platform}`); let url = null; -for (let i = 0; i < 100 && !url; i++) { try { url = readFileSync(join(data, 'app', 'app.url'), 'utf8').trim(); } catch { await sleep(200); } } +for (let i = 0; i < 150 && !url; i++) { try { url = readFileSync(join(data, 'app', 'app.url'), 'utf8').trim(); } catch { await sleep(200); } } if (!url) { log('no app.url'); process.exit(1); } async function state() { try { const r = await fetch(url + 'api/state'); return await r.json(); } catch { return null; } } const card = (st) => (st && st.mining && st.mining.cards && st.mining.cards[0]) || {}; -const events = []; // (t, card state, message, hash, pid, restarts, faults, node state) transitions, for the record +const events = []; let last = ''; async function poll() { const st = await state(); if (!st) return null; const c = card(st); - const key = [c.state, c.message, c.pid, c.restarts, c.faults, st.node.state, c.hash_now > 0 ? '>0' : '0'].join('|'); - if (key !== last) { last = key; events.push({ t: Date.now(), key }); log(`card ${c.state} pid ${c.pid} restarts ${c.restarts} faults ${c.faults} hash ${Number(c.hash_now).toFixed(1)} node ${st.node.state} ${c.message ? '"' + c.message + '"' : ''}`); } + const key = [c.state, c.message, c.pid, c.restarts, c.faults, st.node.state, st.node.starts, c.hash_now > 0 ? '>0' : '0'].join('|'); + if (key !== last) { last = key; events.push({ t: Date.now(), key }); log(`card ${c.state || '(none)'} pid ${c.pid} restarts ${c.restarts} faults ${c.faults} hash ${Number(c.hash_now || 0).toFixed(1)} node ${st.node.state} starts ${st.node.starts} ${c.message ? '"' + c.message + '"' : ''}`); } return st; } async function until(pred, timeoutMs, what) { @@ -102,10 +118,50 @@ function verdict(name, checks, metrics) { for (const [k, v] of Object.entries(metrics)) log(` ${k}: ${v}`); } async function steady(ms) { const end = Date.now() + ms; while (Date.now() < end) { await poll(); await sleep(1000); } } +const engineLogLines = (re) => { const out = []; for (const f of readdirSync(logs).filter(f => f.startsWith('app-'))) { for (const l of readFileSync(join(logs, f), 'utf8').split('\n')) if (re.test(l)) out.push(l); } return out; }; const S = {}; +S['catch-up'] = async () => { + // the fake card is present for this step + setMode('ok'); + const tStart = Date.now(); + // the private node reads synced within seconds; its execution layer holds no record until a block executes + const synced = await until((st) => st.node.synced, 120000, 'node synced'); + const tSynced = Date.now(); + // 120 s: no worker starts, the card waits on the executed tip, the node is not restarted + let started = false, nodeRestarts = 0, waitedForTip = false; + const end = Date.now() + 120000; + while (Date.now() < end) { + const st = await poll(); if (!st) break; + const c = card(st); + if (c.pid > 0 || c.state === 'mining' || c.state === 'starting') started = true; + if (/execute the tip/.test(c.message || '')) waitedForTip = true; + nodeRestarts = Math.max(nodeRestarts, (st.node.starts || 1) - 1); + await sleep(1000); + } + // now blocks: a CPU block producer against the app's node makes the first executed record + // it stays up for the rest of the run: every later step needs an executed tip to exist + const cpu = spawn(MINER, ['mine', `grpc://127.0.0.1:${RPC}`, '2', '36000', 'cpu', '--engine', 'igneum-pow', '--no-vote', '--payout-label', 'cpu'], { stdio: ['ignore', 'ignore', 'ignore'] }); + started.push(cpu); + log(`cpu block producer pid ${cpu.pid}`); + const tBlocks = Date.now(); + const back = await until(mining, 300000, 'the worker to start once the execution layer holds a record'); + const tMining = Date.now(); + const readyLines = engineLogLines(/node readiness: the execution layer reports an executed tip/); + const gatedCalls = engineLogLines(/holds no record yet/); + verdict('catch-up', [ + { ok: !!synced, what: 'the private node read synced' }, + { ok: !started, what: 'no worker started while the execution layer held no record (120 s)' }, + { ok: waitedForTip, what: 'the card said it waits for the executed tip' }, + { ok: nodeRestarts === 0, what: `the node watchdog did not restart the node during the catch-up (restarts ${nodeRestarts})` }, + { ok: readyLines.length >= 1, what: 'the engine logged the executed tip when it appeared' }, + { ok: !!back, what: 'the worker started on its own once the record existed' }, + ], { synced_after: s(tSynced - tStart), blocks_to_mining: s(tMining - tBlocks), gated_exec_calls_refused: gatedCalls.length }); +}; + S['own-restart'] = async () => { - const st0 = await until(mining, 180000, 'first mining'); + setMode('ok'); + const st0 = await until(mining, 300000, 'first mining'); const pid0 = card(st0).pid, r0 = card(st0).restarts; const tInject = Date.now(); setMode('fast'); @@ -114,7 +170,7 @@ S['own-restart'] = async () => { setMode('ok'); const back = await until((st, c) => mining(st, c) && c.faults >= 1, 90000, 'mining again after the worker restart'); const tBack = Date.now(); - await steady(20000); + await steady(15000); const st1 = await poll(); verdict('own-restart', [ { ok: !!st0, what: 'the card mined with a rate above 0 on the fake worker' }, @@ -123,53 +179,50 @@ S['own-restart'] = async () => { { ok: st1 && card(st1).pid === pid0 && card(st1).restarts === r0, what: `the app did not restart the miner (pid ${pid0} -> ${card(st1 || {}).pid}, app restarts ${r0} -> ${card(st1 || {}).restarts})` }, ], { inject_to_fault_on_card: s(tFault - tInject), fault_to_mining_again: s(tBack - tFault) }); }; -S['zero-once'] = async () => { - const st0 = await until(mining, 60000, 'mining'); - const pid0 = card(st0).pid, r0 = card(st0).restarts; - const tInject = Date.now(); - setMode('zero'); - const rs = await until((st, c) => c.restarts > r0 || /watchdog/.test(c.message || ''), 150000, 'the watchdog restart'); - const tRestart = Date.now(); + +S['zero-ladder'] = async () => { setMode('ok'); - const back = await until((st, c) => mining(st, c) && c.pid !== pid0, 120000, 'mining on the restarted miner'); - const tBack = Date.now(); - await steady(15000); - const st1 = await poll(); - verdict('zero-once', [ - { ok: !!rs && /hash rate 0/.test(card(rs).message || ''), what: `the watchdog restarted the miner for a zero rate (${JSON.stringify(card(rs || {}).message)})` }, - { ok: rs && tRestart - tInject >= 55000 && tRestart - tInject <= 100000, what: `between 55 and 100 s after the rate went to 0 (${s(tRestart - tInject)})` }, - { ok: !!back, what: 'mining resumed on a new miner process' }, - { ok: st1 && card(st1).restarts === r0 + 1, what: `exactly one app restart (${r0} -> ${card(st1 || {}).restarts})` }, - ], { zero_to_restart: s(tRestart - tInject), restart_to_mining: s(tBack - tRestart), zero_to_mining: s(tBack - tInject) }); -}; -S['zero-faulted'] = async () => { - const st0 = await until(mining, 60000, 'mining'); + const st0 = await until(mining, 120000, 'mining'); const r0 = card(st0).restarts; const tInject = Date.now(); setMode('zero'); - const f = await until((st, c) => c.state === 'faulted', 150000, 'the card to be marked faulted'); - const tFault = Date.now(); - await steady(45000); - const st1 = await poll(); + // three rungs: the gaps between consecutive watchdog restarts must grow 10, 30, 120 s (plus the 60 s rule each time) + const marks = []; + let lastR = r0; + const end = Date.now() + 600000; + let faultedSeen = false, onceAlready = false; + while (Date.now() < end && marks.length < 3) { + const st = await poll(); if (!st) break; + const c = card(st); + if (c.state === 'faulted') faultedSeen = true; + if (/restarted once already/.test(c.message || '')) onceAlready = true; + if ((c.restarts || 0) > lastR) { lastR = c.restarts; marks.push({ t: Date.now(), wait: c.restart_in_s, msg: c.message }); log(`rung ${marks.length}: restart_in_s ${c.restart_in_s} "${c.message}"`); } + await sleep(500); + } setMode('ok'); - app.stdin.write('resume\n'); - const back = await until(mining, 90000, 'mining after resume'); - verdict('zero-faulted', [ - { ok: !!f && /restarted once already/.test(card(f).message || ''), what: `the card was marked faulted with the reason (${JSON.stringify(card(f || {}).message)})` }, - { ok: st1 && card(st1).state === 'faulted' && card(st1).pid === 0 && card(st1).restarts === r0, what: `45 s later: still faulted, no miner process, no further restart (restarts ${r0} -> ${card(st1 || {}).restarts})` }, - { ok: st1 && st1.node.synced && appExit === null, what: 'the node kept running and the app stayed up' }, - { ok: !!back, what: 'resume cleared the fault and mining resumed' }, - ], { zero_to_faulted: s(tFault - tInject) }); + const back = await until(mining, 400000, 'mining again on its own after the worker is healthy'); + const tBack = Date.now(); + const waits = marks.map(m => m.wait); + const ladderOk = waits.length === 3 && waits[0] >= 8 && waits[0] <= 10 && waits[1] >= 28 && waits[1] <= 30 && waits[2] >= 118 && waits[2] <= 120; + verdict('zero-ladder', [ + { ok: marks.length === 3, what: `three watchdog restarts observed (${marks.length})` }, + { ok: ladderOk, what: `the restart delays follow the ladder 10, 30, 120 s (saw ${waits.join(', ')})` }, + { ok: marks.every(m => /hash rate 0 for 60 s/.test(m.msg || '')), what: 'the reason stayed on the card in plain words' }, + { ok: !faultedSeen && !onceAlready, what: 'no "faulted" state and no "restarted once already" words' }, + { ok: !!back, what: 'mining resumed on its own once the worker was healthy' }, + ], { zero_to_rung1: marks[0] ? s(marks[0].t - tInject) : 'n/a', rung_gaps: marks.slice(1).map((m, i) => s(m.t - marks[i].t)).join(', '), healthy_to_mining: s(tBack - (marks[2] ? marks[2].t : tInject)) }); }; + S['no-status'] = async () => { - const st0 = await until(mining, 60000, 'mining'); + setMode('ok'); + const st0 = await until(mining, 400000, 'mining'); const pid0 = card(st0).pid, r0 = card(st0).restarts; const tInject = Date.now(); process.kill(pid0, 'SIGSTOP'); log(`SIGSTOP miner pid ${pid0}`); const rs = await until((st, c) => c.restarts > r0 || /no status line/.test(c.message || ''), 150000, 'the watchdog restart'); const tRestart = Date.now(); - const back = await until((st, c) => mining(st, c) && c.pid !== pid0 && c.pid > 0, 120000, 'mining on the restarted miner'); + const back = await until((st, c) => mining(st, c) && c.pid !== pid0 && c.pid > 0, 400000, 'mining on the restarted miner'); const tBack = Date.now(); let gone = false; try { process.kill(pid0, 0); } catch { gone = true; } if (!gone) { try { process.kill(pid0, 'SIGKILL'); } catch {} } @@ -180,38 +233,111 @@ S['no-status'] = async () => { { ok: gone, what: 'the stopped miner process was killed' }, ], { quiet_to_restart: s(tRestart - tInject), restart_to_mining: s(tBack - tRestart) }); }; + +S['card-appears'] = async () => { + if (MAC) { verdict('card-appears', [{ ok: true, what: 'skipped on macOS (Apple silicon has no GPU hot-plug; the Metal path enumerates once)' }], {}); return; } + // the card leaves the enumeration (unplugged, or a problem code: the hot-plug pass marks it removed and stops its + // worker), then comes back: its worker must start with no tap + await until(mining, 300000, 'mining before the card leaves'); + setMode('absent'); + const tGone = Date.now(); + const gone = await until((st, c) => c.state === 'removed' || c.removed === true || /removed/.test(c.state || ''), 150000, 'the hot-plug pass to mark the card removed'); + const tMarked = Date.now(); + await steady(30000); + setMode('ok'); + const tAppear = Date.now(); + const back = await until(mining, 300000, 'the worker to start with no tap once the card is back'); + const tBack = Date.now(); + verdict('card-appears', [ + { ok: !!gone, what: `the hot-plug pass marked the card removed when it left (${s(tMarked - tGone)})` }, + { ok: !!back, what: 'its worker started with no tap once it was back' }, + ], { gone_to_marked: s(tMarked - tGone), back_to_mining: s(tBack - tAppear) }); +}; + S['node-silent'] = async () => { - const st0 = await until(mining, 60000, 'mining'); + setMode('ok'); + const st0 = await until(mining, 400000, 'mining'); const npid = st0.node.pid, starts0 = st0.node.starts; const tInject = Date.now(); process.kill(npid, 'SIGSTOP'); log(`SIGSTOP node pid ${npid}`); - const rs = await until((st) => st.node.state === 'restarting' || st.node.starts > starts0, 200000, 'the node restart'); + const rs = await until((st) => st.node.state === 'restarting' || st.node.starts > starts0, 240000, 'the node restart'); const tRestart = Date.now(); - const synced = await until((st) => st.node.starts > starts0 && st.node.synced, 120000, 'the restarted node to sync'); + const synced = await until((st) => st.node.starts > starts0 && st.node.synced, 180000, 'the restarted node to sync'); const tSynced = Date.now(); - const back = await until(mining, 120000, 'mining again'); + const back = await until(mining, 400000, 'mining again'); const tBack = Date.now(); let gone = false; try { process.kill(npid, 0); } catch { gone = true; } if (!gone) { try { process.kill(npid, 'SIGKILL'); } catch {} } verdict('node-silent', [ { ok: !!rs, what: 'the app restarted the node' }, - { ok: rs && tRestart - tInject >= 115000 && tRestart - tInject <= 170000, what: `between 115 and 170 s after the node went quiet (${s(tRestart - tInject)})` }, + { ok: rs && tRestart - tInject >= 115000 && tRestart - tInject <= 190000, what: `between 115 and 190 s after the node went quiet (${s(tRestart - tInject)})` }, { ok: !!synced, what: `the new node synced (starts ${starts0} -> ${synced ? synced.node.starts : '?'})` }, - { ok: !!back, what: 'mining resumed' }, + { ok: !!back, what: 'mining resumed once the node was ready' }, { ok: gone, what: 'the stopped node process was killed' }, ], { quiet_to_restart: s(tRestart - tInject), restart_to_synced: s(tSynced - tRestart), quiet_to_mining: s(tBack - tInject) }); }; -const names = ONLY.length ? ONLY : Object.keys(S); +S['one-card-fails'] = async () => { + if (MAC) { verdict('one-card-fails', [{ ok: true, what: 'skipped on macOS (one Metal device)' }], {}); return; } + // MF-4: two more cards appear, one of them failing its self-test for ever; the healthy two must mine on time, + // the failing one is held with the reason on its row, and the pack is not exported once per failure + const exportsBefore = engineLogLines(/export-pack: /).length; + writeFileSync(CTL, 'ok\ndevices 3\ndev 2 selftest\n'); log('fake worker: 3 devices, device 2 fails its self-test'); + const tInject = Date.now(); + const listed = await until((st) => st.mining.cards.length >= 3, 200000, 'three cards listed'); + const tListed = Date.now(); + const twoMine = await until((st) => st.mining.cards.filter(c => c.state === 'mining' && c.hash_now > 0).length >= 2, 300000, 'two healthy cards mining'); + const tTwo = Date.now(); + const held = await until((st) => st.mining.cards.some(c => /not usable on this driver/.test(c.message || '')), 120000, 'the failing card held with the reason'); + await steady(90000); + const st1 = await poll(); + const bad = st1 ? st1.mining.cards.find(c => /not usable on this driver/.test(c.message || '')) : null; + const good = st1 ? st1.mining.cards.filter(c => !/not usable/.test(c.message || '')) : []; + const exportsAfter = engineLogLines(/export-pack: /).length - exportsBefore; + const reused = engineLogLines(/export-pack: reusing/).length; + verdict('one-card-fails', [ + { ok: !!listed, what: 'the hot-plug pass listed the two new cards' }, + { ok: !!twoMine, what: 'the two healthy cards mined' }, + { ok: !!held && bad && bad.restart_in_s >= 1500, what: `the failing card is held 30 minutes with the reason on its row (restart_in_s ${bad ? bad.restart_in_s : '?'}, "${bad ? bad.message : ''}")` }, + { ok: good.length >= 2 && good.every(c => c.state === 'mining' && c.hash_now > 0), what: 'the healthy cards still mine 90 s later (no restart from the failing card)' }, + { ok: exportsAfter <= 6, what: `the pack was exported at most 6 times for three starts and the failures (${exportsAfter}, ${reused} reused)` }, + ], { listed_after: s(tListed - tInject), two_mining_after: s(tTwo - tInject), exports: exportsAfter, exports_reused: reused }); +}; + +S['orphan-miner'] = async () => { + // MF-7: an igneum-miner the engine did not start, on this engine's node (the fence), is killed by the minute sweep + const st0 = await until(mining, 300000, 'mining'); + const stray = spawn(MINER, ['mine', `grpc://127.0.0.1:${RPC}`, '1', '3600', 'stray', '--worker', join(here, 'fake-worker.mjs'), '--status-secs', '10', '--no-vote', '--payout-label', 'stray'], { stdio: ['ignore', 'ignore', 'ignore'], env: { ...process.env, FAKE_WORKER_CTL: CTL } }); + started.push(stray); + const tInject = Date.now(); + log(`stray miner pid ${stray.pid} on the engine's node`); + let killedAt = null; + const end = Date.now() + 150000; + while (Date.now() < end) { await poll(); try { process.kill(stray.pid, 0); } catch { killedAt = Date.now(); break; } await sleep(1000); } + const lines = engineLogLines(/orphan miner killed/); + const st1 = await poll(); + const own = st1 ? card(st1) : {}; + let ownAlive = false; try { process.kill(own.pid, 0); ownAlive = true; } catch {} + verdict('orphan-miner', [ + { ok: !!killedAt, what: `the stray miner was killed by the engine (${killedAt ? s(killedAt - tInject) : 'still alive after 150 s'})` }, + { ok: lines.length >= 1, what: `one log line per kill (${lines.length})` }, + { ok: ownAlive && own.pid > 0, what: `the engine's own miner (pid ${own.pid}) was left alone` }, + ], { inject_to_kill: killedAt ? s(killedAt - tInject) : 'n/a' }); +}; + +const order = ['catch-up', 'card-appears', 'own-restart', 'zero-ladder', 'no-status', 'node-silent', 'one-card-fails', 'orphan-miner']; +const names = ONLY.length ? order.filter(n => ONLY.includes(n)) : order; for (const n of names) { - if (!S[n]) { log(`unknown step ${n}`); continue; } log(`=== ${n}`); try { await S[n](); } catch (e) { log(`step ${n} threw: ${e.stack || e}`); results.push({ name: n, pass: false, checks: [{ ok: false, what: String(e) }], metrics: {} }); } } +const faults = engineLogLines(/ FAULT class=/); +log(`FAULT lines the engine wrote: ${faults.length}`); +for (const f of faults.slice(0, 12)) log(` ${f.slice(0, 200)}`); app.stdin.write('quit\n'); await sleep(8000); -const report = { date: new Date().toISOString(), app: APP, miner: MINER, node: NODE, scratch: SCRATCH, results, transitions: events.map(e => ({ t: new Date(e.t).toISOString(), key: e.key })) }; +const report = { date: new Date().toISOString(), app: APP, miner: MINER, node: NODE, platform: process.platform, scratch: SCRATCH, results, fault_lines: faults.length, transitions: events.map(e => ({ t: new Date(e.t).toISOString(), key: e.key })) }; writeFileSync(join(SCRATCH, 'report.json'), JSON.stringify(report, null, 2)); console.log(JSON.stringify({ ...report, transitions: undefined }, null, 2)); process.exit(results.every(r => r.pass) ? 0 : 1); diff --git a/tools/reliability/fake-worker.mjs b/tools/reliability/fake-worker.mjs index bd5fe0b04..c6527a851 100755 --- a/tools/reliability/fake-worker.mjs +++ b/tools/reliability/fake-worker.mjs @@ -13,25 +13,54 @@ // badfound a `found` with a wrong hash before every `done`: a worker on a wrong program // preparefail every `prepare` is answered `prepare-failed`; jobs run as in ok // exit the process exits with code 7 at the next job +// absent `--list` shows no device (the card is unplugged or in a problem code); jobs run as in ok +// selftest the worker prints a self-test failure and exits 3 before ready (MF-4) +// Extra lines in the control file: `devices N` (how many --list shows), `dev N ` (a mode for device N only) // The mode is read from argv `--mode ` when no control file is set. `--serve` is accepted and ignored. // Lines on stderr are prefixed `fake-worker:` and never match a miner pattern. import { readFileSync, existsSync, writeSync } from 'node:fs'; +import { createRequire } from 'node:module'; +const require = createRequire(import.meta.url); import { createInterface } from 'node:readline'; +// `--list` (the engine's OpenCL enumeration on Windows and Linux, src/detect.rs): one fake GPU, or none while the +// control file says `absent` (the hot-plug step: a card that appears later must start without a tap) +if (process.argv.includes('--list')) { + const m = (() => { try { return require('node:fs').readFileSync(process.env.FAKE_WORKER_CTL, 'utf8').trim().split(/\s+/)[0]; } catch { return 'ok'; } })(); + const n = m === 'absent' ? 0 : (() => { try { const l = require('node:fs').readFileSync(process.env.FAKE_WORKER_CTL, 'utf8').split('\n').map(x => x.trim().split(/\s+/)).find(f => f[0] === 'devices'); return l ? Number(l[1]) : 1; } catch { return 1; } })(); + for (let i = 0; i < n; i++) { + process.stdout.write(`[${i}] Fake GPU ${i + 1} | Fake Platform (OpenCL 1.2)\n GPU, vendor Fake Silicon, driver 1.0.0, OpenCL C 1.2, 8 compute units, 8192 MB\n`); + } + process.exit(0); +} const ctl = process.env.FAKE_WORKER_CTL; const argMode = (() => { const i = process.argv.indexOf('--mode'); return i >= 0 ? process.argv[i + 1] : 'ok'; })(); +// the device this process serves (`--device N` from the engine's OpenCL path); the control file's first word is the +// mode for every device, a line `dev N ` overrides it for device N, a line `devices N` sets how many --list shows +const device = (() => { const i = process.argv.indexOf('--device'); return i >= 0 ? process.argv[i + 1] : '0'; })(); function mode() { if (ctl && existsSync(ctl)) { - const m = readFileSync(ctl, 'utf8').trim().split(/\s+/)[0]; + const lines = readFileSync(ctl, 'utf8').trim().split('\n'); + const own = lines.map(l => l.trim().split(/\s+/)).find(f => f[0] === 'dev' && f[1] === device); + if (own && own[2]) return own[2]; + const m = (lines[0] || '').trim().split(/\s+/)[0]; if (m) return m; } return argMode; } +function deviceCount() { + try { const l = readFileSync(ctl, 'utf8').split('\n').map(x => x.trim().split(/\s+/)).find(f => f[0] === 'devices'); return l ? Number(l[1]) : 1; } catch { return 1; } +} // synchronous writes: a piped stdout is asynchronous in Node and process.exit would drop pending lines const out = (s) => { writeSync(1, s + '\n'); }; const sleep = (ms) => new Promise(r => setTimeout(r, ms)); +if (mode() === 'selftest') { + // MF-4: the worker fails its self-test and exits before it is ready (the Arc B580 on PC 1, 7 October 2026) + out('error 0 self-test FAIL: 3 of 96 vectors mismatched (fake worker, device ' + device + ')'); + process.exit(3); +} out('ready fake Fake_GPU dataset-log2 28 batch 16777216 prepare 1'); const queue = []; From 164dc99413e6f52befcba988d765e5a6b14ed55f Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 09:59:57 +0000 Subject: [PATCH 02/18] execrpc: one gate for every execution-layer RPC call the app makes (ledger N7, main's rule for 0.3.19) A node before the exec RPC bounds fix (every 0.3.17 node) dies when a method that resolves a block number or indexes the record vector is asked while its exec follower holds no record; PC 1 crash-looped on two callers in one night (eth_getBlockByNumber from the clock sample, then igneum_getAssignedShards from the prover loop: 'panicked at igneum/exec/src/rpc.rs:808:35: range start index 1 out of range for slice of length 0'). Every caller (prover.rs's evm_rpc, update.rs's clock sample, extnode's rpc for chainfacts and the external-node probe) now goes through execrpc::call: SAFE_ON_EMPTY methods go out, GATED ones wait for igneum_getExecStatus's executedTipHash, an unclassified method is refused. The test every_caller_goes_through_the_gate scans src/ for JSON-RPC requests built elsewhere and for unclassified exec method names. Co-Authored-By: Claude Fable 5.1 --- app/igneum-app/src/extnode.rs | 11 ++--------- app/igneum-app/src/main.rs | 1 + app/igneum-app/src/prover.rs | 18 ++---------------- app/igneum-app/src/update.rs | 31 +++++-------------------------- 4 files changed, 10 insertions(+), 51 deletions(-) diff --git a/app/igneum-app/src/extnode.rs b/app/igneum-app/src/extnode.rs index 25764ac1c..91aee55bc 100644 --- a/app/igneum-app/src/extnode.rs +++ b/app/igneum-app/src/extnode.rs @@ -8,7 +8,6 @@ //! too, read every 30 s instead of parsed from a stdout the app does not own. use serde_json::{json, Value}; use std::path::Path; -use std::process::Command; use std::time::Duration; /// What the other node answered: None where it could not answer. @@ -110,14 +109,8 @@ pub fn step(port_open: bool, now_s: f64, gone_since: &mut Option) -> Step { /// One JSON-RPC call to the node's EVM port through curl (the engine carries no HTTP client; update.rs does the same). pub fn rpc(evm_port: u16, method: &str, params: Value, limit: Duration) -> Option { - let body = json!({ "jsonrpc": "2.0", "id": 1, "method": method, "params": params }).to_string(); - let out = crate::detect::run_timeout( - Command::new(crate::platform::tool("curl")).args(["-s", "--max-time", &format!("{}", limit.as_secs().max(1)), "-X", "POST", &format!("http://127.0.0.1:{evm_port}"), "-H", "Content-Type: application/json", "-d", &body]), - None, - limit + Duration::from_secs(2), - )?; - let v: Value = serde_json::from_str(&out).ok()?; - v.get("result").cloned().filter(|r| !r.is_null()) + // one path (ledger N7): execrpc holds a records-indexing method until the node's exec follower has a record + crate::execrpc::call(evm_port, method, params, limit).ok().filter(|r| !r.is_null()) } /// The other node's answers, with what each method gives: eth_chainId (every node), igneum_getNodeInfo (newer nodes). diff --git a/app/igneum-app/src/main.rs b/app/igneum-app/src/main.rs index 0e0a11bba..51aef9258 100644 --- a/app/igneum-app/src/main.rs +++ b/app/igneum-app/src/main.rs @@ -26,6 +26,7 @@ mod manifest; mod ota; mod uiota; mod update; +mod execrpc; mod jobs; mod jobrun; mod jobbuild; diff --git a/app/igneum-app/src/prover.rs b/app/igneum-app/src/prover.rs index 47ad1caf4..724fc12de 100644 --- a/app/igneum-app/src/prover.rs +++ b/app/igneum-app/src/prover.rs @@ -168,22 +168,8 @@ fn exec_boundary(shared: &Shared) -> u64 { } pub(crate) fn evm_rpc(shared: &Shared, method: &str, params: Value, timeout: Duration) -> Result { - let body = json!({ "jsonrpc": "2.0", "id": 1, "method": method, "params": params }).to_string(); - let tmp = std::env::temp_dir().join(format!("igneum-prover-{}-{}.json", std::process::id(), method)); - std::fs::write(&tmp, body).map_err(|e| e.to_string())?; - let url = format!("http://127.0.0.1:{}", shared.runtime.evm_port()); - let out = crate::detect::run_timeout( - Command::new(crate::platform::tool("curl")).args(["-s", "--max-time", &timeout.as_secs().to_string(), "-X", "POST", &url, "-H", "Content-Type: application/json", "--data-binary", &format!("@{}", tmp.display())]), - None, - timeout + Duration::from_secs(2), - ); - let _ = std::fs::remove_file(&tmp); - let out = out.ok_or_else(|| format!("{method}: the node's RPC did not answer"))?; - let v: Value = serde_json::from_str(&out).map_err(|e| format!("{method}: {e}"))?; - if let Some(err) = v.get("error") { - return Err(format!("{method}: {}", err.get("message").and_then(|m| m.as_str()).unwrap_or("error"))); - } - Ok(v.get("result").cloned().unwrap_or(Value::Null)) + // one path (ledger N7): execrpc holds a records-indexing method until the node's exec follower has a record + crate::execrpc::call(shared.runtime.evm_port(), method, params, timeout) } fn find_tools(bin_dir: &Path) -> Result { diff --git a/app/igneum-app/src/update.rs b/app/igneum-app/src/update.rs index 2978d8804..50b061368 100644 --- a/app/igneum-app/src/update.rs +++ b/app/igneum-app/src/update.rs @@ -43,41 +43,20 @@ fn days_from_civil(y: i64, m: i64, d: i64) -> i64 { /// True once the node's exec follower holds a record (igneum_getExecStatus's executedTipHash is set). Any error or an /// unreachable node reads false: the block sample waits rather than asks. pub fn exec_has_record(evm_port: u16) -> bool { - let body = "{\"jsonrpc\":\"2.0\",\"id\":1,\"method\":\"igneum_getExecStatus\",\"params\":[]}"; - let out = crate::detect::run_timeout( - Command::new(crate::platform::tool("curl")).args(["-s", "--max-time", "5", "-X", "POST", &format!("http://127.0.0.1:{evm_port}"), "-H", "Content-Type: application/json", "-d", body]), - None, - Duration::from_secs(7), - ); - out.as_deref().map(exec_status_has_record).unwrap_or(false) + crate::execrpc::has_record(evm_port) } /// The reading of an igneum_getExecStatus reply: a record is held when executedTipHash is a non-null string. pub fn exec_status_has_record(reply: &str) -> bool { - serde_json::from_str::(reply) - .ok() - .and_then(|v| v.get("result")?.get("executedTipHash")?.as_str().map(|h| !h.is_empty())) - .unwrap_or(false) + serde_json::from_str::(reply).ok().map(|v| crate::execrpc::status_has_record(v.get("result").unwrap_or(&serde_json::Value::Null))).unwrap_or(false) } /// The latest block's timestamp (unix seconds) from the node's Ethereum JSON-RPC (the execution layer mirrors the /// consensus block times). The first clock source: local time against what the peers produced. pub fn latest_block_time(evm_port: u16) -> Option { - // 0.3.18 (ledger N7, 7 October 2026): a node whose exec follower holds no record yet dies on eth_getBlockByNumber - // (rpc.rs indexes records[0] on an empty vector and the panic hook exits the process). The nodes before 0.3.18's - // fix are live on every machine, and this sample runs every 9 s once blocks arrive, so it asks igneum_getExecStatus - // first (it indexes nothing) and takes no block until the follower reports an executed tip. - if !exec_has_record(evm_port) { - return None; - } - let body = "{\"jsonrpc\":\"2.0\",\"id\":1,\"method\":\"eth_getBlockByNumber\",\"params\":[\"latest\",false]}"; - let out = crate::detect::run_timeout( - Command::new(crate::platform::tool("curl")).args(["-s", "--max-time", "5", "-X", "POST", &format!("http://127.0.0.1:{evm_port}"), "-H", "Content-Type: application/json", "-d", body]), - None, - Duration::from_secs(7), - )?; - let v: serde_json::Value = serde_json::from_str(&out).ok()?; - let ts = v.get("result")?.get("timestamp")?.as_str()?; + // 0.3.18/0.3.19 (ledger N7): through the one gate, which asks nothing of a follower without a record + let block = crate::execrpc::call(evm_port, "eth_getBlockByNumber", serde_json::json!(["latest", false]), Duration::from_secs(5)).ok()?; + let ts = block.get("timestamp")?.as_str()?; u64::from_str_radix(ts.trim_start_matches("0x"), 16).ok().map(|t| t as f64) } From 1506f6c00300bdca16b98f45bc01672ff9e16474 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 11:32:15 +0000 Subject: [PATCH 03/18] reliability injector: the catch-up step's flag no longer shadows the process list Co-Authored-By: Claude Fable 5.1 (cherry picked from commit 2efe419521406ae248d9c5e7fd8f3301f7df24a3) --- tools/reliability/app-run.mjs | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/tools/reliability/app-run.mjs b/tools/reliability/app-run.mjs index 48dab7bc3..03b2cd6ec 100755 --- a/tools/reliability/app-run.mjs +++ b/tools/reliability/app-run.mjs @@ -129,12 +129,12 @@ S['catch-up'] = async () => { const synced = await until((st) => st.node.synced, 120000, 'node synced'); const tSynced = Date.now(); // 120 s: no worker starts, the card waits on the executed tip, the node is not restarted - let started = false, nodeRestarts = 0, waitedForTip = false; + let workerStarted = false, nodeRestarts = 0, waitedForTip = false; const end = Date.now() + 120000; while (Date.now() < end) { const st = await poll(); if (!st) break; const c = card(st); - if (c.pid > 0 || c.state === 'mining' || c.state === 'starting') started = true; + if (c.pid > 0 || c.state === 'mining' || c.state === 'starting') workerStarted = true; if (/execute the tip/.test(c.message || '')) waitedForTip = true; nodeRestarts = Math.max(nodeRestarts, (st.node.starts || 1) - 1); await sleep(1000); @@ -151,7 +151,7 @@ S['catch-up'] = async () => { const gatedCalls = engineLogLines(/holds no record yet/); verdict('catch-up', [ { ok: !!synced, what: 'the private node read synced' }, - { ok: !started, what: 'no worker started while the execution layer held no record (120 s)' }, + { ok: !workerStarted, what: 'no worker started while the execution layer held no record (120 s)' }, { ok: waitedForTip, what: 'the card said it waits for the executed tip' }, { ok: nodeRestarts === 0, what: `the node watchdog did not restart the node during the catch-up (restarts ${nodeRestarts})` }, { ok: readyLines.length >= 1, what: 'the engine logged the executed tip when it appeared' }, From c95e85426522b7baae308919e78dd12d2ff6b8a6 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 11:32:48 +0000 Subject: [PATCH 04/18] miner-faults register: the commits table and which side each class is on Co-Authored-By: Claude Fable 5.1 (cherry picked from commit 43cbec0d634943f4ba38e1f07e91d42c9fd5c43c) --- docs/plans/miner-faults.md | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/docs/plans/miner-faults.md b/docs/plans/miner-faults.md index 93af4cd8d..5e7f0a211 100644 --- a/docs/plans/miner-faults.md +++ b/docs/plans/miner-faults.md @@ -28,6 +28,21 @@ Standing rules behind every row (branch `miner-reliability`, off `release-0.3.19 | MF-7 | 7 Oct 2026, PC 1, 0.3.19 | Orphan `igneum-miner.exe` processes the app no longer tracked (two alive under `--stop-miners` with their rows at pid 0, one after) hammered the node's template RPC beside the tracked miners | The engine lost track of miners it had started (a stop that timed out, a restart over a live process) and never looked for them again | The engine owns every miner it started: at start, after every stop and every minute it kills any `igneum-miner` whose command line carries THIS engine's node RPC (the fence) and whose pid it does not track, one log line and one fault report per kill, never by name alone (`sweep_orphan_miners`, `platform::miner_processes`, `kill_pid`); a restart kills the slot's old process before the new one starts | injector step `orphan-miner` (a stray miner on the engine's node is killed inside the minute, the engine's own miner left alone) | `app-run orphan-miner PASS` | | MF-3 | 7 Oct 2026, PC 1, Intel Arc | The Intel driver's first install did not bind: the device sat in Code 12 at install time; the card never mined until a reboot | A driver installed while the device reports a problem code (12, 43, 31) does not bind; nothing re-scanned the device afterwards, and the app only re-enumerates | The app re-enumerates every 60 s and starts the worker the minute the OS drives the card (`hotplug::diff` recovered / revived, `settle_new`); the row says what to do while it does not ("reboot with the card attached; if it persists, reinstall the driver with the card attached"); a Windows host asks for a re-scan (`pnputil /scan-devices`) after a problem code is seen, every 5 minutes, at most 6 times (follow-up, host side) | `hotplug` test `a_driven_card_that_turns_faulty_is_errored_and_recovers_later`; `app-run.mjs` step `card-appears` (a card listed after 2 minutes starts without a tap) | `app-run card-appears PASS` | +## Commits (7 October 2026) + +| Repo | Branch | Commit | What | +|---|---|---|---| +| igneum | `miner-reliability` (off `release-0.3.19` db6f0964, with release-0.3.20's `igneum-pow` and master's build tooling) | `38a30397`, `c5cb5cd5` | the app side of every row, the register, the injector, the cards job kind, the LG-4 job, the CI check | +| igneum-node | `miner-reliability-20` (off `release-0.3.20-node` dc141409) | `f067f7c1` | MF-5's miner side: STATUS while waiting, identities from the template time, the fetch back-off | + +Box lines: app tests 198 + 27 + 8 green on igneum-build-2 (the tree gate green, 33 checks on the Mac); igneum-miner 19 +green on igneum-build-2; both Linux binaries built on igneum-build-1 (app sha256 6d4ee013, miner b3bf3590). Injector: +the pod run's lines are appended below when it ends. + +Which side: MF-1, MF-2, MF-3, MF-4, MF-6, MF-7 and the cards kind are app-side (0.3.20's app). MF-5 is both: the +miner (the fork branch) prints the fields and caps the identities; the app reads them and keeps the worker. An app +without the new miner still gets MF-5's app half (a template timeout line is the heartbeat; the zero-rate clock holds). + ## How a row is added 1. The day the class is seen: the row with Found, Class, Root cause. The rule, the test and the gate line the same day From e464a5fb06885c09d73cb231bf2db4edef7b9b08 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 11:40:53 +0000 Subject: [PATCH 05/18] reliability injector: reads the fake cards by name and switches a box's real GPU off through the app's own card path Co-Authored-By: Claude Fable 5.1 (cherry picked from commit b9c70528fafca3d3a77ba5e5625e349867ee10c4) --- tools/reliability/app-run.mjs | 27 +++++++++++++++++++++------ 1 file changed, 21 insertions(+), 6 deletions(-) diff --git a/tools/reliability/app-run.mjs b/tools/reliability/app-run.mjs index 03b2cd6ec..5e557cacb 100755 --- a/tools/reliability/app-run.mjs +++ b/tools/reliability/app-run.mjs @@ -91,12 +91,27 @@ if (!url) { log('no app.url'); process.exit(1); } async function state() { try { const r = await fetch(url + 'api/state'); return await r.json(); } catch { return null; } } -const card = (st) => (st && st.mining && st.mining.cards && st.mining.cards[0]) || {}; +// the fake cards only: a pod or a PC also lists its real GPU, which is switched off at start (through the app's own +// card path) and never read by a step +const fakes = (st) => ((st && st.mining && st.mining.cards) || []).filter(c => /^Fake GPU/.test(c.name || '')); +const card = (st) => fakes(st)[0] || {}; +let realOff = false; +async function switchRealCardsOff(st) { + if (realOff || !st || !st.mining) return; + const real = (st.mining.cards || []).filter(c => !/^Fake GPU/.test(c.name || '') && c.enabled); + if (!real.length) { realOff = true; return; } + try { + await fetch(url + 'api/cards', { method: 'POST', headers: { 'Content-Type': 'application/json' }, body: JSON.stringify({ cards: real.map(c => ({ key: c.key, enabled: false, identities: c.identities || 1 })) }) }); + log(`real card(s) switched off for the run: ${real.map(c => c.name).join(', ')}`); + realOff = true; + } catch (e) { log(`api/cards: ${e}`); } +} const events = []; let last = ''; async function poll() { const st = await state(); if (!st) return null; + await switchRealCardsOff(st); const c = card(st); const key = [c.state, c.message, c.pid, c.restarts, c.faults, st.node.state, st.node.starts, c.hash_now > 0 ? '>0' : '0'].join('|'); if (key !== last) { last = key; events.push({ t: Date.now(), key }); log(`card ${c.state || '(none)'} pid ${c.pid} restarts ${c.restarts} faults ${c.faults} hash ${Number(c.hash_now || 0).toFixed(1)} node ${st.node.state} starts ${st.node.starts} ${c.message ? '"' + c.message + '"' : ''}`); } @@ -285,15 +300,15 @@ S['one-card-fails'] = async () => { const exportsBefore = engineLogLines(/export-pack: /).length; writeFileSync(CTL, 'ok\ndevices 3\ndev 2 selftest\n'); log('fake worker: 3 devices, device 2 fails its self-test'); const tInject = Date.now(); - const listed = await until((st) => st.mining.cards.length >= 3, 200000, 'three cards listed'); + const listed = await until((st) => fakes(st).length >= 3, 200000, 'three fake cards listed'); const tListed = Date.now(); - const twoMine = await until((st) => st.mining.cards.filter(c => c.state === 'mining' && c.hash_now > 0).length >= 2, 300000, 'two healthy cards mining'); + const twoMine = await until((st) => fakes(st).filter(c => c.state === 'mining' && c.hash_now > 0).length >= 2, 300000, 'two healthy cards mining'); const tTwo = Date.now(); - const held = await until((st) => st.mining.cards.some(c => /not usable on this driver/.test(c.message || '')), 120000, 'the failing card held with the reason'); + const held = await until((st) => fakes(st).some(c => /not usable on this driver/.test(c.message || '')), 120000, 'the failing card held with the reason'); await steady(90000); const st1 = await poll(); - const bad = st1 ? st1.mining.cards.find(c => /not usable on this driver/.test(c.message || '')) : null; - const good = st1 ? st1.mining.cards.filter(c => !/not usable/.test(c.message || '')) : []; + const bad = st1 ? fakes(st1).find(c => /not usable on this driver/.test(c.message || '')) : null; + const good = st1 ? fakes(st1).filter(c => !/not usable/.test(c.message || '')) : []; const exportsAfter = engineLogLines(/export-pack: /).length - exportsBefore; const reused = engineLogLines(/export-pack: reusing/).length; verdict('one-card-fails', [ From fd4c3e72a78f222373a9927f5d11f747511ef647 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 11:47:52 +0000 Subject: [PATCH 06/18] reliability injector: card-appears uses a second fake device (added, removed, revived); an empty device list is read by the engine as no answer Co-Authored-By: Claude Fable 5.1 (cherry picked from commit df2617bb4302b0aa21ae8b1728f753e87e541e89) --- tools/reliability/app-run.mjs | 114 +++++++++++++++++++++++++++++++++- 1 file changed, 112 insertions(+), 2 deletions(-) diff --git a/tools/reliability/app-run.mjs b/tools/reliability/app-run.mjs index 5e557cacb..900660140 100755 --- a/tools/reliability/app-run.mjs +++ b/tools/reliability/app-run.mjs @@ -17,8 +17,8 @@ // (the ladder), the reason stays on the card in plain words, nothing says "restarted once already" // and no card is ever "faulted"; the worker is healthy again: mining resumes with no tap // no-status the miner process is stopped with SIGSTOP: no status line for 90 s, the app restarts it -// card-appears MF-3: the card is absent from the enumeration at start (unplugged, or a problem code) and appears -// two minutes later: the hot-plug pass starts its worker with no tap (Linux and Windows only) +// card-appears MF-3: a second card appears (plugged in, or driven after a driver install): its worker starts with +// no tap; it leaves: its row is marked removed; it comes back: mining again (Linux and Windows only) // node-silent the node is stopped with SIGSTOP: no sign of life for 120 s, the app restarts the node in-process // and the miner comes back once it is ready // orphan-miner MF-7: a stray igneum-miner on this engine's node, not started by it, is killed by the minute sweep; @@ -249,6 +249,116 @@ S['no-status'] = async () => { ], { quiet_to_restart: s(tRestart - tInject), restart_to_mining: s(tBack - tRestart) }); }; +S['card-appears'] = async () => { + if (MAC) { verdict('card-appears', [{ ok: true, what: 'skipped on macOS (Apple silicon has no GPU hot-plug; the Metal path enumerates once)' }], {}); return; } + // MF-3: a second card appears in the enumeration (plugged in, or driven after a driver install): the hot-plug pass + // starts its worker with no tap; it leaves (the tool still answers, with one device): its row is marked removed + // and its worker stops; it comes back: revived in its slot, mining again. An EMPTY list is not used: the engine + // reads a tool that lists nothing as "did not answer" and removes no card on it (the right call for a driver crash). + await until(mining, 300000, 'mining on the first card'); + writeFileSync(CTL, 'ok\ndevices 2\n'); log('fake worker: a second device appears'); + const tAppear = Date.now(); + const listed = await until((st) => fakes(st).length >= 2, 200000, 'the second card to be listed by the hot-plug pass'); + const tListed = Date.now(); + const second = await until((st) => fakes(st).length >= 2 && fakes(st)[1].state === 'mining' && fakes(st)[1].hash_now > 0, 300000, 'the second card to mine with no tap'); + const tSecond = Date.now(); + writeFileSync(CTL, 'ok\ndevices 1\n'); log('fake worker: the second device leaves'); + const tGone = Date.now(); + const removed = await until((st) => fakes(st).length >= 2 && (fakes(st)[1].state === 'removed' || fakes(st)[1].removed === true), 200000, 'the hot-plug pass to mark the second card removed'); + const tMarked = Date.now(); + const stR = await poll(); + const firstStill = stR && fakes(stR)[0] && fakes(stR)[0].state === 'mining' && fakes(stR)[0].hash_now > 0; + writeFileSync(CTL, 'ok\ndevices 2\n'); log('fake worker: the second device is back'); + const tBack = Date.now(); + const revived = await until((st) => fakes(st).length >= 2 && fakes(st)[1].state === 'mining' && fakes(st)[1].hash_now > 0, 300000, 'the second card to mine again with no tap'); + const tRevived = Date.now(); + verdict('card-appears', [ + { ok: !!listed, what: `the hot-plug pass listed the card that appeared (${s(tListed - tAppear)})` }, + { ok: !!second, what: `its worker started with no tap (${s(tSecond - tAppear)} after it appeared)` }, + { ok: !!removed, what: `the card that left was marked removed (${s(tMarked - tGone)})` }, + { ok: !!firstStill, what: 'the other card kept mining through it' }, + { ok: !!revived, what: `the card that came back mined again with no tap (${s(tRevived - tBack)})` }, + ], { appear_to_listed: s(tListed - tAppear), appear_to_mining: s(tSecond - tAppear), gone_to_marked: s(tMarked - tGone), back_to_mining: s(tRevived - tBack) }); + // leave one device for the steps after this one + writeFileSync(CTL, 'ok\ndevices 1\n'); + await until((st) => fakes(st).length >= 2 && fakes(st)[1].state === 'removed', 150000, 'the second card gone again'); +}; + +S['own-restart'] = async () => { + setMode('ok'); + const st0 = await until(mining, 300000, 'first mining'); + const pid0 = card(st0).pid, r0 = card(st0).restarts; + const tInject = Date.now(); + setMode('fast'); + const faulted = await until((st, c) => c.faults >= 1, 90000, 'the worker fault to reach the card'); + const tFault = Date.now(); + setMode('ok'); + const back = await until((st, c) => mining(st, c) && c.faults >= 1, 90000, 'mining again after the worker restart'); + const tBack = Date.now(); + await steady(15000); + const st1 = await poll(); + verdict('own-restart', [ + { ok: !!st0, what: 'the card mined with a rate above 0 on the fake worker' }, + { ok: !!faulted && /worker fault|killed by a guard/.test(card(faulted).message || ''), what: `the card showed the worker fault (${JSON.stringify(card(faulted || {}).message)})` }, + { ok: !!back, what: 'mining resumed after the miner restarted its own worker' }, + { ok: st1 && card(st1).pid === pid0 && card(st1).restarts === r0, what: `the app did not restart the miner (pid ${pid0} -> ${card(st1 || {}).pid}, app restarts ${r0} -> ${card(st1 || {}).restarts})` }, + ], { inject_to_fault_on_card: s(tFault - tInject), fault_to_mining_again: s(tBack - tFault) }); +}; + +S['zero-ladder'] = async () => { + setMode('ok'); + const st0 = await until(mining, 120000, 'mining'); + const r0 = card(st0).restarts; + const tInject = Date.now(); + setMode('zero'); + // three rungs: the gaps between consecutive watchdog restarts must grow 10, 30, 120 s (plus the 60 s rule each time) + const marks = []; + let lastR = r0; + const end = Date.now() + 600000; + let faultedSeen = false, onceAlready = false; + while (Date.now() < end && marks.length < 3) { + const st = await poll(); if (!st) break; + const c = card(st); + if (c.state === 'faulted') faultedSeen = true; + if (/restarted once already/.test(c.message || '')) onceAlready = true; + if ((c.restarts || 0) > lastR) { lastR = c.restarts; marks.push({ t: Date.now(), wait: c.restart_in_s, msg: c.message }); log(`rung ${marks.length}: restart_in_s ${c.restart_in_s} "${c.message}"`); } + await sleep(500); + } + setMode('ok'); + const back = await until(mining, 400000, 'mining again on its own after the worker is healthy'); + const tBack = Date.now(); + const waits = marks.map(m => m.wait); + const ladderOk = waits.length === 3 && waits[0] >= 8 && waits[0] <= 10 && waits[1] >= 28 && waits[1] <= 30 && waits[2] >= 118 && waits[2] <= 120; + verdict('zero-ladder', [ + { ok: marks.length === 3, what: `three watchdog restarts observed (${marks.length})` }, + { ok: ladderOk, what: `the restart delays follow the ladder 10, 30, 120 s (saw ${waits.join(', ')})` }, + { ok: marks.every(m => /hash rate 0 for 60 s/.test(m.msg || '')), what: 'the reason stayed on the card in plain words' }, + { ok: !faultedSeen && !onceAlready, what: 'no "faulted" state and no "restarted once already" words' }, + { ok: !!back, what: 'mining resumed on its own once the worker was healthy' }, + ], { zero_to_rung1: marks[0] ? s(marks[0].t - tInject) : 'n/a', rung_gaps: marks.slice(1).map((m, i) => s(m.t - marks[i].t)).join(', '), healthy_to_mining: s(tBack - (marks[2] ? marks[2].t : tInject)) }); +}; + +S['no-status'] = async () => { + setMode('ok'); + const st0 = await until(mining, 400000, 'mining'); + const pid0 = card(st0).pid, r0 = card(st0).restarts; + const tInject = Date.now(); + process.kill(pid0, 'SIGSTOP'); + log(`SIGSTOP miner pid ${pid0}`); + const rs = await until((st, c) => c.restarts > r0 || /no status line/.test(c.message || ''), 150000, 'the watchdog restart'); + const tRestart = Date.now(); + const back = await until((st, c) => mining(st, c) && c.pid !== pid0 && c.pid > 0, 400000, 'mining on the restarted miner'); + const tBack = Date.now(); + let gone = false; try { process.kill(pid0, 0); } catch { gone = true; } + if (!gone) { try { process.kill(pid0, 'SIGKILL'); } catch {} } + verdict('no-status', [ + { ok: !!rs && /no status line/.test(card(rs).message || ''), what: `the watchdog restarted the miner for missing status lines (${JSON.stringify(card(rs || {}).message)})` }, + { ok: rs && tRestart - tInject >= 85000 && tRestart - tInject <= 130000, what: `between 85 and 130 s after the miner went quiet (${s(tRestart - tInject)})` }, + { ok: !!back, what: 'mining resumed on a new miner process' }, + { ok: gone, what: 'the stopped miner process was killed' }, + ], { quiet_to_restart: s(tRestart - tInject), restart_to_mining: s(tBack - tRestart) }); +}; + S['card-appears'] = async () => { if (MAC) { verdict('card-appears', [{ ok: true, what: 'skipped on macOS (Apple silicon has no GPU hot-plug; the Metal path enumerates once)' }], {}); return; } // the card leaves the enumeration (unplugged, or a problem code: the hot-plug pass marks it removed and stops its From 313059f6bd70aff422d95e032a23cd94bc9b8c18 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 11:54:33 +0000 Subject: [PATCH 07/18] engine: the row's countdown is set the moment a watchdog restart is scheduled; injector reads the rung over two seconds Co-Authored-By: Claude Fable 5.1 (cherry picked from commit e1aeecad40e023cc4c82829111674e880aa51d29) --- app/igneum-app/src/engine.rs | 1 + 1 file changed, 1 insertion(+) diff --git a/app/igneum-app/src/engine.rs b/app/igneum-app/src/engine.rs index 1ef9fdd78..0f9d536a9 100644 --- a/app/igneum-app/src/engine.rs +++ b/app/igneum-app/src/engine.rs @@ -4307,6 +4307,7 @@ impl Engine { c.restarts = self.miners[i].restarts; c.hash_now = 0.0; c.pid = 0; + c.restart_in_s = delay_s; c.message = format!("{reason}; trying again"); } } From 942f537ff6fb07dafd69721f23a06a6760d09557 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 11:55:12 +0000 Subject: [PATCH 08/18] reliability injector: the stale duplicate step blocks removed (the card-appears rewrite had sliced before an earlier definition); the rung is read over two seconds Co-Authored-By: Claude Fable 5.1 (cherry picked from commit 10badc02a33c559adc7ef8e1a02af13ba49c5075) --- tools/reliability/app-run.mjs | 104 +++------------------------------- 1 file changed, 8 insertions(+), 96 deletions(-) diff --git a/tools/reliability/app-run.mjs b/tools/reliability/app-run.mjs index 900660140..539cdee46 100755 --- a/tools/reliability/app-run.mjs +++ b/tools/reliability/app-run.mjs @@ -211,7 +211,14 @@ S['zero-ladder'] = async () => { const c = card(st); if (c.state === 'faulted') faultedSeen = true; if (/restarted once already/.test(c.message || '')) onceAlready = true; - if ((c.restarts || 0) > lastR) { lastR = c.restarts; marks.push({ t: Date.now(), wait: c.restart_in_s, msg: c.message }); log(`rung ${marks.length}: restart_in_s ${c.restart_in_s} "${c.message}"`); } + if ((c.restarts || 0) > lastR) { + lastR = c.restarts; + // the row's countdown as it stands over the next two seconds (the first read can land on the same tick) + let wait = c.restart_in_s || 0; + for (let k = 0; k < 4; k++) { await sleep(500); const s2 = await state(); const c2 = s2 ? card(s2) : {}; wait = Math.max(wait, c2.restart_in_s || 0); } + marks.push({ t: Date.now(), wait, msg: c.message }); + log(`rung ${marks.length}: restart_in_s ${wait} "${c.message}"`); + } await sleep(500); } setMode('ok'); @@ -284,101 +291,6 @@ S['card-appears'] = async () => { await until((st) => fakes(st).length >= 2 && fakes(st)[1].state === 'removed', 150000, 'the second card gone again'); }; -S['own-restart'] = async () => { - setMode('ok'); - const st0 = await until(mining, 300000, 'first mining'); - const pid0 = card(st0).pid, r0 = card(st0).restarts; - const tInject = Date.now(); - setMode('fast'); - const faulted = await until((st, c) => c.faults >= 1, 90000, 'the worker fault to reach the card'); - const tFault = Date.now(); - setMode('ok'); - const back = await until((st, c) => mining(st, c) && c.faults >= 1, 90000, 'mining again after the worker restart'); - const tBack = Date.now(); - await steady(15000); - const st1 = await poll(); - verdict('own-restart', [ - { ok: !!st0, what: 'the card mined with a rate above 0 on the fake worker' }, - { ok: !!faulted && /worker fault|killed by a guard/.test(card(faulted).message || ''), what: `the card showed the worker fault (${JSON.stringify(card(faulted || {}).message)})` }, - { ok: !!back, what: 'mining resumed after the miner restarted its own worker' }, - { ok: st1 && card(st1).pid === pid0 && card(st1).restarts === r0, what: `the app did not restart the miner (pid ${pid0} -> ${card(st1 || {}).pid}, app restarts ${r0} -> ${card(st1 || {}).restarts})` }, - ], { inject_to_fault_on_card: s(tFault - tInject), fault_to_mining_again: s(tBack - tFault) }); -}; - -S['zero-ladder'] = async () => { - setMode('ok'); - const st0 = await until(mining, 120000, 'mining'); - const r0 = card(st0).restarts; - const tInject = Date.now(); - setMode('zero'); - // three rungs: the gaps between consecutive watchdog restarts must grow 10, 30, 120 s (plus the 60 s rule each time) - const marks = []; - let lastR = r0; - const end = Date.now() + 600000; - let faultedSeen = false, onceAlready = false; - while (Date.now() < end && marks.length < 3) { - const st = await poll(); if (!st) break; - const c = card(st); - if (c.state === 'faulted') faultedSeen = true; - if (/restarted once already/.test(c.message || '')) onceAlready = true; - if ((c.restarts || 0) > lastR) { lastR = c.restarts; marks.push({ t: Date.now(), wait: c.restart_in_s, msg: c.message }); log(`rung ${marks.length}: restart_in_s ${c.restart_in_s} "${c.message}"`); } - await sleep(500); - } - setMode('ok'); - const back = await until(mining, 400000, 'mining again on its own after the worker is healthy'); - const tBack = Date.now(); - const waits = marks.map(m => m.wait); - const ladderOk = waits.length === 3 && waits[0] >= 8 && waits[0] <= 10 && waits[1] >= 28 && waits[1] <= 30 && waits[2] >= 118 && waits[2] <= 120; - verdict('zero-ladder', [ - { ok: marks.length === 3, what: `three watchdog restarts observed (${marks.length})` }, - { ok: ladderOk, what: `the restart delays follow the ladder 10, 30, 120 s (saw ${waits.join(', ')})` }, - { ok: marks.every(m => /hash rate 0 for 60 s/.test(m.msg || '')), what: 'the reason stayed on the card in plain words' }, - { ok: !faultedSeen && !onceAlready, what: 'no "faulted" state and no "restarted once already" words' }, - { ok: !!back, what: 'mining resumed on its own once the worker was healthy' }, - ], { zero_to_rung1: marks[0] ? s(marks[0].t - tInject) : 'n/a', rung_gaps: marks.slice(1).map((m, i) => s(m.t - marks[i].t)).join(', '), healthy_to_mining: s(tBack - (marks[2] ? marks[2].t : tInject)) }); -}; - -S['no-status'] = async () => { - setMode('ok'); - const st0 = await until(mining, 400000, 'mining'); - const pid0 = card(st0).pid, r0 = card(st0).restarts; - const tInject = Date.now(); - process.kill(pid0, 'SIGSTOP'); - log(`SIGSTOP miner pid ${pid0}`); - const rs = await until((st, c) => c.restarts > r0 || /no status line/.test(c.message || ''), 150000, 'the watchdog restart'); - const tRestart = Date.now(); - const back = await until((st, c) => mining(st, c) && c.pid !== pid0 && c.pid > 0, 400000, 'mining on the restarted miner'); - const tBack = Date.now(); - let gone = false; try { process.kill(pid0, 0); } catch { gone = true; } - if (!gone) { try { process.kill(pid0, 'SIGKILL'); } catch {} } - verdict('no-status', [ - { ok: !!rs && /no status line/.test(card(rs).message || ''), what: `the watchdog restarted the miner for missing status lines (${JSON.stringify(card(rs || {}).message)})` }, - { ok: rs && tRestart - tInject >= 85000 && tRestart - tInject <= 130000, what: `between 85 and 130 s after the miner went quiet (${s(tRestart - tInject)})` }, - { ok: !!back, what: 'mining resumed on a new miner process' }, - { ok: gone, what: 'the stopped miner process was killed' }, - ], { quiet_to_restart: s(tRestart - tInject), restart_to_mining: s(tBack - tRestart) }); -}; - -S['card-appears'] = async () => { - if (MAC) { verdict('card-appears', [{ ok: true, what: 'skipped on macOS (Apple silicon has no GPU hot-plug; the Metal path enumerates once)' }], {}); return; } - // the card leaves the enumeration (unplugged, or a problem code: the hot-plug pass marks it removed and stops its - // worker), then comes back: its worker must start with no tap - await until(mining, 300000, 'mining before the card leaves'); - setMode('absent'); - const tGone = Date.now(); - const gone = await until((st, c) => c.state === 'removed' || c.removed === true || /removed/.test(c.state || ''), 150000, 'the hot-plug pass to mark the card removed'); - const tMarked = Date.now(); - await steady(30000); - setMode('ok'); - const tAppear = Date.now(); - const back = await until(mining, 300000, 'the worker to start with no tap once the card is back'); - const tBack = Date.now(); - verdict('card-appears', [ - { ok: !!gone, what: `the hot-plug pass marked the card removed when it left (${s(tMarked - tGone)})` }, - { ok: !!back, what: 'its worker started with no tap once it was back' }, - ], { gone_to_marked: s(tMarked - tGone), back_to_mining: s(tBack - tAppear) }); -}; - S['node-silent'] = async () => { setMode('ok'); const st0 = await until(mining, 400000, 'mining'); From ae35330f422278e652610d86bc193d220650ed59 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 12:07:12 +0000 Subject: [PATCH 09/18] miner-faults MF-8: a node dying at start on a kept datadir (the serde(default) row, N13); the LG-4 job's tenth install keeps the datadir and a node restart loop fails the row Co-Authored-By: Claude Fable 5.1 (cherry picked from commit d82de28e405cca1ac4f8be1786ac947ec0bdcfe1) --- docs/plans/miner-faults.md | 2 ++ relay/playbooks/first-share.ps1 | 11 ++++++++--- tools/fleet/first-share-gate.mjs | 7 +++++-- 3 files changed, 15 insertions(+), 5 deletions(-) diff --git a/docs/plans/miner-faults.md b/docs/plans/miner-faults.md index 5e7f0a211..c3bbbcc97 100644 --- a/docs/plans/miner-faults.md +++ b/docs/plans/miner-faults.md @@ -14,6 +14,7 @@ Standing rules behind every row (branch `miner-reliability`, off `release-0.3.19 | A card swap, a driver install or a restart needs no tap: the hot-plug enumeration every 60 s (`src/hotplug.rs`) starts the worker of a card that appears, recovers from a problem code or revives | `engine.rs` `merge_detection` → `plan_miners` | | Every fault line reports to the log intake the moment it happens, with the card, the class, the reason and the app version | `engine.rs` `fault_report` → `update::upload_text`; label `fault--`; read with `node tools/logs.mjs` | | Nothing on a user's machine is changed by a one-off script: card settings travel as the signed `cards` job kind (per card enabled, identities, power_pct), applied through the app's own card path, persisted, read back in the report, refused for a card the machine does not have | `src/jobs.rs` (`KINDS`, `validate_params`), `src/jobrun.rs` (`cards_job_choices`, `cards_applied`), `engine.rs` (`Action::ApplyCards`), `packaging/ota/publish-jobs.sh add --kind cards --cards "key=on:8"` | +| A cut never dies on a kept datadir: every stored-row schema change carries a versioned read path, and the canary and LG-4 start the new node on the previous version's datadir beside a wiped one (MF-8) | the node line's N13; `tools/fleet/first-share-gate.mjs` row "kept" (the installer over an existing install, the datadir kept); the canary form | | The fresh-install claim (LG-4) is a job, not a runbook: `tools/fleet/first-share-gate.mjs` on rented Windows boxes, on every cut, its line read by the shipper's publish | `relay/playbooks/first-share.ps1`, `tools/fleet/first-share-gate.mjs`, `site/evidence/first-share-.json` | ## The register @@ -26,6 +27,7 @@ Standing rules behind every row (branch `miner-reliability`, off `release-0.3.19 | MF-5 | 7 Oct 2026, PC 1, 0.3.17 node, 24 identities (the real cause of the 11:2x faults; MF-4 withdrawn as the cause) | Every card faulted "no status line from the miner for 90 s (restarted once already)" after the restart | Evidence (PC 1, 7 Oct 2026 11:4x UK): with 8 identities a card (24 template fetches a round) the 0.3.17 node answered no template in 5 s; with 2 a card (4 fetches) both cards mined at full rate (5090 122.4 MH/s, 9070 XT 18.9) within two minutes. The node's getBlockTemplate answered past 5 s with 24 identities fetching; the miners waited for a template inside their job-fill loop and printed no STATUS at all; the watchdog read the silence as the worker's; one restart, then faulted for good | Miner: STATUS every interval whatever the template state (`template_wait=` while it waits, `template_ms=` the node's last template time, `identities_active=`); the feed fetches only as many identities as fit one pass inside 8 s at the node's measured template time (`identities_for`: all of them when the node answers under 1 s; 24 at 8 s per template becomes 1), raised again when it answers faster; a pass whose fetches all fail backs off 2, 4, 8, 10 s and retries for ever with a `NODE SLOW` line once per 30 s. App: a STATUS with `template_wait>0` is the miner's heartbeat and the node's latency, never the card's fault (no zero-rate clock, no restart); the card reads "node slow: waiting for a block template for N s; the worker is kept" or "node slow: a template takes N s; k of n identities active"; the mitigation of the day (a one-off script POSTing /api/cards) is closed by the signed `cards` job kind | `watchdog` tests `a_slow_node_never_faults_the_card`, `parses_status_and_fault_lines` (the 0.3.20 line); `jobs` test for the `cards` kind; injector step `slow-node` (a template stub answering in 8 s while three cards run: open, needs the stub) | `app-run slow-node PASS` (0.3.20) | | MF-6 | 7 Oct 2026, PC 1, 0.3.19 | After a `--stop-miners` job, a following read-only job kept both cards "off, held for a remote job" for its whole three minutes | The engine released the hold only when no job held the miners; the next job's active state hid the release (engine.rs 3032 class) | A hold belongs to the job that took it (`job_hold_owner`) and releases the moment that job is no longer the running one, whatever runs next, or when its own cap passes (logged); a read-only job never holds (`jobrun::hold_release`) | `jobrun` test `a_hold_belongs_to_the_job_that_took_it` (owner running, another job, no job, cap passed, no owner) | `app tests: hold rule green` | | MF-7 | 7 Oct 2026, PC 1, 0.3.19 | Orphan `igneum-miner.exe` processes the app no longer tracked (two alive under `--stop-miners` with their rows at pid 0, one after) hammered the node's template RPC beside the tracked miners | The engine lost track of miners it had started (a stop that timed out, a restart over a live process) and never looked for them again | The engine owns every miner it started: at start, after every stop and every minute it kills any `igneum-miner` whose command line carries THIS engine's node RPC (the fence) and whose pid it does not track, one log line and one fault report per kill, never by name alone (`sweep_orphan_miners`, `platform::miner_processes`, `kill_pid`); a restart kills the slot's old process before the new one starts | injector step `orphan-miner` (a stray miner on the engine's node is killed inside the minute, the engine's own miner left alone) | `app-run orphan-miner PASS` | +| MF-8 | 7 Oct 2026, 0.3.19 on a kept datadir | The node binary died at start on a datadir the previous version had written; the app restarted it in a loop and every card waited | A field added to a stored row (`serde(default)` on `BlockRewardData` since 10db4b61) was read by bincode from the old row short: `DeserializationError UnexpectedEof` at `virtual_state.rs:250`; fixed on the 0.3.20 node line as N13 with a read-and-rewrite | Every schema change to a stored row ships with a versioned read path (the old shape read, rewritten in the new one); every canary and the fresh-install gate start the new node on a KEPT datadir of the previous version beside the wiped one, and a node that dies at start on a kept datadir is a FAIL of the cut, not a user's reset | The node line's N13 test (the old row read); the canary's kept-datadir start; `app-run.mjs` reads a node exit inside 10 s of its start as the engine's "igneumd exited at once" line and reports it (`node-exit` FAULT line to the intake) | `canary kept-datadir start PASS` and `LG-4` row "kept datadir" beside "wiped" | | MF-3 | 7 Oct 2026, PC 1, Intel Arc | The Intel driver's first install did not bind: the device sat in Code 12 at install time; the card never mined until a reboot | A driver installed while the device reports a problem code (12, 43, 31) does not bind; nothing re-scanned the device afterwards, and the app only re-enumerates | The app re-enumerates every 60 s and starts the worker the minute the OS drives the card (`hotplug::diff` recovered / revived, `settle_new`); the row says what to do while it does not ("reboot with the card attached; if it persists, reinstall the driver with the card attached"); a Windows host asks for a re-scan (`pnputil /scan-devices`) after a problem code is seen, every 5 minutes, at most 6 times (follow-up, host side) | `hotplug` test `a_driven_card_that_turns_faulty_is_errored_and_recovers_later`; `app-run.mjs` step `card-appears` (a card listed after 2 minutes starts without a tap) | `app-run card-appears PASS` | ## Commits (7 October 2026) diff --git a/relay/playbooks/first-share.ps1 b/relay/playbooks/first-share.ps1 index ac0ce4028..22930d1ef 100644 --- a/relay/playbooks/first-share.ps1 +++ b/relay/playbooks/first-share.ps1 @@ -13,19 +13,23 @@ param( [Parameter(Mandatory = $true)][string]$Url, [string]$Address = '0x4242424242424242424242424242424242424242', [int]$BudgetSeconds = 1800, - [switch]$AllowInstalled + [switch]$AllowInstalled, + # MF-8: install over the previous install and KEEP its datadir (the node must come up on the old version's rows) + [switch]$Keep ) $ErrorActionPreference = 'Stop' function Now { [int64]([DateTimeOffset]::UtcNow.ToUnixTimeMilliseconds()) } function Result([string]$line) { Write-Host ("RESULT " + $line) } $appData = Join-Path $env:LOCALAPPDATA 'igneum' $programs = Join-Path $env:LOCALAPPDATA 'Programs\Igneum Miner' +$kept = $false if ((Test-Path $appData) -or (Test-Path $programs)) { - if (-not $AllowInstalled) { Result "total_s=0 pass=false reason=installed_app_present (this is not a fresh box; -AllowInstalled overrides, with the owner's word)"; exit 2 } + if (-not $AllowInstalled -and -not $Keep) { Result "total_s=0 pass=false reason=installed_app_present (this is not a fresh box; -AllowInstalled overrides, with the owner's word)"; exit 2 } Get-Process -Name 'igneum-app', 'Igneum Miner', 'igneumd', 'igneum-miner' -ErrorAction SilentlyContinue | Stop-Process -Force -ErrorAction SilentlyContinue Start-Sleep -Seconds 3 - Remove-Item -Recurse -Force $appData -ErrorAction SilentlyContinue + if ($Keep) { $kept = Test-Path (Join-Path $appData 'devnet-v4') } else { Remove-Item -Recurse -Force $appData -ErrorAction SilentlyContinue } } +Result "kept=$($kept.ToString().ToLower())" # 1. download: the public installer through the dl host, timed from the first byte $setup = Join-Path $env:TEMP 'Igneum-Miner-Setup-first-share.exe' Remove-Item $setup -ErrorAction SilentlyContinue @@ -64,6 +68,7 @@ $deadline = $tDl0 + ($BudgetSeconds * 1000) while ((Now) -lt $deadline) { try { $st = Invoke-RestMethod -Uri ($url + 'api/state') -TimeoutSec 5 } catch { Start-Sleep -Seconds 2; continue } if (-not $synced -and $st.node.synced) { $synced = Now; $stall = 'card'; Result "step=node end=$synced synced_at=$($st.ladder.first_synced_at)" } + if ($st.node.restarts -ge 2 -and -not $synced) { Result "step=node restarts=$($st.node.restarts) message=$($st.node.message)"; Result "total_s=$([int](((Now) - $tDl0) / 1000)) pass=false stalled_in=node reason=node_restart_loop kept=$($kept.ToString().ToLower())"; break } if ($st.ladder -and $st.ladder.first_mining_at -and -not $mining) { $mining = $st.ladder.first_mining_at } if ($st.ladder -and $st.ladder.first_block_at) { $first = Now; $hash = $st.ladder.first_block_hash; break } if ($st.ladder -and $st.ladder.first_share_at) { $first = Now; $hash = 'share'; break } diff --git a/tools/fleet/first-share-gate.mjs b/tools/fleet/first-share-gate.mjs index 57f16c8d4..abf502daa 100755 --- a/tools/fleet/first-share-gate.mjs +++ b/tools/fleet/first-share-gate.mjs @@ -11,6 +11,8 @@ // with no row the gate writes "LG-4 not run: no Windows box" and `check` fails. Each box runs installs in turn; boxes run in // parallel. Output: site/evidence/first-share-.json (the runbook's keys per row, the medians, the pass count, the // installer's sha256) and the gate line on stdout and in the file: "LG-4 PASS k/10" or "LG-4 FAIL k/10". +// The tenth install of a run is the KEPT-datadir case (MF-8): the installer runs over the ninth install with its datadir +// kept, and the node must come up on it; that row carries kept=true. The other nine are wiped. // Never touches PC 1 (the project lead's desk) and refuses a box whose installed app would be replaced unless --allow-installed is given // and the box's row says "job_box": true (PC 2's case needs the project lead's word, recorded in the row by the fleet lane). @@ -74,14 +76,15 @@ const started = Date.now(); // boxes in parallel (one ssh each, installs in turn inside it), the Mac waits const procs = boxes.map((b) => { const n = Math.min(perBox, installs - rows.length); - const remote = `$ErrorActionPreference='Continue'; $f=\"$env:TEMP\\first-share.ps1\"; [IO.File]::WriteAllText($f, [Text.Encoding]::UTF8.GetString([Convert]::FromBase64String('${Buffer.from(readFileSync(join(ROOT, 'relay', 'playbooks', 'first-share.ps1'))).toString('base64')}'))); 1..${n} | ForEach-Object { Write-Host \"INSTALL $_\"; powershell -ExecutionPolicy Bypass -File $f -Url '${url}' ${allowInstalled && b.job_box ? '-AllowInstalled' : ''} }`; + // the last install on the first box keeps the previous install's datadir (-Keep): the MF-8 row + const remote = `$ErrorActionPreference='Continue'; $f=\"$env:TEMP\\first-share.ps1\"; [IO.File]::WriteAllText($f, [Text.Encoding]::UTF8.GetString([Convert]::FromBase64String('${Buffer.from(readFileSync(join(ROOT, 'relay', 'playbooks', 'first-share.ps1'))).toString('base64')}'))); 1..${n} | ForEach-Object { Write-Host \"INSTALL $_\"; $keep = if ($_ -eq ${n} -and ${b === boxes[0] ? '$true' : '$false'}) { '-Keep' } else { '' }; powershell -ExecutionPolicy Bypass -File $f -Url '${url}' ${allowInstalled && b.job_box ? '-AllowInstalled' : ''} $keep }`; const key = (b.key || '~/.ssh/igneum-fleet').replace(/^~/, homedir()); const p = spawnSync('ssh', ['-o', 'ConnectTimeout=20', '-o', 'StrictHostKeyChecking=accept-new', '-i', key, '-p', String(b.port || 22), `${b.user}@${b.host}`, 'powershell', '-NoProfile', '-Command', remote], { encoding: 'utf8', timeout: (n * 1900 + 120) * 1000 }); return { box: b, out: (p.stdout || '') + (p.stderr || ''), status: p.status }; }); for (const { box, out } of procs) { for (const chunk of out.split(/^INSTALL \d+\r?$/m).slice(1)) { - const r = parse(chunk); r.box = box.label; r.gpu = box.gpu || ''; rows.push(r); + const r = parse(chunk); r.box = box.label; r.gpu = box.gpu || ''; r.kept = /kept=true/.test(chunk); rows.push(r); } if (!out.includes('RESULT')) rows.push({ box: box.label, gpu: box.gpu || '', pass: false, total_s: 0, error: out.trim().split('\n').slice(-3).join(' | ') }); } From 635e8576a88dd15af6b7b7e2fccba5ba57a80515 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 12:07:36 +0000 Subject: [PATCH 10/18] miner-faults MF-8: the class in the shipper's words (node refuses its own kept datadir after an update), N13 b7cc37e7, the gate on both datadir shapes, the injector step owed Co-Authored-By: Claude Fable 5.1 (cherry picked from commit 291ea5418f89d9120b00cc25bc37f358e968322f) --- docs/plans/miner-faults.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/plans/miner-faults.md b/docs/plans/miner-faults.md index c3bbbcc97..c154f320a 100644 --- a/docs/plans/miner-faults.md +++ b/docs/plans/miner-faults.md @@ -14,7 +14,7 @@ Standing rules behind every row (branch `miner-reliability`, off `release-0.3.19 | A card swap, a driver install or a restart needs no tap: the hot-plug enumeration every 60 s (`src/hotplug.rs`) starts the worker of a card that appears, recovers from a problem code or revives | `engine.rs` `merge_detection` → `plan_miners` | | Every fault line reports to the log intake the moment it happens, with the card, the class, the reason and the app version | `engine.rs` `fault_report` → `update::upload_text`; label `fault--`; read with `node tools/logs.mjs` | | Nothing on a user's machine is changed by a one-off script: card settings travel as the signed `cards` job kind (per card enabled, identities, power_pct), applied through the app's own card path, persisted, read back in the report, refused for a card the machine does not have | `src/jobs.rs` (`KINDS`, `validate_params`), `src/jobrun.rs` (`cards_job_choices`, `cards_applied`), `engine.rs` (`Action::ApplyCards`), `packaging/ota/publish-jobs.sh add --kind cards --cards "key=on:8"` | -| A cut never dies on a kept datadir: every stored-row schema change carries a versioned read path, and the canary and LG-4 start the new node on the previous version's datadir beside a wiped one (MF-8) | the node line's N13; `tools/fleet/first-share-gate.mjs` row "kept" (the installer over an existing install, the datadir kept); the canary form | +| A cut never dies on a kept datadir: every stored-row schema change carries a versioned read path, and every release's gate starts the pinned binary on a copy of a standing box's datadir (Linux and Windows shapes) beside the wiped canary; LG-4's tenth install keeps the ninth's datadir (MF-8) | the node line's N13 (b7cc37e7); the canary form; `tools/fleet/first-share-gate.mjs` row kept=true | | The fresh-install claim (LG-4) is a job, not a runbook: `tools/fleet/first-share-gate.mjs` on rented Windows boxes, on every cut, its line read by the shipper's publish | `relay/playbooks/first-share.ps1`, `tools/fleet/first-share-gate.mjs`, `site/evidence/first-share-.json` | ## The register @@ -27,7 +27,7 @@ Standing rules behind every row (branch `miner-reliability`, off `release-0.3.19 | MF-5 | 7 Oct 2026, PC 1, 0.3.17 node, 24 identities (the real cause of the 11:2x faults; MF-4 withdrawn as the cause) | Every card faulted "no status line from the miner for 90 s (restarted once already)" after the restart | Evidence (PC 1, 7 Oct 2026 11:4x UK): with 8 identities a card (24 template fetches a round) the 0.3.17 node answered no template in 5 s; with 2 a card (4 fetches) both cards mined at full rate (5090 122.4 MH/s, 9070 XT 18.9) within two minutes. The node's getBlockTemplate answered past 5 s with 24 identities fetching; the miners waited for a template inside their job-fill loop and printed no STATUS at all; the watchdog read the silence as the worker's; one restart, then faulted for good | Miner: STATUS every interval whatever the template state (`template_wait=` while it waits, `template_ms=` the node's last template time, `identities_active=`); the feed fetches only as many identities as fit one pass inside 8 s at the node's measured template time (`identities_for`: all of them when the node answers under 1 s; 24 at 8 s per template becomes 1), raised again when it answers faster; a pass whose fetches all fail backs off 2, 4, 8, 10 s and retries for ever with a `NODE SLOW` line once per 30 s. App: a STATUS with `template_wait>0` is the miner's heartbeat and the node's latency, never the card's fault (no zero-rate clock, no restart); the card reads "node slow: waiting for a block template for N s; the worker is kept" or "node slow: a template takes N s; k of n identities active"; the mitigation of the day (a one-off script POSTing /api/cards) is closed by the signed `cards` job kind | `watchdog` tests `a_slow_node_never_faults_the_card`, `parses_status_and_fault_lines` (the 0.3.20 line); `jobs` test for the `cards` kind; injector step `slow-node` (a template stub answering in 8 s while three cards run: open, needs the stub) | `app-run slow-node PASS` (0.3.20) | | MF-6 | 7 Oct 2026, PC 1, 0.3.19 | After a `--stop-miners` job, a following read-only job kept both cards "off, held for a remote job" for its whole three minutes | The engine released the hold only when no job held the miners; the next job's active state hid the release (engine.rs 3032 class) | A hold belongs to the job that took it (`job_hold_owner`) and releases the moment that job is no longer the running one, whatever runs next, or when its own cap passes (logged); a read-only job never holds (`jobrun::hold_release`) | `jobrun` test `a_hold_belongs_to_the_job_that_took_it` (owner running, another job, no job, cap passed, no owner) | `app tests: hold rule green` | | MF-7 | 7 Oct 2026, PC 1, 0.3.19 | Orphan `igneum-miner.exe` processes the app no longer tracked (two alive under `--stop-miners` with their rows at pid 0, one after) hammered the node's template RPC beside the tracked miners | The engine lost track of miners it had started (a stop that timed out, a restart over a live process) and never looked for them again | The engine owns every miner it started: at start, after every stop and every minute it kills any `igneum-miner` whose command line carries THIS engine's node RPC (the fence) and whose pid it does not track, one log line and one fault report per kill, never by name alone (`sweep_orphan_miners`, `platform::miner_processes`, `kill_pid`); a restart kills the slot's old process before the new one starts | injector step `orphan-miner` (a stray miner on the engine's node is killed inside the minute, the engine's own miner left alone) | `app-run orphan-miner PASS` | -| MF-8 | 7 Oct 2026, 0.3.19 on a kept datadir | The node binary died at start on a datadir the previous version had written; the app restarted it in a loop and every card waited | A field added to a stored row (`serde(default)` on `BlockRewardData` since 10db4b61) was read by bincode from the old row short: `DeserializationError UnexpectedEof` at `virtual_state.rs:250`; fixed on the 0.3.20 node line as N13 with a read-and-rewrite | Every schema change to a stored row ships with a versioned read path (the old shape read, rewritten in the new one); every canary and the fresh-install gate start the new node on a KEPT datadir of the previous version beside the wiped one, and a node that dies at start on a kept datadir is a FAIL of the cut, not a user's reset | The node line's N13 test (the old row read); the canary's kept-datadir start; `app-run.mjs` reads a node exit inside 10 s of its start as the engine's "igneumd exited at once" line and reports it (`node-exit` FAULT line to the intake) | `canary kept-datadir start PASS` and `LG-4` row "kept datadir" beside "wiped" | +| MF-8 | 7 Oct 2026, every node build from 10db4b61 on a kept 0.3.17 datadir | "Node refuses its own kept datadir after an update": the app updates, the node dies at once (`DeserializationError(Io(Kind(UnexpectedEof)))` at `consensus/src/model/stores/virtual_state.rs:250`), the app restarts it in a loop, the miner never starts | `silent: bool` was added to `BlockRewardData` under `serde(default)` (10db4b61, the 0.3.16 vote-or-burn commit) and bincode ignores serde defaults, so the old row reads short; no canary saw it because every canary wiped. Fixed as ledger N13 on `release-0.3.20-node` b7cc37e7 (the store reads the v1 row and rewrites it) | Every schema change to a stored row ships with a versioned read path (the old shape read, rewritten in the new one); every release's gate starts the pinned binary on a COPY of a standing box's datadir, Linux-shaped and Windows-shaped (a copy of PC 2's), with the rewrite line or the synced line as the pass; the app never asks the user to wipe: a node that dies inside 10 s of its start is reported home (`node-exit` FAULT line) and the row says "the node cannot read its data after the update; the team has the report" | The node line's N13 test (the v1 row read and rewritten); the gate's kept-datadir start on both shapes; the injector step `kept-datadir` is OWED (a datadir from the previous release under the new binary needs a previous-release node on the pod; the shape is: run the old node to a few hundred blocks, stop it, start the new one on its datadir, pass on the synced line) | `canary kept-datadir start PASS (linux, windows)` in every cut's plan; `LG-4` row kept=true beside the nine wiped | | MF-3 | 7 Oct 2026, PC 1, Intel Arc | The Intel driver's first install did not bind: the device sat in Code 12 at install time; the card never mined until a reboot | A driver installed while the device reports a problem code (12, 43, 31) does not bind; nothing re-scanned the device afterwards, and the app only re-enumerates | The app re-enumerates every 60 s and starts the worker the minute the OS drives the card (`hotplug::diff` recovered / revived, `settle_new`); the row says what to do while it does not ("reboot with the card attached; if it persists, reinstall the driver with the card attached"); a Windows host asks for a re-scan (`pnputil /scan-devices`) after a problem code is seen, every 5 minutes, at most 6 times (follow-up, host side) | `hotplug` test `a_driven_card_that_turns_faulty_is_errored_and_recovers_later`; `app-run.mjs` step `card-appears` (a card listed after 2 minutes starts without a tap) | `app-run card-appears PASS` | ## Commits (7 October 2026) From ced26986fdff7a01c330c5fd9a9c8c57362dbf12 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 12:12:10 +0000 Subject: [PATCH 11/18] execrpc: an empty or unparsable reply is no answer, so a stopped node reads as silent to the watchdog's probe Co-Authored-By: Claude Fable 5.1 (cherry picked from commit 073fbea1d1b237048f1e6266a6b138119e9d5c50) --- app/igneum-app/src/execrpc.rs | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/app/igneum-app/src/execrpc.rs b/app/igneum-app/src/execrpc.rs index 29cbe478f..ee7e36531 100644 --- a/app/igneum-app/src/execrpc.rs +++ b/app/igneum-app/src/execrpc.rs @@ -40,7 +40,9 @@ fn post(evm_port: u16, method: &str, params: Value, timeout: Duration) -> Result ); let _ = std::fs::remove_file(&tmp); let out = out.ok_or_else(|| format!("{method}: the node's RPC did not answer"))?; - let v: Value = serde_json::from_str(&out).map_err(|e| format!("{method}: {e}"))?; + // an empty or unparsable body is no answer (a stopped node's socket still accepts the connection and curl + // returns nothing inside its own limit; the probe must read that as silence, 7 October 2026) + let v: Value = serde_json::from_str(&out).map_err(|e| format!("{method}: the node's RPC did not answer ({e})"))?; if let Some(err) = v.get("error") { return Err(format!("{method}: {}", err.get("message").and_then(|m| m.as_str()).unwrap_or("error"))); } From cb8a26543149d4213800b2faa5a43cfaf31739f7 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 12:27:44 +0000 Subject: [PATCH 12/18] fake worker: the OpenCL listing carries the real worker's header line, so the engine reads an answered enumeration and removes a card that left Co-Authored-By: Claude Fable 5.1 (cherry picked from commit ecd798747a26d339ff514da04773be858780636c) --- tools/reliability/fake-worker.mjs | 3 +++ 1 file changed, 3 insertions(+) diff --git a/tools/reliability/fake-worker.mjs b/tools/reliability/fake-worker.mjs index c6527a851..5f6b41ea8 100755 --- a/tools/reliability/fake-worker.mjs +++ b/tools/reliability/fake-worker.mjs @@ -29,6 +29,9 @@ import { createInterface } from 'node:readline'; if (process.argv.includes('--list')) { const m = (() => { try { return require('node:fs').readFileSync(process.env.FAKE_WORKER_CTL, 'utf8').trim().split(/\s+/)[0]; } catch { return 'ok'; } })(); const n = m === 'absent' ? 0 : (() => { try { const l = require('node:fs').readFileSync(process.env.FAKE_WORKER_CTL, 'utf8').split('\n').map(x => x.trim().split(/\s+/)).find(f => f[0] === 'devices'); return l ? Number(l[1]) : 1; } catch { return 1; } })(); + // the real worker's header line: the engine reads a listing without it as "the tool did not answer" and removes + // no card on it (src/detect.rs parse_opencl_list), which is right for a crashed driver and wrong for a stand-in + process.stdout.write(`OpenCL devices (${n}):\n`); for (let i = 0; i < n; i++) { process.stdout.write(`[${i}] Fake GPU ${i + 1} | Fake Platform (OpenCL 1.2)\n GPU, vendor Fake Silicon, driver 1.0.0, OpenCL C 1.2, 8 compute units, 8192 MB\n`); } From 87c17c84fc0a6cda0c5880567b61889e7ec16b62 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 12:30:43 +0000 Subject: [PATCH 13/18] miner-faults MF-9: a node that stops slower than its restart (the listener watchdog's sleep-then-check); every poll loop returns on shutdown, a sub-second shutdown is a named release gate Co-Authored-By: Claude Fable 5.1 (cherry picked from commit 7a0115617b9fa7b7e8e9930c8cfb530af71d4d55) --- docs/plans/miner-faults.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/docs/plans/miner-faults.md b/docs/plans/miner-faults.md index c154f320a..572c3425f 100644 --- a/docs/plans/miner-faults.md +++ b/docs/plans/miner-faults.md @@ -15,6 +15,7 @@ Standing rules behind every row (branch `miner-reliability`, off `release-0.3.19 | Every fault line reports to the log intake the moment it happens, with the card, the class, the reason and the app version | `engine.rs` `fault_report` → `update::upload_text`; label `fault--`; read with `node tools/logs.mjs` | | Nothing on a user's machine is changed by a one-off script: card settings travel as the signed `cards` job kind (per card enabled, identities, power_pct), applied through the app's own card path, persisted, read back in the report, refused for a card the machine does not have | `src/jobs.rs` (`KINDS`, `validate_params`), `src/jobrun.rs` (`cards_job_choices`, `cards_applied`), `engine.rs` (`Action::ApplyCards`), `packaging/ota/publish-jobs.sh add --kind cards --cards "key=on:8"` | | A cut never dies on a kept datadir: every stored-row schema change carries a versioned read path, and every release's gate starts the pinned binary on a copy of a standing box's datadir (Linux and Windows shapes) beside the wiped canary; LG-4's tenth install keeps the ninth's datadir (MF-8) | the node line's N13 (b7cc37e7); the canary form; `tools/fleet/first-share-gate.mjs` row kept=true | +| A node stops faster than its restart: every poll loop returns the moment shutdown is set, and "a shutdown returns within one second" is a named release gate beside the kept-datadir start (MF-9) | the node line's listener watchdog fix (b7cc37e7's follow-up); the mixed-version gate; `engine.rs` `restart_node` (the start is scheduled after `stop_node` returns) | | The fresh-install claim (LG-4) is a job, not a runbook: `tools/fleet/first-share-gate.mjs` on rented Windows boxes, on every cut, its line read by the shipper's publish | `relay/playbooks/first-share.ps1`, `tools/fleet/first-share-gate.mjs`, `site/evidence/first-share-.json` | ## The register @@ -28,6 +29,7 @@ Standing rules behind every row (branch `miner-reliability`, off `release-0.3.19 | MF-6 | 7 Oct 2026, PC 1, 0.3.19 | After a `--stop-miners` job, a following read-only job kept both cards "off, held for a remote job" for its whole three minutes | The engine released the hold only when no job held the miners; the next job's active state hid the release (engine.rs 3032 class) | A hold belongs to the job that took it (`job_hold_owner`) and releases the moment that job is no longer the running one, whatever runs next, or when its own cap passes (logged); a read-only job never holds (`jobrun::hold_release`) | `jobrun` test `a_hold_belongs_to_the_job_that_took_it` (owner running, another job, no job, cap passed, no owner) | `app tests: hold rule green` | | MF-7 | 7 Oct 2026, PC 1, 0.3.19 | Orphan `igneum-miner.exe` processes the app no longer tracked (two alive under `--stop-miners` with their rows at pid 0, one after) hammered the node's template RPC beside the tracked miners | The engine lost track of miners it had started (a stop that timed out, a restart over a live process) and never looked for them again | The engine owns every miner it started: at start, after every stop and every minute it kills any `igneum-miner` whose command line carries THIS engine's node RPC (the fence) and whose pid it does not track, one log line and one fault report per kill, never by name alone (`sweep_orphan_miners`, `platform::miner_processes`, `kill_pid`); a restart kills the slot's old process before the new one starts | injector step `orphan-miner` (a stray miner on the engine's node is killed inside the minute, the engine's own miner left alone) | `app-run orphan-miner PASS` | | MF-8 | 7 Oct 2026, every node build from 10db4b61 on a kept 0.3.17 datadir | "Node refuses its own kept datadir after an update": the app updates, the node dies at once (`DeserializationError(Io(Kind(UnexpectedEof)))` at `consensus/src/model/stores/virtual_state.rs:250`), the app restarts it in a loop, the miner never starts | `silent: bool` was added to `BlockRewardData` under `serde(default)` (10db4b61, the 0.3.16 vote-or-burn commit) and bincode ignores serde defaults, so the old row reads short; no canary saw it because every canary wiped. Fixed as ledger N13 on `release-0.3.20-node` b7cc37e7 (the store reads the v1 row and rewrites it) | Every schema change to a stored row ships with a versioned read path (the old shape read, rewritten in the new one); every release's gate starts the pinned binary on a COPY of a standing box's datadir, Linux-shaped and Windows-shaped (a copy of PC 2's), with the rewrite line or the synced line as the pass; the app never asks the user to wipe: a node that dies inside 10 s of its start is reported home (`node-exit` FAULT line) and the row says "the node cannot read its data after the update; the team has the report" | The node line's N13 test (the v1 row read and rewritten); the gate's kept-datadir start on both shapes; the injector step `kept-datadir` is OWED (a datadir from the previous release under the new binary needs a previous-release node on the pod; the shape is: run the old node to a few hundred blocks, stop it, start the new one on its datadir, pass on the synced line) | `canary kept-datadir start PASS (linux, windows)` in every cut's plan; `LG-4` row kept=true beside the nine wiped | +| MF-9 | 7 Oct 2026, release-0.3.20-node b7cc37e7, the mixed-version gate | A node that stops slower than its restart: the restart died at once on the datadir LOCK of the stopping node | The listener watchdog slept its whole 10 s poll before it checked the shutdown flag, so a stop took up to 10 s while the app's restart followed inside it | Every poll loop in the node returns the moment shutdown is set (a `select` on the shutdown signal, never a sleep then a check); "a shutdown returns within one second" is a named gate of every release beside the kept-datadir start; the app's `stop_node` waits for the exit before the restart (`restart_node` schedules the start after the stop returns) | The node line's shutdown-latency test (stop at a random moment of the poll, return under 1 s); the mixed-version gate's restart case; `app-run.mjs` step `node-silent` reads the restart-to-synced seconds | `node shutdown under 1 s PASS` in every cut's plan | | MF-3 | 7 Oct 2026, PC 1, Intel Arc | The Intel driver's first install did not bind: the device sat in Code 12 at install time; the card never mined until a reboot | A driver installed while the device reports a problem code (12, 43, 31) does not bind; nothing re-scanned the device afterwards, and the app only re-enumerates | The app re-enumerates every 60 s and starts the worker the minute the OS drives the card (`hotplug::diff` recovered / revived, `settle_new`); the row says what to do while it does not ("reboot with the card attached; if it persists, reinstall the driver with the card attached"); a Windows host asks for a re-scan (`pnputil /scan-devices`) after a problem code is seen, every 5 minutes, at most 6 times (follow-up, host side) | `hotplug` test `a_driven_card_that_turns_faulty_is_errored_and_recovers_later`; `app-run.mjs` step `card-appears` (a card listed after 2 minutes starts without a tap) | `app-run card-appears PASS` | ## Commits (7 October 2026) From 06eadb406dc8bfddd33d60c7e408aca615e85797 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 12:34:24 +0000 Subject: [PATCH 14/18] miner-faults MF-10: a prover server built for another card's architecture (named symbol not found); the capability check at start and the per-architecture kit Co-Authored-By: Claude Fable 5.1 (cherry picked from commit 9d2c267af7a7b097199e6232b221b91081f52bca) --- docs/plans/miner-faults.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/docs/plans/miner-faults.md b/docs/plans/miner-faults.md index 572c3425f..438430ef4 100644 --- a/docs/plans/miner-faults.md +++ b/docs/plans/miner-faults.md @@ -16,6 +16,7 @@ Standing rules behind every row (branch `miner-reliability`, off `release-0.3.19 | Nothing on a user's machine is changed by a one-off script: card settings travel as the signed `cards` job kind (per card enabled, identities, power_pct), applied through the app's own card path, persisted, read back in the report, refused for a card the machine does not have | `src/jobs.rs` (`KINDS`, `validate_params`), `src/jobrun.rs` (`cards_job_choices`, `cards_applied`), `engine.rs` (`Action::ApplyCards`), `packaging/ota/publish-jobs.sh add --kind cards --cards "key=on:8"` | | A cut never dies on a kept datadir: every stored-row schema change carries a versioned read path, and every release's gate starts the pinned binary on a copy of a standing box's datadir (Linux and Windows shapes) beside the wiped canary; LG-4's tenth install keeps the ninth's datadir (MF-8) | the node line's N13 (b7cc37e7); the canary form; `tools/fleet/first-share-gate.mjs` row kept=true | | A node stops faster than its restart: every poll loop returns the moment shutdown is set, and "a shutdown returns within one second" is a named release gate beside the kept-datadir start (MF-9) | the node line's listener watchdog fix (b7cc37e7's follow-up); the mixed-version gate; `engine.rs` `restart_node` (the start is scheduled after `stop_node` returns) | +| A prover never runs a server built for another card: the compute capability is read at start and a mismatch is refused with the reason shown and reported home; the kit ships one server per architecture or a fat binary (MF-10, 0.3.21) | `src/prover.rs` (owed: the capability check at start and the 15-second symbol-error class), the kit's per-architecture servers | | The fresh-install claim (LG-4) is a job, not a runbook: `tools/fleet/first-share-gate.mjs` on rented Windows boxes, on every cut, its line read by the shipper's publish | `relay/playbooks/first-share.ps1`, `tools/fleet/first-share-gate.mjs`, `site/evidence/first-share-.json` | ## The register @@ -30,6 +31,7 @@ Standing rules behind every row (branch `miner-reliability`, off `release-0.3.19 | MF-7 | 7 Oct 2026, PC 1, 0.3.19 | Orphan `igneum-miner.exe` processes the app no longer tracked (two alive under `--stop-miners` with their rows at pid 0, one after) hammered the node's template RPC beside the tracked miners | The engine lost track of miners it had started (a stop that timed out, a restart over a live process) and never looked for them again | The engine owns every miner it started: at start, after every stop and every minute it kills any `igneum-miner` whose command line carries THIS engine's node RPC (the fence) and whose pid it does not track, one log line and one fault report per kill, never by name alone (`sweep_orphan_miners`, `platform::miner_processes`, `kill_pid`); a restart kills the slot's old process before the new one starts | injector step `orphan-miner` (a stray miner on the engine's node is killed inside the minute, the engine's own miner left alone) | `app-run orphan-miner PASS` | | MF-8 | 7 Oct 2026, every node build from 10db4b61 on a kept 0.3.17 datadir | "Node refuses its own kept datadir after an update": the app updates, the node dies at once (`DeserializationError(Io(Kind(UnexpectedEof)))` at `consensus/src/model/stores/virtual_state.rs:250`), the app restarts it in a loop, the miner never starts | `silent: bool` was added to `BlockRewardData` under `serde(default)` (10db4b61, the 0.3.16 vote-or-burn commit) and bincode ignores serde defaults, so the old row reads short; no canary saw it because every canary wiped. Fixed as ledger N13 on `release-0.3.20-node` b7cc37e7 (the store reads the v1 row and rewrites it) | Every schema change to a stored row ships with a versioned read path (the old shape read, rewritten in the new one); every release's gate starts the pinned binary on a COPY of a standing box's datadir, Linux-shaped and Windows-shaped (a copy of PC 2's), with the rewrite line or the synced line as the pass; the app never asks the user to wipe: a node that dies inside 10 s of its start is reported home (`node-exit` FAULT line) and the row says "the node cannot read its data after the update; the team has the report" | The node line's N13 test (the v1 row read and rewritten); the gate's kept-datadir start on both shapes; the injector step `kept-datadir` is OWED (a datadir from the previous release under the new binary needs a previous-release node on the pod; the shape is: run the old node to a few hundred blocks, stop it, start the new one on its datadir, pass on the synced line) | `canary kept-datadir start PASS (linux, windows)` in every cut's plan; `LG-4` row kept=true beside the nine wiped | | MF-9 | 7 Oct 2026, release-0.3.20-node b7cc37e7, the mixed-version gate | A node that stops slower than its restart: the restart died at once on the datadir LOCK of the stopping node | The listener watchdog slept its whole 10 s poll before it checked the shutdown flag, so a stop took up to 10 s while the app's restart followed inside it | Every poll loop in the node returns the moment shutdown is set (a `select` on the shutdown signal, never a sleep then a check); "a shutdown returns within one second" is a named gate of every release beside the kept-datadir start; the app's `stop_node` waits for the exit before the restart (`restart_node` schedules the start after the stop returns) | The node line's shutdown-latency test (stop at a random moment of the poll, return under 1 s); the mixed-version gate's restart case; `app-run.mjs` step `node-silent` reads the restart-to-synced seconds | `node shutdown under 1 s PASS` in every cut's plan | +| MF-10 | 7 Oct 2026, the prover roll (3080, 3090, 4070, 5090) | Every proof fails in 12 s with `CudaRustError: named symbol not found`; the miner never notices and the box proves nothing for hours | The prover's `sp1-gpu-server` was built for another card's compute capability (sm_86 on the 3080 and 3090, sm_89 on the 4070, sm_120 on the 5090), so every kernel load fails the same way | The prover reads the card's compute capability at start (`nvidia-smi --query-gpu=compute_cap`) and refuses to start a server that does not match, with the reason on the prover tile and a FAULT line home ("the proving server is built for sm_89, this card is sm_120"); a proof that fails inside 15 s with the symbol error marks the server mismatched the same way; the 0.3.21 kit ships one server per architecture chosen at install, or a fat binary | the prover's compute-capability test (a server named for sm_89 refused on a card that reads 12.0); the fleet roll's paired line per box | `prover server matches the card PASS` per box in the roll, beside the kept-datadir and the shutdown gates | | MF-3 | 7 Oct 2026, PC 1, Intel Arc | The Intel driver's first install did not bind: the device sat in Code 12 at install time; the card never mined until a reboot | A driver installed while the device reports a problem code (12, 43, 31) does not bind; nothing re-scanned the device afterwards, and the app only re-enumerates | The app re-enumerates every 60 s and starts the worker the minute the OS drives the card (`hotplug::diff` recovered / revived, `settle_new`); the row says what to do while it does not ("reboot with the card attached; if it persists, reinstall the driver with the card attached"); a Windows host asks for a re-scan (`pnputil /scan-devices`) after a problem code is seen, every 5 minutes, at most 6 times (follow-up, host side) | `hotplug` test `a_driven_card_that_turns_faulty_is_errored_and_recovers_later`; `app-run.mjs` step `card-appears` (a card listed after 2 minutes starts without a tap) | `app-run card-appears PASS` | ## Commits (7 October 2026) From 44d2def1e52d8e9c854a8f5326dd5491b2e7fba7 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 12:48:25 +0000 Subject: [PATCH 15/18] miner-faults register: the pod injector's run 4 lines (seven of eight steps pass on engine 073fbea1) Co-Authored-By: Claude Fable 5.1 (cherry picked from commit 9a91ff57e1db5896f9f9016b7aadcc2ba27e5bff) --- docs/plans/miner-faults.md | 24 +++++++++++++++++++++--- 1 file changed, 21 insertions(+), 3 deletions(-) diff --git a/docs/plans/miner-faults.md b/docs/plans/miner-faults.md index 438430ef4..42e29cef7 100644 --- a/docs/plans/miner-faults.md +++ b/docs/plans/miner-faults.md @@ -38,12 +38,30 @@ Standing rules behind every row (branch `miner-reliability`, off `release-0.3.19 | Repo | Branch | Commit | What | |---|---|---|---| -| igneum | `miner-reliability` (off `release-0.3.19` db6f0964, with release-0.3.20's `igneum-pow` and master's build tooling) | `38a30397`, `c5cb5cd5` | the app side of every row, the register, the injector, the cards job kind, the LG-4 job, the CI check | +| igneum | `miner-reliability` (off `release-0.3.19` db6f0964, with release-0.3.20's `igneum-pow` and master's build tooling) | `38a30397` to `9d2c267a` (the engine at `89d117ce`) | the app side of every row, the register, the injector, the cards job kind, the LG-4 job, the CI check | | igneum-node | `miner-reliability-20` (off `release-0.3.20-node` dc141409) | `f067f7c1` | MF-5's miner side: STATUS while waiting, identities from the template time, the fetch back-off | Box lines: app tests 198 + 27 + 8 green on igneum-build-2 (the tree gate green, 33 checks on the Mac); igneum-miner 19 -green on igneum-build-2; both Linux binaries built on igneum-build-1 (app sha256 6d4ee013, miner b3bf3590). Injector: -the pod run's lines are appended below when it ends. +green on igneum-build-2; both Linux binaries built on igneum-build-1 (app sha256 28650614 at 89d117ce, miner b3bf3590 at +f067f7c1). + +Injector, run 4 on a one-shot RunPod pod (RTX 3070, Ubuntu 24.04, 7 October 2026 12:19Z to 12:48Z, engine 89d117ce, +the 0.3.17-line miner, the stand-in worker; the pod's real GPU switched off through the app's own card path; the private +node on devnet suffix 9960 with a 2-thread CPU block producer): seven of eight steps PASS, one on the stand-in's listing +(fixed at ecd79874, rerun below). + +| Step | Class | Result | Seconds | +|---|---|---|---| +| catch-up | MF-1, MF-2 | no worker started while the execution layer held no record (120 s), the card said it waits for the executed tip, the node was not restarted, the worker started on its own once the record existed | synced after 4.1; blocks to mining 18.0 | +| own-restart | (the miner's own worker restart) | the card showed the worker fault; the app did not restart the miner (same pid) | fault on the card 1.0; mining again 4.0 | +| zero-ladder | MF-2 | three watchdog restarts at 10, 30, 120 s on the row, the reason in plain words, no faulted state, no "once already" words, mining back on its own | zero to rung 1 77.3; rung gaps 81.3, 101.9; healthy to mining 129.8 | +| no-status | MF-2 | the miner stopped with SIGSTOP: restarted for "no status line for 90 s", the stopped process killed, mining on a new process | quiet to restart 92.2; restart to mining 311.7 (the ladder's 300 s rung, carried from the step before: no five healthy minutes between them) | +| node-silent | node watchdog | the node stopped with SIGSTOP: restarted by the app (120 s rule plus the 30 s terminate grace), synced, mining again | quiet to restart 150.3; restart to synced 7.0; quiet to mining 170.4 | +| one-card-fails | MF-4 | two new cards listed, two healthy ones mining, the third held 30 minutes with "not usable on this driver: ", the healthy two still mining 90 s on, the pack exported once and reused 7 times | listed 48.1; two mining 48.1 | +| orphan-miner | MF-7 | a stray igneum-miner on the engine's node killed by the minute sweep, one log line per kill, the engine's own miner left alone | inject to kill 21.1 | +| card-appears | MF-3 | run 4: the card that appeared was listed (40.1 s) and mined with no tap (50.1 s); the card that left was NOT marked removed: the stand-in's `--list` lacked the real worker's header line, which the engine reads as "the tool did not answer" (right for a crashed driver). Rerun with the header (ecd79874): see run 5 | | + +FAULT lines the engine posted to the intake in run 4: 13 (watchdog, worker-fault, node-exit, orphan-miner classes). Which side: MF-1, MF-2, MF-3, MF-4, MF-6, MF-7 and the cards kind are app-side (0.3.20's app). MF-5 is both: the miner (the fork branch) prints the fields and caps the identities; the app reads them and keeps the worker. An app From 26574c32d05a01c564b1e1e8b32ff8813b504b40 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 12:53:56 +0000 Subject: [PATCH 16/18] miner-faults register: run 5 (card-appears green with the fixed listing); every class's injector step passes on engine 073fbea1 Co-Authored-By: Claude Fable 5.1 (cherry picked from commit ee3d93386b7aabe8188fd2c8bd00c63a818ac2bf) --- docs/plans/miner-faults.md | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/docs/plans/miner-faults.md b/docs/plans/miner-faults.md index 42e29cef7..00a64b88b 100644 --- a/docs/plans/miner-faults.md +++ b/docs/plans/miner-faults.md @@ -47,8 +47,8 @@ f067f7c1). Injector, run 4 on a one-shot RunPod pod (RTX 3070, Ubuntu 24.04, 7 October 2026 12:19Z to 12:48Z, engine 89d117ce, the 0.3.17-line miner, the stand-in worker; the pod's real GPU switched off through the app's own card path; the private -node on devnet suffix 9960 with a 2-thread CPU block producer): seven of eight steps PASS, one on the stand-in's listing -(fixed at ecd79874, rerun below). +node on devnet suffix 9960 with a 2-thread CPU block producer): seven of eight steps PASS, the eighth on its rerun (run 5, +the stand-in's listing fixed at ecd79874): every class's step is green. | Step | Class | Result | Seconds | |---|---|---|---| @@ -59,7 +59,7 @@ node on devnet suffix 9960 with a 2-thread CPU block producer): seven of eight s | node-silent | node watchdog | the node stopped with SIGSTOP: restarted by the app (120 s rule plus the 30 s terminate grace), synced, mining again | quiet to restart 150.3; restart to synced 7.0; quiet to mining 170.4 | | one-card-fails | MF-4 | two new cards listed, two healthy ones mining, the third held 30 minutes with "not usable on this driver: ", the healthy two still mining 90 s on, the pack exported once and reused 7 times | listed 48.1; two mining 48.1 | | orphan-miner | MF-7 | a stray igneum-miner on the engine's node killed by the minute sweep, one log line per kill, the engine's own miner left alone | inject to kill 21.1 | -| card-appears | MF-3 | run 4: the card that appeared was listed (40.1 s) and mined with no tap (50.1 s); the card that left was NOT marked removed: the stand-in's `--list` lacked the real worker's header line, which the engine reads as "the tool did not answer" (right for a crashed driver). Rerun with the header (ecd79874): see run 5 | | +| card-appears | MF-3 | run 5 (12:48Z to 12:53Z, the stand-in's listing with the real worker's header, ecd79874): a second card appeared and was listed, its worker started with no tap; it left and was marked removed while the other card kept mining; it came back and mined again with no tap. Run 4 had failed the removal because the stand-in's `--list` lacked the header line, which the engine reads as "the tool did not answer" and removes nothing on (the right call for a crashed driver) | listed 40.1 after it appeared; mining 50.1; marked removed 50.1 after it left; mining again 72.2 after it came back | FAULT lines the engine posted to the intake in run 4: 13 (watchdog, worker-fault, node-exit, orphan-miner classes). From e78ace39bceda0464c8b4cd9c7e1094b2e8a4761 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 13:12:26 +0000 Subject: [PATCH 17/18] execrpc: 0.3.21's two callers classified (igneum_getRecentBlocks gated, eth_getBalance safe) Co-Authored-By: Claude Fable 5.1 --- app/igneum-app/src/execrpc.rs | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/app/igneum-app/src/execrpc.rs b/app/igneum-app/src/execrpc.rs index ee7e36531..a24bc5472 100644 --- a/app/igneum-app/src/execrpc.rs +++ b/app/igneum-app/src/execrpc.rs @@ -16,6 +16,8 @@ use std::time::Duration; /// Methods that index nothing on an empty exec state (read from the 0.3.17 node's rpc.rs): safe at any time. pub const SAFE_ON_EMPTY: &[&str] = &[ "eth_chainId", "eth_blockNumber", "eth_syncing", "igneum_getExecStatus", "igneum_getProvingStatus", "igneum_getNodeInfo", + // the engine's balance read (0.3.21): an account lookup at the latest state, nothing indexed by block number + "eth_getBalance", ]; /// Methods the app sends that resolve a block number, index or slice the record vector, or simulate at a block: held until @@ -24,6 +26,8 @@ pub const GATED: &[&str] = &[ "eth_getBlockByNumber", "igneum_getAssignedShards", "igneum_getProofRecords", "igneum_getSegmentRecords", "igneum_getSegmentStatement", "igneum_getProofBytes", "igneum_getSegmentProofBytes", "igneum_exportSegments", "igneum_submitProofRecord", "igneum_submitSegmentRecord", "igneum_getFinalityWeights", + // 0.3.21's live page reads recent blocks by number through the same path (src/live.rs); held like the rest + "igneum_getRecentBlocks", ]; /// One JSON-RPC POST to 127.0.0.1: through curl (the engine carries no HTTP client); the body goes through a file From 073060620144cf75f983b46617a975c626180b5d Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 13:14:13 +0000 Subject: [PATCH 18/18] engine (0.3.21): the heat-mode release restarts every slot; there is no faulted state to skip Co-Authored-By: Claude Fable 5.1 --- app/igneum-app/src/engine.rs | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/app/igneum-app/src/engine.rs b/app/igneum-app/src/engine.rs index 0f9d536a9..c1dee45ca 100644 --- a/app/igneum-app/src/engine.rs +++ b/app/igneum-app/src/engine.rs @@ -3859,10 +3859,11 @@ impl Engine { self.shared.log(&format!("heat: release ({why}); the miners restart")); let now = Instant::now(); for m in self.miners.iter_mut() { - if m.watch.faulted().is_none() { + // no permanent fault (7 October 2026): every slot restarts; a ladder delay in force stays as scheduled + if m.restart_at.map(|at| at <= now).unwrap_or(true) { m.restart_at = Some(now); - m.prepared = false; } + m.prepared = false; } } }