diff --git a/app/igneum-app/src/engine.rs b/app/igneum-app/src/engine.rs index 887b6d75..22b7b714 100644 --- a/app/igneum-app/src/engine.rs +++ b/app/igneum-app/src/engine.rs @@ -1858,6 +1858,11 @@ impl Engine { match procs::spawn(Source::Miner(card_idx), &self.bins.miner, &args, cwd.as_deref(), &log, &self.lines_tx, &envs) { Ok(p) => { self.shared.log(&format!("miner {} started (pid {}): {}", self.miners[i].label, p.pid(), p.cmdline)); + if self.miners[i].starts == 1 { + // one Activity line per card at its first start (an OTA relaunch shows every card coming back) + let name = self.st().mining.cards.get(self.miners[i].card).map(|c| c.name.clone()).unwrap_or_else(|| self.miners[i].label.clone()); + self.shared.event("info", &format!("{name}: worker starting (pid {})", p.pid())); + } let mut st = self.st(); if let Some(c) = st.mining.cards.get_mut(card_idx) { c.pid = p.pid(); @@ -4114,7 +4119,15 @@ impl Engine { } if synced && !was_synced { self.shared.event("ok", &format!("node synced: {blocks} blocks, {peers} peer(s)")); + let t = self.secs(now); + let names: Vec = self.st().mining.cards.iter().map(|c| c.name.clone()).collect(); for m in self.miners.iter_mut() { + // a card faulted for silence while the node was not ready starts again now (PC 1, 7 October 2026) + if m.watch.faulted_by_node() { + m.watch.event(t, crate::watchdog::Event::NodeSynced); + let name = names.get(m.card).cloned().unwrap_or_else(|| m.label.clone()); + self.shared.event("info", &format!("{name}: starts again now the node is synced (it was faulted for silence while the node was syncing)")); + } if m.proc.is_none() && m.restart_at.is_none() && m.watch.faulted().is_none() { m.restart_at = Some(now); } diff --git a/app/igneum-app/src/watchdog.rs b/app/igneum-app/src/watchdog.rs index 369ad822..ded5fd83 100644 --- a/app/igneum-app/src/watchdog.rs +++ b/app/igneum-app/src/watchdog.rs @@ -17,6 +17,11 @@ /// No status line from a running miner for this long: restart it. pub const NO_STATUS_S: f64 = 90.0; +/// A worker that gives no status within this many seconds of its start, with the node synced, is one restart then +/// faulted (PC 1, 7 October 2026 04:51Z: the row asks for a fault line at 60 s). +pub const START_S: f64 = 60.0; +/// The start of a fault reason the node's readiness, not the card, explains: released when the node syncs. +const NODE_FAULTS: [&str; 2] = ["no status line from the miner", "the worker gave no status within"]; /// Hash rate 0 while the node is synced and the worker is ready for this long: restart the miner. pub const ZERO_RATE_S: f64 = 60.0; /// The miner is restarting its own worker: the app waits this long for `ready` before it steps in. @@ -86,6 +91,9 @@ pub enum Event<'a> { Stopped, /// The user changed the card's settings: a faulted card may try again. Reset, + /// The node became synced: a card faulted for silence while the node was not ready tries again (PC 1, 7 October + /// 2026 04:51Z: every card faulted "no status line" during the post-relaunch sync and never came back). + NodeSynced, } #[derive(Debug, Clone, PartialEq)] @@ -207,9 +215,20 @@ impl CardWatch { *self = Self::default(); Action::None } + Event::NodeSynced => { + if self.faulted.as_deref().map(|f| NODE_FAULTS.iter().any(|p| f.starts_with(p))).unwrap_or(false) { + *self = Self::default(); + } + Action::None + } } } + /// True when the card is faulted for a reason the node's readiness explains (released by Event::NodeSynced). + pub fn faulted_by_node(&self) -> bool { + self.faulted.as_deref().map(|f| NODE_FAULTS.iter().any(|p| f.starts_with(p))).unwrap_or(false) + } + /// Called every engine tick while the miner process is alive. pub fn tick(&mut self, now_s: f64, node_synced: bool) -> Action { if self.faulted.is_some() { @@ -223,14 +242,23 @@ impl CardWatch { } return Action::None; } + if !node_synced { + // a miner is judged only while the node is synced: with no template it cannot print status, so the silence + // clocks (start, no status, zero rate) hold at now until the node is back (PC 1, 7 October 2026) + self.started_s = Some(now_s); + if self.last_status_s.is_some() { + self.last_status_s = Some(now_s); + } + self.zero_since = None; + return Action::None; + } + if !self.ready && self.last_status_s.is_none() && now_s - started > START_S { + return self.escalate(format!("the worker gave no status within {} s of starting", START_S as u64)); + } let last = self.last_status_s.unwrap_or(started); if now_s - last > NO_STATUS_S { return self.escalate(format!("no status line from the miner for {} s", NO_STATUS_S as u64)); } - if !node_synced { - // a zero rate while the node syncs is expected; the timer starts again once it is synced - self.zero_since = None; - } if node_synced && self.ready { if let Some(z) = self.zero_since { if now_s - z >= ZERO_RATE_S { @@ -488,6 +516,48 @@ mod tests { assert!(matches!(w.tick(240.0, true), Action::Restart(_))); } + /// PC 1, 7 October 2026 04:51:43Z (docs/plans/release-0.3.17.md, the PC 1 row): the app relaunched after the OTA + /// apply, started its workers while the node was still syncing, and every card ended "faulted: no status line from + /// the miner for 90 s (restarted once already)", pid 0, 0.0 MH/s from then on. Today's watchdog clocks the silence + /// while the node is not synced and never releases the fault. + #[test] + fn a_worker_started_while_the_node_syncs_is_not_faulted_and_a_faulted_one_returns_at_sync() { + let mut w = CardWatch::new(); + w.event(0.0, Event::Started); + // the node reads unsynced for 200 s (the relaunch's sync): no verdict at all + let mut t = 5.0; + while t <= 200.0 { + assert_eq!(w.tick(t, false), Action::None, "at {t}: a miner is judged only while the node is synced"); + t += 10.0; + } + // the node syncs; the worker answers inside 60 s and is healthy + assert_eq!(w.tick(210.0, true), Action::None); + w.event(230.0, Event::Ready); + healthy(&mut w, 240.0, 400.0); + assert_eq!(w.faulted(), None); + // the recorded state: a card faulted for silence while the node was not ready + let mut f = CardWatch::new(); + f.event(0.0, Event::Started); + assert!(matches!(f.tick(100.0, true), Action::Restart(_))); + f.event(101.0, Event::Started); + let a = f.tick(200.0, true); + assert!(matches!(a, Action::Fault(ref r) if (r.starts_with("no status line from the miner") || r.starts_with("the worker gave no status within")) && r.contains("restarted once already")), "{a:?}"); + assert!(f.faulted_by_node()); + f.event(300.0, Event::NodeSynced); + assert_eq!(f.faulted(), None, "the node syncing releases a silence fault"); + assert_eq!(f.watchdog_restarts(), 0); + // a fault the card owns (zero rate while synced) stays through NodeSynced + let mut z = CardWatch::new(); + z.event(0.0, Event::Started); + z.event(1.0, Event::Ready); + let zero = parse_status(STATUS_OK.replace("now=124.10 MH/s", "now=0.00 MH/s").as_str()).unwrap(); + for t in [10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 70.0, 80.0] { z.event(t, Event::Status(&zero)); z.tick(t, true); } + assert!(z.faulted().is_none() || !z.faulted_by_node()); + let before = z.faulted().map(|s| s.to_string()); + z.event(500.0, Event::NodeSynced); + assert_eq!(z.faulted().map(|s| s.to_string()), before); + } + #[test] fn no_status_for_90s_restarts() { let mut w = CardWatch::new(); @@ -502,10 +572,11 @@ mod tests { #[test] fn no_status_from_the_start() { + // a worker silent from its start, with the node synced, is judged at START_S (60 s), not NO_STATUS_S let mut w = CardWatch::new(); w.event(0.0, Event::Started); - assert_eq!(w.tick(89.0, true), Action::None); - assert!(matches!(w.tick(91.0, true), Action::Restart(_))); + assert_eq!(w.tick(59.0, true), Action::None); + assert!(matches!(w.tick(61.0, true), Action::Restart(ref r) if r.contains("gave no status within 60 s"))); } #[test] diff --git a/docs/plans/miner-ui-4.md b/docs/plans/miner-ui-4.md index c0a781bb..533e3386 100644 --- a/docs/plans/miner-ui-4.md +++ b/docs/plans/miner-ui-4.md @@ -107,3 +107,9 @@ Tests: UI 39; the app crate on the box 152 passed (the N4 pair, the shot path, t - UI: nodeWords' behind case names the cause. 42 UI tests, 172 box tests. Nothing of this on the Mac tonight. - Decided (main, 7 October 2026, 05:00 UK): `igneum_getPeers` stays on 0.3.19; the observer-based merge check ships in 0.3.18. The Mac check when it takes 0.3.18: the hands moved to the build box at 00:18 UK, so hand node 1 no longer holds 26610; the app, left in external mode, reads "node stopped" and starts its own node only on relaunch (the apply). Check after the apply: own node on 26610/26611, api/state node.mode own, digest_source rpc, synced; the miner stays paused. Read-only; nothing else that night. - Node side landed: 6e4ace3f on build/ca3-v4-0318 (04:53 UK) adds `blockrate: { bps, ghostdagK, mergeDepth, finalityDepth, pruningDepth }` to igneum_getNodeInfo (devnet: bps 1, ghostdagK 18, mergeDepth 3600). The app reads blockrate.mergeDepth (b7d33e8e) and keeps the 3,600 fallback, since whether 6e4ace3f rides in 0.3.18 is the shipper's call. `igneum_getPeers` (lastDeliveredBlueScore) stays on 0.3.19. + +## 11. Workers after an OTA relaunch (7 October 2026, 05:20 UK; 0.3.18) + +- Known-failed case: PC 1 at the 0.3.17 relaunch (04:51:43Z, docs/plans/release-0.3.17.md PC 1 row): the app started its workers while the node was still syncing; the card watchdog clocked the silence anyway (a miner with no template prints no status), restarted each once, then faulted every card "no status line from the miner for 90 s (restarted once already)", pid 0, 0.0 MH/s on; faulted cards were never restarted, including at the synced transition. +- Fix (src/watchdog.rs, src/engine.rs): a miner is judged only while the node is synced (the start, no-status and zero-rate clocks hold at now while it is not); a worker silent from its start with the node synced is one restart then a fault at 60 s (`START_S`), 90 s stays for a running one; a fault the node's readiness explains (`faulted_by_node`) is released by `Event::NodeSynced` when the node syncs, the card starts again with an Activity line; one Activity line per card at its first worker start. Test `a_worker_started_while_the_node_syncs_is_not_faulted_and_a_faulted_one_returns_at_sync` (recorded state), 173 box tests. +- The pre-apply card set already persists (cards' enabled flags live in settings and the relaunch planned every enabled card); the fault, not the plan, was the failure.