diff --git a/app/igneum-app/src/engine.rs b/app/igneum-app/src/engine.rs index 8831769bf..84d356d5f 100644 --- a/app/igneum-app/src/engine.rs +++ b/app/igneum-app/src/engine.rs @@ -449,6 +449,36 @@ fn jitter_secs(seed: u64) -> u64 { 5 + (seed.wrapping_mul(2654435761) >> 7) % 56 } +/// Seconds after a resume before every enabled card must be mining (a worker takes 10 to 60 s to its first STATUS +/// line with a hash rate; the pack export before it a few seconds more). +pub(crate) const RESUME_CHECK_SECS: u64 = 90; + +/// What the resume rule reads from a miner slot. +#[derive(Clone, Debug, PartialEq, Eq)] +pub(crate) struct ResumeSlot { + /// the watchdog marked the card faulted + pub faulted: bool, + /// a worker process is alive on the slot + pub live: bool, +} + +/// The resume rule: every slot without a live worker is re-armed, faulted or not. (The rule before 5 October 2026 +/// re-armed only faulted slots; `stop_miners("paused")` had cleared every slot's `restart_at`, so a healthy paused +/// card never restarted: PC 2 at 21:25:11Z, the Mac that afternoon.) +pub(crate) fn slots_to_rearm_on_resume(slots: &[ResumeSlot]) -> Vec { + slots.iter().enumerate().filter(|(_, s)| !s.live).map(|(i, _)| i).collect() +} + +/// The check RESUME_CHECK_SECS after a resume: every enabled, present, non-faulted card must report a hash rate +/// above 0 or its name is returned with its state (the caller logs one line per card). +pub(crate) fn resume_check(cards: &[CardState]) -> Vec { + cards + .iter() + .filter(|c| c.enabled && c.present() && c.state != "faulted" && c.hash_now <= 0.0) + .map(|c| format!("{} is not mining {RESUME_CHECK_SECS} s after resume (state {}, pid {}{})", c.name, c.state, c.pid, if c.message.is_empty() { String::new() } else { format!(", {}", c.message) })) + .collect() +} + struct MinerSlot { card: usize, label: String, @@ -477,6 +507,8 @@ pub struct Engine { node: Option, node_external: bool, node_restart_at: Option, + /// resume rule (5 October 2026): when due, every enabled card must be mining or its name goes to the log + resume_check_at: Option, node_started_at: Instant, node_starts: u32, node_restarts: u32, @@ -579,6 +611,7 @@ impl Engine { node: None, node_external: false, node_restart_at: None, + resume_check_at: None, node_started_at: now, node_starts: 0, node_restarts: 0, @@ -765,13 +798,23 @@ impl Engine { self.shared.save_settings(); self.st().mining.paused = false; self.shared.event("ok", "mining resumed"); - // a user action: faulted cards try again - for m in self.miners.iter_mut() { + // the resume rule (5 October 2026, PC 2 at 21:25:11Z and the Mac that afternoon): stop_miners("paused") + // cleared every slot's restart_at and this arm re-armed only FAULTED slots, so a card whose worker had + // simply been stopped stayed "off" at 0 MH/s until the app was relaunched. Now every slot without a + // live worker is re-armed (a faulted one reset first), its pack is exported again before the start + // (prepared = false: the hour may have turned while paused), and a check 90 s later names any + // enabled card that is not mining (`resume_check`). + let views: Vec = self.miners.iter().map(|m| ResumeSlot { faulted: m.watch.faulted().is_some(), live: m.proc.is_some() }).collect(); + let now = Instant::now(); + for i in slots_to_rearm_on_resume(&views) { + let m = &mut self.miners[i]; if m.watch.faulted().is_some() { m.watch.event(0.0, crate::watchdog::Event::Reset); - m.restart_at = Some(Instant::now()); } + m.restart_at = Some(now); + m.prepared = false; } + self.resume_check_at = Some(now + Duration::from_secs(RESUME_CHECK_SECS)); if !self.running { self.start(); } @@ -2193,6 +2236,18 @@ impl Engine { self.tick_node(now); self.tick_watch(now); self.tick_miners(now); + if let Some(at) = self.resume_check_at { + if now >= at { + self.resume_check_at = None; + let (cards, paused) = { let st = self.st(); (st.mining.cards.clone(), st.mining.paused) }; + if !paused { + for line in resume_check(&cards) { + self.shared.log(&format!("resume: {line}")); + self.shared.event("error", &format!("resume: {line}")); + } + } + } + } self.tick_telemetry(now); self.tick_sweep(now); } @@ -3392,6 +3447,44 @@ pub fn parse_race(body: &str) -> Option { Some(r) } +#[cfg(test)] +mod resume_tests { + use super::*; + + fn card(name: &str, enabled: bool, state: &str, hash: f64) -> CardState { + CardState { name: name.into(), vendor: "nvidia".into(), kind: "discrete".into(), enabled, state: state.into(), hash_now: hash, ..Default::default() } + } + + /// The state machine: paused (every slot stopped, restart_at cleared) -> resumed -> every slot without a live + /// worker is re-armed, faulted or not. + #[test] + fn resume_rearms_every_stopped_slot() { + let slots = [ResumeSlot { faulted: false, live: false }, ResumeSlot { faulted: true, live: false }, ResumeSlot { faulted: false, live: true }]; + assert_eq!(slots_to_rearm_on_resume(&slots), vec![0, 1], "the healthy stopped slot and the faulted one restart; the live one is left alone"); + } + + /// The known-failed case (PC 2, 5 October 2026, 21:25:11Z, app 0.3.9: `[ok] mining resumed`, then `0.00 MH/s, + /// waiting` for 20 minutes): one healthy slot, stopped by the pause, not faulted. The old rule re-armed only + /// faulted slots and returned nothing for it; the new rule returns it. + #[test] + fn the_pc2_resume_of_21_25_11z_restarts_under_the_new_rule_and_not_the_old() { + let old_rule = |slots: &[ResumeSlot]| -> Vec { slots.iter().enumerate().filter(|(_, s)| s.faulted).map(|(i, _)| i).collect() }; + let pc2 = [ResumeSlot { faulted: false, live: false }]; + assert!(old_rule(&pc2).is_empty(), "the 0.3.9 rule left the 5090's slot unarmed: this is the defect"); + assert_eq!(slots_to_rearm_on_resume(&pc2), vec![0]); + } + + /// The check 90 s after a resume names every enabled card without a hash rate, and nothing else. + #[test] + fn resume_check_names_the_cards_not_mining() { + let cards = [card("NVIDIA GeForce RTX 5090", true, "off", 0.0), card("AMD Radeon(TM) Graphics", true, "mining", 3.4), card("Intel UHD", false, "off", 0.0), card("RTX 3060", true, "faulted", 0.0)]; + let lines = resume_check(&cards); + assert_eq!(lines.len(), 1, "{lines:?}"); + assert!(lines[0].starts_with("NVIDIA GeForce RTX 5090 is not mining 90 s after resume (state off"), "{}", lines[0]); + assert!(resume_check(&[card("RTX 5090", true, "mining", 118.9)]).is_empty()); + } +} + #[cfg(test)] mod tests { use super::{digest_from_line, parse_race, switches_of, sync_decision, Reading}; diff --git a/docs/plans/proving-v1.md b/docs/plans/proving-v1.md index fb0735e67..4dce96a98 100644 --- a/docs/plans/proving-v1.md +++ b/docs/plans/proving-v1.md @@ -112,6 +112,8 @@ Reading. Nobody pays an aggregator as a separate role: Aztec's 30% goes to whoev | Apple silicon default | **off** | the gate was "a shard under 60 s with the miner running": the M5 Max CPU took 41.3 and 55.4 s for EMPTY shards under tonight's load and 272 s for a 200-pgas shard on 4 October; a full shard at `S_p` was never under 60 s. Settings switches it on | bench-log "proving v1" CPU chain row; 4 October CPU rows | | The prover profile per card and the 12 GB and 16 GB gates (Josh: "make sure we can prove on 12gb cards"; "is there any way we can make 12gb cards mine and prove?") | **measured on PC 2, the rows below** | the SP1 6.8.1 GPU server reads `ELEMENT_THRESHOLD`, `HEIGHT_THRESHOLD`, `SHARD_SIZE` and the `SP1_WORKER_NUM_*`/`BUFFER_SIZE` knobs from the environment it inherits (`sp1-core-executor-6.8.1/src/opts.rs`, `sp1-prover-6.8.1/src/worker/config.rs`); the app passes a profile per card (`provedefault.rs`) and the host forwards it | the sweep job `memsweep-pc2-pv1` and the miner-on run | +The resume path (5 October 2026, the 0.3.11 app): `POST /api/resume` on 0.3.9 re-armed only FAULTED cards (`stop_miners("paused")` clears every slot's `restart_at`), so a healthy paused card stayed "off" at 0 MH/s until the app was relaunched: PC 2 at 21:25:11Z (the aggregation-cost job's pause and resume; `[ok] mining resumed` then `0.00 MH/s, waiting` for 20 minutes), the Mac that afternoon. Now every slot without a live worker is re-armed and its pack exported again before the start, and 90 s later `resume_check` logs `resume: is not mining 90 s after resume (state ..., pid ...)` for every enabled card without a hash rate (`engine.rs`, three unit tests: the state machine, the 21:25:11Z case against the old rule, the check). + A prover job on a shared card runs as the app's user or cleans its socket (`pkill -f sp1-gpu-server; rm -f /tmp/sp1-cuda-*.sock` at the start and the end; `tools/ci/prover-socket-check.sh`): the root-socket fault of 20:00Z, bench-log. ### The prover profiles: the tiers from the S_p curve (bench-log, "proving v1", the sweep, the miner-on pair and the curve) diff --git a/tools/proving-v1/pc2-socket-fix.ps1 b/tools/proving-v1/pc2-socket-fix.ps1 new file mode 100644 index 000000000..80eb0d283 --- /dev/null +++ b/tools/proving-v1/pc2-socket-fix.ps1 @@ -0,0 +1,31 @@ +# The root-socket fix for PC 2 (5 October 2026): a measurement job that ran igneum-prove-host as root inside WSL2 left +# /tmp/sp1-cuda-0.sock owned by root (and possibly a root sp1-gpu-server), so the app's prover (its own WSL user) fails +# every shard with "CudaClientError: Connect(PermissionDenied)". This job, as root: kills every sp1-gpu-server, removes +# the sockets, prints their owners before and after, then switches the app's prover off and on so its next shard +# starts a fresh server under the app's user. Never stops the miners. About 60 s. +$ErrorActionPreference = 'Continue' +$urlFile = if ($env:IGNEUM_APP_DIR) { Join-Path $env:IGNEUM_APP_DIR 'app.url' } else { Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' } +if (-not (Test-Path $urlFile)) { $urlFile = Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' } +$base = (Get-Content $urlFile -Raw).Trim().TrimEnd('/') +function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') } +function Prove($on) { try { (Invoke-RestMethod -Method Post -Uri "$base/api/prove" -ContentType 'application/json' -Body (@{on=$on} | ConvertTo-Json -Compress) -TimeoutSec 10) | ConvertTo-Json -Compress } catch { "error: $_" } } +"RESULT start $(Stamp)" +$bash = @' +echo "RESULT sockets_before $(ls -la /tmp/sp1-cuda-*.sock 2>&1 | tr '\n' ' ')" +echo "RESULT servers_before $(ps -eo user,pid,cmd | grep -E '[s]p1-gpu-server' | tr '\n' ' ' || echo none)" +pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock +echo "RESULT sockets_after $(ls -la /tmp/sp1-cuda-*.sock 2>&1 | tr '\n' ' ')" +echo "RESULT servers_after $(ps -eo user,pid,cmd | grep -E '[s]p1-gpu-server' | tr '\n' ' ' || echo none)" +'@ +$job = $env:IGNEUM_JOB_DIR; if (-not $job) { $job = $env:TEMP } +$bashFile = Join-Path $job 'socket-fix.sh' +[IO.File]::WriteAllText($bashFile, ($bash -replace "`r`n", "`n"), (New-Object System.Text.UTF8Encoding $false)) +$wslPath = (& wsl.exe -d Ubuntu-24.04 -u root -- wslpath -a ($bashFile -replace '\\', '/') 2>$null) +if (-not $wslPath) { $wslPath = '/mnt/c' + ($bashFile.Substring(2) -replace '\\', '/') } +& wsl.exe -d Ubuntu-24.04 -u root -- bash (($wslPath -replace "`0", '').Trim()) 2>&1 | ForEach-Object { ($_ -replace "`0", '') } +"RESULT prove_off $(Stamp) $(Prove $false)" +Start-Sleep -Seconds 15 +"RESULT prove_on $(Stamp) $(Prove $true)" +Start-Sleep -Seconds 40 +$st = $null; try { $st = Invoke-RestMethod -Uri "$base/api/state" -TimeoutSec 20 } catch {} +"RESULT end $(Stamp) proving status: $($st.proving.status) message: $($st.proving.message)"