App: the resume path re-arms every stopped card, re-exports its pack and checks 90 s later that every enabled card mines (PC 2 at 21:25:11Z and the Mac that afternoon stayed at 0 MH/s after resume); the PC 2 socket-fix job
The known-failed case, from PC 2's 0.3.9 log (run win-1ccfe586-20261005-200114): 1791234223 pause -> 'stopping the miners (paused)' (every slot's restart_at cleared, the 5090 'off'); 1791235511 '[ok] mining resumed'; then '0.00 MH/s, waiting' at every 30-s status line until the 0.3.10 restart at 21:49:41Z. Cause: Cmd::Resume re-armed only slots whose watchdog said faulted; the 5090's slot was healthy and stopped, so nothing restarted it. The test the_pc2_resume_of_21_25_11z_restarts_under_the_new_rule_and_not_the_old encodes that slot (faulted false, live false): the old rule returns [] (the defect), the new rule [0]. cargo test -p igneum-app resume: 3 passed; provedefault: 6 passed. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
parent
d7c28ec695
commit
c4da08d302
3 changed files with 129 additions and 3 deletions
|
|
@ -449,6 +449,36 @@ fn jitter_secs(seed: u64) -> u64 {
|
|||
5 + (seed.wrapping_mul(2654435761) >> 7) % 56
|
||||
}
|
||||
|
||||
/// Seconds after a resume before every enabled card must be mining (a worker takes 10 to 60 s to its first STATUS
|
||||
/// line with a hash rate; the pack export before it a few seconds more).
|
||||
pub(crate) const RESUME_CHECK_SECS: u64 = 90;
|
||||
|
||||
/// What the resume rule reads from a miner slot.
|
||||
#[derive(Clone, Debug, PartialEq, Eq)]
|
||||
pub(crate) struct ResumeSlot {
|
||||
/// the watchdog marked the card faulted
|
||||
pub faulted: bool,
|
||||
/// a worker process is alive on the slot
|
||||
pub live: bool,
|
||||
}
|
||||
|
||||
/// The resume rule: every slot without a live worker is re-armed, faulted or not. (The rule before 5 October 2026
|
||||
/// re-armed only faulted slots; `stop_miners("paused")` had cleared every slot's `restart_at`, so a healthy paused
|
||||
/// card never restarted: PC 2 at 21:25:11Z, the Mac that afternoon.)
|
||||
pub(crate) fn slots_to_rearm_on_resume(slots: &[ResumeSlot]) -> Vec<usize> {
|
||||
slots.iter().enumerate().filter(|(_, s)| !s.live).map(|(i, _)| i).collect()
|
||||
}
|
||||
|
||||
/// The check RESUME_CHECK_SECS after a resume: every enabled, present, non-faulted card must report a hash rate
|
||||
/// above 0 or its name is returned with its state (the caller logs one line per card).
|
||||
pub(crate) fn resume_check(cards: &[CardState]) -> Vec<String> {
|
||||
cards
|
||||
.iter()
|
||||
.filter(|c| c.enabled && c.present() && c.state != "faulted" && c.hash_now <= 0.0)
|
||||
.map(|c| format!("{} is not mining {RESUME_CHECK_SECS} s after resume (state {}, pid {}{})", c.name, c.state, c.pid, if c.message.is_empty() { String::new() } else { format!(", {}", c.message) }))
|
||||
.collect()
|
||||
}
|
||||
|
||||
struct MinerSlot {
|
||||
card: usize,
|
||||
label: String,
|
||||
|
|
@ -477,6 +507,8 @@ pub struct Engine {
|
|||
node: Option<Proc>,
|
||||
node_external: bool,
|
||||
node_restart_at: Option<Instant>,
|
||||
/// resume rule (5 October 2026): when due, every enabled card must be mining or its name goes to the log
|
||||
resume_check_at: Option<Instant>,
|
||||
node_started_at: Instant,
|
||||
node_starts: u32,
|
||||
node_restarts: u32,
|
||||
|
|
@ -579,6 +611,7 @@ impl Engine {
|
|||
node: None,
|
||||
node_external: false,
|
||||
node_restart_at: None,
|
||||
resume_check_at: None,
|
||||
node_started_at: now,
|
||||
node_starts: 0,
|
||||
node_restarts: 0,
|
||||
|
|
@ -765,13 +798,23 @@ impl Engine {
|
|||
self.shared.save_settings();
|
||||
self.st().mining.paused = false;
|
||||
self.shared.event("ok", "mining resumed");
|
||||
// a user action: faulted cards try again
|
||||
for m in self.miners.iter_mut() {
|
||||
// the resume rule (5 October 2026, PC 2 at 21:25:11Z and the Mac that afternoon): stop_miners("paused")
|
||||
// cleared every slot's restart_at and this arm re-armed only FAULTED slots, so a card whose worker had
|
||||
// simply been stopped stayed "off" at 0 MH/s until the app was relaunched. Now every slot without a
|
||||
// live worker is re-armed (a faulted one reset first), its pack is exported again before the start
|
||||
// (prepared = false: the hour may have turned while paused), and a check 90 s later names any
|
||||
// enabled card that is not mining (`resume_check`).
|
||||
let views: Vec<ResumeSlot> = self.miners.iter().map(|m| ResumeSlot { faulted: m.watch.faulted().is_some(), live: m.proc.is_some() }).collect();
|
||||
let now = Instant::now();
|
||||
for i in slots_to_rearm_on_resume(&views) {
|
||||
let m = &mut self.miners[i];
|
||||
if m.watch.faulted().is_some() {
|
||||
m.watch.event(0.0, crate::watchdog::Event::Reset);
|
||||
m.restart_at = Some(Instant::now());
|
||||
}
|
||||
m.restart_at = Some(now);
|
||||
m.prepared = false;
|
||||
}
|
||||
self.resume_check_at = Some(now + Duration::from_secs(RESUME_CHECK_SECS));
|
||||
if !self.running {
|
||||
self.start();
|
||||
}
|
||||
|
|
@ -2193,6 +2236,18 @@ impl Engine {
|
|||
self.tick_node(now);
|
||||
self.tick_watch(now);
|
||||
self.tick_miners(now);
|
||||
if let Some(at) = self.resume_check_at {
|
||||
if now >= at {
|
||||
self.resume_check_at = None;
|
||||
let (cards, paused) = { let st = self.st(); (st.mining.cards.clone(), st.mining.paused) };
|
||||
if !paused {
|
||||
for line in resume_check(&cards) {
|
||||
self.shared.log(&format!("resume: {line}"));
|
||||
self.shared.event("error", &format!("resume: {line}"));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
self.tick_telemetry(now);
|
||||
self.tick_sweep(now);
|
||||
}
|
||||
|
|
@ -3392,6 +3447,44 @@ pub fn parse_race(body: &str) -> Option<RaceParsed> {
|
|||
Some(r)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod resume_tests {
|
||||
use super::*;
|
||||
|
||||
fn card(name: &str, enabled: bool, state: &str, hash: f64) -> CardState {
|
||||
CardState { name: name.into(), vendor: "nvidia".into(), kind: "discrete".into(), enabled, state: state.into(), hash_now: hash, ..Default::default() }
|
||||
}
|
||||
|
||||
/// The state machine: paused (every slot stopped, restart_at cleared) -> resumed -> every slot without a live
|
||||
/// worker is re-armed, faulted or not.
|
||||
#[test]
|
||||
fn resume_rearms_every_stopped_slot() {
|
||||
let slots = [ResumeSlot { faulted: false, live: false }, ResumeSlot { faulted: true, live: false }, ResumeSlot { faulted: false, live: true }];
|
||||
assert_eq!(slots_to_rearm_on_resume(&slots), vec![0, 1], "the healthy stopped slot and the faulted one restart; the live one is left alone");
|
||||
}
|
||||
|
||||
/// The known-failed case (PC 2, 5 October 2026, 21:25:11Z, app 0.3.9: `[ok] mining resumed`, then `0.00 MH/s,
|
||||
/// waiting` for 20 minutes): one healthy slot, stopped by the pause, not faulted. The old rule re-armed only
|
||||
/// faulted slots and returned nothing for it; the new rule returns it.
|
||||
#[test]
|
||||
fn the_pc2_resume_of_21_25_11z_restarts_under_the_new_rule_and_not_the_old() {
|
||||
let old_rule = |slots: &[ResumeSlot]| -> Vec<usize> { slots.iter().enumerate().filter(|(_, s)| s.faulted).map(|(i, _)| i).collect() };
|
||||
let pc2 = [ResumeSlot { faulted: false, live: false }];
|
||||
assert!(old_rule(&pc2).is_empty(), "the 0.3.9 rule left the 5090's slot unarmed: this is the defect");
|
||||
assert_eq!(slots_to_rearm_on_resume(&pc2), vec![0]);
|
||||
}
|
||||
|
||||
/// The check 90 s after a resume names every enabled card without a hash rate, and nothing else.
|
||||
#[test]
|
||||
fn resume_check_names_the_cards_not_mining() {
|
||||
let cards = [card("NVIDIA GeForce RTX 5090", true, "off", 0.0), card("AMD Radeon(TM) Graphics", true, "mining", 3.4), card("Intel UHD", false, "off", 0.0), card("RTX 3060", true, "faulted", 0.0)];
|
||||
let lines = resume_check(&cards);
|
||||
assert_eq!(lines.len(), 1, "{lines:?}");
|
||||
assert!(lines[0].starts_with("NVIDIA GeForce RTX 5090 is not mining 90 s after resume (state off"), "{}", lines[0]);
|
||||
assert!(resume_check(&[card("RTX 5090", true, "mining", 118.9)]).is_empty());
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{digest_from_line, parse_race, switches_of, sync_decision, Reading};
|
||||
|
|
|
|||
|
|
@ -112,6 +112,8 @@ Reading. Nobody pays an aggregator as a separate role: Aztec's 30% goes to whoev
|
|||
| Apple silicon default | **off** | the gate was "a shard under 60 s with the miner running": the M5 Max CPU took 41.3 and 55.4 s for EMPTY shards under tonight's load and 272 s for a 200-pgas shard on 4 October; a full shard at `S_p` was never under 60 s. Settings switches it on | bench-log "proving v1" CPU chain row; 4 October CPU rows |
|
||||
| The prover profile per card and the 12 GB and 16 GB gates (the project lead: "make sure we can prove on 12gb cards"; "is there any way we can make 12gb cards mine and prove?") | **measured on PC 2, the rows below** | the SP1 6.8.1 GPU server reads `ELEMENT_THRESHOLD`, `HEIGHT_THRESHOLD`, `SHARD_SIZE` and the `SP1_WORKER_NUM_*`/`BUFFER_SIZE` knobs from the environment it inherits (`sp1-core-executor-6.8.1/src/opts.rs`, `sp1-prover-6.8.1/src/worker/config.rs`); the app passes a profile per card (`provedefault.rs`) and the host forwards it | the sweep job `memsweep-pc2-pv1` and the miner-on run |
|
||||
|
||||
The resume path (5 October 2026, the 0.3.11 app): `POST /api/resume` on 0.3.9 re-armed only FAULTED cards (`stop_miners("paused")` clears every slot's `restart_at`), so a healthy paused card stayed "off" at 0 MH/s until the app was relaunched: PC 2 at 21:25:11Z (the aggregation-cost job's pause and resume; `[ok] mining resumed` then `0.00 MH/s, waiting` for 20 minutes), the Mac that afternoon. Now every slot without a live worker is re-armed and its pack exported again before the start, and 90 s later `resume_check` logs `resume: <card> is not mining 90 s after resume (state ..., pid ...)` for every enabled card without a hash rate (`engine.rs`, three unit tests: the state machine, the 21:25:11Z case against the old rule, the check).
|
||||
|
||||
A prover job on a shared card runs as the app's user or cleans its socket (`pkill -f sp1-gpu-server; rm -f /tmp/sp1-cuda-*.sock` at the start and the end; `tools/ci/prover-socket-check.sh`): the root-socket fault of 20:00Z, bench-log.
|
||||
|
||||
### The prover profiles: the tiers from the S_p curve (bench-log, "proving v1", the sweep, the miner-on pair and the curve)
|
||||
|
|
|
|||
31
tools/proving-v1/pc2-socket-fix.ps1
Normal file
31
tools/proving-v1/pc2-socket-fix.ps1
Normal file
|
|
@ -0,0 +1,31 @@
|
|||
# The root-socket fix for PC 2 (5 October 2026): a measurement job that ran igneum-prove-host as root inside WSL2 left
|
||||
# /tmp/sp1-cuda-0.sock owned by root (and possibly a root sp1-gpu-server), so the app's prover (its own WSL user) fails
|
||||
# every shard with "CudaClientError: Connect(PermissionDenied)". This job, as root: kills every sp1-gpu-server, removes
|
||||
# the sockets, prints their owners before and after, then switches the app's prover off and on so its next shard
|
||||
# starts a fresh server under the app's user. Never stops the miners. About 60 s.
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$urlFile = if ($env:IGNEUM_APP_DIR) { Join-Path $env:IGNEUM_APP_DIR 'app.url' } else { Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
if (-not (Test-Path $urlFile)) { $urlFile = Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
$base = (Get-Content $urlFile -Raw).Trim().TrimEnd('/')
|
||||
function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') }
|
||||
function Prove($on) { try { (Invoke-RestMethod -Method Post -Uri "$base/api/prove" -ContentType 'application/json' -Body (@{on=$on} | ConvertTo-Json -Compress) -TimeoutSec 10) | ConvertTo-Json -Compress } catch { "error: $_" } }
|
||||
"RESULT start $(Stamp)"
|
||||
$bash = @'
|
||||
echo "RESULT sockets_before $(ls -la /tmp/sp1-cuda-*.sock 2>&1 | tr '\n' ' ')"
|
||||
echo "RESULT servers_before $(ps -eo user,pid,cmd | grep -E '[s]p1-gpu-server' | tr '\n' ' ' || echo none)"
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock
|
||||
echo "RESULT sockets_after $(ls -la /tmp/sp1-cuda-*.sock 2>&1 | tr '\n' ' ')"
|
||||
echo "RESULT servers_after $(ps -eo user,pid,cmd | grep -E '[s]p1-gpu-server' | tr '\n' ' ' || echo none)"
|
||||
'@
|
||||
$job = $env:IGNEUM_JOB_DIR; if (-not $job) { $job = $env:TEMP }
|
||||
$bashFile = Join-Path $job 'socket-fix.sh'
|
||||
[IO.File]::WriteAllText($bashFile, ($bash -replace "`r`n", "`n"), (New-Object System.Text.UTF8Encoding $false))
|
||||
$wslPath = (& wsl.exe -d Ubuntu-24.04 -u root -- wslpath -a ($bashFile -replace '\\', '/') 2>$null)
|
||||
if (-not $wslPath) { $wslPath = '/mnt/c' + ($bashFile.Substring(2) -replace '\\', '/') }
|
||||
& wsl.exe -d Ubuntu-24.04 -u root -- bash (($wslPath -replace "`0", '').Trim()) 2>&1 | ForEach-Object { ($_ -replace "`0", '') }
|
||||
"RESULT prove_off $(Stamp) $(Prove $false)"
|
||||
Start-Sleep -Seconds 15
|
||||
"RESULT prove_on $(Stamp) $(Prove $true)"
|
||||
Start-Sleep -Seconds 40
|
||||
$st = $null; try { $st = Invoke-RestMethod -Uri "$base/api/state" -TimeoutSec 20 } catch {}
|
||||
"RESULT end $(Stamp) proving status: $($st.proving.status) message: $($st.proving.message)"
|
||||
Loading…
Reference in a new issue