Merge remote-tracking branch 'origin/miner-reliability-21' into release-0.3.21

This commit is contained in:
igneum-labs 2026-10-07 13:46:25 +00:00
commit 1e1e257580
4 changed files with 142 additions and 6 deletions

View file

@ -203,3 +203,48 @@ mod tests {
assert_eq!(PROVE_UNDER_12GB_LINE, "proving needs a 12 GB card; mining continues");
}
}
/// MF-10 (docs/plans/miner-faults.md, 7 October 2026): a `sp1-gpu-server` built for another card's architecture fails
/// every proof in 12 s with `CudaRustError: named symbol not found` and nothing notices. `card_cap` is the card's
/// compute capability as nvidia-smi prints it ("12.0", "8.9", "8.6"); `server_archs` are the `sm_NN` words found in
/// the server binary (empty = unknown, which passes: an SDK server may carry no arch string). Some(the sentence) when
/// the server names architectures and none is the card's.
pub fn server_mismatch(card_cap: &str, server_archs: &[String]) -> Option<String> {
let cap = card_cap.trim();
if cap.is_empty() || server_archs.is_empty() {
return None;
}
let want = format!("sm_{}", cap.replace('.', ""));
if server_archs.iter().any(|a| a.trim() == want) {
return None;
}
Some(format!("the proving server is built for {} and this card is {want} (compute capability {cap}); proving stays off here until a server for this card is installed", server_archs.join(", ")))
}
/// A proof failure line that names the architecture class (MF-10): the server's kernels do not load on this card.
pub fn is_arch_failure(text: &str) -> bool {
text.contains("named symbol not found") || text.contains("no kernel image is available")
}
#[cfg(test)]
mod arch_tests {
use super::*;
/// The prover roll, 7 October 2026: sm_86 servers on a 4070 (8.9) and a 5090 (12.0) failed every proof; the
/// 3080 and 3090 (8.6) proved.
#[test]
fn a_server_for_another_card_is_refused() {
let sm86 = vec!["sm_86".to_string()];
assert!(server_mismatch("8.9", &sm86).unwrap().contains("built for sm_86 and this card is sm_89"));
assert!(server_mismatch("12.0", &sm86).unwrap().contains("sm_120"));
assert_eq!(server_mismatch("8.6", &sm86), None);
// a fat binary names every card
let fat = vec!["sm_86".to_string(), "sm_89".to_string(), "sm_120".to_string()];
assert_eq!(server_mismatch("8.9", &fat), None);
// unknown on either side passes (the proof failure line is the second guard)
assert_eq!(server_mismatch("", &sm86), None);
assert_eq!(server_mismatch("8.9", &[]), None);
assert!(is_arch_failure("CudaRustError: named symbol not found"));
assert!(!is_arch_failure("PermissionDenied"));
}
}

View file

@ -384,6 +384,8 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
read_ids(&shared, &t);
}
}
// MF-10: the server architecture check, once per tools probe (None = not read yet; Some(None) = matches)
let mut arch_check: Option<Option<String>> = None;
loop {
std::thread::sleep(Duration::from_secs(10));
let enabled = shared.settings.lock().unwrap().prove;
@ -468,6 +470,23 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
continue;
}
}
// MF-10 (7 October 2026, the prover roll): a GPU server built for another card's architecture fails every
// proof; the card's compute capability and the server's sm_ words are read once per tools probe, a mismatch
// refuses the GPU path with the reason on the tile
if t.cuda {
if arch_check.is_none() {
arch_check = Some(server_arch_check(&shared, t));
}
if let Some(Some(line)) = arch_check.as_ref() {
set(&shared, |p| {
p.enabled = true;
p.available = false;
p.status = "off".into();
p.message = line.clone();
});
continue;
}
}
set(&shared, |p| {
p.enabled = true;
p.available = true;
@ -702,7 +721,10 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
let last = out.lines().rev().find(|l| l.contains("RESULT") || l.contains("rror")).unwrap_or("failed").to_string();
// the root-socket class (5 October 2026, PC 2 at 20:00Z and 21:25Z): a job that ran the host as root
// inside WSL2 left /tmp/sp1-cuda-0.sock owned by root, and this user's client cannot open it
let hint = if last.contains("PermissionDenied") { " (a GPU-server socket /tmp/sp1-cuda-*.sock owned by another user, left by a job that ran the prover as root: remove it as that user, or run the socket-fix job)" } else { "" };
let hint = if crate::provedefault::is_arch_failure(&last) { " (MF-10: the proving server is built for another card's architecture; proving stays off here until a server for this card is installed)" } else if last.contains("PermissionDenied") { " (a GPU-server socket /tmp/sp1-cuda-*.sock owned by another user, left by a job that ran the prover as root: remove it as that user, or run the socket-fix job)" } else { "" };
if crate::provedefault::is_arch_failure(&last) {
shared.log(&format!("FAULT class=prover-arch card=\"gpu\" app={} reason=\"{}\"", crate::engine::VERSION, last.replace('"', "'")));
}
return Err(format!("prover: {last}{hint}"));
}
let res: Value = serde_json::from_str(&std::fs::read_to_string(&results).map_err(|e| e.to_string())?).map_err(|e| e.to_string())?;
@ -1175,3 +1197,23 @@ mod tests {
assert!(probe_message(Path::new("C:\\Igneum"), false, true).contains("Ubuntu-24.04 did not answer"));
}
}
/// MF-10: the card's compute capability and the GPU server's architecture words, read through the same shell the
/// tools live in (WSL2 on Windows). None = no mismatch (or nothing readable); Some(line) = refuse with this reason.
fn server_arch_check(shared: &Shared, t: &Tools) -> Option<String> {
let script = "cap=$(nvidia-smi --query-gpu=compute_cap --format=csv,noheader 2>/dev/null | head -1); srv=\"$HOME/.sp1/bin/sp1-gpu-server\"; archs=$( [ -f \"$srv\" ] && strings \"$srv\" 2>/dev/null | grep -o 'sm_[0-9]*' | sort -u | tr '\\n' ' '); echo \"CAP=$cap\"; echo \"ARCHS=$archs\"";
let out = if t.wsl {
let file = crate::wslhost::write_script("prove-arch", script).ok()?;
crate::platform::quiet(&mut crate::wslhost::command(&crate::platform::tool("wsl"), crate::wslhost::DISTRO, None, &file.path, true, &[])).output().ok().map(|o| String::from_utf8_lossy(&o.stdout).to_string())?
} else {
crate::detect::run_timeout(std::process::Command::new("bash").args(["-c", script]), None, Duration::from_secs(20))?
};
let cap = out.lines().find_map(|l| l.strip_prefix("CAP=")).unwrap_or("").trim().to_string();
let archs: Vec<String> = out.lines().find_map(|l| l.strip_prefix("ARCHS=")).unwrap_or("").split_whitespace().map(|s| s.to_string()).collect();
shared.log(&format!("prover: card compute capability {}, server architectures {}", if cap.is_empty() { "unknown" } else { &cap }, if archs.is_empty() { "unknown".to_string() } else { archs.join(" ") }));
let line = crate::provedefault::server_mismatch(&cap, &archs);
if let Some(l) = &line {
shared.log(&format!("FAULT class=prover-arch card=\"gpu\" app={} reason=\"{}\"", crate::engine::VERSION, l.replace('"', "'")));
}
line
}

View file

@ -16,7 +16,7 @@ Standing rules behind every row (branch `miner-reliability`, off `release-0.3.19
| Nothing on a user's machine is changed by a one-off script: card settings travel as the signed `cards` job kind (per card enabled, identities, power_pct), applied through the app's own card path, persisted, read back in the report, refused for a card the machine does not have | `src/jobs.rs` (`KINDS`, `validate_params`), `src/jobrun.rs` (`cards_job_choices`, `cards_applied`), `engine.rs` (`Action::ApplyCards`), `packaging/ota/publish-jobs.sh add --kind cards --cards "key=on:8"` |
| A cut never dies on a kept datadir: every stored-row schema change carries a versioned read path, and every release's gate starts the pinned binary on a copy of a standing box's datadir (Linux and Windows shapes) beside the wiped canary; LG-4's tenth install keeps the ninth's datadir (MF-8) | the node line's N13 (b7cc37e7); the canary form; `tools/fleet/first-share-gate.mjs` row kept=true |
| A node stops faster than its restart: every poll loop returns the moment shutdown is set, and "a shutdown returns within one second" is a named release gate beside the kept-datadir start (MF-9) | the node line's listener watchdog fix (b7cc37e7's follow-up); the mixed-version gate; `engine.rs` `restart_node` (the start is scheduled after `stop_node` returns) |
| A prover never runs a server built for another card: the compute capability is read at start and a mismatch is refused with the reason shown and reported home; the kit ships one server per architecture or a fat binary (MF-10, 0.3.21) | `src/prover.rs` (owed: the capability check at start and the 15-second symbol-error class), the kit's per-architecture servers |
| A prover never runs a server built for another card: the compute capability is read at start and a mismatch is refused with the reason shown and reported home; the kit ships one server per architecture or a fat binary (MF-10, 0.3.21) | `src/provedefault.rs` `server_mismatch` and `is_arch_failure` (tested), `src/prover.rs` `server_arch_check` (the card's `compute_cap` and the server's `sm_` words read through the tools' own shell at the first probe; a mismatch refuses the GPU path with the reason on the tile and a `FAULT class=prover-arch` line; a proof failing with the symbol error is logged as the same class), the kit's per-architecture servers (0.3.21 kit, owed) |
| The fresh-install claim (LG-4) is a job, not a runbook: `tools/fleet/first-share-gate.mjs` on rented Windows boxes, on every cut, its line read by the shipper's publish | `relay/playbooks/first-share.ps1`, `tools/fleet/first-share-gate.mjs`, `site/evidence/first-share-<version>.json` |
## The register
@ -29,9 +29,11 @@ Standing rules behind every row (branch `miner-reliability`, off `release-0.3.19
| MF-5 | 7 Oct 2026, PC 1, 0.3.17 node, 24 identities (the real cause of the 11:2x faults; MF-4 withdrawn as the cause) | Every card faulted "no status line from the miner for 90 s (restarted once already)" after the restart | Evidence (PC 1, 7 Oct 2026 11:4x UK): with 8 identities a card (24 template fetches a round) the 0.3.17 node answered no template in 5 s; with 2 a card (4 fetches) both cards mined at full rate (5090 122.4 MH/s, 9070 XT 18.9) within two minutes. The node's getBlockTemplate answered past 5 s with 24 identities fetching; the miners waited for a template inside their job-fill loop and printed no STATUS at all; the watchdog read the silence as the worker's; one restart, then faulted for good | Miner: STATUS every interval whatever the template state (`template_wait=<s>` while it waits, `template_ms=` the node's last template time, `identities_active=`); the feed fetches only as many identities as fit one pass inside 8 s at the node's measured template time (`identities_for`: all of them when the node answers under 1 s; 24 at 8 s per template becomes 1), raised again when it answers faster; a pass whose fetches all fail backs off 2, 4, 8, 10 s and retries for ever with a `NODE SLOW` line once per 30 s. App: a STATUS with `template_wait>0` is the miner's heartbeat and the node's latency, never the card's fault (no zero-rate clock, no restart); the card reads "node slow: waiting for a block template for N s; the worker is kept" or "node slow: a template takes N s; k of n identities active"; the mitigation of the day (a one-off script POSTing /api/cards) is closed by the signed `cards` job kind | `watchdog` tests `a_slow_node_never_faults_the_card`, `parses_status_and_fault_lines` (the 0.3.20 line); `jobs` test for the `cards` kind; injector step `slow-node` (a template stub answering in 8 s while three cards run: open, needs the stub) | `app-run slow-node PASS` (0.3.20) |
| MF-6 | 7 Oct 2026, PC 1, 0.3.19 | After a `--stop-miners` job, a following read-only job kept both cards "off, held for a remote job" for its whole three minutes | The engine released the hold only when no job held the miners; the next job's active state hid the release (engine.rs 3032 class) | A hold belongs to the job that took it (`job_hold_owner`) and releases the moment that job is no longer the running one, whatever runs next, or when its own cap passes (logged); a read-only job never holds (`jobrun::hold_release`) | `jobrun` test `a_hold_belongs_to_the_job_that_took_it` (owner running, another job, no job, cap passed, no owner) | `app tests: hold rule green` |
| MF-7 | 7 Oct 2026, PC 1, 0.3.19 | Orphan `igneum-miner.exe` processes the app no longer tracked (two alive under `--stop-miners` with their rows at pid 0, one after) hammered the node's template RPC beside the tracked miners | The engine lost track of miners it had started (a stop that timed out, a restart over a live process) and never looked for them again | The engine owns every miner it started: at start, after every stop and every minute it kills any `igneum-miner` whose command line carries THIS engine's node RPC (the fence) and whose pid it does not track, one log line and one fault report per kill, never by name alone (`sweep_orphan_miners`, `platform::miner_processes`, `kill_pid`); a restart kills the slot's old process before the new one starts | injector step `orphan-miner` (a stray miner on the engine's node is killed inside the minute, the engine's own miner left alone) | `app-run orphan-miner PASS` |
| MF-8 | 7 Oct 2026, every node build from 10db4b61 on a kept 0.3.17 datadir | "Node refuses its own kept datadir after an update": the app updates, the node dies at once (`DeserializationError(Io(Kind(UnexpectedEof)))` at `consensus/src/model/stores/virtual_state.rs:250`), the app restarts it in a loop, the miner never starts | `silent: bool` was added to `BlockRewardData` under `serde(default)` (10db4b61, the 0.3.16 vote-or-burn commit) and bincode ignores serde defaults, so the old row reads short; no canary saw it because every canary wiped. Fixed as ledger N13 on `release-0.3.20-node` b7cc37e7 (the store reads the v1 row and rewrites it) | Every schema change to a stored row ships with a versioned read path (the old shape read, rewritten in the new one); every release's gate starts the pinned binary on a COPY of a standing box's datadir, Linux-shaped and Windows-shaped (a copy of PC 2's), with the rewrite line or the synced line as the pass; the app never asks the user to wipe: a node that dies inside 10 s of its start is reported home (`node-exit` FAULT line) and the row says "the node cannot read its data after the update; the team has the report" | The node line's N13 test (the v1 row read and rewritten); the gate's kept-datadir start on both shapes; the injector step `kept-datadir` is OWED (a datadir from the previous release under the new binary needs a previous-release node on the pod; the shape is: run the old node to a few hundred blocks, stop it, start the new one on its datadir, pass on the synced line) | `canary kept-datadir start PASS (linux, windows)` in every cut's plan; `LG-4` row kept=true beside the nine wiped |
| MF-8 | 7 Oct 2026, every node build from 10db4b61 on a kept 0.3.17 datadir | "Node refuses its own kept datadir after an update": the app updates, the node dies at once (`DeserializationError(Io(Kind(UnexpectedEof)))` at `consensus/src/model/stores/virtual_state.rs:250`), the app restarts it in a loop, the miner never starts | `silent: bool` was added to `BlockRewardData` under `serde(default)` (10db4b61, the 0.3.16 vote-or-burn commit) and bincode ignores serde defaults, so the old row reads short; no canary saw it because every canary wiped. Fixed as ledger N13 on `release-0.3.20-node` b7cc37e7 (the store reads the v1 row and rewrites it) | Every schema change to a stored row ships with a versioned read path (the old shape read, rewritten in the new one); every release's gate starts the pinned binary on a COPY of a standing box's datadir, Linux-shaped and Windows-shaped (a copy of PC 2's), with the rewrite line or the synced line as the pass; the app never asks the user to wipe: a node that dies inside 10 s of its start is reported home (`node-exit` FAULT line) and the row says "the node cannot read its data after the update; the team has the report" | The node line's N13 test (the v1 row read and rewritten); the gate's kept-datadir start on both shapes; the injector step `kept-datadir` (`app-run.mjs --node-old <previous igneumd>`: the previous release's node writes the datadir to 150 blocks and stops, the engine's new node must open it, read synced, with one start and no exit line; its pod seconds are appended with the 0.3.21 run) | `canary kept-datadir start PASS (linux, windows)` in every cut's plan; `LG-4` row kept=true beside the nine wiped |
| MF-9 | 7 Oct 2026, release-0.3.20-node b7cc37e7, the mixed-version gate | A node that stops slower than its restart: the restart died at once on the datadir LOCK of the stopping node | The listener watchdog slept its whole 10 s poll before it checked the shutdown flag, so a stop took up to 10 s while the app's restart followed inside it | Every poll loop in the node returns the moment shutdown is set (a `select` on the shutdown signal, never a sleep then a check); "a shutdown returns within one second" is a named gate of every release beside the kept-datadir start; the app's `stop_node` waits for the exit before the restart (`restart_node` schedules the start after the stop returns) | The node line's shutdown-latency test (stop at a random moment of the poll, return under 1 s); the mixed-version gate's restart case; `app-run.mjs` step `node-silent` reads the restart-to-synced seconds | `node shutdown under 1 s PASS` in every cut's plan |
| MF-10 | 7 Oct 2026, the prover roll (3080, 3090, 4070, 5090) | Every proof fails in 12 s with `CudaRustError: named symbol not found`; the miner never notices and the box proves nothing for hours | The prover's `sp1-gpu-server` was built for another card's compute capability (sm_86 on the 3080 and 3090, sm_89 on the 4070, sm_120 on the 5090), so every kernel load fails the same way | The prover reads the card's compute capability at start (`nvidia-smi --query-gpu=compute_cap`) and refuses to start a server that does not match, with the reason on the prover tile and a FAULT line home ("the proving server is built for sm_89, this card is sm_120"); a proof that fails inside 15 s with the symbol error marks the server mismatched the same way; the 0.3.21 kit ships one server per architecture chosen at install, or a fat binary | the prover's compute-capability test (a server named for sm_89 refused on a card that reads 12.0); the fleet roll's paired line per box | `prover server matches the card PASS` per box in the roll, beside the kept-datadir and the shutdown gates |
| MF-10 | 7 Oct 2026, the prover roll (3080, 3090, 4070, 5090) | Every proof fails in 12 s with `CudaRustError: named symbol not found`; the miner never notices and the box proves nothing for hours | The prover's `sp1-gpu-server` was built for another card's compute capability (sm_86 on the 3080 and 3090, sm_89 on the 4070, sm_120 on the 5090), so every kernel load fails the same way | The prover reads the card's compute capability at start (`nvidia-smi --query-gpu=compute_cap`) and refuses to start a server that does not match, with the reason on the prover tile and a FAULT line home ("the proving server is built for sm_89, this card is sm_120"); a proof that fails inside 15 s with the symbol error marks the server mismatched the same way; the 0.3.21 kit ships one server per architecture chosen at install, or a fat binary | `provedefault` test `a_server_for_another_card_is_refused` (sm_86 on 8.9 and 12.0 refused, 8.6 passes, a fat binary passes, unknowns pass); the fleet roll's paired line per box | `prover server matches the card PASS` per box in the roll, beside the kept-datadir and the shutdown gates |
| MF-11 | 7 Oct 2026, PC 2 (1ccfe586), silent 10:46Z to 13:38Z (Kernel-Power 41 and 6008 "unexpected shutdown", no bugcheck, no dump; two such events on that box today; the 0.3.19 update-now at 10:34:38Z had returned cleanly, so the update is not the cause) | The machine goes silent and nothing tells anyone: no job, no relay task, no line reaches the team for three hours; the restart path today is a hand at the PC | A power loss took the whole PC; the relay agent runs under the app's lifetime instead of beside it, and no watcher raises a line when a machine stops reporting | An app or agent that does not report within 10 minutes raises a line somewhere that still runs (the intake's own silence watch per machine id: a FAULT line "no report from <id8> for N min" to the team's channel); the relay agent runs as a service that survives the app and restarts the app on boot (the relay lane, 0.3.21); the app's first act after any start is the read-back line "app <version> up, node <commit>" to the intake | the intake's silence-watch test (a machine whose last upload is older than 10 min is listed once); the relay lane's service test (the app killed, the agent still answers; the box rebooted, the app back) | `silence watch` line in the team channel per quiet box; PC 2's own line when it is back. Note for MF-8: PC 2's datadir, cut off mid-write twice by power loss, started clean on c4459193 (the PC 2 job's PASS) |
| MF-12 | 7 Oct 2026, the fleet lane's ten-member pool window (ten rented 3070s, `igneum-miner mine none ... --pool pool-1:4463`, daemon 03457d96, miner 9829bdf7); the pool lane's row (its branch pool-mf-row 343dd83b called it MF-11; renumbered here so the register has one number per class) | A pool member without a node of its own stopped hashing at the epoch boundary 78 to 79 and never resumed: every member printed `POOL SEEDS epoch c1fc2c7c... class 3` and `worker: info prepare started for epoch c1fc2c7c160f8a19 ... (NVRTC sm_86 in the background)`, that prepare never answered `prepared` or `prepare-failed`, the worker sat at 0 percent GPU serving the epoch-78 pack, the daemon's STATUS read `workers=9 accepted=0`, vardiff eased every member from shift 10 to 27 with no share, and no error line was printed anywhere (12:50Z to 13:08Z; cleared only by a restart on a pack exported from pool-1's node) | Open (the pool lane): the member's `prepare` line is the solo miner's shape (`igneum/miner/src/pool.rs` `prepare_line`), so the fault is either the member's pack write under `--prepare-packs` or the worker's NVRTC prepare on that pack; the fleet lane holds the worker lines and a solo control on the same box is the split | A member that sent a `prepare` and heard nothing for 120 s treats it as `prepare-failed`: it re-exports the pack, re-sends the prepare once, and on a second silence restarts its worker with the reason on its STATUS line and in the pool's `stats`; the pool daemon flags a member whose accepted count stays 0 across an epoch roll (`EPOCH STALL` line, the member's `online` false on the page); no pool member is ever silent across a boundary | owed (the pool lane): the pool's measure harness crosses two epoch boundaries with a member on `none` and a worker whose prepare is held (the stand-in worker of `tools/reliability/fake-worker.mjs`), and reads shares on both sides | the ten-member window's capture: every member's accepted count rises across every boundary the window crosses |
| MF-3 | 7 Oct 2026, PC 1, Intel Arc | The Intel driver's first install did not bind: the device sat in Code 12 at install time; the card never mined until a reboot | A driver installed while the device reports a problem code (12, 43, 31) does not bind; nothing re-scanned the device afterwards, and the app only re-enumerates | The app re-enumerates every 60 s and starts the worker the minute the OS drives the card (`hotplug::diff` recovered / revived, `settle_new`); the row says what to do while it does not ("reboot with the card attached; if it persists, reinstall the driver with the card attached"); a Windows host asks for a re-scan (`pnputil /scan-devices`) after a problem code is seen, every 5 minutes, at most 6 times (follow-up, host side) | `hotplug` test `a_driven_card_that_turns_faulty_is_errored_and_recovers_later`; `app-run.mjs` step `card-appears` (a card listed after 2 minutes starts without a tap) | `app-run card-appears PASS` |
## Commits (7 October 2026)

View file

@ -21,6 +21,8 @@
// no tap; it leaves: its row is marked removed; it comes back: mining again (Linux and Windows only)
// node-silent the node is stopped with SIGSTOP: no sign of life for 120 s, the app restarts the node in-process
// and the miner comes back once it is ready
// kept-datadir MF-8 (with --node-old): the previous release's node ran this datadir to a few hundred blocks and was
// stopped; the engine's new node must open it and sync (a node that dies inside 10 s is the class)
// orphan-miner MF-7: a stray igneum-miner on this engine's node, not started by it, is killed by the minute sweep;
// the engine's own miner is left alone
// one-card-fails MF-4: two more cards appear, one failing its self-test for ever: the healthy two mine, the failing
@ -38,8 +40,12 @@ const opt = (n, d) => { const i = args.indexOf(n); return i >= 0 ? args[i + 1] :
const APP = opt('--app');
const MINER = opt('--miner');
const NODE = opt('--node');
// MF-8: a previous release's igneumd; the kept-datadir pre-phase runs it first on the app's node dir, stops it, and the
// engine's own (new) node must come up on that datadir
const NODE_OLD = opt('--node-old');
const ONLY = opt('--only', '').split(',').filter(Boolean);
const SCRATCH = process.env.SCRATCH || `/tmp/igneum-reliability-app-${process.pid}`;
const CHAIN_ID_ENV = {};
const RPC = 29960, P2P = 29961, SUFFIX = 9960;
const MAC = process.platform === 'darwin';
for (const [k, v] of Object.entries({ APP, MINER, NODE })) if (!v || !existsSync(v)) { console.error(`missing ${k} (${v})`); process.exit(2); }
@ -75,6 +81,31 @@ const env = {
// easy genesis bits so the CPU block producer of the catch-up step makes blocks on two threads
IGNEUM_DEVNET_GENESIS_BITS: '0x1f100000',
};
// MF-8 pre-phase: the previous release writes the datadir first
let keptPre = null;
if (NODE_OLD) {
if (!existsSync(NODE_OLD)) { console.error(`missing --node-old ${NODE_OLD}`); process.exit(2); }
const ndir = join(data, `devnet-${SUFFIX}`); mkdirSync(ndir, { recursive: true });
const nodeArgs = ['--devnet', `--devnet-suffix=${SUFFIX}`, '--nodnsseed', '--disable-upnp', '--nologfiles', '--enable-unsynced-mining', '--outpeers=0',
`--appdir=${ndir}`, `--rpclisten=127.0.0.1:${RPC}`, `--evm-rpclisten=127.0.0.1:${RPC + 180}`, `--listen=127.0.0.1:${P2P}`, '--loglevel=warn', '--yes'];
const oldLog = join(SCRATCH, 'node-old.log');
const oldNode = spawn(NODE_OLD, nodeArgs, { stdio: ['ignore', 'pipe', 'pipe'], env: { ...env } });
oldNode.stdout.on('data', d => appendFileSync(oldLog, d)); oldNode.stderr.on('data', d => appendFileSync(oldLog, d));
log(`kept-datadir: the previous release's node pid ${oldNode.pid} on ${ndir}`);
await sleep(5000);
const producer = spawn(MINER, ['mine', `grpc://127.0.0.1:${RPC}`, '2', '600', 'cpu0', '--engine', 'igneum-pow', '--no-vote', '--payout-label', 'cpu0'], { stdio: ['ignore', 'pipe', 'ignore'] });
let blocks = 0; producer.stdout.on('data', d => { blocks += (d.toString().match(/ACCEPTED block/g) || []).length; });
const t0b = Date.now();
while (blocks < 150 && Date.now() - t0b < 400000) await sleep(2000);
try { producer.kill('SIGTERM'); } catch {}
await sleep(3000);
const tStop = Date.now();
oldNode.kill('SIGTERM');
let gone = false; for (let i = 0; i < 60 && !gone; i++) { await sleep(500); try { process.kill(oldNode.pid, 0); } catch { gone = true; } }
if (!gone) { oldNode.kill('SIGKILL'); await sleep(1000); }
keptPre = { blocks, stop_s: (Date.now() - tStop) / 1000, gone };
log(`kept-datadir: ${blocks} blocks written by the previous release, its node stopped in ${keptPre.stop_s.toFixed(1)} s`);
}
const app = spawn(APP, ['--no-open'], { stdio: ['pipe', 'pipe', 'pipe'], env });
const appOut = join(SCRATCH, 'app.out');
app.stdout.on('data', d => appendFileSync(appOut, d)); app.stderr.on('data', d => appendFileSync(appOut, d));
@ -363,8 +394,24 @@ S['orphan-miner'] = async () => {
], { inject_to_kill: killedAt ? s(killedAt - tInject) : 'n/a' });
};
const order = ['catch-up', 'card-appears', 'own-restart', 'zero-ladder', 'no-status', 'node-silent', 'one-card-fails', 'orphan-miner'];
const names = ONLY.length ? order.filter(n => ONLY.includes(n)) : order;
S['kept-datadir'] = async () => {
if (!NODE_OLD) { verdict('kept-datadir', [{ ok: false, what: 'not run: no --node-old given (a previous release\'s igneumd)' }], {}); return; }
const tStart = Date.now();
const synced = await until((st) => st.node.synced && st.node.starts >= 1, 240000, 'the new node to come up on the kept datadir');
const tSynced = Date.now();
const st1 = await poll();
const died = engineLogLines(/igneumd exited at once|igneumd exited with code/);
const rewrite = engineLogLines(/virtual state|rewrit|v1 row/i);
verdict('kept-datadir', [
{ ok: keptPre && keptPre.blocks >= 100, what: `the previous release wrote the datadir (${keptPre ? keptPre.blocks : 0} blocks) and stopped in ${keptPre ? keptPre.stop_s.toFixed(1) : '?'} s` },
{ ok: !!synced, what: `the new node opened the kept datadir and read synced (${s(tSynced - tStart)} after the engine started)` },
{ ok: died.length === 0, what: `no node exit at start (${died.length} exit lines)` },
{ ok: st1 && st1.node.starts === 1, what: `one node start, no restart loop (starts ${st1 ? st1.node.starts : '?'})` },
], { blocks_by_old_node: keptPre ? keptPre.blocks : 0, old_node_stop_s: keptPre ? keptPre.stop_s.toFixed(1) : 'n/a', new_node_synced_after: s(tSynced - tStart), rewrite_lines: rewrite.length });
};
const order = ['kept-datadir', 'catch-up', 'card-appears', 'own-restart', 'zero-ladder', 'no-status', 'node-silent', 'one-card-fails', 'orphan-miner'];
const names = (ONLY.length ? order.filter(n => ONLY.includes(n)) : order).filter(n => n !== 'kept-datadir' || NODE_OLD);
for (const n of names) {
log(`=== ${n}`);
try { await S[n](); } catch (e) { log(`step ${n} threw: ${e.stack || e}`); results.push({ name: n, pass: false, checks: [{ ok: false, what: String(e) }], metrics: {} }); }