From df2617bb4302b0aa21ae8b1728f753e87e541e89 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 11:47:52 +0000 Subject: [PATCH] reliability injector: card-appears uses a second fake device (added, removed, revived); an empty device list is read by the engine as no answer Co-Authored-By: Claude Fable 5.1 --- tools/reliability/app-run.mjs | 114 +++++++++++++++++++++++++++++++++- 1 file changed, 112 insertions(+), 2 deletions(-) diff --git a/tools/reliability/app-run.mjs b/tools/reliability/app-run.mjs index 5e557cac..90066014 100755 --- a/tools/reliability/app-run.mjs +++ b/tools/reliability/app-run.mjs @@ -17,8 +17,8 @@ // (the ladder), the reason stays on the card in plain words, nothing says "restarted once already" // and no card is ever "faulted"; the worker is healthy again: mining resumes with no tap // no-status the miner process is stopped with SIGSTOP: no status line for 90 s, the app restarts it -// card-appears MF-3: the card is absent from the enumeration at start (unplugged, or a problem code) and appears -// two minutes later: the hot-plug pass starts its worker with no tap (Linux and Windows only) +// card-appears MF-3: a second card appears (plugged in, or driven after a driver install): its worker starts with +// no tap; it leaves: its row is marked removed; it comes back: mining again (Linux and Windows only) // node-silent the node is stopped with SIGSTOP: no sign of life for 120 s, the app restarts the node in-process // and the miner comes back once it is ready // orphan-miner MF-7: a stray igneum-miner on this engine's node, not started by it, is killed by the minute sweep; @@ -249,6 +249,116 @@ S['no-status'] = async () => { ], { quiet_to_restart: s(tRestart - tInject), restart_to_mining: s(tBack - tRestart) }); }; +S['card-appears'] = async () => { + if (MAC) { verdict('card-appears', [{ ok: true, what: 'skipped on macOS (Apple silicon has no GPU hot-plug; the Metal path enumerates once)' }], {}); return; } + // MF-3: a second card appears in the enumeration (plugged in, or driven after a driver install): the hot-plug pass + // starts its worker with no tap; it leaves (the tool still answers, with one device): its row is marked removed + // and its worker stops; it comes back: revived in its slot, mining again. An EMPTY list is not used: the engine + // reads a tool that lists nothing as "did not answer" and removes no card on it (the right call for a driver crash). + await until(mining, 300000, 'mining on the first card'); + writeFileSync(CTL, 'ok\ndevices 2\n'); log('fake worker: a second device appears'); + const tAppear = Date.now(); + const listed = await until((st) => fakes(st).length >= 2, 200000, 'the second card to be listed by the hot-plug pass'); + const tListed = Date.now(); + const second = await until((st) => fakes(st).length >= 2 && fakes(st)[1].state === 'mining' && fakes(st)[1].hash_now > 0, 300000, 'the second card to mine with no tap'); + const tSecond = Date.now(); + writeFileSync(CTL, 'ok\ndevices 1\n'); log('fake worker: the second device leaves'); + const tGone = Date.now(); + const removed = await until((st) => fakes(st).length >= 2 && (fakes(st)[1].state === 'removed' || fakes(st)[1].removed === true), 200000, 'the hot-plug pass to mark the second card removed'); + const tMarked = Date.now(); + const stR = await poll(); + const firstStill = stR && fakes(stR)[0] && fakes(stR)[0].state === 'mining' && fakes(stR)[0].hash_now > 0; + writeFileSync(CTL, 'ok\ndevices 2\n'); log('fake worker: the second device is back'); + const tBack = Date.now(); + const revived = await until((st) => fakes(st).length >= 2 && fakes(st)[1].state === 'mining' && fakes(st)[1].hash_now > 0, 300000, 'the second card to mine again with no tap'); + const tRevived = Date.now(); + verdict('card-appears', [ + { ok: !!listed, what: `the hot-plug pass listed the card that appeared (${s(tListed - tAppear)})` }, + { ok: !!second, what: `its worker started with no tap (${s(tSecond - tAppear)} after it appeared)` }, + { ok: !!removed, what: `the card that left was marked removed (${s(tMarked - tGone)})` }, + { ok: !!firstStill, what: 'the other card kept mining through it' }, + { ok: !!revived, what: `the card that came back mined again with no tap (${s(tRevived - tBack)})` }, + ], { appear_to_listed: s(tListed - tAppear), appear_to_mining: s(tSecond - tAppear), gone_to_marked: s(tMarked - tGone), back_to_mining: s(tRevived - tBack) }); + // leave one device for the steps after this one + writeFileSync(CTL, 'ok\ndevices 1\n'); + await until((st) => fakes(st).length >= 2 && fakes(st)[1].state === 'removed', 150000, 'the second card gone again'); +}; + +S['own-restart'] = async () => { + setMode('ok'); + const st0 = await until(mining, 300000, 'first mining'); + const pid0 = card(st0).pid, r0 = card(st0).restarts; + const tInject = Date.now(); + setMode('fast'); + const faulted = await until((st, c) => c.faults >= 1, 90000, 'the worker fault to reach the card'); + const tFault = Date.now(); + setMode('ok'); + const back = await until((st, c) => mining(st, c) && c.faults >= 1, 90000, 'mining again after the worker restart'); + const tBack = Date.now(); + await steady(15000); + const st1 = await poll(); + verdict('own-restart', [ + { ok: !!st0, what: 'the card mined with a rate above 0 on the fake worker' }, + { ok: !!faulted && /worker fault|killed by a guard/.test(card(faulted).message || ''), what: `the card showed the worker fault (${JSON.stringify(card(faulted || {}).message)})` }, + { ok: !!back, what: 'mining resumed after the miner restarted its own worker' }, + { ok: st1 && card(st1).pid === pid0 && card(st1).restarts === r0, what: `the app did not restart the miner (pid ${pid0} -> ${card(st1 || {}).pid}, app restarts ${r0} -> ${card(st1 || {}).restarts})` }, + ], { inject_to_fault_on_card: s(tFault - tInject), fault_to_mining_again: s(tBack - tFault) }); +}; + +S['zero-ladder'] = async () => { + setMode('ok'); + const st0 = await until(mining, 120000, 'mining'); + const r0 = card(st0).restarts; + const tInject = Date.now(); + setMode('zero'); + // three rungs: the gaps between consecutive watchdog restarts must grow 10, 30, 120 s (plus the 60 s rule each time) + const marks = []; + let lastR = r0; + const end = Date.now() + 600000; + let faultedSeen = false, onceAlready = false; + while (Date.now() < end && marks.length < 3) { + const st = await poll(); if (!st) break; + const c = card(st); + if (c.state === 'faulted') faultedSeen = true; + if (/restarted once already/.test(c.message || '')) onceAlready = true; + if ((c.restarts || 0) > lastR) { lastR = c.restarts; marks.push({ t: Date.now(), wait: c.restart_in_s, msg: c.message }); log(`rung ${marks.length}: restart_in_s ${c.restart_in_s} "${c.message}"`); } + await sleep(500); + } + setMode('ok'); + const back = await until(mining, 400000, 'mining again on its own after the worker is healthy'); + const tBack = Date.now(); + const waits = marks.map(m => m.wait); + const ladderOk = waits.length === 3 && waits[0] >= 8 && waits[0] <= 10 && waits[1] >= 28 && waits[1] <= 30 && waits[2] >= 118 && waits[2] <= 120; + verdict('zero-ladder', [ + { ok: marks.length === 3, what: `three watchdog restarts observed (${marks.length})` }, + { ok: ladderOk, what: `the restart delays follow the ladder 10, 30, 120 s (saw ${waits.join(', ')})` }, + { ok: marks.every(m => /hash rate 0 for 60 s/.test(m.msg || '')), what: 'the reason stayed on the card in plain words' }, + { ok: !faultedSeen && !onceAlready, what: 'no "faulted" state and no "restarted once already" words' }, + { ok: !!back, what: 'mining resumed on its own once the worker was healthy' }, + ], { zero_to_rung1: marks[0] ? s(marks[0].t - tInject) : 'n/a', rung_gaps: marks.slice(1).map((m, i) => s(m.t - marks[i].t)).join(', '), healthy_to_mining: s(tBack - (marks[2] ? marks[2].t : tInject)) }); +}; + +S['no-status'] = async () => { + setMode('ok'); + const st0 = await until(mining, 400000, 'mining'); + const pid0 = card(st0).pid, r0 = card(st0).restarts; + const tInject = Date.now(); + process.kill(pid0, 'SIGSTOP'); + log(`SIGSTOP miner pid ${pid0}`); + const rs = await until((st, c) => c.restarts > r0 || /no status line/.test(c.message || ''), 150000, 'the watchdog restart'); + const tRestart = Date.now(); + const back = await until((st, c) => mining(st, c) && c.pid !== pid0 && c.pid > 0, 400000, 'mining on the restarted miner'); + const tBack = Date.now(); + let gone = false; try { process.kill(pid0, 0); } catch { gone = true; } + if (!gone) { try { process.kill(pid0, 'SIGKILL'); } catch {} } + verdict('no-status', [ + { ok: !!rs && /no status line/.test(card(rs).message || ''), what: `the watchdog restarted the miner for missing status lines (${JSON.stringify(card(rs || {}).message)})` }, + { ok: rs && tRestart - tInject >= 85000 && tRestart - tInject <= 130000, what: `between 85 and 130 s after the miner went quiet (${s(tRestart - tInject)})` }, + { ok: !!back, what: 'mining resumed on a new miner process' }, + { ok: gone, what: 'the stopped miner process was killed' }, + ], { quiet_to_restart: s(tRestart - tInject), restart_to_mining: s(tBack - tRestart) }); +}; + S['card-appears'] = async () => { if (MAC) { verdict('card-appears', [{ ok: true, what: 'skipped on macOS (Apple silicon has no GPU hot-plug; the Metal path enumerates once)' }], {}); return; } // the card leaves the enumeration (unplugged, or a problem code: the hot-plug pass marks it removed and stops its