Merge remote-tracking branch 'origin/miner-reliability-21' into release-0.3.21
This commit is contained in:
commit
03608266ce
2 changed files with 28 additions and 2 deletions
|
|
@ -68,6 +68,25 @@ the stand-in's listing fixed at ecd79874): every class's step is green.
|
|||
|
||||
FAULT lines the engine posted to the intake in run 4: 13 (watchdog, worker-fault, node-exit, orphan-miner classes).
|
||||
|
||||
Injector on 0.3.21's own binaries (7 October 2026 15:44Z to 16:11Z, a one-shot RunPod 3070: app 0.3.21 sha256 786c3d37
|
||||
built on igneum-build-2 from release-0.3.21 8ed08dcf, node c4459193 sha256 48acf4aa with N13 and the shutdown watchdog,
|
||||
the 0.3.17-line igneumd 5899f603 as the previous release for `--node-old`): nine steps, eight PASS in the run and the
|
||||
ninth (catch-up) on its fresh-datadir rerun (CATCHUP021).
|
||||
|
||||
| Step | Class | Result on 0.3.21 | Seconds |
|
||||
|---|---|---|---|
|
||||
| kept-datadir | MF-8 | the previous release's node wrote 151 blocks and stopped in 1.0 s (MF-9's sub-second shutdown, read on the OLD node); the new node opened the kept datadir, read synced, one start, no exit line | new node synced 4.1 after the engine started |
|
||||
| catch-up | MF-1, MF-2 | in the run the datadir carried records from the first second (kept), so the worker started at once, which is right; the fresh-datadir rerun is the class's own read: CATCHUP021 | |
|
||||
| card-appears | MF-3 | the card that appeared was listed and mined with no tap; the one that left was marked removed; back, it mined again | listed 58.2, mining 69.2; removed 49.1 after it left; mining again 72.2 after it came back |
|
||||
| own-restart | (the miner's own) | the fault on the card, no app restart | 1.0 / 3.0 |
|
||||
| zero-ladder | MF-2 | rungs 10, 30, 120 s, the reason on the row, mining back on its own | zero to rung 1 76.8; gaps 81.3, 101.9; healthy to mining 129.8 |
|
||||
| no-status | MF-2 | the stopped miner restarted at 90 s, mining on a new process | quiet to restart 93.2; restart to mining 310.8 (the carried 300 s rung) |
|
||||
| node-silent | node watchdog | the stopped node restarted by the app, synced, mining again | quiet to restart 151.4; restart to synced 3.0; quiet to mining 170.4 |
|
||||
| one-card-fails | MF-4 | two healthy cards mined, the third held 30 min with the self-test reason, 2 exports with 2 reuses | listed 47.1; two mining 57.1 |
|
||||
| orphan-miner | MF-7 | a stray miner killed by the sweep, the engine's own left alone | inject to kill 14.0 |
|
||||
|
||||
FAULT lines posted in the 0.3.21 run: 9.
|
||||
|
||||
Which side: MF-1, MF-2, MF-3, MF-4, MF-6, MF-7 and the cards kind are app-side (0.3.20's app). MF-5 is both: the
|
||||
miner (the fork branch) prints the fields and caps the identities; the app reads them and keeps the worker. An app
|
||||
without the new miner still gets MF-5's app half (a template timeout line is the heartbeat; the zero-rate clock holds).
|
||||
|
|
|
|||
|
|
@ -195,10 +195,17 @@ S['catch-up'] = async () => {
|
|||
const tMining = Date.now();
|
||||
const readyLines = engineLogLines(/node readiness: the execution layer reports an executed tip/);
|
||||
const gatedCalls = engineLogLines(/holds no record yet/);
|
||||
// after the kept-datadir pre-phase the datadir already holds executed blocks: the record exists from the first second
|
||||
// and the worker may start at once; the rule then reads as "the readiness line came before the first worker start"
|
||||
const readyLine = engineLogLines(/node readiness: the execution layer reports an executed tip/)[0];
|
||||
const startLine = engineLogLines(/: worker starting \(pid/)[0];
|
||||
const order = readyLine && startLine ? (readyLine.split(' ')[0] <= startLine.split(' ')[0]) : false;
|
||||
verdict('catch-up', [
|
||||
{ ok: !!synced, what: 'the private node read synced' },
|
||||
{ ok: !workerStarted, what: 'no worker started while the execution layer held no record (120 s)' },
|
||||
{ ok: waitedForTip, what: 'the card said it waits for the executed tip' },
|
||||
keptPre
|
||||
? { ok: order, what: 'the datadir carried records (kept): the readiness line preceded the first worker start' }
|
||||
: { ok: !workerStarted, what: 'no worker started while the execution layer held no record (120 s)' },
|
||||
keptPre ? { ok: true, what: 'the wait for the executed tip does not apply on a kept datadir' } : { ok: waitedForTip, what: 'the card said it waits for the executed tip' },
|
||||
{ ok: nodeRestarts === 0, what: `the node watchdog did not restart the node during the catch-up (restarts ${nodeRestarts})` },
|
||||
{ ok: readyLines.length >= 1, what: 'the engine logged the executed tip when it appeared' },
|
||||
{ ok: !!back, what: 'the worker started on its own once the record existed' },
|
||||
|
|
|
|||
Loading…
Reference in a new issue