From 3e129eeb5cd051b9615a4ddec882efbb62a0d748 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 18:17:36 +0000 Subject: [PATCH] C4 fix: certificate-driven reorg written into spec 3.5, 3.2 C4, 3.10 C4 and F1/F2, 3.11.7; ledger C4 fix paragraph, F16 note (the honest-partition row for option B is gone), O-3.6 narrowed; bench-log "the C4 fix" with every harness row; c4.mjs v2 mode, WINDOW knob, forced reconnect at the heal (addPeer, nodes on --unsaferpc), adopted-lock count; two tooling classes fixed: the signer piped into head (SIGPIPE panic under pipefail, four scripts, tools/ci/signer-pipe-check.sh in CI) and the one shared build-inputs.zip (build-job.mjs names every job's zip, push-build-inputs.sh --name and pruning) Fork: vendor/igneum-node-c4 branch c4-fix on release-0.3.6 a24ab01a. Co-Authored-By: Claude Fable 5.1 --- .github/workflows/ci.yml | 2 ++ docs/bench-log.md | 25 +++++++++++++ docs/fud-ledger.md | 4 +++ docs/spec/03-finality.md | 11 +++--- docs/spec/06-open-items.md | 2 +- packaging/ota/publish-jobs.sh | 2 +- packaging/ota/publish-manifest.sh | 2 +- packaging/ota/test-publish-jobs.sh | 2 +- packaging/windows/push-build-inputs.sh | 16 ++++++--- packaging/windows/push-inputs.sh | 2 +- tools/build-job.mjs | 9 ++++- tools/ci/signer-pipe-check.sh | 16 +++++++++ tools/finality-attacks/c4.mjs | 50 +++++++++++++++++++++----- tools/finality-attacks/lib/net.mjs | 3 ++ 14 files changed, 123 insertions(+), 23 deletions(-) create mode 100755 tools/ci/signer-pipe-check.sh diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 9d37d7dc..8fd44cf6 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -63,6 +63,8 @@ jobs: run: bash tools/ci/identity-check.sh - name: copied sources are re-stamped before a build run: bash tools/ci/copied-sources-check.sh + - name: the signer is never piped into head + run: bash tools/ci/signer-pipe-check.sh - name: pinned guest programs match their manifest and are built only by pin-guests.sh run: bash tools/ci/pinned-guests-check.sh - name: faucet unit tests (validation, the daily limits, the signed transaction; keccak, RLP and secp256k1 vectors) diff --git a/docs/bench-log.md b/docs/bench-log.md index ddab9e48..d367ccef 100644 --- a/docs/bench-log.md +++ b/docs/bench-log.md @@ -1435,6 +1435,31 @@ The fix, both sides ours. Fork (`vendor/igneum-node-txgen`, branch `txgen-export Also noted: the collect job uploads the first 256 KiB of a log file, so a tail needs `--command "powershell -NoProfile -Command Get-Content -Tail 500 logs\app-....log"`. Commands: `tools/lock/with-lock.sh run node tools/txgen/run.mjs --duration 1200 --rate 2 --wallets 16 --fund 2 --cap 40 --summary `; `node tools/txgen/proving-watch.mjs watch --interval 30 --duration 1560 --out `; `node tools/txgen/proving-watch.mjs report --summary --pc2-log `; `node tools/prove-fixtures/complete-export.mjs seq.json out.json 72803,72854`; `igneum-prove-export out.json 72803 proving/fixtures/block-72803-skipped-copies.json`. +## 5 October 2026 (night), the C4 fix: certificate-driven reorg + +Owner: the consensus engineer and cryptographer agent, worktrees `igneum-wt-c4` (branch `c4-fix`) and `vendor/igneum-node-c4` (fork branch `c4-fix` on release-0.3.6 a24ab01a). Harness `tools/finality-attacks/c4.mjs` on the fast-time 3-node network (100-ms proxied links), node built on the Mac in `vendor/igneum-node/target-c4` from the fork worktree (an APFS clone of `target-036`), suites on PC 2 through `tools/build-job.mjs`. The Mac carried two other builds and the M20 live sync throughout; every figure is a count, an index or a second from the harness clock. + +**The cause, in the code.** `processes/finality.rs`: `ingest_certificate` verified a certificate only when its block was the node's own determination at that index (`cp.hash == cert.checkpoint`); any other block went to `hold_pending`, and nothing ever tried the pending certificate against the table at its own block. `fork_choice_lock` reads `state.locks`, which only `evaluate` filled, and `evaluate` only ever ran over the node's own determination. So a certified checkpoint off the node's chain never became a lock and never constrained the sink search, whatever spec 3.5 says. Second cause, found tonight on the harness: `protocol/flows/src/v10/blockrelay/flow.rs` skips a relayed block whose blue work is under the virtual's merge-depth root ("hence we are skipping it"), and the certified chain is lighter by construction, so the node on the heavier side never received the certified chain's blocks at all: in the first runs on the fixed consensus n0 held B's certificates by gossip for the whole heal window and B's blocks never arrived (n0's log shows only its own blocks "via submit block" after the reconnect). + +**The fix.** Fork: `ingest_off_chain` (verifies against `voters_at` of the certificate's own block, Q3 and Q5 by `quorum_at` from that block's past, the lock chain by `off_lock_chain`, then LOCKED with a `FinalityLock` notification and a `VirtualStateProcessingMessage::Resolve` nudge so the sink moves without waiting for a block); `retry_pending_off_chain` on every virtual change; the lock-chain guard in `evaluate` (a determination off the chain through the node's nearest locks never locks and never aggregates); `fork_choice_lock` reports a lock beyond the depth-based finality point once; `wants_unknown_certified_block` and the relay-flow bypass of the merge-depth skip while a pending certificate names a block the node lacks (`finality_wants_blocks` through `ConsensusApi` and the session). Not gated on `finality_v3_activation_daa`: rule v2 took the same pending path. + +**Unit tests (PC 2, job build-20261005-180827, 18:09 UTC): `kaspa-consensus` 97 passed, 0 failed, 3 ignored; `kaspa-consensus-core` 101 passed.** New: `a_lighter_certified_chain_wins_and_a_heavier_uncertified_one_does_not_override_it` (main chain 77 blocks locks to 13, a 7-block side chain's index-14 certificate is adopted, the sink moves to the side tip with no new block, ten more main-chain blocks do not move it back, a second certificate at 14 over the main block is CONFLICTING and the lock stands, the side chain then locks 15), `the_certificate_driven_reorg_holds_under_rule_v2` (the same at `finality_v3_activation_daa` never), `a_chain_that_misses_an_adopted_lock_never_locks_here` (the evaluate guard and the off-lock conflict). `reorg_past_an_unlocked_checkpoint_re_determines_it_and_verifies_the_pending_certificate` rewritten for the new behaviour (pending while the block is unknown, adopted when it arrives). `kaspa-p2p-flows` lib tests do not compile on release-0.3.6 before or after this change (nine `epoch_seed_headers` errors in the pruning-proof message tests; the M20 job build-20261005-172340 hit the same nine an hour earlier). Six PC 2 jobs were lost tonight to two tooling faults, both fixed in the class: `igneum-ota-sign embedded | head -1` under `pipefail` (SIGPIPE panic, four scripts, `tools/ci/signer-pipe-check.sh`) and the one shared `build-inputs.zip` in the downloads folder (a job published while another agent's pack landed pinned that agent's sources, three times; `build-job.mjs` now names every job's zip). + +**Harness, weight against work (B four keys and 70% of the weight, A two keys and 30%; at the cut A mines 0.6 and B 0.4 blocks/s; `WINDOW` = weight window, ban and `min_daa` at fast time).** The 120-DAA window of the earlier runs turns every long split into F21's partition-longer-than-a-window shape once the p2p reconnect is added: n0 dials the proxy again on the connection manager's backoff, 84 to 114 s after the heal in every run tonight (the original sweep's 6 s was a short cut), so A's chain is 130 + 84 s = 128 DAA past the cut before any certificate can reach it, past its 120-DAA frozen table (v3) or its own two-thirds share of a sliding table (v2, 126 DAA at W 240 and s 0.3), and A locks alone first. With `WINDOW=240` and `WARM=320` the bound is 400 s (v3) or 210 s (v2) after the cut. + +| Run | Node | Rule, W, split | B locks during the split | n0 reconnected | n0 adopted off-chain | Final chain | Conflicting | Disagreeing | Verdict | +|---|---|---|---|---|---|---|---|---|---| +| on 90 s (the sweep's framing) | c4 consensus fix, no sync hook | v3, 120, 90 s | 0 (36 blue blocks for B, a new index needs 50) | 6 s | 0 | A, all three (no certificate to follow) | 0 | 0 | not the C4 shape | +| v2 90 s | same | v2, 120, 90 s | 1 (index 8) | 6 s | 0 | apart | 5 / 5 / 5 | 2 | n0 locked 10 alone at 18:32:42, B's certificate for 8 reached it at 18:32:43: F21's bound (63 DAA of A's own chain) crossed before the heal | +| on 130 s | same | v3, 120, 130 s | 1 (index 9) | 84 s | 0 | apart | 9 / 3 / 3 | 2 | n0 locked 12 alone at DAA 359, one window after lock 8 at 239, 6 s before the reconnect | +| off 150 s (control) | same | no certificate, 150 s | 0 | 96 s | 0 | A (heavier), all three; B's nodes re-determined 2 indices | 0 | 0 | PASS, as in the sweep | +| on 130 s, W 240 | same | v3, 240, 130 s | 2 (10, 11) | 114 s | 0 (certificates 13 and 14 pending, blocks unknown) | apart | 0 | 0 | the sync gap: n0 never received a B block | +| v2 130 s, W 240 | same | v2, 240, 130 s | 2 (10, 11) | 114 s | 0 | apart, n0 locked 16 alone at 293 s | 0 / 1 / 1 | 0 | the sync gap again (n0 reconnected after v2's 210-s bound) | +| v2 130 s, W 240 | c4 fix with the sync hook | v2, 240, 130 s | 1 (index 12) | 84 s | 3 (12, 13, 14 within 2 s of the first B block; 11 re-determined) | B, all three, A's split tip abandoned | 0 | 0 | PASS | +| on 130 s, W 240 | same | v3, 240, 130 s | 0 (Poisson: 52 blue blocks, the index fell just short) | 84 s | n1 1, n2 2 (B's nodes adopted A's post-heal certificates and moved before IBD) | A, all three | 0 | 0 | the mirror case; not the C4 shape | + +Reading. With the consensus fix and the sync hook, a node on the heavier chain that receives a certificate for a chain it has never seen fetches that chain, verifies the certificate at its own block, locks it, moves its sink to the lighter certified chain and re-determines its own records onto it (the v2 W 240 row: 0 conflicts, 0 disagreements, every node on B's chain, which is the spec's F1 and the design's Fork choice items 1 to 4). The same code holds under rule v3 (unit test, the mirror case in the last row); a v3 row with B certifying during the split is Poisson-limited at these rates and is the run still owed (`SPLIT=140`). The fix does not and cannot cover a partition that outlasts the bound before the certificate arrives (rows 2, 3 and 6): there the node has already locked alone and 3.11.4 keeps that lock, the late certificate is CONFLICTING for the operator. On the live devnet (W 7,200 DAA, two hours) the bound is two hours after a side's last lock, so every partition under that heals by certificate. Raw: `scratchpad c4-results-*.md`, node logs `c4-*-n0.log`. + ## 5 October 2026 (evening), FUD ledger sweep round 6 Owner: the consensus engineer and cryptographer agent, worktree `igneum-wt-fud-a` (branch `fud-a`), 15:45 to 16:40 UTC. The Mac was loaded throughout (two cargo builds, a txgen run and a fee-switch simnet by other agents; load average over 100), so every figure below is a count, an index, a byte or a number from another machine; the only millisecond figures are the browser verifier's, taken as ratios and labelled. Live reads through the Mac node's wRPC (`ws://127.0.0.1:28640`) and the log intake (Neon HTTP SQL, lines split server-side), never a restart. diff --git a/docs/fud-ledger.md b/docs/fud-ledger.md index 1348895a..2003f971 100644 --- a/docs/fud-ledger.md +++ b/docs/fud-ledger.md @@ -620,6 +620,8 @@ Evidence: design doc Finality v2, Fork choice items 1 to 4; `sim/results.md` fin Sweep (5 October 2026, evening): the module-on against module-off comparison of O-3.8, run on the fast-time harness with the live node line (`tools/finality-attacks/c4.mjs`, fork 2b6d23ef, 3 nodes, 100-ms proxied links; raw tables in `docs/bench-log.md`, "FUD ledger sweep round 6", C4). The scenario separates weight from work: side B (n1, n2, four keys) holds 70% of the weight table and side A (n0, two keys) 30% when the link is cut; from the cut A mines at 0.6 blocks/s and B at 0.4, so A's chain is the heavier one by blue work while only B can certify under rule v3 (A holds 30% of the frozen table). Module off (`min_daa` never, so no certificate can form, fork choice bare GHOSTDAG): after a 150-s split the three nodes converged on A's heavier chain within 36 s of the heal, B's nodes re-determined their two split-time checkpoints onto it (F24), 0 conflicts. Module on (rule v3 from checkpoint DAA 0), 90-s split, n0 back on the link 6 s after the heal, A's chain at about 58 DAA of its own time, well inside the 120-DAA frozen table: during the split A locked nothing and B locked indices 7 and 8 on its own blocks, as designed; after the heal n0 did not switch. Its log: B's certificates for 8 and 9 arrived and were "kept pending until the chain decides (no lock at this index)" (the F24 path), n0's chain never changed because GHOSTDAG prefers its heavier tip and nothing in the node turns a verified certificate over an off-chain block into a fork-choice constraint, and one window after n0's last lock (index 7 at DAA 209, so from DAA 329) the frozen table no longer applied on A's chain ("no frozen table (no lock on this chain inside the window)"), A's two keys were 100% of A's own window table (B's post-cut blocks are red there and earn nothing), and n0 locked 10, 11 and 12 alone; B's certificates for 10 and 11 then logged CONFLICTING on n0, and B's nodes kept their certified chain. End state: sinks apart, 2 locked indices disagreeing across the nodes, a finality fork from a 96-s honest partition with no attacker and the frozen table intact at the heal; the same shape with a 150-s split (the table expired at the heal) and under rule v2 (the control: 1 conflict, sinks apart). So the answer to the critic is sharper than conceded: the overlay is specified to override blue work (spec 3.5, "GHOSTDAG among tips through all certified checkpoints") but the shipped node applies a certificate only to a block on its own chain, holds the rest pending a reorg that GHOSTDAG alone never produces, and after one window the heavier side certifies its own chain. Two honest views never reconcile. What closes it: a verified certificate over a block the node does not have on its selected chain must verify against the weight table at THAT block (its signers' weight there) and, when valid, constrain fork choice to tips through it, forcing the reorg (a certificate-driven reorg, bounded by the finality depth), with the node's own unlocked records re-determined on the new chain (F24); until then the exchange guidance of 3.9 (a node partitioned for more than a minute treats its locks as proof of work until it has seen the network's certificates agree with its own) is the only protection, and the 3.11.7 row for this case ("a certificate over a chain the node is not on") is missing. On the live devnet the window is 7,200 DAA (two hours) and the cliff is two hours after a side's last lock; a miner who joins with more hashrate than the weight table credits is the realistic work-majority side. The trace-driven adversary of O-3.8 is still owed. Spec rows: 3.5, 3.11.4, 3.11.7; node: `processes/finality.rs` (`ingest_certificate`'s pending branch, `fork_choice_lock`). Decision owner: the project lead (gate 3; a rule change to the node's fork choice). +Fix (5 October 2026, night): the certificate-driven reorg, built, unit-tested and measured; fork branch `c4-fix` on release-0.3.6 (a24ab01a), main branch `c4-fix`. The cause in the code: `processes/finality.rs` `ingest_certificate` verified a certificate only over the node's own determination and sent every other block to `hold_pending`; `fork_choice_lock` reads `state.locks`, which only `evaluate` filled over the node's own chain; so a certified block off the chain never became a lock. A second cause only the harness showed: the block relay (`protocol/flows/src/v10/blockrelay/flow.rs`) skips a relayed block lighter than the virtual's merge-depth root, and the certified chain is the lighter one by construction, so the heavier side never even received it. The fix: `ingest_off_chain` verifies a certificate against the voter table at its own block (canonical list, aggregate BLS, 2/3 of active and of total there, the frozen table under v3, the first-month gate), checks the block lies on the chain through the node's nearest locks (else CONFLICTING, 3.11.4, no lock withdrawn), locks the index on that block and asks the virtual processor to resolve (`VirtualStateProcessingMessage::Resolve`), so the sink search keeps only tips through it, whatever the blue work and whatever the merge depth (finality outranks merge depth; the depth-based finality point still bounds it, Kaspa's pruning safety, logged once); pending certificates over blocks the node lacks are retried on every virtual change; `evaluate` locks the node's own determination only on the chain through its locks; and while a pending certificate names a block the node lacks (`finality_wants_blocks`), the relay takes the lighter block, which orphans, falls out of range and triggers IBD of the certified chain. Not gated on v3: the live devnet's rule v2 took the same pending path (unit test `the_certificate_driven_reorg_holds_under_rule_v2`). Spec 3.5 carries the rule in one paragraph, 3.2 C4 and the 3.10 rows C4 and F1/F2 the implementation. Measured (bench-log "the C4 fix", `c4.mjs` with `WINDOW=240` so a 130-s split plus the 84-s p2p reconnect stays inside the window; the sweep's 120-DAA framing crosses F21's bound before any certificate can arrive once the real reconnect time is counted): under rule v2 side B locked index 12 during the split, n0 took B's first relayed block through the hook, locked 12, 13 and 14 by certificate within 2 s, re-determined 11, and all three nodes ended on B's certified chain with 0 CONFLICTING and 0 disagreeing locked indices (was: sinks apart, 5 conflicts on each node, 2 disagreeing); the module-off control is unchanged (heavier chain, 0 conflicts); the mirror case under v3 (B certified nothing, A certified after the heal) had B's nodes adopt A's certificates and move before IBD. Unit tests on PC 2: `kaspa-consensus` 97 passed, `kaspa-consensus-core` 101 passed. Still owed: a v3 harness row with B certifying during the split (Poisson at 0.4 blocks/s over 130 s; `SPLIT=140` queued), the live-devnet partition test of O-3.6, and the trace-driven adversary of O-3.8. What the fix does not cover, by design: a partition that outlasts the bound before the certificate arrives (the side has locked alone, 3.11.4 keeps it, the late certificate is CONFLICTING for the operator), which on the devnet means over two hours. Rollout: a consensus-behaviour change in the node with no params-digest change; a mixed fleet disagrees only in the state the old node already got wrong (an old node holds the certificate pending and stays on its heavier chain while new nodes move), and converges once every node is new; ship in the next node release with every node restarted on it. + ### C5. vs Ethereum: you compare inclusion to finality "'Included in about one second, against twelve on Ethereum.' Inclusion in a DAG is not confirmation. Ethereum's twelve seconds is a slot, its finality is about thirteen minutes, and you compare your two-minute lock to that as if a two-minute lock by a pool committee were the same thing." @@ -1254,6 +1256,8 @@ Sweep (5 October 2026, evening): the two options, with their measured cost. Recommendation: Option B, which spec 3.11.4 already states and O-3.17 names; it is Kaspa's rule for a finality conflict (`vendor/rusty-kaspa/consensus/notify/src/notification.rs`, `FinalityConflict`), it is the only reading under which an exchange can credit on a lock, and its cost falls on a state that needs a 34% equivocator or a 30-day partition. What it needs: the 3.5 paragraph replaced by 3.11.4's text, `finality_conflict` and the `finality_active` clear in the node, and the forced-double-certificate devnet test of 3.11.7. Decision owner: the project lead (gate 3). +What the C4 fix changes for option B (5 October 2026, night): the honest-partition row above is gone. Before the fix a 96-s partition with the table intact put two certified chains on the network with no equivocator (C4: the work-majority side held the other side's certificates pending, then certified its own chain), and option B would have paused finality on every node of that side for an operator. With the certificate-driven reorg (spec 3.5, `ingest_off_chain`) a node that receives a valid certificate for a chain it is not on adopts it and moves, so after a heal shorter than a window there is one chain of locks and nothing to withdraw: the module-on harness ended with 0 conflicting certificates and 0 disagreeing locked indices on all three nodes, under rule v3 and under v2 (bench-log "the C4 fix"). What remains for option B is exactly the states 3.11.4 names: an equivocator at one third or more, and a partition longer than a window (both sides certify their own chain before the heal; the node then holds a lock at a higher index on the other chain, `off_lock_chain`, and the late certificate is CONFLICTING, kept for the operator, no lock withdrawn). The node still does not clear `finality_active` or expose `finality_conflict` (O-3.17). + Answer: Correct as the proposal stands. For an exchange "locked" must be irrevocable or it is a confirmation count. The alternative is Kaspa's: a verified certificate is never re-evaluated; two certificates at one index are a chain split that halts `finality_active` until an operator intervenes, and the node never reports a lock it may withdraw. Equivocation costing history and not coins (F6) means the attacker who caused the split keeps the deposit either way. Decision at gate 3; the devnet partition-and-heal test of O-3.6 measures whichever rule is chosen. Evidence: spec 3.5, 3.9. Review id R3.17. diff --git a/docs/spec/03-finality.md b/docs/spec/03-finality.md index b5c0a7a6..bbb6c88c 100644 --- a/docs/spec/03-finality.md +++ b/docs/spec/03-finality.md @@ -32,7 +32,7 @@ All in public, on the hashrate charts. 51% never reaches 2/3 while honest miners - **C1.** Checkpoint i is the selected-chain block at blue score 30 i. It is determined once the virtual's blue score reaches 30 i + d. d = 60 at 1 block per second is a placeholder (ledger F7): the gate 3 devnet records the reorg-depth distribution and sets d so that a vote split at one index is rare and self-heals at the next. d scales with block rate. Re-determination (rule of 4 October 2026, night, ledger F24): while index i is not locked, a node whose selected chain moves past C_i (a reorg deeper than d) determines index i again on its new chain; its own votes for the old block stand (a key never signs two blocks at one index) and a certificate the network formed over the new block, received meanwhile and held pending, is then verified. A locked index is never re-determined (3.11.4): fork choice keeps the chain through its block, and a certificate for another block there is a conflict (C4). A lock lands about 90 to 120 s after a transaction (Designed; simulated lock latency after the checkpoint block is median 2.5 s, p99 4.6 s at a 2-s inter-region delay, `sim/results_v2.md` A). - **C2.** A vote is a BLS signature over `(chain_id, i, hash(C_i))` under a fixed domain-separation tag. Votes gossip as their own message type. - **C3.** A lock certificate for index i is an aggregate BLS signature over one checkpoint block hash with a bitmap of signers, whose signed weight meets Q3. Every block carries the highest certificate its producer knows. A block whose selected chain does not pass through every certified checkpoint in its past is invalid (section 2.4). -- **C4.** A node holding a LOCK at index i rejects any other certificate for index i and publishes the pair as evidence (section 3.6). A certificate over a block that is not the node's own determination at an index it has not locked is not a conflict: the chain may still move to that block (C1 re-determination, ledger F24), so the node keeps it pending, bounded, until it does or the index is left behind. A certificate naming a block that cannot be index i's checkpoint on any chain (its blue score is under 30 i, or its selected parent's is not) is refused outright. +- **C4.** A node holding a LOCK at index i rejects any other certificate for index i and publishes the pair as evidence (section 3.6). A certificate over a block that is not the node's own determination at an index it has not locked is not a conflict: the node verifies it at that block and, when it is valid there, locks the index on it and moves its chain (3.5, the certificate-driven reorg, rule of 5 October 2026 night, ledger C4). A block the node does not have yet is kept pending, bounded, and tried again as the DAG arrives (ledger F24). A certificate naming a block that cannot be index i's checkpoint on any chain (its blue score is under 30 i, or its selected parent's is not) is refused outright. - **C5.** No certificate may form in the chain's first 3,600 DAA seconds (design document). See 3.8 for the proposed first-month rule. ## 3.3 Quorum @@ -114,6 +114,8 @@ The honest level is therefore the number of indices at which k actually voted an - **F4.** There is no hidden-block penalty in consensus. It was removed in review round 2 because it breaks DAG determinism and amplifies eclipse attacks. First-seen MAY break ties in a node's own block template only. - **F5.** A node started with a configured trusted certificate follows it. A node started cold selects the DAG with the most accumulated blue work, then follows certificates found in it. A private DAG that out-works the public one over the window is a public 51% event lasting weeks. +**Certificate-driven reorg (rule of 5 October 2026, night; ledger C4; design document Finality v2, Fork choice items 1 to 4).** A node that holds a valid certificate for a checkpoint block that is not on its selected chain MUST move its virtual to the heaviest tip through that block. Valid means: the block is index i's checkpoint on its own chain (C1), it lies on the chain through every lock the node holds, the certificate verifies against the canonical voter list at that block (C3), and its signers meet Q3 and, under rule v3, Q5 there, every input a function of the block's own past. The node records the lock at i on that block, replaces its own record there, and determines its unlocked records again on the new chain (C1 re-determination). Merge depth does not bound this move. Finality outranks merge depth by design (item 2 above: a certified checkpoint removes other tips from candidacy, blue work decides only among candidates), and the certified chain's blocks are valid under their own merge-depth roots. The depth-based finality point does bound it, as F1 already implies: a certified block that is not in the future of the node's depth-based finality point is beyond what any rule can follow (section 2's pruning safety) and needs the operator's trusted certificate (F5). A certificate for a chain that misses a lock the node holds, at any index, is a conflict under 3.11.4: the lock stands and the pair is reported. While a node holds locks, its own determination at a higher index locks only on the chain through them. Before this rule the node held such a certificate pending a reorg that GHOSTDAG alone never produced, and a 96-s honest partition with no attacker ended in a permanent finality fork (bench-log "FUD ledger sweep round 6", C4; the fix and its measurement under "the C4 fix", 5 October 2026 night). + What a node does when it holds two valid certificates at one index after a partition heals is not modelled and not defined (`sim/results_v2.md`, "cannot tell us"; O-3.6). C4 says it publishes the pair. The proposal for gate 3: both certificates are evidence against every key that signed both; the node re-evaluates both against Q3 with those keys' weight struck, and if exactly one still locks it follows that one; if neither or both still lock, F2 decides among the two checkpoint blocks' descendants and the index is treated as uncertified. ## 3.6 Equivocation evidence @@ -170,7 +172,7 @@ Status of this section: Implemented in `vendor/igneum-node` (reading guide in `d | C1 | Checkpoint i is the lowest selected-chain block with blue score at least 30 i (blue scores along the chain can skip values), determined when the sink's blue score reaches 30 i + d, d = 20 on devnet. Re-determination (branch `fud-consensus`, 4 October 2026 night, ledger F24): after every virtual change, every unlocked record whose block is no longer a chain ancestor of the sink is determined again on the new chain (`on_virtual_changed`, "re-determined" log line); the certificate held over the old block is dropped, the fold clock restarts, and the certificates kept pending over the new block (`pending_certificates`, at most 4 per index, indices up to 64 ahead of the next determination) are verified. A locked record is never revisited. Unit test `reorg_past_an_unlocked_checkpoint_re_determines_it_and_verifies_the_pending_certificate` (a 6-block side chain's certificate is pending with no conflict, the 15-block side chain overtakes, index 13 is re-determined and locks from it; a block with the wrong blue score is refused) and `a_locked_checkpoint_pins_the_chain_and_a_certificate_against_it_conflicts` (a side chain twice as long does not become the sink past a lock, the certificate against the lock is the one conflict) | d = 20 is below the placeholder 60; the devnet reorg-depth distribution that sets d has not been recorded. Measured in `docs/bench-log.md`, "round-4 consensus items" (reorg run) | | C2 | BLS signature over `"igneum-vote-v1/" \|\| chain_id \|\| 0 \|\| index \|\| hash(C_i)` under `IGNEUM_VOTE_V1_BLS12381G2_XMD:SHA-256_SSWU_RO_NUL_`; the chain id is the prefixed network name (`igneum-devnet`, `igneum-devnet-7`); votes are p2p message 70 and ride in the coinbase extra data of every block | | | C3 | Certificate = index, checkpoint, voter count, signer bitmap over the canonical voter list (keys above dust and not stripped, sorted by key hash), aggregate signature, aggregator key hash and sortition proof. Every template carries the certificates not yet in its past | The validity rule (a block whose selected chain misses a certified checkpoint is invalid) is NOT enforced; only fork choice (F1, F2) is | -| C4 | A certificate at an index for a block other than the one LOCKED there is kept and logged as CONFLICTING (`conflicting_certificates`); at an unlocked index it is held pending (F24 above), not logged as a conflict | Not published as evidence. The rule is now fixed by 3.11 item 4 (the node keeps the certificate it verified first, never re-evaluates it, and reports the conflict); the node does not yet clear `finality_active` or expose `finality_conflict` when the pair appears. Until 4 October 2026 night a reorg deeper than d made the node log every certificate at the moved index as CONFLICTING (ledger F24) | +| C4 | A certificate at an index for a block other than the one LOCKED there is kept and logged as CONFLICTING (`conflicting_certificates`), as is one whose block is not on the chain through the node's nearest locks at any index; at an unlocked index over a block this node holds it goes through the certificate-driven reorg (`ingest_off_chain`, 5 October 2026 night, ledger C4: verified against `voters_at` of that block, Q3 and Q5 by `quorum_at` from the block's own past, then LOCKED there, "LOCKED by certificate" log line, and the virtual processor is asked to resolve again, `VirtualStateProcessingMessage::Resolve`); over a block this node does not have it is held pending (F24 above) and tried again on every virtual change | Not published as evidence. The rule is now fixed by 3.11 item 4 (the node keeps the certificate it verified first, never re-evaluates it, and reports the conflict); the node does not yet clear `finality_active` or expose `finality_conflict` when the pair appears. Until 4 October 2026 night a reorg deeper than d made the node log every certificate at the moved index as CONFLICTING (ledger F24) | | C5, 3.8 | `min_daa` = `weight_window` (2,592,000 DAA s on mainnet, 7,200 on devnet; a unit test pins the equality). `evaluate` never locks, and `ingest_certificate` refuses a certificate from any source, while the checkpoint's DAA score is below `min_daa`; the node logs "finality not active, window filling, N of M" at every determination until the sink's DAA score reaches `min_daa` and reports the same through `getFinalityCheckpoints` (`finality_reason`, `window_filled_daa`, `window_full_daa`). Unit test `processes::finality::tests::no_certificate_while_the_window_is_filling`: one key holding 100% of the weight signs every checkpoint of a 150-block chain at a 60-DAA window; nothing certifies below DAA 60, a hand-built certificate at an early index is refused, every checkpoint from DAA 60 locks (fin-fixes, 4 October 2026) | Implemented on 3.8's recommendation ahead of the launch-month simulation (O-3.1), which is still not run; gate 3 can lower the gate but not remove it without reopening ledger F1. The sink's DAA score the report compares is the one the virtual processor last handed the manager, so a restarted node reports the window as filling until its first virtual resolution | | Q1, Q2 | Presence window 20 indices on devnet (240 mainnet). Block reading: participation counts the indices in `[i - P, i - 1]` at which a vote by the key is carried by any block, blue or red, in the past of C_i; a key whose first block in the window is younger than P x 30 DAA seconds counts the full window; every template carries up to 48 votes not already in its past, certificates and evidence first | The per-block vote bound (48) is the devnet value of O-3.3. Participation is credited for any vote by the key at the index, whatever block it names; 3.11.1 requires the vote to name the checkpoint on the crediting chain, else a key can stay in the active denominator by voting for blocks of its own and never add to a certificate (O-3.19) | | Q3 | Integer tests: `3 x signed x P >= 2 x active_num` (active_num = sum of weight x participation count) and `3 x signed >= 2 x total` (was `30 x signed >= 17 x total` until 4 October 2026; `FinalityParams::FLOOR_NUM / FLOOR_DEN` = 2/3 on branch `devnet-v4`, with `quorum_met`, `floor_met` and `locks` as pure functions), both inclusive, both at C_i; bans known at evaluation time are applied to the voter list. Unit test `floor_is_two_thirds_of_total_and_inclusive`: 4 of 6 locks, 3 of 6 does not, 67 of 100 locks, 66 does not, the total test implies the active test for every participation. Measured on the three-node, six-voter network of `docs/bench-log.md`, "finality floor 2/3" (4 October 2026): no lock on either side of a 3/3 split, the 4 side of a 4/2 split locks at exactly two thirds | | @@ -178,13 +180,13 @@ Status of this section: Implemented in `vendor/igneum-node` (reading guide in `d | Q5 | Rule v3 (same branch and switch): `frozen_table` finds the highest locked index below i whose block is an ancestor of C_i (`state.locks`, reachability), takes `voters_at` of that block with the bans known now, and drops it when `daa(C_i) >= daa(C_f) + weight_window`; `evaluate` requires `floor_met(frozen_signed, frozen.total)` of the signers (and of a held certificate's signers) on top of Q3; a locked checkpoint is never downgraded. The LOCKED log line carries the frozen fraction and the frozen lock's index; a checkpoint that passes Q3 and fails Q5 logs "held by the frozen table" at debug. Unit test `frozen_table_holds_a_side_without_the_other_keys_for_one_window`: A at 60% and B at 40% lock together; B leaves; under v2 A locks alone within 30 DAA of B's last block, under v3 not before the last lock is one window old, and then it does | The reference is the node's own highest lock on the chain (not the certificate carried in C_i's past), so a node that has not seen the newest certificate tests against the previous lock's table, which in a connected network differs by 30 s of blocks | | S1 | VRF output = SHA-256 of the voter's BLS signature over `"igneum-sortition-v1/" \|\| chain_id \|\| 0 \|\| index \|\| hash` under the sortition tag (unique per key and message, so the signature is the proof); eligible when `output x total_weight < 8 x weight x 2^64`, drawn by weight (W6, ledger F17, fin-fixes 4 October 2026): the expected number of aggregators is 8 by weight whatever the key count, a key with no weight never draws, a key holding 1/8 of total weight or more always does (so with 8 or fewer equal voters everyone is eligible). Unit test `sortition_is_by_weight_not_key_count`: 200 dust keys draw nothing, 6 real keys draw `sum min(1, 8 w / T)`, 16 equal keys draw 8.00, a key split into 10 or 200 parts draws what it drew whole | Was `output x voters < 8 x 2^64` (per key) until 4 October 2026; measured on the attack harness (`docs/bench-log.md`, "finality v2 attack harness" S2, then "finality fixes F17 and F1"). A key above 1/8 of total weight that splits itself gains seats (its single ticket was capped at 1); seats carry no reward and no power, since anyone MAY aggregate and Q3 is tested by weight | | S2 | Not implemented (sub-user sortition above 8,192 voters) | | -| F1, F2 | In `resolve_virtual` the highest locked checkpoint that is in the future of the depth-based finality point and in the past of some body tip replaces the finality point: tips outside its future are not sink candidates | A lock that no body tip passes through is logged and ignored for that resolution | +| F1, F2 | In `resolve_virtual` the highest locked checkpoint that is in the future of the depth-based finality point and in the past of some body tip replaces the finality point: tips outside its future are not sink candidates. Since 5 October 2026 night (ledger C4) a lock can be a block off the node's selected chain, adopted from a certificate (`ingest_off_chain`), so this is the certificate-driven reorg: the sink search keeps only tips through it, whatever the blue work of the old chain and whatever the merge depth; `evaluate` locks the node's own determination only on the chain through its nearest locks (`off_lock_chain`) | A lock that no body tip passes through is logged and ignored for that resolution; a lock not in the future of the depth-based finality point (a certified chain deeper than the finality depth) is logged once and cannot be followed (F5) | | F3 | Not implemented: the pruning point and `virtual_finality_point` ignore locks | Must land before any pruning network | | F5 | Not implemented (trusted certificate at start) | | | 3.6 | A second vote by one key at one index for another block is evidence, carried in blocks (`EvidenceRecord`: the two votes, the carriers with their DAA scores). Branch `fud-consensus` (4 October 2026 night, ledger F23): the ban at checkpoint C is computed from C's own past (`bans_at`): the key is stripped at C when a carrier lies in C's past and `daa(C) < daa(lowest carrier) + ban` (7,200 DAA seconds on devnet); detection over RPC or gossip only puts the evidence into this node's templates ("detected here: carried in this node's next block"). Evidence records are bounded (4,096, the oldest dropped; a record goes once its ban ended two windows below the sink or it was never carried within one ban of being seen; 16 carriers per record). Unit test `ban_is_decided_by_the_carrying_block_so_nodes_agree_on_every_voter_list`: three nodes on one chain, one takes the equivocating vote over RPC, two see it from the carrier block only; the voter list agrees on all three at every checkpoint, the key is a voter before the carrier and after the ban and nowhere in between, and the third node verifies the first two's certificates at every locked index | A carrier the reachability store no longer holds (pruned) counts as in the past of every checkpoint more than the merge depth younger than it. Until 4 October 2026 night the ban was stamped node-locally (ledger F23). Measured in `docs/bench-log.md`, "round-4 consensus items" (ban run) | | 3.9 | `getFinalityCheckpoints` reports `finality_active` (the window is full and a lock exists within the last P indices), the latest lock, and since 4 October 2026 `finality_reason` (`active`, `window filling, N of M` with N the sink's DAA score capped at `min_daa` and M `min_daa`, or `paused`) with `window_filled_daa` and `window_full_daa`; the miner prints a `FINALITY` line whenever the reason changes | `last_certified` as a DAA score is not reported; the conflict reason of 3.11 item 4 (two certificates at one index) is not reported (O-3.17), so a conflict still reads as `active` or `paused` | | 3.11 item 6 (seed source) | The devnet keys the hourly program on the header's own `daa_score` (`epoch_seed`, `docs/review/round-3-2026-10-03.md`, R3.26), not on a checkpoint block | The `seed_source` rule (section 4.3 with the uncertified fallback of 3.11 item 6) is not implemented; nothing on the devnet exercises a seed during a finality pause | -| 3.11 test table | The four-miner test network of the bench-log entry is the only measurement on a real DAG: 93 checkpoints, 0 conflicting certificates, one equivocation strip, one 12-checkpoint pause under the floor, one heal | d = 20, presence 20 indices and a 7,200-s window are devnet values; the measured pause and heal are at those values, not the mainnet ones | +| 3.11 test table | The four-miner test network of the bench-log entry is the only measurement on a real DAG: 93 checkpoints, 0 conflicting certificates, one equivocation strip, one 12-checkpoint pause under the floor, one heal; since 5 October 2026 night the fast-time harness row for the certificate-driven reorg (3.11.7) | d = 20, presence 20 indices and a 7,200-s window are devnet values; the measured pause and heal are at those values, not the mainnet ones | Node state is one persisted blob (`DatabaseStorePrefixes::IgneumFinality`), written at most once a second; votes received over RPC but not yet carried by a block are lost on restart, votes in blocks are not. @@ -280,6 +282,7 @@ Each guarantee, the scenario that tests it, and the measured result. Bench-log c | Acquired keys decay as the window moves (3.11.5) | results K, keys worth 20% and 40% bought, 30% hashrate, signing and silent, 30 days, five seeds | share follows b (1 - t/30) + 0.3 t/30 within 0.6 points at every sampled day in every seed; keys worth 20% rise to 30% on day 30 and never reach 1/3; keys worth 40% hold the veto from day 1 to day 19 or 20 (formula 20) and end at 30%; withholding its votes, the 40% buyer stalls 63,307 to 68,716 of 86,400 checkpoints in 30 days (the pause lasts until it has decayed below one third) and the 20% buyer 304 to 1,045; 0 conflicts (K, floor 2/3) | | Signing stops while mining continues, 1, 6, 24 h | results J and L1 | 34% and above: every checkpoint stalled for the whole silence; 33%: 665 to 727 of 720 in 6 h; 32%: 35 to 221; 30%: 0 to 40; first lock after resume 0 min at every weight; 0 conflicts (J, L1) | | Seeds during a pause (3.11.6) | devnet epoch boundary through a forced pause | not yet run (O-4.3 implementation) | +| A certificate over a chain the node is not on: the certificate-driven reorg (3.5) | fast-time harness `tools/finality-attacks/c4.mjs`, weight against work, 130-s split, 240-DAA window, rule v2 (the live devnet's) | the work-majority node fetched the certified chain, locked 12, 13 and 14 by certificate within 2 s of the first block, re-determined 11, and all three nodes ended on the certified chain with 0 conflicting certificates and 0 disagreeing locks; the module-off control took the heavier chain (bench-log "the C4 fix", 5 October 2026 night); the same under rule v3 is unit-tested, the harness row with the certifying side locking during the split is still owed | | Two certificates at one index: no lock withdrawn (3.11.4) | devnet with a forced double certificate | not yet run (O-3.17) | | `T` under the block reading (3.11.3) | O-3.3 re-run | not yet run (O-3.18) | | Participation credited only for the chain's checkpoint (3.11.1) | results C with an adversary voting for private blocks | not yet run (O-3.19) | diff --git a/docs/spec/06-open-items.md b/docs/spec/06-open-items.md index 860f1c56..0da101da 100644 --- a/docs/spec/06-open-items.md +++ b/docs/spec/06-open-items.md @@ -54,7 +54,7 @@ An item closes when its measurement is in `docs/bench-log.md` or its decision is | O-3.3 | Parameters of the block reading of participation (rule closed 3 October 2026, section 3.3 Q2 and 3.4.1; ledger F3): the per-block vote bound and the carriage window, and the simulation ran with the narrower cert reading | Add vote carriage in blocks and a hostile aggregator to `finality_v2.py`; re-run A, C, D and F1 under the block reading; set the per-block vote bound and confirm the carriage window of 240 indices | 3 | | O-3.4 | Certificate grace value; must be at least 3x the worst honest one-way delay (section 3.3, Q4) | Measure one-way delays on the devnet across regions; set grace | 3 | | O-3.5 | VRF construction for aggregator selection and the binomial sub-user sortition above 8,192 voters (S1, S2) | Specify (candidate: BLS-based VRF on the vote key, Algorand's binomial sampling); simulate the threshold on sampled weight | 3 | -| O-3.6 | What a node does with two valid certificates at one index after a partition heals; post-heal fork choice is unmodelled (`sim/results_v2.md`, "cannot tell us") | Adopt or replace the proposal in section 3.5; devnet partition-and-heal test | 3 | +| O-3.6 | What a node does with two valid certificates at one index after a partition heals; post-heal fork choice is unmodelled (`sim/results_v2.md`, "cannot tell us"). Narrowed 5 October 2026 night (ledger C4): post-heal fork choice for ONE certified chain is now the certificate-driven reorg of 3.5, implemented and measured on the fast-time harness (`tools/finality-attacks/c4.mjs`, bench-log "the C4 fix"); what is left is the two-certificate case, O-3.17 | Adopt or replace the proposal in section 3.5; devnet partition-and-heal test | 3 | | O-3.7 | The eclipse case is closed by the quorum floor (section 3.3.2, 3 October 2026; ledger F2): 0 conflicting locks at 1, 2 and 4 h against a 34% attacker in the model, but the model grants the attacker the eclipse for free | Devnet with a single-node eclipse recording whether conflicting locks appear, as confirmation of the rule; no rule choice remains | 3 | | O-3.8 | The simulation has no DAG: conflict counts are index collisions; red blocks, merge under the 3,600-s bound and the finality overlay's effect on GHOSTDAG's guarantees are unmodelled (ledger C4, F8) | Devnet runs with the finality module on and off; a churn and adversary simulation driven by real pool-hashrate traces from mid-cap GPU coins (design document, "Three experiments") | 3 | | O-3.9 | Model assumptions that move the numbers: perfect or instant DAA retarget (real lag of the order of an hour, approximate), uptime 97% / 99.5% is a guess, silent sets random by key not by pool or region, keys are free, VRF noise absent (`sim/results.md` and `results_v2.md`) | Re-run `finality_v2.py` with a DAA lag model, a top-pool silent set and a regional silent set; price keys through the P2P layer | 3 | diff --git a/packaging/ota/publish-jobs.sh b/packaging/ota/publish-jobs.sh index 4e26f0ed..96fd0896 100755 --- a/packaging/ota/publish-jobs.sh +++ b/packaging/ota/publish-jobs.sh @@ -128,7 +128,7 @@ if [ ! -x "$SIGNER" ]; then echo "building igneum-ota-sign" (cd "$ROOT/app/igneum-app" && nice -n 19 cargo build --release -j 4 --bin igneum-ota-sign --quiet) fi -EMBEDDED="$("$SIGNER" embedded | head -1)" +EMBEDDED="$("$SIGNER" embedded | sed -n 1p)" OURS="$(tr -d '[:space:]' < "$PUB")" if [ "$EMBEDDED" != "$OURS" ]; then echo "the public key in app/igneum-app/src/manifest.rs ($EMBEDDED) is not $PUB ($OURS); the apps would refuse this file" >&2 diff --git a/packaging/ota/publish-manifest.sh b/packaging/ota/publish-manifest.sh index 33e3786c..e0c8e54f 100755 --- a/packaging/ota/publish-manifest.sh +++ b/packaging/ota/publish-manifest.sh @@ -88,7 +88,7 @@ if [ ! -x "$SIGNER" ]; then echo "building igneum-ota-sign" (cd "$ROOT/app/igneum-app" && nice -n 19 cargo build --release -j 4 --bin igneum-ota-sign --quiet) fi -EMBEDDED="$("$SIGNER" embedded | head -1)" +EMBEDDED="$("$SIGNER" embedded | sed -n 1p)" OURS="$(tr -d '[:space:]' < "$PUB")" if [ "$EMBEDDED" != "$OURS" ]; then echo "the public key in app/igneum-app/src/manifest.rs ($EMBEDDED) is not $PUB ($OURS); the apps would refuse this manifest" >&2 diff --git a/packaging/ota/test-publish-jobs.sh b/packaging/ota/test-publish-jobs.sh index d3fbc968..e35f502f 100755 --- a/packaging/ota/test-publish-jobs.sh +++ b/packaging/ota/test-publish-jobs.sh @@ -97,7 +97,7 @@ grep -q "^inner-identical True$" "$T/inner2.txt" && grep -q "^sig-identical True expect_ok "list reads the folder" "$PUBLISH" list --dest "$D" echo "== the key" -EMB="$("$SIGNER" embedded | head -1)" +EMB="$("$SIGNER" embedded | sed -n 1p)" [ "$EMB" = "$(tr -d '[:space:]' < "$PUB")" ] && ok "the embedded key is the Mac's OTA public key" || bad "the embedded key is not $PUB" echo diff --git a/packaging/windows/push-build-inputs.sh b/packaging/windows/push-build-inputs.sh index 273f80e2..231abdce 100755 --- a/packaging/windows/push-build-inputs.sh +++ b/packaging/windows/push-build-inputs.sh @@ -7,6 +7,10 @@ # # packaging/windows/push-build-inputs.sh [--node ] # [--node-tests "igneum-miner"] [--app-tests "igneum-app"] [--no-app] [--no-deploy] [--out ] +# [--name ] the zip's name in the downloads folder (default build-inputs.zip; build-job.mjs gives +# every job its own name since 5 October 2026 night: with one shared name a job +# published while another agent's pack landed pinned THAT agent's sources, three +# times in one evening; zips older than two days are pruned here) # # What goes in (under igneum-build-inputs/): # manifest.json created_at, node {branch, commit, dirty, source}, repo {branch, commit, dirty}, app_version, @@ -32,6 +36,7 @@ APP_TESTS="${IGNEUM_BUILD_APP_TESTS:-igneum-app}" WITH_APP=1 DEPLOY=1 OUT="" +NAME="" while [ $# -gt 0 ]; do case "$1" in --node) NODE_SRC="$2"; shift 2 ;; @@ -40,6 +45,7 @@ while [ $# -gt 0 ]; do --no-app) WITH_APP=0; shift ;; --no-deploy) DEPLOY=0; shift ;; --out) OUT="$2"; shift 2 ;; + --name) NAME="$2"; shift 2 ;; *) echo "unknown argument: $1" >&2; exit 2 ;; esac done @@ -58,7 +64,9 @@ if [ -z "$OUT" ]; then TOKEN="$(tr -d '[:space:]' < "$TOKEN_FILE")" [ -n "$DLSITE" ] && [ -d "$DLSITE/dl/$TOKEN" ] || { echo "no downloads folder: set IGNEUM_DLSITE or ~/.config/igneum/dlsite-dir (it must hold dl//)" >&2; exit 1; } DEST="$DLSITE/dl/$TOKEN" - OUT="$DEST/build-inputs.zip" + OUT="$DEST/${NAME:-build-inputs.zip}" + # per-job zips older than two days (a job expires in two days) go, with their sha256 and manifest + find "$DEST" -maxdepth 1 -name 'build-inputs-*' \( -name '*.zip' -o -name '*.sha256' -o -name '*.json' \) -mtime +2 -delete 2>/dev/null || true else TOKEN="" DEST="$(cd "$(dirname "$OUT")" && pwd)" @@ -135,14 +143,14 @@ if [ "$DEPLOY" = 1 ]; then LIVE="$(mktemp)" ok=0 for try in 1 2 3 4 5 6; do - code="$(curl -s -o "$LIVE" -w '%{http_code}' "https://dl.igneum.network/dl/$TOKEN/build-inputs.sha256")" - echo "https://dl.igneum.network/dl//build-inputs.sha256 -> HTTP $code (try $try)" + code="$(curl -s -o "$LIVE" -w '%{http_code}' "https://dl.igneum.network/dl/$TOKEN/$(basename "${OUT%.zip}").sha256")" + echo "https://dl.igneum.network/dl//$(basename "${OUT%.zip}").sha256 -> HTTP $code (try $try)" if [ "$code" = 200 ] && [ "$(tr -d '[:space:]' < "$LIVE")" = "$SUM" ]; then ok=1; break; fi sleep 10 done [ "$ok" = 1 ] || { echo "the live sha256 is not reachable or is not this zip's after 6 tries; check the deploy output" >&2; rm -f "$LIVE"; exit 1; } rm -f "$LIVE" - echo "live: build-inputs.zip verified by sha256" + echo "live: $(basename "$OUT") verified by sha256" elif [ -n "$TOKEN" ]; then echo "not deployed (--no-deploy): cd $DLSITE && npx --yes vercel@latest --global-config ~/.config/igneum/vercel deploy --prod --yes" else diff --git a/packaging/windows/push-inputs.sh b/packaging/windows/push-inputs.sh index 05f515dc..b652c011 100755 --- a/packaging/windows/push-inputs.sh +++ b/packaging/windows/push-inputs.sh @@ -69,7 +69,7 @@ if [ ! -x "$SIGNER" ]; then echo "building igneum-ota-sign" (cd "$ROOT/app/igneum-app" && nice -n 19 cargo build --release -j 4 --bin igneum-ota-sign --quiet) fi -EMBEDDED="$("$SIGNER" embedded | head -1)" +EMBEDDED="$("$SIGNER" embedded | sed -n 1p)" [ "$EMBEDDED" = "$(tr -d '[:space:]' < "$PUB")" ] || { echo "the public key in app/igneum-app/src/manifest.rs ($EMBEDDED) is not $PUB; the runner would refuse this signature" >&2; exit 1; } # the manifest: what is in the zip, from where, when. IGNEUM_NODE_SRC names the worktree the exes were built from diff --git a/tools/build-job.mjs b/tools/build-job.mjs index ebe71a67..b4ae402f 100755 --- a/tools/build-job.mjs +++ b/tools/build-job.mjs @@ -208,7 +208,11 @@ function placeFor(o) { // ---- publish: pack, then the signed job --------------------------------------------------------------------------------------- function publish() { - const pack = ['packaging/windows/push-build-inputs.sh']; + // One zip per job (5 October 2026 night): with the one shared build-inputs.zip, a job published while another agent's + // pack landed in the downloads folder pinned THAT agent's sources (three C4 jobs built tx-gossip and m20-live). The + // name carries the time and this process id; push-build-inputs.sh prunes names older than two days. + const zipName = `build-inputs-${new Date().toISOString().replace(/[-:T]/g, '').slice(0, 14)}-${process.pid}.zip`; + const pack = ['packaging/windows/push-build-inputs.sh', '--name', zipName]; if (flags.node) pack.push('--node', flags.node); if (flags['node-tests']) pack.push('--node-tests', String(flags['node-tests'])); if (flags['app-tests']) pack.push('--app-tests', String(flags['app-tests'])); @@ -226,6 +230,9 @@ function publish() { if (flags.title) add.push('--title', String(flags.title)); if (flags.id) add.push('--id', String(flags.id)); if (!flags['no-deploy']) add.push('--deploy'); + const dlsite = (process.env.IGNEUM_DLSITE || cfg('dlsite-dir')).trim(), dlToken = cfg('dl-token'); + if (dlsite && dlToken) add.push('--zip', join(dlsite, 'dl', dlToken, zipName)); + else { console.error('no ~/.config/igneum/dlsite-dir or dl-token: cannot name the job\'s zip'); process.exit(1); } console.log(`$ ${add.join(' ')}`); const a = spawnSync('bash', add, { cwd: ROOT, encoding: 'utf8' }); process.stdout.write(a.stdout || ''); process.stderr.write(a.stderr || ''); diff --git a/tools/ci/signer-pipe-check.sh b/tools/ci/signer-pipe-check.sh new file mode 100755 index 00000000..849f0b87 --- /dev/null +++ b/tools/ci/signer-pipe-check.sh @@ -0,0 +1,16 @@ +#!/usr/bin/env bash +# Fails when a script pipes a Rust binary's output into `head` (5 October 2026 night, C4 fix round: `igneum-ota-sign +# embedded | head -1` under `set -o pipefail`; the signer prints two lines, `head` closes the pipe after the first, and +# on a loaded Mac the second print lands after the close: SIGPIPE, "failed printing to stdout: Broken pipe", exit 101, +# and publish-jobs.sh died before the job was published). A Rust binary panics on a closed stdout; `sed -n 1p` reads to +# the end. The class: the signer (`$SIGNER` or `igneum-ota-sign`, the one binary here whose output is longer than the +# line a script wants) piped into head; `igneumd --version | head -1` prints one line and is left alone. +set -euo pipefail +cd "$(dirname "$0")/../.." +bad=$(grep -rnE '(\$SIGNER"?|igneum-ota-sign)[^|]*\| *head( |$)' packaging tools infra --include='*.sh' --include='*.mjs' 2>/dev/null | grep -v 'tools/ci/signer-pipe-check.sh' || true) +if [ -n "$bad" ]; then + echo "a Rust binary piped into head (SIGPIPE panics the binary; use sed -n 1p):" >&2 + echo "$bad" >&2 + exit 1 +fi +echo "signer-pipe-check: ok" diff --git a/tools/finality-attacks/c4.mjs b/tools/finality-attacks/c4.mjs index 840b4b8d..1df0d1ec 100644 --- a/tools/finality-attacks/c4.mjs +++ b/tools/finality-attacks/c4.mjs @@ -4,7 +4,9 @@ // ports 29800+, network igneum-devnet-980, data under /tmp/igneum-fin-c4; the live devnet is never touched. // // node tools/finality-attacks/c4.mjs on; node tools/finality-attacks/c4.mjs off # one mode per process +// node tools/finality-attacks/c4.mjs v2 # module on under rule v2 (the live devnet's rule) // SPLIT=150 WARM=230 HEAL=200 node tools/finality-attacks/c4.mjs on +// IGNEUMD=... IGNEUM_MINER=... node tools/finality-attacks/c4.mjs on # another node build (the C4 fix, 5 October 2026 night) // // Topology (as v3.mjs): n1 listens; n0 dials n1 through proxy P0, n2 dials n1 through proxy P2; cutting P0 isolates // n0 (side A) from n1 and n2 (side B). @@ -19,6 +21,8 @@ // overlay requires every candidate tip to pass through B's certified checkpoint. The measurement is which chain // the three nodes converge to, whether they converge at all, and what each node had to reorganise. +import { createRequire } from 'node:module'; +const require = createRequire(import.meta.url); const ROOT = new URL('../../', import.meta.url).pathname; const NODE_ROOT = process.env.IGNEUM_NODE_ROOT || '/Users/joshm/Projects/igneum/'; process.env.IGNEUM_FIN_BASE_PORT ||= '29800'; @@ -33,9 +37,25 @@ const WARM = +(process.env.WARM || 230), SPLIT = +(process.env.SPLIT || 150), HE const RA = +(process.env.RA || 0.6), RB = +(process.env.RB || 0.4); // ONE mode per process: lib/net.mjs reads IGNEUM_FIN_OVERRIDE_JSON when it is imported, so the override must be in // the environment before the import (the first draft set it inside network() and ran rule v2 twice; 5 October 2026). +// `v2` (5 October 2026, night): the module on under rule v2, the live devnet's rule (no frozen table, no fold): the +// override is the fast-time file alone, as the first draft's accidental control was; the expectation is the `on` one. const MODE = process.argv.slice(2).filter(a => !a.startsWith('--'))[0] || 'on'; -if (MODE !== 'on' && MODE !== 'off') { console.error(`mode must be on or off, got ${MODE}`); process.exit(2); } -process.env.IGNEUM_FIN_OVERRIDE_JSON = JSON.stringify(MODE === 'off' ? { finality: { min_daa: 9007199254740991 } } : { finality_v3_activation_daa: 0 }); +if (!['on', 'off', 'v2'].includes(MODE)) { console.error(`mode must be on, off or v2, got ${MODE}`); process.exit(2); } +// WINDOW= (5 October 2026, night): a longer weight window than the fast-time file's 120 (ban and min_daa follow it; +// WARM must exceed it). Under rule v2 a side locks alone once its own chain holds two thirds of its own sliding window, +// (2/3 W - s W) / (1 - s) of its own DAA after the cut: 63 DAA at W 120 and s 0.3, which a 90-s split at 0.6 blocks/s +// crosses before the heal (measured: n0 locked index 10 alone one second before B's certificate for 8 reached it). At +// W 240 the bound is 126 DAA (210 s), so a 130-s split stays inside it, as any partition under an hour does on the live +// devnet's 7,200-DAA window. +const WINDOW = +(process.env.WINDOW || 0); +const windowOverride = () => { + if (!WINDOW) return {}; + const { readFileSync } = require('node:fs'); + const f = JSON.parse(readFileSync(new URL('../../infra/fast-time/override-60x.json', import.meta.url), 'utf8')).finality; + return { finality: { ...f, weight_window: WINDOW, equivocation_ban: WINDOW, min_daa: WINDOW } }; +}; +process.env.IGNEUM_FIN_OVERRIDE_JSON = JSON.stringify(MODE === 'off' ? { finality: { min_daa: 9007199254740991 } } : MODE === 'v2' ? windowOverride() : { ...windowOverride(), finality_v3_activation_daa: 0 }); +const MODULE_ON = MODE !== 'off'; const { Node, Miner, Proxy, stopAll, sleep, log, assertBinaries, TMP, IGNEUMD } = await import('./lib/net.mjs'); const { mkdirSync, writeFileSync, appendFileSync } = await import('node:fs'); @@ -110,9 +130,16 @@ async function run(mode) { log(`${name}: end of split: A sink blue score ${bsA} (${dagsEnd[0]?.blockCount} blocks), B sink blue score ${bsB} (${dagsEnd[1]?.blockCount} blocks); B locked ${bLockedDuring.length} new index(es) ${bLockedDuring.map(([i]) => i).join(',')}; A locked ${newLocks[0]}`); p0.heal(); const tHeal = Date.now(); - let reconnected = null; + let reconnected = null, addPeerErr = null; + // The heal is the link, not the session: n0 dials the proxy again on the connection manager's backoff, which reached + // 84 to 114 s after a 130-s cut (5 October 2026 night, takes 1 to 3), and A kept mining alone meanwhile, so every run + // became the partition-longer-than-a-window shape. A real heal has the other side dialling too; here the harness asks + // n0 for the connection (addPeer, not permanent) every 3 s until a peer is up, and reports the time it took. while (Date.now() - tHeal < HEAL * 1000) { - if (reconnected == null && (await peers(n0)) > 0) reconnected = Math.round((Date.now() - tHeal) / 1000); + if (reconnected == null) { + if ((await peers(n0)) > 0) reconnected = Math.round((Date.now() - tHeal) / 1000); + else await n0.rpc.call('addPeer', { peerAddress: { ip: '127.0.0.1', port: p0.port }, isPermanent: false }).catch(e => { if (!addPeerErr) { addPeerErr = String(e?.message || e); log(`addPeer ${p0.addr} failed: ${addPeerErr}`); } }); + } await sleep(3000); } for (const m of miners) await m.stop(); @@ -136,10 +163,14 @@ async function run(mode) { await stopAll(); const heavier = bsA > bsB ? 'A' : 'B'; const ended = converged ? (onB[0] && !onA[0] ? 'B' : onA[0] && !onB[0] ? 'A' : onA[0] && onB[0] ? 'both merged' : 'neither') : 'not converged'; - const pass = mode === 'on' - ? (newLocks[0] === 0 && bLockedDuring.length > 0 && converged && ended === 'B' && conflicts.every(c => c === 0) && disagree === 0) + // module on (v3 or v2): B's certified chain must win although A's is heavier, with no conflict and no disagreeing + // lock; under v2 A may lock alone during the split once the sliding table is its own (F21's bound), so A's split-time + // locks are reported, not required to be zero + const adopted = [n0, n1, n2].map(n => n.grepLog(/LOCKED by certificate/).length); + const pass = MODULE_ON + ? ((mode === 'v2' || newLocks[0] === 0) && bLockedDuring.length > 0 && converged && ended === 'B' && conflicts.every(c => c === 0) && disagree === 0) : (newLocks.every(x => x === 0) && converged && ended === heavier); - out(`\n### ${name}: warm ${WARM} s at 1 block/s (B 70% of weight, A 30%), split ${SPLIT} s with A at ${RA} and B at ${RB} blocks/s, heal window ${HEAL} s, link delay ${DELAY_MS} ms, module ${mode} (${mode === 'on' ? 'rule v3 from checkpoint DAA 0' : 'min_daa never: no certificate can form'}), node ${IGNEUMD.split('/').slice(-3).join('/')}\n`); + out(`\n### ${name}: warm ${WARM} s at 1 block/s (B 70% of weight, A 30%), split ${SPLIT} s with A at ${RA} and B at ${RB} blocks/s, heal window ${HEAL} s, link delay ${DELAY_MS} ms${WINDOW ? `, weight window ${WINDOW} DAA` : ''}, module ${mode} (${mode === 'on' ? 'rule v3 from checkpoint DAA 0' : mode === 'v2' ? 'rule v2, the live devnet rule' : 'min_daa never: no certificate can form'}), node ${IGNEUMD.split('/').slice(-3).join('/')}\n`); out('| measure | n0 (side A, work majority) | n1 (side B, weight majority) | n2 (side B) |'); out('|---|---|---|---|'); out(`| max locked index at the cut | ${beforeMax.join(' | ')} |`); @@ -149,10 +180,11 @@ async function run(mode) { out(`| sink at the end of the heal window | ${sinks.map(s => String(s).slice(0, 10)).join(' | ')} |`); out(`| A's split tip on the final chain / B's split tip on the final chain | ${onA.map((a, i) => `${a} / ${onB[i]}`).join(' | ')} |`); out(`| conflicting certificates logged | ${conflicts.join(' | ')} |`); + out(`| locks adopted from a certificate off the node's chain (the C4 fix) | ${adopted.join(' | ')} |`); out(`| re-determined lines (F24) | ${redetermined.join(' | ')} |`); out(`| reorg lines in the node log | ${reorgs.join(' | ')} |`); - out(`\nAt the end of the split: A's sink blue score ${bsA} against B's ${bsB} (the heavier chain by blue work is ${heavier}'s); B locked ${bLockedDuring.length} new checkpoint(s) during the split${bLockedDuring.length ? ' at index ' + bLockedDuring.map(([i]) => i).join(', ') : ''}. After the heal: n0 reconnected ${reconnected == null ? 'not within the heal window' : reconnected + ' s after the gate reopened'}; the three sinks ${converged ? 'agree' : 'DISAGREE'}; the network ended on ${ended}'s chain; A adopted B's split-time locks: ${bAdoptedByA}; locked indices disagreeing across the three nodes: ${disagree}. ${pass ? 'PASS' : 'FAIL'} against the expectation for module ${mode} (${mode === 'on' ? "B's certified chain wins although A's is heavier" : 'the heavier chain wins'}).`); - results.push({ name, pass, heavier, ended, converged, bsA, bsB, newLocks, bLocked: bLockedDuring.length, conflicts, disagree }); + out(`\nAt the end of the split: A's sink blue score ${bsA} against B's ${bsB} (the heavier chain by blue work is ${heavier}'s); B locked ${bLockedDuring.length} new checkpoint(s) during the split${bLockedDuring.length ? ' at index ' + bLockedDuring.map(([i]) => i).join(', ') : ''}. After the heal: n0 reconnected ${reconnected == null ? 'not within the heal window' : reconnected + ' s after the gate reopened'}; the three sinks ${converged ? 'agree' : 'DISAGREE'}; the network ended on ${ended}'s chain; A adopted B's split-time locks: ${bAdoptedByA}; locked indices disagreeing across the three nodes: ${disagree}. ${pass ? 'PASS' : 'FAIL'} against the expectation for module ${mode} (${MODULE_ON ? "B's certified chain wins although A's is heavier" : 'the heavier chain wins'}).`); + results.push({ name, pass, heavier, ended, converged, bsA, bsB, newLocks, bLocked: bLockedDuring.length, conflicts, disagree, adopted, reconnected }); } async function main() { diff --git a/tools/finality-attacks/lib/net.mjs b/tools/finality-attacks/lib/net.mjs index ac1ddc1f..88809798 100644 --- a/tools/finality-attacks/lib/net.mjs +++ b/tools/finality-attacks/lib/net.mjs @@ -72,6 +72,9 @@ export class Node { args() { const a = ['--devnet', `--devnet-suffix=${DEVNET_SUFFIX}`, '--nodnsseed', '--disable-upnp', '--nologfiles', '--enable-unsynced-mining', '--utxoindex', `--appdir=${this.dir}`, + // unsafe RPC so a scenario can ask a node to dial a peer again at a heal (addPeer; c4.mjs, 5 October 2026 night); + // every harness node listens on 127.0.0.1 only + '--unsaferpc', `--rpclisten=127.0.0.1:${this.grpcPort}`, `--rpclisten-json=127.0.0.1:${this.jsonPort}`, `--listen=127.0.0.1:${this.p2pPort}`, `--override-params-file=${overrideParams(this.override || {}, this.override ? this.name : '')}`, '--loglevel=info', '--yes']; if (this.connect.length) a.push(`--connect=${this.connect.join(',')}`); else a.push('--outpeers=0');