diff --git a/app/igneum-app/src/engine.rs b/app/igneum-app/src/engine.rs index 850754324..cb05672fc 100644 --- a/app/igneum-app/src/engine.rs +++ b/app/igneum-app/src/engine.rs @@ -1183,6 +1183,7 @@ impl Engine { } fn miner_args(&self, slot: &MinerSlot, card: &CardState) -> Vec { + let pack_arg = format!("--pack packs\\{}", pack_dir_name(slot.card)); let s = self.shared.settings.lock().unwrap(); let r = &self.shared.runtime; let mut a = vec![ @@ -1236,7 +1237,9 @@ impl Engine { if !wargs.is_empty() { wargs.push(' '); } - wargs.push_str("--pack packs\\devnet"); + // one pack directory per card (ledger X22, 5 October 2026): two miners used to truncate the same + // packs\devnet while the other card's worker was reading it + wargs.push_str(&pack_arg); } if !wargs.is_empty() { a.push("--worker-args".into()); @@ -1333,7 +1336,7 @@ impl Engine { let vendor = self.st().mining.cards.get(card_idx).map(|c| c.vendor.clone()).unwrap_or_default(); let worker = self.miners[i].worker.clone(); std::thread::spawn(move || { - let r = if build { build_worker_from_source(&shared, &bins, &vendor) } else { export_pack(&shared, &bins).map(|_| worker) }; + let r = if build { build_worker_from_source(&shared, &bins, &vendor, card_idx) } else { export_pack(&shared, &bins, card_idx).map(|_| worker) }; shared.send(Cmd::WorkerBuilt(card_idx, r)); }); } @@ -2290,7 +2293,14 @@ impl Engine { if self.bins_worker_missing(card_idx) { self.miners[i].needs_rebuild = true; } - self.miners[i].restart_at = Some(now); + // ledger X22 (5 October 2026): while the node is still syncing the template's epoch moves + // with every batch of headers, and an export-and-rebuild per exit 42 was a loop; the + // miner itself no longer exits 42 during a sync, and the app waits a minute as well + let synced = self.st().node.synced; + self.miners[i].restart_at = Some(if synced { now } else { now + Duration::from_secs(60) }); + if !synced { + self.shared.log("the node is still syncing; the worker restarts in 60 s"); + } } else if verdict != crate::watchdog::Action::None { // exit 43: the miner gave up on its worker; once more, then the card is faulted self.watchdog_verdict(i, verdict, &tail); @@ -3108,7 +3118,17 @@ pub fn parse_race(body: &str) -> Option { #[cfg(test)] mod tests { - use super::{parse_race, sync_decision, Reading}; + use super::{pack_dir_name, parse_race, sync_decision, Reading}; + + /// Ledger X22: every card exports into its own pack directory, so no two workers read one directory while an + /// export truncates it. + #[test] + fn pack_directories_are_per_card() { + assert_eq!(pack_dir_name(0), "devnet-0"); + assert_eq!(pack_dir_name(3), "devnet-3"); + let names: std::collections::HashSet = (0..8).map(pack_dir_name).collect(); + assert_eq!(names.len(), 8); + } #[test] fn race_line_parses() { @@ -3192,10 +3212,17 @@ fn civil_from_days(z: i64) -> (i64, u32, u32) { (if m <= 2 { y + 1 } else { y }, m, d) } -/// Exports this hour's program pack from the node to \packs\devnet (the prebuilt workers read it with --pack). +/// The pack directory of one card under \packs: `devnet-` (ledger X22, 5 October 2026: one +/// directory per card, so an export for one card never truncates the files another card's worker is reading). +pub(crate) fn pack_dir_name(card: usize) -> String { + format!("devnet-{card}") +} + +/// Exports this hour's program pack from the node to \packs\devnet- (the prebuilt workers read it +/// with --pack). #[allow(unused_variables)] -fn export_pack(shared: &Arc, bins: &Bins) -> Result<(), String> { - let pack = shared.runtime.app_dir.join("packs").join("devnet"); +fn export_pack(shared: &Arc, bins: &Bins, card: usize) -> Result<(), String> { + let pack = shared.runtime.app_dir.join("packs").join(pack_dir_name(card)); let _ = std::fs::create_dir_all(&pack); let out = crate::detect::run_timeout(std::process::Command::new(&bins.miner).args(["export-pack", &shared.runtime.rpc_url(), &pack.display().to_string()]), None, Duration::from_secs(120)).unwrap_or_default(); shared.log(&format!("export-pack: {}", out.lines().last().unwrap_or("no output"))); @@ -3205,7 +3232,7 @@ fn export_pack(shared: &Arc, bins: &Bins) -> Result<(), String> { /// Today's Windows path when no prebuilt worker ships: export the program pack, run proto-cuda\build.bat (or /// proto-opencl\build.bat) in the MSVC environment, and use the exe it writes. Untested from the Mac. #[allow(unused_variables)] -fn build_worker_from_source(shared: &Arc, bins: &Bins, vendor: &str) -> Result { +fn build_worker_from_source(shared: &Arc, bins: &Bins, vendor: &str, card: usize) -> Result { #[cfg(not(windows))] { Err("no GPU worker is installed".into()) @@ -3213,7 +3240,7 @@ fn build_worker_from_source(shared: &Arc, bins: &Bins, vendor: &str) -> #[cfg(windows)] { use std::process::Command; - let pack = shared.runtime.app_dir.join("packs").join("devnet"); + let pack = shared.runtime.app_dir.join("packs").join(pack_dir_name(card)); let _ = std::fs::create_dir_all(&pack); let out = crate::detect::run_timeout(Command::new(&bins.miner).args(["export-pack", &shared.runtime.rpc_url(), &pack.display().to_string()]), None, Duration::from_secs(120)).unwrap_or_default(); shared.log(&format!("export-pack: {}", out.lines().last().unwrap_or(""))); diff --git a/docs/bench-log.md b/docs/bench-log.md index 55918d0e2..62659d289 100644 --- a/docs/bench-log.md +++ b/docs/bench-log.md @@ -1673,3 +1673,47 @@ Renter weight share at the end of the day, finality sim, seed 7 (DAA form / medi | Kaspa sampled DAA | 2.8 / 0.5 | 5.7 / 1.1 | 8.5 / 1.6 | 14.2 / 2.7 | 19.8 / 3.8 | 28.3 / 5.5 | 33.9 / 6.6 | 42.4 / 8.2 | 56.6 / 11.0 | 70.9 / 13.7 | 86.0 / 16.5 | Reading. The amplifier the critic names, weight share over hash share, is under 1.0 in every cell: the lag never pays the pulse more than par. What the controller decides is how close to par it gets. Under Kaspa's rule the pulse takes 87.4% of the blocks for 89.3% of the hashes (0.96 of par), the chain runs at 1,716 to 1,753 blocks in the peak minute and crawls with gaps up to 262 s, and the DAA-second window hands the renter a third of the weight on day 12 and 86% on day 30: the critic's week is a fortnight, and it costs the whole pulse, 89% of the network's hashes. Under the Igneum rule v2 the same pulse takes 23.3% of the blocks (0.036 blocks per hash against the base's, 0.22 of par), peaks at 201 blocks a minute, and never reaches a third under either form (19.6% at day 30 under the DAA form). The median-time form of F14 caps the renter at 16.5 to 16.8% under BOTH controllers, 0.18 to 0.19 of par, because 10 minutes of blocks is 10 minutes of weight whatever the block count, and the crawl after the pulse costs the honest side only the empty buckets. So the F14 rule change is defence in depth that holds even under Kaspa's controller, and the rule v2 controller alone keeps the renter under a third. Under all four cells 0 checkpoints stalled and 0 conflicting locks (the renter signs). Not modelled: the renter's own blocks are all blue (GHOSTDAG abstracted, as the rest of the simulator), the honest network does not react to the pulse, the median-time window is the wall clock (every stamp is honest), and the DAA form's window edge moves by 10-minute buckets. Raw: `/tmp/igneum-daa-trace/scenario-n.out`, traces `/tmp/igneum-daa-trace/pulse-*.npz`. + +## 5 October 2026 (night), ledger close round 1: fork fixes X20, X19, P15, M25, M28, X22 + +Owner: the ledger-fork agent, docs worktree `igneum-wt-ledger-fork` (branch `ledger-fork`), fork worktree `vendor/igneum-node-ledger` (branch `ledger-fixes` from `release-0.3.6` a24ab01a; commits 9e109c65 X20, fc0cb938 X19 and the M25 node half, b6f381e2 P15, 3d4ec451 the miner: M28, X22, the M25 miner half). Every number below is a count, a pass or a fail; nothing here is a rate. + +**Build (Mac, build lock).** `cargo build --release -j 4 -p kaspad -p igneum-miner --features kaspad/igneum-pow`, `CARGO_TARGET_DIR` inside the worktree, seeded from the 0.3.6 target: 7 min 34 s for the last pass, `igneumd` 41,000,864 bytes, `igneum-miner` 8,581,184 bytes. + +**Suites (PC 2, 1ccfe586, job `build-20261005-190808`, `--node-tests "kaspa-consensus kaspa-consensus-core kaspa-p2p-flows kaspa-mining igneum-exec igneum-miner"`, `--targets linux --no-app`).** Six attempts, two of them the evidence. `build-20261005-190808` (3d4ec451): a compile error in the new P15 test (`RpcErr` has no `Debug`), fixed in f6f43532. `build-20261005-191711` (f6f43532): 9 compile errors in `protocol/flows/src/ibd/proof.rs`'s tests, pre-existing on `release-0.3.6` since d35b00cf (M20 changed the wire messages and the proof type and left the tests behind), so the `kaspa-p2p-flows` test target did not compile on the base commit; test-only fix in 815c7d64, checked with `cargo check --tests` of all six crates on the Mac (3 min 28 s) before the next job. `build-20261005-192537` (815c7d64) ran ANOTHER agent's inputs: `build-inputs.zip` is one shared path on the download host, another publish replaced the zip and its `.sha256` while the job waited in the queue (26 min), and the PC's "sha256 ok" compared the replaced file against itself (it reported `devnet-v4 3bfe346f`, app 0.3.9, 8,123,614 bytes against my 7,740,161); the C4 agent hit the same race earlier tonight and its worktree's tooling (`igneum-wt-c4`, `--name`/`--zip`, a zip per job) was used from `build-20261005-200049` on. `build-20261005-195526` (815c7d64, shared zip, landed): kaspa-consensus 92 passed 3 failed, all three `PowCacheQueueFull` (the finality tests racing the PC's PoW cache queue under the parallel test runner: `frozen_table_holds_a_side_without_the_other_keys_for_one_window`, `a_shallower_sink_un_determines_the_indices_it_cannot_reach`, `no_certificate_while_the_window_is_filling`; none of them mine), consensus-core 100 passed 1 failed (`fast_time_60x_file_is_the_devnet_at_60x` reads the MAIN checkout's `override-60x.json`, which still carries the dead field this line refuses; with `IGNEUM_FAST_TIME_FILE` on the `ledger-fork` copy it passes, 101/0 on the Mac). The evidence: + +| Job | Commit | Crate | Passed | Failed | Ignored | +|---|---|---|---|---|---| +| `build-20261005-200049` (kaspa-consensus alone, per-job zip) | 815c7d64 | kaspa-consensus | 95 | 0 | 3 | +| `build-20261005-200606` (the other five, per-job zip) | 815c7d64 | kaspa-consensus-core | 101 | 0 | 2 | +| | | kaspa-mining | 52 | 0 | 0 | +| | | igneum-exec | 13 | 0 | 0 | +| | | igneum-miner | 17 | 0 | 0 | +| | | kaspa-p2p-flows | 32 | 1 | 0 | +| Mac (build lock, debug profile, `cargo test -p kaspa-p2p-flows --lib`) | bd1b676a | kaspa-p2p-flows | 33 | 0 | 0 | + +The one PC failure is the new `one_line_per_minute_with_the_median_skew` test: its loop advanced the clock without advancing the refused headers, so the median drifted (the monitor is right, the test's inputs were not); bd1b676a keeps every refused header 60 s ahead of its own clock. Not re-sent to PC 2 (the coordinator's one-job rule for tonight); the Mac run is labelled as such. Mac fallback runs of the whole set, debug profile, under the build lock (the shared-zip race made the PC result untrustworthy for an hour): igneum-exec 13/0, igneum-miner 17/0, kaspa-consensus 95/0 (3 ignored), kaspa-mining 52/0, kaspa-consensus-core 101/0 with the worktree's fast-time file. + +**M25 mismatch re-run (Mac, run lock, `scratchpad/m25-rerun.sh`, the batch3 recipe of 01:05 UTC against the `ledger-fixes` build: one node with real PoW, genesis bits 0x1f010000, `pow_day_ms` 86,400,000, ports 29410 to 29412, `--devnet-suffix=931`; a control miner, then a miner started with `IGNEUM_POW_DAY_MS=1440000`; 2 CPU threads, 60 s each; the dead `timestamp_deviation_tolerance` field stripped from the override, which the new node refuses).** + +| Miner | Environment | Day it built for | Found | Rejected | Node `PoW accepted` / `PoW rejected` | +|---|---|---|---|---|---| +| ctl | none | 20,731 (the node's) | 65 | 0 | 130 / 0 over both runs | +| mis | `IGNEUM_POW_DAY_MS=1440000` | 20,731 (installed from the template: `day 86400000 ms`) | 65 | 0 | | + +Reading: on `release-0.3.6` the miner already takes the day from the template (G12), so the batch3 failure of 01:05 UTC (0 accepted, 4 rejected as `BlockInvalid`) does not reproduce on this line; the fix adds the live day length to `next_pair` and the day-naming rejection text, which this run could not trigger. + +**M28, the NVRTC worker's CPU emulation (Mac, build slot, `proto-cuda/nvrtc/emu/test.sh`, 19:50 UTC).** Packs A and B stamped as the miner stamps them (`kernel_sha256` over the six kernel files); pack T = pack A with one line appended to `kernel_bound.cu` after the stamp. + +| Case | Result | +|---|---| +| `--check` pack A | `check PASS` in 1,445 ms; self-test PASS (cache head, last line, FNV-1a 64; dataset head, word [268435455], 64 samples; 96 of 96 vector lanes); 2 `source check PASS` lines | +| `--check` pack T (tampered after the stamp) | refused: `kernel_bound.cu does not hash to the value the miner derived from the seed (program.json kernel_sha256): tampered or stale pack`; 0 `source check PASS` lines (nothing reached the compiler) | +| `--serve` A, prepare B, swap | PASS: ready + prepare 1; 64 + 64 + 32 found on A, B prepared with self-test PASS, swapped, 32 found on B, job 5 refused (no pair); 15 sampled hashes equal to `igneum-pow hash-bound`; 4 `source check PASS` in serve, 2 in check | + +Owed: the tampered pack against the real worker on PC 2's RTX 5090 (the job tooling runs cargo suites, not a script); `pf_load` is shared by inclusion, so the refusal is the same code path. + +**X22 app test (Mac, build slot, `cargo test pack_directories_are_per_card` in `app/igneum-app`).** 1 passed, 0 failed (76 filtered). + +**Not run.** The X20 fast-time simnet timing (a simnet chain is a few hundred blocks; the unit test's step count is the evidence, the 10^6-block chain is still the owed experiment) and the X19 two-node skew run (needs an `attack-switches` build of the node). + diff --git a/docs/fud-ledger.md b/docs/fud-ledger.md index 776f2f5f2..4c3ac50aa 100644 --- a/docs/fud-ledger.md +++ b/docs/fud-ledger.md @@ -173,6 +173,8 @@ Answer: Correct, and this is the most dangerous entry in the ledger. With the ac Evidence: arithmetic above (not yet in `sim/`); design doc, hostile review table row "No stake exists in the first month". Simulation of the launch month: not yet. +Fix (5 October 2026, night), the harness text only: `tools/finality-attacks/run.mjs` names the 2/3-of-total floor in the s6 comments and criterion and in the s5 result line, where it still said 56.7%; the scenario logic is untouched. Branch `ledger-fork`. + ### F2. The two-hour presence window is an eclipse vector "Cut the big pools' vote gossip for two hours, not their blocks, and the remaining keys become 100% of active weight. A faction with a fifth of the weight locks alone. You chose liveness over safety and called it a feature." @@ -1334,12 +1336,14 @@ Evidence: spec 5.1; `docs/design/execution-layer.md` 4.1, 4.3, 9.1. Review id R3 ### P15. RPC blocks are segments, so `gasUsed` can exceed `gasLimit` "A segment holds up to 180 blocks and your RPC block is the segment. Indexers assert `gasUsed <= gasLimit`." -Status: Open, minor, extends R10. Sweep (5 October 2026): confirmed in code on the `proving` branch (`igneum/exec/src/rpc.rs:355-356`: `gasLimit` is the single-block `BLOCK_EXECUTION_GAS_LIMIT` while `gasUsed` is the segment record's total). Not exercised: a segment over the limit needs about 1,400 transfers inside 180 blocks. Fix row in section 2.5. +Status: Fixed on a branch, pending merge (5 October 2026, night): fork branch `ledger-fixes` b6f381e2, suites PC 2 jobs `build-20261005-200049` (kaspa-consensus 95 passed, 0 failed, 3 ignored) and `build-20261005-200606` (kaspa-consensus-core 101/0, kaspa-mining 52/0, igneum-exec 13/0, igneum-miner 17/0, kaspa-p2p-flows 32 passed 1 failed: the new clock-skew test's own arithmetic, corrected in bd1b676a and 33/0 on the Mac), both on 815c7d64; the tip is bd1b676a. Was: Open, minor, extends R10. Sweep (5 October 2026): confirmed in code on the `proving` branch (`igneum/exec/src/rpc.rs:355-356`: `gasLimit` is the single-block `BLOCK_EXECUTION_GAS_LIMIT` while `gasUsed` is the segment record's total). Not exercised: a segment over the limit needs about 1,400 transfers inside 180 blocks. Fix row in section 2.5. Answer: Correct; spec 7.1 states that a segment's total can exceed `gaslimit`. Fix: report the segment's limit as `k x B_e` in the RPC block, or document the invariant break for Blockscout (R10). Evidence: spec 7.1; `docs/design/execution-layer.md` 8.2. Review id R3.12. +Fix (5 October 2026, night): `igneum/exec/src/rpc.rs` reports `gasLimit` as k x `BLOCK_EXECUTION_GAS_LIMIT` for a k-block segment (k = the record's mergeset length, at least 1), beside the segment's `gasUsed`, so `gasUsed <= gasLimit` holds for an indexer; the per-block limit and k sit under `igneum.blockGasLimit` and `igneum.segmentBlocks`. Unit test `rpc_block_gas_limit_is_the_segment_limit` (a three-block segment with `gasUsed` over one block's limit). Fork commit b6f381e2. + ### E9. The specification's year is 365 days; the code's is 365.25 "Spec 2.5: 31,536,000 DAA seconds a year, halving every 63,072,000, about 31.7098 IGN a second. `igneum.rs`: 31,557,600, 63,115,200, 31.688 IGN. Which one is the money?" @@ -1406,12 +1410,14 @@ Evidence: spec 8.2. Review id R3.21. ### G10. The signalling default on first run "8.3 says the default is the choice the user last made. On first run there is none." -Status: Open, minor. Sweep (5 October 2026): the app carries no signalling control today (`grep -i signal app/igneum-app/src`: none), so nothing is signalled on first run; the rule binds when the control is built. +Status: Rule written (5 October 2026): spec 8.3 item 2; the control is unbuilt in the app. Was: Open, minor. Sweep (5 October 2026): the app carries no signalling control today (`grep -i signal app/igneum-app/src`: none), so nothing is signalled on first run; the rule binds when the control is built. Answer: Correct. Fix: first run signals nothing until the user chooses, shown in the interface. Evidence: spec 8.3 item 2. Review id R3.22. +Fix (5 October 2026, night): `docs/spec/08-client-security.md` 8.3 item 2 now ends: on first run there is no last choice, the client signals nothing until the user chooses, and the interface shows that nothing is being signalled. The app carries no signalling control today, so the rule binds the control when it is built; nothing to test until then. Branch `ledger-fork`. + ### X12. Tonight's numbers are quoted before they are logged, and the worker efficiency gap is unexplained "The 50x step, the trough near 9 million, 5.5 blocks a second and two-thirds efficiency over eight processes are in the brief and not in the bench-log. And the chain saw 70 to 83 MH/s from a card that benches 229, which is a third, not two thirds." @@ -1886,30 +1892,36 @@ Evidence: the first run's `/tmp/igneum-redteam-fin/n0/node.log` parse line (kept ### X19. Operational knobs and silences in the shipped node "A slow-clock node disconnects every peer on every relayed block and never says why; the handshake's `time_offset` is computed and unused; `IGNEUM_ATTACK_TS_OFFSET_MS` and `IGNEUM_POW_STRIKES` are compiled into the live binary; `timestamp_deviation_tolerance` is dead and still accepted." -Status: Open, minor (4 October 2026). Sweep (5 October 2026): the app half is done (bench-log, "a node 60 s behind the clock is silently dead": the node card shows the skew with a Sync clock button, Start is blocked past 10 s, checked with `IGNEUM_APP_FAKE_SKEW=-60`); the node's one WARN line is filed in `docs/plans/node-changes.md` section 1 and not written; `faketime` is not installed on this Mac, so the node experiment was not run. +Status: Fixed on a branch, pending merge (5 October 2026, night): fork branch `ledger-fixes` fc0cb938, suites PC 2 jobs `build-20261005-200049` (kaspa-consensus 95 passed, 0 failed, 3 ignored) and `build-20261005-200606` (kaspa-consensus-core 101/0, kaspa-mining 52/0, igneum-exec 13/0, igneum-miner 17/0, kaspa-p2p-flows 32 passed 1 failed: the new clock-skew test's own arithmetic, corrected in bd1b676a and 33/0 on the Mac), both on 815c7d64; the tip is bd1b676a. Was: Open, minor (4 October 2026). Sweep (5 October 2026): the app half is done (bench-log, "a node 60 s behind the clock is silently dead": the node card shows the skew with a Sync clock button, Start is blocked past 10 s, checked with `IGNEUM_APP_FAKE_SKEW=-60`); the node's one WARN line is filed in `docs/plans/node-changes.md` section 1 and not written; `faketime` is not installed on this Mac, so the node experiment was not run. Answer: Correct. `blockrelay/flow.rs:201`, `router.rs:215-224`, `flow_context.rs:819`, `peer.rs:13`; `virtual_processor/processor.rs:1663-1670` (merged in `baa8bc8a`); `pow_guard.rs:28`. The 10 s bound itself is right in shape (two constants, both directions, not overridable). Fix: one WARN from `time_offset` at handshake, the attack switches behind a feature flag, the dead field refused. Review ids R4.1.6 to R4.1.8. Evidence: the files above. Experiment: `igneumd` under `faketime -60s` logs one clear line and does not churn. +Fix (5 October 2026, night), the node half, fork commit fc0cb938. (a) `protocol/flows/src/flowcontext/clock_skew.rs`: the relay path (`charge_pow_strikes`, which sees every block result) keeps the (header timestamp minus local now) of the headers refused as `TimeTooFarIntoTheFuture` over the last minute and logs ONE WARN per minute, `clock skew: local time is N s behind the median peer block time (blocks are being refused; set the clock)`, with the count refused and the tolerance; the per-block error text is unchanged, so the app's parser keeps working (`docs/plans/node-changes.md` section 1). The mirror case (local clock ahead) is refused on the peers and not visible here; not written. (b) `IGNEUM_ATTACK_TS_OFFSET_MS` (both template-time sites) and `IGNEUM_POW_STRIKES` (the PoW guard) are read only under the new cargo feature `attack-switches` (`kaspad` forwards it to `kaspa-consensus`, `kaspa-mining`, `kaspa-p2p-flows`); a release build returns the clock's time and the default strike limit and never reads the environment. (c) `OverrideParams` refuses a file that carries `timestamp_deviation_tolerance` at load, with a message naming the field, and never writes it; `infra/fast-time/override-60x.json` on branch `ledger-fork` no longer carries it (a node from this line refuses the master copy of that file until the two merge together). Tests: `clock_skew` unit tests (a 60 s skew gives one line, nothing more inside the minute, the median after it), `release_build_ignores_the_strike_env` (compiled only without the feature), the override refusal in `params.rs`. The two-node fast-time run with one node's clock offset was not run (the offset needs an `attack-switches` build of the node, a second build); the WARN path is covered by the unit test on the monitor and by review of the hook, which runs on every relayed block result. + ### X20. Cold-sync checkpoint determination is indices times chain length "A fresh node starts `next_index` at 0 and walks the selected chain from the sink for every index. At mainnet length it never finishes its first resolution." -Status: Open, minor now (4 October 2026). Sweep (5 October 2026): unchanged on `finality-fixes` (`processes/finality.rs:204` starts `next_index` at 1 and walks the chain per index). A 10^6-block simnet is 11.6 days of chain at 1 block/s and was not built. Fix row in section 2.5. +Status: Fixed on a branch, pending merge (5 October 2026, night): fork branch `ledger-fixes` 9e109c65, suites PC 2 jobs `build-20261005-200049` (kaspa-consensus 95 passed, 0 failed, 3 ignored) and `build-20261005-200606` (kaspa-consensus-core 101/0, kaspa-mining 52/0, igneum-exec 13/0, igneum-miner 17/0, kaspa-p2p-flows 32 passed 1 failed: the new clock-skew test's own arithmetic, corrected in bd1b676a and 33/0 on the Mac), both on 815c7d64; the tip is bd1b676a. Was: Open, minor now (4 October 2026). Sweep (5 October 2026): unchanged on `finality-fixes` (`processes/finality.rs:204` starts `next_index` at 1 and walks the chain per index). A 10^6-block simnet is 11.6 days of chain at 1 block/s and was not built. Fix row in section 2.5. Answer: Correct. `processes/finality.rs:413-421`. About 3 x 10^12 store reads at a 10^7-block chain, approximate. Fix: start from the last certified index carried in headers, walk once. Review id R4.1.10. Evidence: the file above. Experiment: a fresh node against a 10^6-block simnet chain, time to first resolution. +Fix (5 October 2026, night): `consensus/src/processes/finality.rs` `on_virtual_changed` resolves every index the sink can determine in ONE descending walk of the selected chain (`chain_blocks_at`, targets highest first), where it walked from the sink once per index. Cost per cold sync from indices x chain length store reads to one pass over the chain. Locked indices above the next one (F24) are passed over as before. On this line headers carry no certified index (certificates ride in coinbase payloads and are ingested as blocks arrive), so the walk starts at `next_index`; the ledger's "start from the last certified index carried in headers" has no field to start from and is noted here, not built. Unit test `cold_sync_determines_every_index_in_one_walk`: 150-block chain, interval 5, depth 2, 29 indices; the one walk takes at most 150 steps against a per-index sum over 10 times larger, and names the same block as `chain_block_at` for every index. The fast-time simnet timing was not run: a simnet chain is a few hundred blocks, where the two walks differ by milliseconds; the 10^6-block chain of the experiment line is still owed. Fork commit 9e109c65. + ### M25. The miner takes the day length from its environment, and the schedule global can tear "The template carries epoch and lead but not the day; the miner reads `IGNEUM_POW_DAY_MS` from its environment. And `install_pow_schedule` stores day, lead, epoch while `pow_schedule` loads epoch, lead, so a miner switched between schedules can wrap `pow_epoch_seed_score`." -Status: Open, minor (4 October 2026). Sweep (5 October 2026): the schedule now rides in the override file as consensus params (`a5ef8b07`, in `devnet-v4`); the mismatch run was done (5 October 2026, 01:05 UTC, `scratchpad/runs/batch3.sh` under the run lock: one `finality-fixes` node with real PoW at genesis bits 2^16 on ports 29410 to 29412, `pow_day_ms` 86,400,000 in its override file; a control miner, then a miner started with `IGNEUM_POW_DAY_MS=1440000`, 2 CPU threads, 60 s each). Confirmed: the miner still takes the day from its environment (it built its cache for day 1,243,862 while the node and the control miner were on day 20,731); control 7 blocks accepted, 0 rejected; mismatched 0 accepted, 4 rejected, each as `Reject(BlockInvalid)` on the miner and `block has invalid proof-of-work` on the node, so the miner says that it is rejected and not why. Fix row 127: the day length in the template beside the epoch and lead, and a rejection reason that names the day. The environment fallback is G12, owned by the fud-consensus branch. +Status: Fixed on a branch, pending merge (5 October 2026, night): fork branch `ledger-fixes` 3d4ec451 and fc0cb938, suites PC 2 jobs `build-20261005-200049` (kaspa-consensus 95 passed, 0 failed, 3 ignored) and `build-20261005-200606` (kaspa-consensus-core 101/0, kaspa-mining 52/0, igneum-exec 13/0, igneum-miner 17/0, kaspa-p2p-flows 32 passed 1 failed: the new clock-skew test's own arithmetic, corrected in bd1b676a and 33/0 on the Mac), both on 815c7d64; the tip is bd1b676a. Was: Open, minor (4 October 2026). Sweep (5 October 2026): the schedule now rides in the override file as consensus params (`a5ef8b07`, in `devnet-v4`); the mismatch run was done (5 October 2026, 01:05 UTC, `scratchpad/runs/batch3.sh` under the run lock: one `finality-fixes` node with real PoW at genesis bits 2^16 on ports 29410 to 29412, `pow_day_ms` 86,400,000 in its override file; a control miner, then a miner started with `IGNEUM_POW_DAY_MS=1440000`, 2 CPU threads, 60 s each). Confirmed: the miner still takes the day from its environment (it built its cache for day 1,243,862 while the node and the control miner were on day 20,731); control 7 blocks accepted, 0 rejected; mismatched 0 accepted, 4 rejected, each as `Reject(BlockInvalid)` on the miner and `block has invalid proof-of-work` on the node, so the miner says that it is rejected and not why. Fix row 127: the day length in the template beside the epoch and lead, and a rejection reason that names the day. The environment fallback is G12, owned by the fud-consensus branch. Answer: Correct. `igneum/miner/src/main.rs:590-596`; `consensus/core/src/igneum.rs:156-172, 198-204`. Fix: the day length in the template; one atomic struct swap. Review id R4.1.9. Evidence: the files above. Experiment: `igneum-miner` with `IGNEUM_POW_DAY_MS=1440000` against a default node accepts zero blocks and says why. +Fix (5 October 2026, night). Read on `release-0.3.6` first: the template already carries `day_ms`, `day_index` and `next_day_index` beside the epoch fields (`RpcPowEpochInfo`, G12 merged into 0.3.6), the miner installs the node's three values from every template and never reads `IGNEUM_POW_DAY_MS` (only the node does, on devnet and simnet). Two gaps remained and are closed: `next_pair` in `igneum/miner/src/main.rs` took the devnet constant `DAY_MS` for the day-change lead instead of the live `pow_day_ms()` (wrong on any non-default day length, fork commit 3d4ec451); and the node's rejection named nothing (`PoW rejected ... (daa, nonce)`, `RuleError::InvalidPoW` bare). Now `RuleError::InvalidPoW { header_day, engine_day, day_ms }` reads `block has invalid proof-of-work (header day D from its timestamp at M ms per day, engine day E; a miner on another day length or epoch seed computes another dataset)` and the log line adds the epoch seed (fork commit fc0cb938). Mismatch re-run (the batch3 recipe: one `ledger-fixes` node with real PoW at genesis bits 0x1f010000 on ports 29410 to 29412, `pow_day_ms` 86,400,000, a control miner then a miner started with `IGNEUM_POW_DAY_MS=1440000`, 2 CPU threads, 60 s each, under the run lock): the miner started with `IGNEUM_POW_DAY_MS=1440000` installed the node's schedule from the first template (`node PoW schedule: 60 DAA per epoch, lead 10, day 86400000 ms`), built its cache for day 20,731 like the node and the control, and was accepted: control 65 found, 0 rejected; mismatched 65 found, 0 rejected; node 130 `PoW accepted`, 0 `PoW rejected` (19:07 to 19:09 UTC). The environment is ignored, so the rejection path was not exercised live; its text is the unit-level evidence (`RuleError` message and the log line). Bench-log, "5 October 2026 (night), ledger close round 1". + ### M26. The interval fault guard freezes its baseline and loops "On a trip you skip the STATUS print, so the baseline it would have updated stays frozen, and you roll the counters back to it. Any healthy rate over ten times a slow first interval trips again every interval, forever, with no STATUS line and no `faults=` for the app to read." @@ -1935,12 +1947,14 @@ Evidence: the files above. Experiment: a fast-time simnet node patched to flip ` ### M28. The kernel text is bound only to its own directory "The worker compiles whatever `kernel_bound.cu` it finds in a directory whose `seeds.txt` matches; the self-test checks the GPU against a `vectors.h` from the same directory. Nothing commits the text to what the generator would emit for the seed. 'Source check PASS' is a prefix equality." -Status: Open, minor (4 October 2026). The writer is the local miner and the directory is the user's own, so no privilege boundary is crossed. Sweep (5 October 2026): no kernel hash in `program.json` on any branch (`git grep -i 'kernel_hash|program_hash'` on `miner-reliability` finds nothing); the tampered-pack test needs a GPU worker. Fix row in section 2.5. +Status: Fixed on a branch, pending merge (5 October 2026, night): fork branch `ledger-fixes` 3d4ec451 (miner) and branch `ledger-fork` (workers), suites PC 2 jobs `build-20261005-200049` (kaspa-consensus 95 passed, 0 failed, 3 ignored) and `build-20261005-200606` (kaspa-consensus-core 101/0, kaspa-mining 52/0, igneum-exec 13/0, igneum-miner 17/0, kaspa-p2p-flows 32 passed 1 failed: the new clock-skew test's own arithmetic, corrected in bd1b676a and 33/0 on the Mac), both on 815c7d64; the tip is bd1b676a. Was: Open, minor (4 October 2026). The writer is the local miner and the directory is the user's own, so no privilege boundary is crossed. Sweep (5 October 2026): no kernel hash in `program.json` on any branch (`git grep -i 'kernel_hash|program_hash'` on `miner-reliability` finds nothing); the tampered-pack test needs a GPU worker. Fix row in section 2.5. Answer: Correct. `packfile.h:~262-283, 306-345`; `worker.cpp:414-416, 596-620`; `emu/test.sh:57-66`. The CPU re-check (`main.rs:1250-1256`) stops a wrong program from earning, not from running. Fix: a hash of the emitted kernel in `program.json`, derived from the seed by the miner and checked by both workers; one sentence in the docs that the chain commits to the seed, not the text. Review id R4.2.4. Evidence: the files above. Experiment: a tampered pack with matching vectors must be refused. +Fix (5 October 2026, night). The miner stamps `program.json` with `kernel_sha256`, the SHA-256 of every kernel file it emitted from the seed (`kernel.cu`, `kernel_bound.cu`, `kernel.cl`, `kernel_bound.cl`, `program.h`, `memhard.h`), on `export-pack` and on every `prepare` pack (`stamp_kernel_hashes`, fork commit 3d4ec451). Both workers check the text before anything compiles: `proto-cuda/nvrtc/packfile.h` `pf_check_kernel_hashes` (a SHA-256 in C, checked against python's on three vectors) runs inside `pf_load`, so the NVRTC worker's `--pack`, `--check` and `prepare` paths and the OpenCL worker's `--pack` start refuse a pack whose files do not hash to the stamp, or that carries no stamp; the OpenCL `prepare` path now calls `pf_load` and fails the prepare with the reason (before it only skipped the self-test). One sentence in `proto-cuda/README.md`: the chain commits to the seed, not to the text. Tests: the miner's `tampered_kernel_text_is_refused_by_the_stamp` (stamp, verify, re-stamp is idempotent, one constant edited in `kernel_bound.cu` is refused with the file named, a pack without the stamp is refused); the NVRTC worker's CPU emulation test (`proto-cuda/nvrtc/emu/test.sh`, Mac, 5 October 2026, 19:50 UTC) stamps packs A and B as the miner does and adds pack T, pack A with a line appended to `kernel_bound.cu` after the stamp: pack A `check PASS` as before, pack T refused with `kernel_bound.cu does not hash to the value the miner derived from the seed` and no `source check PASS` line (the text never reached the compiler), the serve round unchanged (`PASS: ready + prepare 1; 64 + 64 + 32 found on pack A, prepared B ... 32 found on B`, 15 sampled hashes equal to `igneum-pow hash-bound`). Owed: the same tampered pack against the real worker on PC 2's RTX 5090 (the `build` job tooling compiles crates and runs cargo suites, it does not run a script on the PC); the vectors still match on pack T by construction, so the only thing the GPU run adds is that the real NVRTC path shares `pf_load`, which it does by inclusion. A pack from `igneum-pow export` (the CLI, not the miner) carries no stamp and is refused by the one-click workers until it is stamped; the emu test shows the stamping. + ### X21. A wrong program burns power with a green rate "The CPU re-check counts mismatches and does nothing: no threshold, no stop. The app reads `hash`, `now`, `template_age` and `synced` from STATUS and nothing else, so `mismatched=` and `WORKER FAULT` never reach the card." @@ -1954,12 +1968,14 @@ Evidence: the files above. Experiment: one constant edited in `kernel_bound.cu` ### X22. Worker restart paths, minor "Two miners truncate the same `packs\devnet` while workers read it; every restart begins on the stale first pack; the restart loop has no cap; the 60 s stall guard wraps the self-heal build; a node can crash the miner through an `expect`; a worker without prepare support loops on export during IBD." -Status: Open, minor (4 October 2026). Sweep (5 October 2026): two of the six fixed on `miner-reliability` (`501363e0`: a dead worker's stdin no longer spins the fill loop; `945153ab`: a silent worker still runs the guards and STATUS); the shared `packs\devnet` truncation, the stale first pack, the `expect` crash and the IBD export loop remain; restart timing needs the PCs. +Status: Fixed on a branch, pending merge (5 October 2026, night): fork branch `ledger-fixes` 3d4ec451 (miner) and branch `ledger-fork` (app), suites PC 2 jobs `build-20261005-200049` (kaspa-consensus 95 passed, 0 failed, 3 ignored) and `build-20261005-200606` (kaspa-consensus-core 101/0, kaspa-mining 52/0, igneum-exec 13/0, igneum-miner 17/0, kaspa-p2p-flows 32 passed 1 failed: the new clock-skew test's own arithmetic, corrected in bd1b676a and 33/0 on the Mac), both on 815c7d64; the tip is bd1b676a. Was: Open, minor (4 October 2026). Sweep (5 October 2026): two of the six fixed on `miner-reliability` (`501363e0`: a dead worker's stdin no longer spins the fill loop; `945153ab`: a silent worker still runs the guards and STATUS); the shared `packs\devnet` truncation, the stale first pack, the `expect` crash and the IBD export loop remain; restart timing needs the PCs. Answer: Correct on each: `engine.rs:~2047-2055` and `emit.rs:1156-1163`; `main.rs:1206-1228`; `worker.cpp:684-703`; `main.rs:~581, 893`; `main.rs:1373-1380` with `engine.rs:~1405-1411` (the 14:20 export during IBD, bench-log). Each costs a restart, none loses the run. Review ids R4.2.5 to R4.2.10. Evidence: the files above. Experiment: time from worker restart to first accepted job on both PCs. +Fix (5 October 2026, night), the four remaining paths. (1) The shared `packs\devnet` truncation: the app exports one pack directory per card, `packs\devnet-` (`pack_dir_name`, `export_pack`, `build_worker_from_source`, `miner_args` in `app/igneum-app/src/engine.rs`), so an export for one card never truncates what another card's worker is reading; app test `pack_directories_are_per_card` (passed on the Mac, 1 of 1). (2) The stale first pack: a restarted worker is spawned with `--pack` pointing at the pack of the CURRENT seeds under `--prepare-packs` when the miner wrote one (`worker_args_with_pack`, `pack_dir_for`), not at the launcher's first pack, which cost a build of the old program and then a foreground self-heal build of the current one; unit test `restart_worker_args_take_the_current_pack`. (3) The `expect` crash: `template()` skips a template whose block does not convert and the legacy seed walk returns `None` on any RPC error or missing verbose data (the three call paths skip the template, the two one-shot commands exit with a message); no `expect` on the node's answers remains in the mining loops (the ones left are on the user's own address and on process start). (4) The IBD export loop: a worker without prepare support no longer makes the miner exit 42 on a seed change while the template is not synced (the change is printed once, `last_seeds` keeps the worker's pair, so the first change after the sync completes still exits 42 once); the app waits 60 s before restarting a miner that exited 42 while the node is not synced. Fork commit 3d4ec451 (miner), branch `ledger-fork` (app). The restart timing on both PCs is still owed (hardware). + ### E16. The 20% pool is burned on the live chain, and the text says it pays provers "Your coinbase sends the 20% to an OP_RETURN tagged `igneum-proving-pool-v0`. Your own comment says provably burned. Your cap test counts it as supply. Your litepaper says, present tense, that it pays a standing prover population." diff --git a/docs/spec/08-client-security.md b/docs/spec/08-client-security.md index ab7fd7112..d668f98c8 100644 --- a/docs/spec/08-client-security.md +++ b/docs/spec/08-client-security.md @@ -21,7 +21,7 @@ The two findings it answers. An app that auto-updates on ten thousand machines i ## 8.3 The client cannot change consensus 1. Consensus rules change only through the upgrade path of section 5.7: new code activates when 90% of blue blocks in the signalling window carry the signal. A client release carries code; it carries no activation. The chain, not the app, decides when a rule takes effect, and 90% of miners have to say so in their blocks. -2. The client MUST NOT set a signalling bit without a user choice shown in the interface, and the default MUST be the choice the user last made, never the release's preference. +2. The client MUST NOT set a signalling bit without a user choice shown in the interface, and the default MUST be the choice the user last made, never the release's preference. On first run there is no last choice: the client signals nothing until the user chooses, and the interface shows that nothing is being signalled (rule written 5 October 2026, ledger G10; the control is not yet built in the app). 3. A release that changed a consensus rule without an activation signal would fork its users off the chain, which is the only thing a hostile release key can do to consensus, and it is visible to everyone within one block. ## 8.4 Distribution diff --git a/infra/fast-time/README.md b/infra/fast-time/README.md index c02d3c0ca..fbbc3ab49 100644 --- a/infra/fast-time/README.md +++ b/infra/fast-time/README.md @@ -13,7 +13,9 @@ igneumd --devnet --devnet-suffix=950 --override-params-file=infra/fast-time/over ``` The file is complete: every field `OverrideParams` accepts is present (`consensus/core/src/config/params.rs`), so a -reader sees the whole profile in one place. Merge it with `skip_proof_of_work: true` for a network without miners +reader sees the whole profile in one place. `timestamp_deviation_tolerance` is not among them: the field is dead +(nothing in consensus reads it) and a node from the `ledger-fixes` line refuses a file that carries it (ledger X19, +5 October 2026). Merge it with `skip_proof_of_work: true` for a network without miners (the harnesses do this), or lower `genesis_bits` for CPU miners (the devnet value 0x1d100000 is the GPU difficulty). Harnesses: `node tools/finality-attacks/run.mjs s3 --fast-time`, `node tools/harness/run.mjs s3 s4 --fast-time` diff --git a/infra/fast-time/override-60x.json b/infra/fast-time/override-60x.json index d5db07547..0269c1b80 100644 --- a/infra/fast-time/override-60x.json +++ b/infra/fast-time/override-60x.json @@ -1,5 +1,4 @@ { - "timestamp_deviation_tolerance": 132, "past_median_time_window_size": 27, "difficulty_window_size": 661, "min_difficulty_window_size": 150, diff --git a/proto-cuda/README.md b/proto-cuda/README.md index 61af86168..746b1760f 100644 --- a/proto-cuda/README.md +++ b/proto-cuda/README.md @@ -29,7 +29,11 @@ compiler, redistributable, shipped next to the exe with `nvrtc-builtins64_128.dl `LoadLibrary`/`GetProcAddress` (`nvrtc/cuda_api.h`), reads a pack directory at run time (`nvrtc/packfile.h`: program.h, seeds.txt, vectors.h), hands NVRTC the pack's `kernel.cu` and `kernel_bound.cu` up to the host launch wrappers with `program.h` and `memhard.h` as named headers, byte for byte, loads the cubin through the driver, fills the cache, builds -the dataset and self-tests both plus the three vector warps against `vectors.h` before it serves a job. It speaks the +the dataset and self-tests both plus the three vector warps against `vectors.h` before it serves a job. The chain +commits to the seed, not to the text: `igneum-miner` writes the SHA-256 of every kernel file it emitted from the seed +into the pack's `program.json` (`kernel_sha256`, 5 October 2026, ledger M28), and both workers refuse a pack whose +files no longer hash to those values before anything is compiled (`packfile.h`, `pf_check_kernel_hashes`), so an +edited or stale kernel text never runs. It speaks the same `--serve` protocol as `host.cu` (jobs, `prepare` in the background for the hourly swap, `found`/`done`); a job on seeds it has no pair for makes it look for the miner's pack by `seeds.txt` and build it in the foreground. `host.cu` stays the ahead-of-time harness and the launcher's fallback when the prebuilt worker is missing and a toolkit exists. diff --git a/proto-cuda/nvrtc/emu/test.sh b/proto-cuda/nvrtc/emu/test.sh index 642ccb0cd..e832e9f80 100755 --- a/proto-cuda/nvrtc/emu/test.sh +++ b/proto-cuda/nvrtc/emu/test.sh @@ -33,6 +33,30 @@ B_DAY="69676e65756d2d6461792ffa51000000000000" "$POW" export --epoch-hex "$B_EPOCH" --day-hex "$B_DAY" --out "$OUT/pack-b" > "$OUT/export-b.log"; head -2 "$OUT/export-b.log" printf 'epoch_seed_hex %s\nday_seed_hex %s\nday_index 1\n' "$B_EPOCH" "$B_DAY" > "$OUT/pack-b/seeds.txt" +# Ledger M28 (5 October 2026): igneum-miner stamps program.json with the SHA-256 of the kernel files it wrote from the +# seed, and the workers refuse a pack whose files do not hash to it. Packs from igneum-pow export carry no stamp, so +# the test stamps them the way the miner does (one "kernel_sha256" line after the opening brace). +stamp_pack() { # pack dir + python3 - "$1" <<'PYS' +import hashlib, sys, os +d = sys.argv[1] +names = ["kernel.cu", "kernel_bound.cu", "kernel.cl", "kernel_bound.cl", "program.h", "memhard.h"] +entries = ['"%s": "%s"' % (n, hashlib.sha256(open(os.path.join(d, n), 'rb').read()).hexdigest()) for n in names if os.path.exists(os.path.join(d, n))] +p = os.path.join(d, "program.json") +lines = [l for l in open(p).read().split("\n") if not l.strip().startswith('"kernel_sha256')] +i = next(k for k, l in enumerate(lines) if l.strip() == "{") +lines[i + 1:i + 1] = [' "kernel_sha256": {%s},' % ", ".join(entries), + ' "kernel_sha256_rule": "SHA-256 of each file as the miner wrote it from the seed; the chain commits to the seed, not to this text",'] +open(p, "w").write("\n".join(lines)) +PYS +} +stamp_pack "$OUT/pack-a" +stamp_pack "$OUT/pack-b" +# Pack T: pack A with one constant of kernel_bound.cu edited after the stamp (the vectors still match, the text does +# not). It must be refused before anything compiles. +cp -R "$OUT/pack-a" "$OUT/pack-t" +printf '\n// one constant edited after the stamp (ledger M28 test)\n' >> "$OUT/pack-t/kernel_bound.cu" + # The kernels as the emulation runs them: the launch syntax rewritten (as emu/emu.sh does), each pack in its own namespace. emu_kernel() { # pack dir, namespace, file stem { echo '#include '; echo '#include '; echo "namespace $2 {" @@ -56,6 +80,12 @@ echo "== --check pack A" grep -q '^check PASS' "$OUT/check-a.log" grep -c 'source check PASS' "$OUT/check-a.log" | grep -qx 2 +echo "== --check pack T (ledger M28): a kernel edited after the miner's stamp must be refused, with the file named" +if "$OUT/igneum-worker-cuda-emu" --check --pack "$OUT/pack-t" > "$OUT/check-t.log" 2>&1; then echo "FAIL: the tampered pack was accepted:"; cat "$OUT/check-t.log"; exit 1; fi +grep -q 'kernel_bound.cu does not hash to the value the miner derived from the seed' "$OUT/check-t.log" || { echo "FAIL: wrong refusal:"; cat "$OUT/check-t.log"; exit 1; } +if grep -q 'source check PASS' "$OUT/check-t.log"; then echo "FAIL: the tampered text reached the compiler"; exit 1; fi +echo "tampered pack refused: $(grep -o 'kernel_bound.cu does not hash[^(]*' "$OUT/check-t.log" | head -1)" + echo "== the annotation rule: without -default-device the stand-in must reject program.h's igneum_launch_* declarations as host code (what the RTX 5090 reported on 4 October 2026)" if IGNEUM_EMU_NO_DEFAULT_DEVICE=1 "$OUT/igneum-worker-cuda-emu" --check --pack "$OUT/pack-a" > "$OUT/check-nodefault.log" 2>&1; then echo "FAIL: the unannotated program.h declarations were not flagged"; exit 1; fi grep -q 'program.h(4[0-9]): cudaError_t igneum_launch_cache_fill.*host functions are not allowed in JIT mode' "$OUT/check-nodefault.log" || { echo "FAIL: wrong rejection:"; cat "$OUT/check-nodefault.log"; exit 1; } diff --git a/proto-cuda/nvrtc/packfile.h b/proto-cuda/nvrtc/packfile.h index fa407f93c..23a598420 100644 --- a/proto-cuda/nvrtc/packfile.h +++ b/proto-cuda/nvrtc/packfile.h @@ -236,6 +236,100 @@ static int pf_seeds_line(const char* text, const char* key, char* out, size_t ca static int pf_fail(char* err, size_t cap, const char* msg) { if (err && cap) { strncpy(err, msg, cap - 1); err[cap - 1] = 0; } return 0; } +// ---- SHA-256 (FIPS 180-4), for the kernel text check of ledger M28 (5 October 2026) ---- +static const uint32_t pf_sha256_k[64] = { + 0x428a2f98u, 0x71374491u, 0xb5c0fbcfu, 0xe9b5dba5u, 0x3956c25bu, 0x59f111f1u, 0x923f82a4u, 0xab1c5ed5u, + 0xd807aa98u, 0x12835b01u, 0x243185beu, 0x550c7dc3u, 0x72be5d74u, 0x80deb1feu, 0x9bdc06a7u, 0xc19bf174u, + 0xe49b69c1u, 0xefbe4786u, 0x0fc19dc6u, 0x240ca1ccu, 0x2de92c6fu, 0x4a7484aau, 0x5cb0a9dcu, 0x76f988dau, + 0x983e5152u, 0xa831c66du, 0xb00327c8u, 0xbf597fc7u, 0xc6e00bf3u, 0xd5a79147u, 0x06ca6351u, 0x14292967u, + 0x27b70a85u, 0x2e1b2138u, 0x4d2c6dfcu, 0x53380d13u, 0x650a7354u, 0x766a0abbu, 0x81c2c92eu, 0x92722c85u, + 0xa2bfe8a1u, 0xa81a664bu, 0xc24b8b70u, 0xc76c51a3u, 0xd192e819u, 0xd6990624u, 0xf40e3585u, 0x106aa070u, + 0x19a4c116u, 0x1e376c08u, 0x2748774cu, 0x34b0bcb5u, 0x391c0cb3u, 0x4ed8aa4au, 0x5b9cca4fu, 0x682e6ff3u, + 0x748f82eeu, 0x78a5636fu, 0x84c87814u, 0x8cc70208u, 0x90befffau, 0xa4506cebu, 0xbef9a3f7u, 0xc67178f2u }; +#define PF_ROTR(x, n) (((x) >> (n)) | ((x) << (32 - (n)))) +static void pf_sha256_block(uint32_t h[8], const uint8_t* p) { + uint32_t w[64], a, b, c, d, e, f, g, hh, t1, t2; + int i; + for (i = 0; i < 16; ++i) w[i] = ((uint32_t)p[4 * i] << 24) | ((uint32_t)p[4 * i + 1] << 16) | ((uint32_t)p[4 * i + 2] << 8) | (uint32_t)p[4 * i + 3]; + for (i = 16; i < 64; ++i) { + uint32_t s0 = PF_ROTR(w[i - 15], 7) ^ PF_ROTR(w[i - 15], 18) ^ (w[i - 15] >> 3); + uint32_t s1 = PF_ROTR(w[i - 2], 17) ^ PF_ROTR(w[i - 2], 19) ^ (w[i - 2] >> 10); + w[i] = w[i - 16] + s0 + w[i - 7] + s1; + } + a = h[0]; b = h[1]; c = h[2]; d = h[3]; e = h[4]; f = h[5]; g = h[6]; hh = h[7]; + for (i = 0; i < 64; ++i) { + t1 = hh + (PF_ROTR(e, 6) ^ PF_ROTR(e, 11) ^ PF_ROTR(e, 25)) + ((e & f) ^ (~e & g)) + pf_sha256_k[i] + w[i]; + t2 = (PF_ROTR(a, 2) ^ PF_ROTR(a, 13) ^ PF_ROTR(a, 22)) + ((a & b) ^ (a & c) ^ (b & c)); + hh = g; g = f; f = e; e = d + t1; d = c; c = b; b = a; a = t1 + t2; + } + h[0] += a; h[1] += b; h[2] += c; h[3] += d; h[4] += e; h[5] += f; h[6] += g; h[7] += hh; +} +static void pf_sha256_hex(const uint8_t* data, size_t len, char out[65]) { + uint32_t h[8] = { 0x6a09e667u, 0xbb67ae85u, 0x3c6ef372u, 0xa54ff53au, 0x510e527fu, 0x9b05688cu, 0x1f83d9abu, 0x5be0cd19u }; + uint8_t tail[128]; + size_t i, rem = len % 64, pad; + uint64_t bits = (uint64_t)len * 8u; + for (i = 0; i + 64 <= len; i += 64) pf_sha256_block(h, data + i); + memcpy(tail, data + i, rem); + tail[rem] = 0x80; + pad = (rem < 56) ? 64 : 128; + memset(tail + rem + 1, 0, pad - rem - 1); + for (i = 0; i < 8; ++i) tail[pad - 1 - i] = (uint8_t)(bits >> (8 * i)); + pf_sha256_block(h, tail); + if (pad == 128) pf_sha256_block(h, tail + 64); + for (i = 0; i < 8; ++i) snprintf(out + 8 * i, 9, "%08x", h[i]); +} + +// Ledger M28 (5 October 2026): program.json carries "kernel_sha256": {"kernel.cu": "<64 hex>", ...}, stamped by the +// miner over the text it emitted from the seed (the chain commits to the seed, never to the text). Every file named +// there that exists in the pack must hash to its value, else the pack is refused: a kernel edited after the export, +// or a stale pack, never reaches the compiler. A pack without the stamp is refused (re-export it with the miner). +static int pf_check_kernel_hashes(const char* dir, char* err, size_t cap) { + char* json = pf_read_pack_file(dir, "program.json", NULL); + const char* p; + const char* end; + int checked = 0; + if (!json) return pf_fail(err, cap, "cannot read program.json (the kernel hashes live there; re-export the pack with the miner)"); + p = strstr(json, "\"kernel_sha256\""); + if (!p) { free(json); return pf_fail(err, cap, "program.json carries no kernel_sha256 (the miner stamps it; re-export the pack)"); } + p = strchr(p, '{'); + end = p ? strchr(p, '}') : NULL; + if (!p || !end) { free(json); return pf_fail(err, cap, "program.json kernel_sha256 is malformed"); } + while (p < end) { + char name[128], want[65], got[65], m[600]; + const char* q; + size_t n, flen; + uint8_t* bytes; + p = strchr(p, '"'); + if (!p || p >= end) break; + q = strchr(p + 1, '"'); + if (!q || q >= end) break; + n = (size_t)(q - p - 1); + if (n == 0 || n >= sizeof(name)) break; + memcpy(name, p + 1, n); name[n] = 0; + p = strchr(q + 1, '"'); + if (!p || p >= end) break; + q = strchr(p + 1, '"'); + if (!q || q >= end || q - p - 1 != 64) { free(json); snprintf(m, sizeof(m), "program.json kernel_sha256 entry for %s is not 64 hex characters", name); return pf_fail(err, cap, m); } + memcpy(want, p + 1, 64); want[64] = 0; + p = q + 1; + if (strchr(name, '/') || strchr(name, '\\')) continue; + bytes = (uint8_t*)pf_read_pack_file(dir, name, &flen); + if (!bytes) continue; /* a file the miner stamped but this pack does not carry (another worker's kernel) */ + pf_sha256_hex(bytes, flen, got); + free(bytes); + if (strcmp(got, want) != 0) { + free(json); + snprintf(m, sizeof(m), "%s does not hash to the value the miner derived from the seed (program.json kernel_sha256): tampered or stale pack", name); + return pf_fail(err, cap, m); + } + ++checked; + } + free(json); + if (checked == 0) return pf_fail(err, cap, "program.json kernel_sha256 names no file this pack carries"); + return 1; +} + // Reads program.h, seeds.txt (optional) and vectors.h (optional) of a pack directory. Returns 1 on success. static int pf_load(const char* dir, PfPack* pk, char* err, size_t cap) { char* prog; @@ -316,6 +410,8 @@ static int pf_load(const char* dir, PfPack* pk, char* err, size_t cap) { } free(vec); } + // ledger M28: the kernel text must be the text the miner emitted from the seed + if (!pf_check_kernel_hashes(dir, err, cap)) return 0; return 1; } diff --git a/proto-opencl/host.c b/proto-opencl/host.c index 457ec2090..79fc5774a 100644 --- a/proto-opencl/host.c +++ b/proto-opencl/host.c @@ -1070,8 +1070,14 @@ static void prepareRun(PrepareTask* t) { ServePair* p = (ServePair*)calloc(1, sizeof(ServePair)); cl_command_queue q = NULL; double tb; + PfPack pk; + char perr[512]; strncpy(p->epochHex, t->epochHex, 64); p->epochHex[64] = 0; strncpy(p->dayHex, t->dayHex, sizeof(p->dayHex) - 1); + /* ledger M28 (5 October 2026): the pack is read and checked (seeds, and the kernel text against the hashes the + * miner stamped in program.json) BEFORE anything is compiled; a refused pack fails the prepare with the reason, + * where before only the self-test was skipped */ + if (!pf_load(t->packDir, &pk, perr, sizeof(perr))) { snprintf(t->error, sizeof(t->error), "pack %s refused: %s", t->packDir, perr); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; } snprintf(path, sizeof(path), "%s/kernel_bound.cl", t->packDir); src = readFile(path, &srcLen); if (!src) { snprintf(t->error, sizeof(t->error), "cannot read %s", path); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; } diff --git a/tools/finality-attacks/run.mjs b/tools/finality-attacks/run.mjs index 3e276f58d..e8e7d21bc 100644 --- a/tools/finality-attacks/run.mjs +++ b/tools/finality-attacks/run.mjs @@ -143,14 +143,14 @@ async function s1() { // --------------------------------------------------------------------------------------------- // Scenario 6: partition with the floor. Two nodes over a proxy, 6 keys known from a shared warmup. Part A: 3/3 -// split, neither side holds 56.7% of total, so zero new locks on either side; heal, locks resume. Part B: 4/2 +// split, neither side holds 2/3 of total, so zero new locks on either side; heal, locks resume. Part B: 4/2 // split, the 4 side holds 66.7% and locks, the 2 side does not. async function s6() { const name = 's6-partition-floor'; // The floor protects a partition only while the majority side's fresh blocks stay small against its weight // window (devnet window 7,200 DAA). So warm up long enough to fill most of the window, then split briefly. At // ~12 bps the window fills in about 600 s; a 3-miner side adds ~3 bps, so a split under ~300 s keeps the - // majority below the 56.7% floor (3.3.1 on a real DAG; spec 3.7 item 8). --quick shrinks both, which exposes + // majority below the 2/3 floor (3.3.1 on a real DAG; spec 3.7 item 8). --quick shrinks both, which exposes // the under-filled-window breach instead (reported honestly as the time-to-breach). const warm = dur(420), split = dur(150), healWin = dur(150); @@ -192,11 +192,11 @@ async function s6() { return { before0, before1, maxNew0, maxNew1, newLocks, breach0, breach1, after0, after1, healed, conflicts, daa: w0.daaScore, voters: w0.voters }; } - // Part A: 3/3. Neither side holds 56.7% of the (filled) total, so no side should lock during a short split. + // Part A: 3/3. Neither side holds 2/3 of the (filled) total, so no side should lock during a short split. const a = await partition(['a0', 'a1', 'a2'], ['b0', 'b1', 'b2'], 'A(3/3)'); const passA = a.newLocks === 0 && a.healed && a.conflicts.every(c => c === 0); results.push({ name: name + '-A(3/3)', secs: warm + split + healWin, - criterion: '3/3 split on a filled window: zero new locks on either side (neither holds 56.7% of total); locks resume after healing; no conflicting certificates', + criterion: '3/3 split on a filled window: zero new locks on either side (neither holds 2/3 of total); locks resume after healing; no conflicting certificates', result: `warmup window daa ~${a.daa} (voters ${a.voters}); new locks during ${split}s split ${a.newLocks} (time to first new lock side0=${a.breach0 ?? 'none'}s side1=${a.breach1 ?? 'none'}s); resumed after heal ${a.healed}; conflicting certs ${a.conflicts.join('/')}`, pass: passA }); @@ -313,7 +313,7 @@ async function s5() { results.push({ name, secs, criterion: 'the burst earns weight proportional to its block share over the window (no retarget amplification, W2/F14) and cannot lock alone', result: burstWeightShare != null - ? `burster weight share ${(burstWeightShare * 100).toFixed(1)}% vs block share ${(burstBlockShare * 100).toFixed(1)}% (ratio ${ratio.toFixed(3)}, want ~1.0 = no amplification); below the 56.7% floor so cannot lock alone; locks with <2 votes ${soloLocks}; conflicting certs ${conflicts}` + ? `burster weight share ${(burstWeightShare * 100).toFixed(1)}% vs block share ${(burstBlockShare * 100).toFixed(1)}% (ratio ${ratio.toFixed(3)}, want ~1.0 = no amplification); below the 2/3 floor so cannot lock alone; locks with <2 votes ${soloLocks}; conflicting certs ${conflicts}` : `could not identify the burster key; conflicting certs ${conflicts}`, pass }); await stopAll();