From b826993b0870ed8d74360620e86c1ec233e6a263 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Tue, 6 Oct 2026 23:41:08 +0000 Subject: [PATCH 01/10] Pool daemon: refuses to start half-alive (listeners bound first, exit 2; a node that answers no template, exit 3; a STATUS line every 30 s) The fleet agent's row from the 6 October night: the daemon ran 20 minutes as a process serving nothing while its node answered no template and its member port had been held. Now main binds the member and API listeners before anything else (server::bind_listener; a failure is "POOL NOT STARTED: cannot bind the members listener on : ", exit 2), probes the node for a template with pow_epoch, retried for --node-wait-secs (default 60) with a line per attempt, else exit 3, and prints a STATUS line every 30 s (members, jobs issued, shares, blocks, template failures, the node's state: ok, NODE NOT ANSWERING, NO TEMPLATE YET). fetch_job counts template failures and the last success. Test server::bind_tests: a held port is refused with the address in the message, the freed port binds. Suite 23 of 23 on igneum-build-1. Protocol and API unchanged. Co-Authored-By: Claude Fable 5.1 (cherry picked from commit a3b5cd48ecfce35a20c6e11aea9987e280cd0443) --- docs/plans/pool.md | 14 ++++++++++++++ pool/README.md | 4 ++++ pool/src/api.rs | 6 +----- pool/src/config.rs | 6 +++++- pool/src/main.rs | 46 ++++++++++++++++++++++++++++++++++++++++++++-- pool/src/node.rs | 39 +++++++++++++++++++++++++++++++++++---- pool/src/pool.rs | 4 ++++ pool/src/server.rs | 30 +++++++++++++++++++++++++----- 8 files changed, 132 insertions(+), 17 deletions(-) diff --git a/docs/plans/pool.md b/docs/plans/pool.md index 30c21215d..9742307af 100644 --- a/docs/plans/pool.md +++ b/docs/plans/pool.md @@ -236,3 +236,17 @@ a wrong-size cache and mismatched silently); a pool on a class v4 network needs Building the pool crate on igneum-build-1: the pool reads the fork through the `vendor/igneum-node` symlink; lib.sh syncs the fork worktree it points at (`vendor/igneum-node-pr`) as a whole repository, and the symlink itself is created once on the box by hand (`ln -s igneum-node-pr /srv/builds//vendor/igneum-node`), the one step the scripts do not do. + +### 9.1 7 October 2026: the daemon refuses to start half-alive + +The fleet agent's night (6 October, 23:01Z to 23:32Z): the rebased pair never ran as a test. The daemon was started while +the node still held its member port, then four members (the wind-down's weight rule freed four pods, not ten) sat +connected with no job while every template fetch timed out; nothing in the daemon's own log said so except one line per +member. The row from it, now code (pool `269364a`+): the member and API listeners are bound in `main` before anything +else and a port that cannot be bound ends the start with exit code 2 and `POOL NOT STARTED: cannot bind the members +listener on : `; the node must answer a template with its `pow_epoch` before the pool serves anyone, +retried for `--node-wait-secs` (default 60, each attempt logged), else exit code 3; a `STATUS` line every 30 s +(`members= jobs_issued= shares_accepted= shares_rejected= blocks= template_failures=; node ok | NODE NOT ANSWERING | +NO TEMPLATE YET`). Test: `server::bind_tests` (a held port refused with the address in the message, the freed port binds). +The 10-member run is still owed: a clean 30-minute window, the daemon bound and its `node ... answers templates` line +printed before the first member, on the next pods the weight rule frees (the table must read over 75 percent signed first). diff --git a/pool/README.md b/pool/README.md index 9d042524b..3b10392fb 100644 --- a/pool/README.md +++ b/pool/README.md @@ -170,3 +170,7 @@ Every Linux build and suite runs on the box through `tools/build-remote.sh` from crate reads the fork through the `vendor/igneum-node` symlink; the build library syncs the fork worktree it points at as a whole repository, and the symlink is made once on the box by hand: `ln -s /srv/builds//vendor/igneum-node`. Artefacts land in `pool/target-remote/release/igneum-pool` (x86_64 Linux); the Mac builds no Linux binary. + +Start-up contract (7 October 2026): the daemon exits 2 when `--listen` or `--http` cannot be bound, exits 3 when the node +answers no template with `pow_epoch` within `--node-wait-secs` (default 60), and prints `node answers templates: ...` +before it accepts a member; a `STATUS` line every 30 s names the member count, jobs issued, shares and the node's state. diff --git a/pool/src/api.rs b/pool/src/api.rs index 0a32424ad..05401ea5b 100644 --- a/pool/src/api.rs +++ b/pool/src/api.rs @@ -199,11 +199,7 @@ pub fn route(pool: &Pool, path: &str) -> Vec { } } -pub async fn serve(pool: Arc) { - let listener = tokio::net::TcpListener::bind(&pool.cfg.http).await.unwrap_or_else(|e| { - eprintln!("cannot listen on {}: {e}", pool.cfg.http); - std::process::exit(1) - }); +pub async fn serve(pool: Arc, listener: tokio::net::TcpListener) { println!("{} pool: stats API and page on http://{}/", crate::state::unix_ms(), pool.cfg.http); loop { let Ok((sock, _)) = listener.accept().await else { continue }; diff --git a/pool/src/config.rs b/pool/src/config.rs index 26dac402f..48b066a51 100644 --- a/pool/src/config.rs +++ b/pool/src/config.rs @@ -33,12 +33,14 @@ pub struct Config { /// Shares sampled at shifts above this are still verified in v0 (spec sample_shift 8 is an allowance, not a duty) pub verify_threads: usize, pub snapshot_interval_s: u64, + /// How long the start-up waits for the node to answer a template before the daemon refuses to start (exit 3) + pub node_wait_secs: u64, } impl Config { pub fn usage() -> ! { eprintln!( - "usage: igneum-pool [--node grpc://127.0.0.1:26610]... [--evm-rpc http://127.0.0.1:26790] [--listen 0.0.0.0:4463]\n\ + "usage: igneum-pool [--node grpc://127.0.0.1:26610]... [--evm-rpc http://127.0.0.1:26790] [--listen 0.0.0.0:4463] [--node-wait-secs 60]\n\ \x20 [--http 127.0.0.1:4480] [--data-dir ./data] [--payout-key ] [--fee-percent 1] [--min-payout 1.0]\n\ \x20 [--pplns-window 2.0] [--payout-interval-s 60] [--dry-run] [--network devnet|testnet|mainnet]\n\ \x20 [--share-interval-s 10] [--min-shift 0] [--max-shift 60] [--stale-grace-ms 2000] [--orphan-after-daa 120]\n\ @@ -70,6 +72,7 @@ impl Config { public_url: String::new(), verify_threads: 2, snapshot_interval_s: 15, + node_wait_secs: 60, }; let mut i = 0; while i < args.len() { @@ -100,6 +103,7 @@ impl Config { "--public-url" => c.public_url = val(), "--verify-threads" => c.verify_threads = val().parse::().unwrap_or_else(|_| Config::usage()).max(1), "--snapshot-interval-s" => c.snapshot_interval_s = val().parse().unwrap_or_else(|_| Config::usage()), + "--node-wait-secs" => c.node_wait_secs = val().parse().unwrap_or_else(|_| Config::usage()), "-h" | "--help" => Config::usage(), _ => Config::usage(), } diff --git a/pool/src/main.rs b/pool/src/main.rs index a96459160..206dcec8c 100644 --- a/pool/src/main.rs +++ b/pool/src/main.rs @@ -27,6 +27,7 @@ use config::Config; use kaspa_addresses::{Address, Prefix, Version}; use kaspa_grpc_client::GrpcClient; use kaspa_pow::igneum::IgneumEngine; +use kaspa_rpc_core::api::rpc::RpcApi; use pool::{NetInfo, Pool}; use state::{State, WEI_PER_IGN}; use std::collections::HashMap; @@ -38,6 +39,17 @@ use std::time::{Duration, Instant}; async fn main() { let args: Vec = std::env::args().skip(1).collect(); let cfg = Config::from_args(&args); + // The two listeners first (7 October 2026, the fleet agent's row from the 6 October night: a daemon whose member + // port another process still held ran 20 minutes as a process with no socket): a port that cannot be bound ends + // the start here, exit code 2, before a node is contacted or a member is let down. + let member_listener = server::bind_listener("members", &cfg.listen).await.unwrap_or_else(|e| { + eprintln!("POOL NOT STARTED: {e}"); + std::process::exit(2) + }); + let api_listener = server::bind_listener("stats API", &cfg.http).await.unwrap_or_else(|e| { + eprintln!("POOL NOT STARTED: {e}"); + std::process::exit(2) + }); std::fs::create_dir_all(&cfg.data_dir).unwrap_or_else(|e| { eprintln!("data dir {}: {e}", cfg.data_dir.display()); std::process::exit(1) @@ -68,6 +80,33 @@ async fn main() { } } let node = clients.remove(0); + // The node must answer a template with its pow_epoch before the pool serves anyone: a pool whose node answers no + // template issues no job (the 6 October night: four members connected, "template timed out" per member, no job + // for 20 minutes). Retried for --node-wait-secs, saying so; then exit code 3. + { + let probe_address = Address::new(Prefix::Devnet, Version::PubKey, &[0u8; 32]); + let deadline = Instant::now() + Duration::from_secs(cfg.node_wait_secs); + let mut attempt = 0u32; + loop { + attempt += 1; + let why = match tokio::time::timeout(Duration::from_secs(10), node.get_block_template(probe_address.clone(), Vec::new())).await { + Ok(Ok(t)) if t.pow_epoch.is_some() => { + let i = t.pow_epoch.unwrap(); + println!("{} node {} answers templates: epoch {} class {} era {} daa {}", state::unix_ms(), cfg.nodes[0], i.epoch_index, i.class().name(), i.era_seed.map(|h| h.to_string()).unwrap_or_else(|| "-".into()), i.virtual_daa_score); + break; + } + Ok(Ok(_)) => "the template carries no pow_epoch (a node before the 0.3.x line); the pool cannot name a program".to_string(), + Ok(Err(e)) => e.to_string(), + Err(_) => "template timed out after 10 s".to_string(), + }; + if Instant::now() >= deadline { + eprintln!("POOL NOT STARTED: node {} answered no usable template in {} s ({attempt} attempts; last: {why})", cfg.nodes[0], cfg.node_wait_secs); + std::process::exit(3) + } + eprintln!("{} waiting for node {} to answer a template (attempt {attempt}: {why}); giving up at {} s", state::unix_ms(), cfg.nodes[0], cfg.node_wait_secs); + tokio::time::sleep(Duration::from_secs(5)).await; + } + } // the UTXO-side coinbase address: derived from the pool's EVM address; the execution layer pays the EVM side // (igneum/exec/src/executor.rs), the UTXO output is not spent by v0 (docs/plans/pool.md) let prefix = match cfg.network.as_str() { @@ -99,6 +138,8 @@ async fn main() { verify_permits: tokio::sync::Semaphore::new(cfg.verify_threads), started: Instant::now(), cfg: cfg.clone(), + template_failures: AtomicU64::new(0), + last_template_ok_ms: AtomicU64::new(0), }); let chain = evm.call("eth_chainId", serde_json::json!([])).await.ok().and_then(|v| v.as_str().map(|s| s.to_string())); println!( @@ -119,8 +160,9 @@ async fn main() { tokio::spawn(node::confirm_loop(pool.clone())); tokio::spawn(node::net_loop(pool.clone())); tokio::spawn(node::vardiff_loop(pool.clone())); - tokio::spawn(server::listen(pool.clone())); - tokio::spawn(api::serve(pool.clone())); + tokio::spawn(node::status_loop(pool.clone())); + tokio::spawn(server::listen(pool.clone(), member_listener)); + tokio::spawn(api::serve(pool.clone(), api_listener)); { let (pool, payer) = (pool.clone(), payer.clone()); tokio::spawn(async move { diff --git a/pool/src/node.rs b/pool/src/node.rs index 513813aa3..24cb8ae37 100644 --- a/pool/src/node.rs +++ b/pool/src/node.rs @@ -39,10 +39,18 @@ pub fn extra_data(member: &Member, pool_address: &[u8; 20]) -> Vec { /// cache build), so it runs under `spawn_blocking`. pub async fn fetch_job(pool: &Arc, member: &Arc) -> Result<(), String> { let t0 = Instant::now(); - let tmpl = tokio::time::timeout(Duration::from_secs(5), pool.node.get_block_template(pool.pay_address.clone(), extra_data(member, &pool.pool_address))) - .await - .map_err(|_| "template timed out".to_string())? - .map_err(|e| e.to_string())?; + let tmpl = match tokio::time::timeout(Duration::from_secs(5), pool.node.get_block_template(pool.pay_address.clone(), extra_data(member, &pool.pool_address))).await { + Ok(Ok(t)) => t, + Ok(Err(e)) => { + pool.template_failures.fetch_add(1, Ordering::Relaxed); + return Err(e.to_string()); + } + Err(_) => { + pool.template_failures.fetch_add(1, Ordering::Relaxed); + return Err("template timed out".to_string()); + } + }; + pool.last_template_ok_ms.store(unix_ms(), Ordering::Relaxed); let mut raw = tmpl.block; raw.header.vote_key_hash = member.key_hash; let block: Block = raw.clone().try_into().map_err(|e| format!("block convert: {e}"))?; @@ -344,6 +352,29 @@ pub async fn confirm_loop(pool: Arc) { } } +/// The daemon's own status line every 30 s (7 October 2026: a night where four members sat connected with no job for +/// 20 minutes was visible only as per-member "template timed out" lines). +pub async fn status_loop(pool: Arc) { + loop { + tokio::time::sleep(Duration::from_secs(30)).await; + let members = pool.members_snapshot().len(); + let failures = pool.template_failures.load(Ordering::Relaxed); + let last_ok = pool.last_template_ok_ms.load(Ordering::Relaxed); + let jobs = pool.next_job_id.load(Ordering::Relaxed); + let (accepted, rejected, blocks) = { + let s = pool.state.lock().unwrap(); + (s.accepted, s.rejected, s.blocks_found) + }; + let node_state = if last_ok == 0 { + "NO TEMPLATE YET from the node: no job has been issued".to_string() + } else { + let ago = unix_ms().saturating_sub(last_ok) / 1000; + if ago > 30 { format!("NODE NOT ANSWERING: last template {ago} s ago") } else { format!("node ok, last template {ago} s ago") } + }; + println!("{} STATUS members={members} jobs_issued={jobs} shares_accepted={accepted} shares_rejected={rejected} blocks={blocks} template_failures={failures}; {node_state}", now()); + } +} + /// Network numbers every 5 s. pub async fn net_loop(pool: Arc) { loop { diff --git a/pool/src/pool.rs b/pool/src/pool.rs index 61548c0ba..e44682589 100644 --- a/pool/src/pool.rs +++ b/pool/src/pool.rs @@ -109,6 +109,10 @@ pub struct Pool { pub want_templates: tokio::sync::Notify, pub verify_permits: tokio::sync::Semaphore, pub started: Instant, + /// Template fetches that failed or timed out since start, and the unix ms of the last one that succeeded (0 = + /// never): a daemon whose node answers no template issues no job, and the STATUS line in the log says so + pub template_failures: AtomicU64, + pub last_template_ok_ms: AtomicU64, } impl Pool { diff --git a/pool/src/server.rs b/pool/src/server.rs index 900922652..5f235f5a1 100644 --- a/pool/src/server.rs +++ b/pool/src/server.rs @@ -19,11 +19,14 @@ fn now() -> String { format!("{}.{:03}", t.as_secs(), t.subsec_millis()) } -pub async fn listen(pool: Arc) { - let listener = tokio::net::TcpListener::bind(&pool.cfg.listen).await.unwrap_or_else(|e| { - eprintln!("cannot listen on {}: {e}", pool.cfg.listen); - std::process::exit(1) - }); +/// Binds a listener or says exactly why not. Called from `main` BEFORE anything else starts (7 October 2026, the fleet +/// agent's row from the 6 October night: a daemon whose member port another process still held ran for 20 minutes as a +/// process with no socket; now a bind that fails ends the process with exit code 2 before a member can be let down). +pub async fn bind_listener(what: &str, addr: &str) -> Result { + tokio::net::TcpListener::bind(addr).await.map_err(|e| format!("cannot bind the {what} listener on {addr}: {e} (another process holds the port, or the address is not this machine's)")) +} + +pub async fn listen(pool: Arc, listener: tokio::net::TcpListener) { println!("{} pool: members on {} (chain id {}, {})", now(), pool.cfg.listen, pool.cfg.chain_id(), pool.cfg.network); loop { match listener.accept().await { @@ -323,3 +326,20 @@ async fn handle_share(pool: &Arc, m: &Arc, job_id: u64, nonce: u64 tokio::spawn(submit_block(pool.clone(), m.clone(), raw, nonce, shift, daa, blue)); } } + +#[cfg(test)] +mod bind_tests { + use super::bind_listener; + + /// Known good: a free port binds. Known failed: the same port, held, is refused with the address in the message + /// (the shape the fleet saw on 6 October: a node still holding 4463). + #[tokio::test] + async fn a_held_port_is_refused_loudly_and_a_free_one_binds() { + let held = bind_listener("members", "127.0.0.1:0").await.expect("a free port binds"); + let addr = held.local_addr().unwrap().to_string(); + let err = bind_listener("members", &addr).await.err().expect("the held port is refused"); + assert!(err.contains(&addr) && err.contains("members"), "{err}"); + drop(held); + bind_listener("members", &addr).await.expect("released, it binds again"); + } +} From deb7d37100bf4603a5c78d6b759b24a78cb3ed02 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 06:10:39 +0000 Subject: [PATCH 02/10] Pool daemon: --template-timeout-s (default 20, was a fixed 5 s): pool-1's node builds a template in 1.6 to 1.8 s under load, four parallel member fetches overran 5 s and no job was issued for 7 minutes (7 October 2026, 06:08Z) Co-Authored-By: Claude Fable 5.1 (cherry picked from commit d3f674b807e319902501df4ffe140240a9904e24) --- pool/src/config.rs | 8 +++++++- pool/src/node.rs | 4 ++-- 2 files changed, 9 insertions(+), 3 deletions(-) diff --git a/pool/src/config.rs b/pool/src/config.rs index 48b066a51..cdfe690c4 100644 --- a/pool/src/config.rs +++ b/pool/src/config.rs @@ -35,12 +35,16 @@ pub struct Config { pub snapshot_interval_s: u64, /// How long the start-up waits for the node to answer a template before the daemon refuses to start (exit 3) pub node_wait_secs: u64, + /// Per-member template fetch timeout. 7 October 2026, 06:08Z: pool-1's node built a template in 1.6 to 1.8 s under + /// its own miner plus four members, the fixed 5 s gave up on the parallel fetches, no job was issued for 7 minutes, + /// every block the stale templates found was an orphan, and vardiff could not act (a new target rides with a job) + pub template_timeout_s: u64, } impl Config { pub fn usage() -> ! { eprintln!( - "usage: igneum-pool [--node grpc://127.0.0.1:26610]... [--evm-rpc http://127.0.0.1:26790] [--listen 0.0.0.0:4463] [--node-wait-secs 60]\n\ + "usage: igneum-pool [--node grpc://127.0.0.1:26610]... [--evm-rpc http://127.0.0.1:26790] [--listen 0.0.0.0:4463] [--node-wait-secs 60] [--template-timeout-s 20]\n\ \x20 [--http 127.0.0.1:4480] [--data-dir ./data] [--payout-key ] [--fee-percent 1] [--min-payout 1.0]\n\ \x20 [--pplns-window 2.0] [--payout-interval-s 60] [--dry-run] [--network devnet|testnet|mainnet]\n\ \x20 [--share-interval-s 10] [--min-shift 0] [--max-shift 60] [--stale-grace-ms 2000] [--orphan-after-daa 120]\n\ @@ -73,6 +77,7 @@ impl Config { verify_threads: 2, snapshot_interval_s: 15, node_wait_secs: 60, + template_timeout_s: 20, }; let mut i = 0; while i < args.len() { @@ -104,6 +109,7 @@ impl Config { "--verify-threads" => c.verify_threads = val().parse::().unwrap_or_else(|_| Config::usage()).max(1), "--snapshot-interval-s" => c.snapshot_interval_s = val().parse().unwrap_or_else(|_| Config::usage()), "--node-wait-secs" => c.node_wait_secs = val().parse().unwrap_or_else(|_| Config::usage()), + "--template-timeout-s" => c.template_timeout_s = val().parse::().unwrap_or_else(|_| Config::usage()).max(1), "-h" | "--help" => Config::usage(), _ => Config::usage(), } diff --git a/pool/src/node.rs b/pool/src/node.rs index 24cb8ae37..a3a92d6ae 100644 --- a/pool/src/node.rs +++ b/pool/src/node.rs @@ -39,7 +39,7 @@ pub fn extra_data(member: &Member, pool_address: &[u8; 20]) -> Vec { /// cache build), so it runs under `spawn_blocking`. pub async fn fetch_job(pool: &Arc, member: &Arc) -> Result<(), String> { let t0 = Instant::now(); - let tmpl = match tokio::time::timeout(Duration::from_secs(5), pool.node.get_block_template(pool.pay_address.clone(), extra_data(member, &pool.pool_address))).await { + let tmpl = match tokio::time::timeout(Duration::from_secs(pool.cfg.template_timeout_s), pool.node.get_block_template(pool.pay_address.clone(), extra_data(member, &pool.pool_address))).await { Ok(Ok(t)) => t, Ok(Err(e)) => { pool.template_failures.fetch_add(1, Ordering::Relaxed); @@ -47,7 +47,7 @@ pub async fn fetch_job(pool: &Arc, member: &Arc) -> Result<(), Str } Err(_) => { pool.template_failures.fetch_add(1, Ordering::Relaxed); - return Err("template timed out".to_string()); + return Err(format!("template timed out after {} s (--template-timeout-s)", pool.cfg.template_timeout_s)); } }; pool.last_template_ok_ms.store(unix_ms(), Ordering::Relaxed); From a21c4bb1c88a8f93633b339246101d466c68bfc8 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 06:18:53 +0000 Subject: [PATCH 03/10] Pool daemon: the confirmation walk on its own gRPC connection, never restarted from the pruning point, bounded per tick 7 October 2026, 06:01Z and 06:12:55Z on pool-1: confirm_loop restarted its chain walk from the pruning point on any failed getVirtualChainFromBlock and then fetched every chain block since (about 120,000) over the one connection the templates used; one timed-out request under four parallel template fetches started it, every template request after it timed out, no job was issued, 105 blocks on stale templates were orphans. Now: a second GrpcClient (pool.walker) for the walk and the network numbers; a failed chain call keeps its cursor (the sink only when the node no longer knows it; cursor_after_failure, node::walk_tests); at most 600 chain blocks per tick with a line saying so; a start with pending blocks walks from the sink and says that older ones resolve by the orphan rule. docs/plans/pool.md section 9.2: the re-run's record (the class question closed on pm-1 and pm-2, 71,240 shares, 0 mismatches), the cause, what stays open. Suite 24 of 24 on igneum-build-1. Protocol and API unchanged. Co-Authored-By: Claude Fable 5.1 (cherry picked from commit bacc407f170b769a1352f45ce85bb18a753f8b45) --- docs/plans/pool.md | 39 ++++++++++++++++++ pool/src/main.rs | 12 ++++++ pool/src/node.rs | 100 ++++++++++++++++++++++++++++++++++----------- pool/src/pool.rs | 4 ++ 4 files changed, 131 insertions(+), 24 deletions(-) diff --git a/docs/plans/pool.md b/docs/plans/pool.md index 9742307af..4c28d914c 100644 --- a/docs/plans/pool.md +++ b/docs/plans/pool.md @@ -250,3 +250,42 @@ retried for `--node-wait-secs` (default 60, each attempt logged), else exit code NO TEMPLATE YET`). Test: `server::bind_tests` (a held port refused with the address in the message, the freed port binds). The 10-member run is still owed: a clean 30-minute window, the daemon bound and its `node ... answers templates` line printed before the first member, on the next pods the weight rule frees (the table must read over 75 percent signed first). + +### 9.2 7 October 2026, 06:00Z to 06:33Z: the re-run (five members), what it settled and what it found + +Settled: the class question. pm-1 and pm-2 (RTX 3070, grpc url `none`) at +7 min: 36,917 and 34,323 shares accepted, 0 +rejected, 0 WORKER MISMATCH, 0 refused, on the class v3 program of the pool's node's epoch (`5530a50d...`, era = the devnet +genesis hash), the genesis day (20729) and dataset size (2^28) installed from the daemon's `seeds` line. The 6 October +failure does not reproduce. + +Found: the daemon stalls on a real chain. From 06:01Z (the first block found) no job was issued for seven minutes, pm-3 +and pm-4 never got one, every block found on the stale templates (55, then 105, 480 DAA behind the virtual) was an +orphan, vardiff could not act (a new share target rides with a job) so the two hashing members sent about 90 shares a +second each and the share check read 9.8 ms on pool-1's cores (1.35 ms on the Mac); after a swap to a 20 s template +timeout (f808b3f3) with the data dir kept, the stall returned within a minute. The node built templates in 0.6 ms all +the while (its prewarm line) and served its own solo miner. + +The cause, from the daemon's code: `confirm_loop` restarted its chain walk from the PRUNING POINT on any failed +`getVirtualChainFromBlock`, and then called `get_block` for every chain block since (about 120,000 on the devnet) over +the ONE gRPC connection the templates used (the client is one request stream per connection, 5 s request timeout); one +timed-out request under four parallel template fetches started the walk, every later request waited behind it and +timed out, which restarted the walk again. The kept state file (105 pending blocks) restarted it after the swap. The +private measurement run (section 5) never saw it: a 600-block chain walks in a moment. + +Fixed (pool commit after bd49c2a9): the walk and the network numbers run on a second gRPC connection (`pool.walker`); a +failed chain call keeps its cursor (moved to the sink only when the node no longer knows it, never to the pruning +point; test `node::walk_tests`); a tick walks at most 600 chain blocks and says how many; a start with pending blocks +walks from the sink and says that older ones resolve by the orphan rule. `--template-timeout-s` stays (default 20). + +Still open from the night, for the next window: the node log showed a new gRPC connection about every 0.8 s with the +count steady at 5 (something opens and closes one each time; the daemon's re-subscribe loop is the suspect, the +subscribe-line count in the daemon log decides it); vardiff's new target should apply to the member's current work +without waiting for a job (a one-line member change, needs a miner rebuild); the 9.8 ms share check on the pool box's +cores against 1.35 ms on the Mac (the verifier's day cache is 1 GiB and random reads on a rented box's memory are the +cost; two verify threads saturate at about 200 shares a second, which vardiff must keep far away); and the 10-member +load figure itself, not taken. + +Consequences per tier: a pool operator needs a box whose memory serves the 1 GiB cache at speed (a rented 2 vCPU box +verifies about 100 shares a second per core at 9.8 ms; at one share per 10 s per member that is 1,000 members per +core, so the verifier is not the limit once vardiff holds); a member on any card is unaffected by any of this; the +chain walk now costs the node at most 600 `get_block` calls per 5 s on its own connection. diff --git a/pool/src/main.rs b/pool/src/main.rs index 206dcec8c..43033f1ee 100644 --- a/pool/src/main.rs +++ b/pool/src/main.rs @@ -80,6 +80,17 @@ async fn main() { } } let node = clients.remove(0); + let walker = match tokio::time::timeout(Duration::from_secs(10), GrpcClient::connect(cfg.nodes[0].clone())).await { + Ok(Ok(c)) => Arc::new(c), + Ok(Err(e)) => { + eprintln!("node {} (second connection): {e}", cfg.nodes[0]); + std::process::exit(1) + } + Err(_) => { + eprintln!("node {} (second connection): connect timed out", cfg.nodes[0]); + std::process::exit(1) + } + }; // The node must answer a template with its pow_epoch before the pool serves anyone: a pool whose node answers no // template issues no job (the 6 October night: four members connected, "template timed out" per member, no job // for 20 minutes). Retried for --node-wait-secs, saying so; then exit code 3. @@ -129,6 +140,7 @@ async fn main() { next_template_id: AtomicU64::new(0), next_job_id: AtomicU64::new(0), node, + walker, extra_nodes: clients, pool_address, pool_address_hex: pool_address_hex.clone(), diff --git a/pool/src/node.rs b/pool/src/node.rs index a3a92d6ae..6f5ebf767 100644 --- a/pool/src/node.rs +++ b/pool/src/node.rs @@ -284,9 +284,22 @@ pub async fn submit_block(pool: Arc, member: Arc, mut raw: RpcRawB pool.want_templates.notify_one(); } +/// Where the confirmation walk continues from after a failed `getVirtualChainFromBlock`: the same cursor, retried +/// next tick. The first version restarted from the pruning point, and on a 120,000-block devnet chain that is a +/// `get_block` per chain block through the connection the templates shared: one RPC timeout under load (7 October 2026, +/// 06:01Z) started a walk that starved every template request for the rest of the window. A cursor the node no +/// longer knows (pruned, or from a state file of another chain) moves to the sink, and the pending blocks older than +/// it resolve by the orphan rule. +pub fn cursor_after_failure(cursor: Hash, sink: Hash, cursor_known: bool) -> Hash { + if cursor_known { cursor } else { sink } +} + +/// At most this many chain blocks (and `get_block` calls) per 5 s tick; the rest continue next tick. +pub const WALK_MAX_PER_TICK: usize = 600; + /// Confirms pending blocks by walking the selected chain: a block is paid when it is a chain block or in the /// mergeset blues of one (the execution layer pays exactly those, executor.rs `b.is_blue`); a pending block more -/// than `orphan_after_daa` behind the virtual with no blue merge is an orphan. +/// than `orphan_after_daa` behind the virtual with no blue merge is an orphan. On its own connection (`pool.walker`). pub async fn confirm_loop(pool: Arc) { let mut last_chain: Option = None; loop { @@ -295,42 +308,64 @@ pub async fn confirm_loop(pool: Arc) { let s = pool.state.lock().unwrap(); s.blocks.iter().filter(|b| b.status == "pending").map(|b| (b.hash.clone(), b.daa_score)).collect() }; - let info = match pool.node.get_block_dag_info().await { + let info = match pool.walker.get_block_dag_info().await { Ok(i) => i, - Err(_) => continue, + Err(e) => { + eprintln!("{} confirm: getBlockDagInfo failed ({e})", now()); + continue; + } }; - if last_chain.is_none() { - last_chain = Some(info.pruning_point_hash); - } if pending.is_empty() { // keep the cursor near the tip so the first pending block costs one short walk last_chain = Some(info.sink); continue; } - let pending_set: HashSet = pending.iter().map(|p| p.0.clone()).collect(); - let low = last_chain.unwrap(); - let chain = match pool.node.get_virtual_chain_from_block(low, false, None).await { + // a first tick with pending blocks (a state file kept across a restart): from the sink, never the pruning + // point; blocks older than the sink that were blue are a payout lost to the restart, said once + let low = match last_chain { + Some(h) => h, + None => { + println!("{} confirm: {} pending block(s) at start; the walk begins at the sink, older ones resolve by the orphan rule", now(), pending.len()); + last_chain = Some(info.sink); + info.sink + } + }; + let t0 = Instant::now(); + let chain = match pool.walker.get_virtual_chain_from_block(low, false, None).await { Ok(c) => c, Err(e) => { - eprintln!("{} confirm: getVirtualChainFromBlock failed ({e}); restarting from the pruning point", now()); - last_chain = Some(info.pruning_point_hash); + let known = pool.walker.get_block(low, false).await.is_ok(); + let next = cursor_after_failure(low, info.sink, known); + eprintln!("{} confirm: getVirtualChainFromBlock from {} failed ({e}); cursor {}", now(), low, if next == low { "kept, retried next tick".to_string() } else { format!("unknown to the node, moved to the sink {next}") }); + last_chain = Some(next); continue; } }; + let added = &chain.added_chain_block_hashes; + let take = added.len().min(WALK_MAX_PER_TICK); let mut blues: HashSet = HashSet::new(); - for h in &chain.added_chain_block_hashes { + let mut fetched = 0usize; + for h in &added[..take] { blues.insert(h.to_string()); - if let Ok(b) = pool.node.get_block(*h, false).await - && let Some(v) = b.verbose_data - { - for m in v.merge_set_blues_hashes { - blues.insert(m.to_string()); + match pool.walker.get_block(*h, false).await { + Ok(b) => { + fetched += 1; + if let Some(v) = b.verbose_data { + for m in v.merge_set_blues_hashes { + blues.insert(m.to_string()); + } + } + } + Err(e) => { + eprintln!("{} confirm: getBlock {h} failed ({e}); the walk stops here and continues next tick", now()); + break; } } - } - if let Some(h) = chain.added_chain_block_hashes.last() { last_chain = Some(*h); } + if added.len() > WALK_MAX_PER_TICK || t0.elapsed() > Duration::from_secs(2) { + println!("{} confirm: walked {fetched} of {} chain blocks in {:.0} ms ({} pending)", now(), added.len(), t0.elapsed().as_secs_f64() * 1e3, pending.len()); + } for (hash, daa) in pending { if blues.contains(&hash) { let r = pool.state.lock().unwrap().confirm_block(&hash, pool.cfg.fee_percent); @@ -348,7 +383,6 @@ pub async fn confirm_loop(pool: Arc) { println!("{} ORPHAN {} (daa {}, virtual {}): not blue within {} DAA", now(), &hash[..16], daa, info.virtual_daa_score, pool.cfg.orphan_after_daa); } } - let _ = pending_set; } } @@ -379,17 +413,17 @@ pub async fn status_loop(pool: Arc) { pub async fn net_loop(pool: Arc) { loop { let mut n = NetInfo { network: pool.cfg.network.clone(), chain_id: pool.cfg.chain_id(), ..Default::default() }; - if let Ok(i) = pool.node.get_block_dag_info().await { + if let Ok(i) = pool.walker.get_block_dag_info().await { n.difficulty = i.difficulty; n.daa_score = i.virtual_daa_score; n.block_count = i.block_count; n.network = i.network.to_string(); } - n.hashrate = pool.node.estimate_network_hashes_per_second(1000, None).await.ok().map(|h| h as f64); - if let Ok(b) = pool.node.get_sink_blue_score().await { + n.hashrate = pool.walker.estimate_network_hashes_per_second(1000, None).await.ok().map(|h| h as f64); + if let Ok(b) = pool.walker.get_sink_blue_score().await { n.blue_score = b; } - if let Ok(i) = pool.node.get_info().await { + if let Ok(i) = pool.walker.get_info().await { n.synced = i.is_synced; n.node_version = i.server_version; } @@ -438,3 +472,21 @@ pub async fn vardiff_loop(pool: Arc) { let _ = pool.next_member_id.load(Ordering::Relaxed); } } + +#[cfg(test)] +mod walk_tests { + use super::*; + + /// The cursor rule after a failed chain call: known cursor kept (the 6 October shape restarted from the pruning + /// point and walked 120,000 blocks), unknown cursor moved to the sink. + #[test] + fn a_failed_chain_call_keeps_a_known_cursor_and_never_goes_to_the_pruning_point() { + let cursor = Hash::from_bytes([1u8; 32]); + let sink = Hash::from_bytes([2u8; 32]); + let pruning = Hash::from_bytes([3u8; 32]); + assert_eq!(cursor_after_failure(cursor, sink, true), cursor); + assert_eq!(cursor_after_failure(cursor, sink, false), sink); + assert_ne!(cursor_after_failure(cursor, sink, false), pruning); + assert!(WALK_MAX_PER_TICK <= 1000, "a tick's walk stays bounded"); + } +} diff --git a/pool/src/pool.rs b/pool/src/pool.rs index e44682589..3e801b74d 100644 --- a/pool/src/pool.rs +++ b/pool/src/pool.rs @@ -100,6 +100,10 @@ pub struct Pool { pub next_template_id: AtomicU64, pub next_job_id: AtomicU64, pub node: Arc, + /// A second connection to the same node for the confirmation walk and the network numbers (7 October 2026, + /// 06:01Z and 06:12:55Z: the walk's `get_block` calls shared the template connection and starved every template + /// request; the gRPC client is one request stream per connection) + pub walker: Arc, pub extra_nodes: Vec>, /// The pool's EVM coinbase address, lowercase 0x hex, named in every template's `IGNA` field pub pool_address: [u8; 20], From edb90502039695747ea136538ce3c0d8a836e95d Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 06:41:12 +0000 Subject: [PATCH 04/10] Pool vardiff: the idle easing no longer consumes the sized first correction; docs/plans/pool.md 9.3: the re-run's close and the member rows pm-1, 7 October 2026: the member was idle for four intervals while its worker compiled, the first-phase idle easing set first_done, and the measured 190 shares a second then walked down one step per 30 s for 5.5 minutes (the share check at 10 to 11 ms on pool-1's cores). The sized correction now stays owed through idle easings (test an_idle_easing_does_not_consume_the_sized_first_correction). Section 9.3: 207 found / 144 confirmed / 56 own orphans under 2f6c0358, the residual as the node's 1 to 1.6 s template latency (a node-lane row), the member findings. Co-Authored-By: Claude Fable 5.1 (cherry picked from commit 98bef42ca451678e7723bac9e842032326198f56) --- docs/plans/pool.md | 37 +++++++++++++++++++++++++++++++++++++ pool/src/vardiff.rs | 28 +++++++++++++++++++++++++--- 2 files changed, 62 insertions(+), 3 deletions(-) diff --git a/docs/plans/pool.md b/docs/plans/pool.md index 4c28d914c..732241fbc 100644 --- a/docs/plans/pool.md +++ b/docs/plans/pool.md @@ -289,3 +289,40 @@ Consequences per tier: a pool operator needs a box whose memory serves the 1 GiB verifies about 100 shares a second per core at 9.8 ms; at one share per 10 s per member that is 1,000 members per core, so the verifier is not the limit once vardiff holds); a member on any card is unaffected by any of this; the chain walk now costs the node at most 600 `get_block` calls per 5 s on its own connection. + +### 9.3 The re-run's close (06:33:33Z) and the member rows from its logs + +After 2f6c0358 (06:20:34Z to 06:33:33Z, 13 minutes, four members): 207 blocks found, 144 confirmed, 69 orphaned (13 +inherited pending at the swap, 56 on its own jobs: a 27 percent residual at 1 bps), 20,404 shares accepted, 0 rejected, +0 mismatches, 0 refused, every STATUS `node ok`, the gRPC connection churn gone (ids #3 to #5 once each in a minute; the +193 connections in 150 s were under f808b3f3's pruning-point walk), re-subscribe lines 2 (one per daemon start: the +re-subscribe loop was never the churn). Before it, f808b3f3 had worked through the walk by about 06:14Z: 245 blocks, +69 confirmed, 163 orphaned. First CONFIRMED line 06:24:32Z, 4.80 IGN to the finder. Logs: `~/Desktop/fleet/pool-run-0607/`. + +The residual orphans are template latency: the daemon's STATUS reads `last template 1 s ago` and the solo miner on the +same node reads `template_ms=1632`; pool-1's node answers a template request in 1 to 1.6 s although its mining manager +answers a per-member request from its cache by a coinbase rewrite (`modify_block_template`) and its prewarm builds in +0.6 ms. At 1 bps a template that arrives 1.6 s old loses about a quarter of the blocks found on it, pool or solo. That +latency is the node's RPC path, a row for the node lane (measure `get_block_template` round trips on a devnet node under +a miner and a pool; the suspect is the RPC service's `pow_epoch` derivation per request). 82 "RPC request timeout" lines +after the 2f6c0358 swap are the gRPC client's own 5 s request timeout on that path; `--template-timeout-s` cannot +lengthen it. + +From pm-1's log (1.09 GB): 4,342,515 lines are `worker: error N epoch seed mismatch: this worker holds epoch eea66ce7...`, +one per re-queued job, for the 40 s the worker compiled the new epoch's pack (NVRTC 40,327 ms) and again after every +reconnect, because a session respawned the worker (a second process beside the first, the same compile again) and +the member fed jobs for a pair the worker did not hold yet. Vardiff from the same log: shift 10 to 11 (idle easing +while the worker compiled, which consumed the sized first correction), then 11 down to 0 one step per 30 s over 5.5 +minutes at about 190 shares a second, the load that put the share check at 10 to 11 ms on pool-1's cores. + +Fixed in the member (fork commit after b8070476): one worker process per run, jobs held for a pair whose prepare is +pending and never re-prepared once held (`WorkerMemory`, test `jobs_are_held_while_a_pair_is_being_prepared`), worker +error lines summarised (one per class, then a count per 1,000), `set_target` applied to the current work at once. +Fixed in the pool (`vardiff.rs`): the idle easing no longer consumes the sized first correction (test +`an_idle_easing_does_not_consume_the_sized_first_correction`). + +Consequences per tier: a member on a 3070-class card loses 40 s of hash at every epoch roll to the NVRTC compile unless +its pack is prepared ahead (the pool's `seeds` line carries `next_epoch_seed` for that, the member's prepare-ahead is +the next member row); a member on a reconnect now keeps its worker and its compiled pair; a pool operator's verifier +sees the sized correction within 10 s of a member's first shares (one share per 10 s per member from then on) instead of +5 minutes of a 190-share-a-second flood; the 10-member load figure is still owed and now has a clean form to run in. diff --git a/pool/src/vardiff.rs b/pool/src/vardiff.rs index 7811cee14..4f86908d6 100644 --- a/pool/src/vardiff.rs +++ b/pool/src/vardiff.rs @@ -79,8 +79,11 @@ impl Vardiff { return None; } if n == 0 { - // nothing yet after an interval: wait up to three intervals, then one easier step - if (now_s - self.started_s) < 3.0 * self.interval_s { + // nothing yet after an interval: wait up to three intervals, then one easier step. The sized + // correction stays owed (first_done stays false): 7 October 2026, pm-1's worker compiled its pack for + // 40 s, the idle easing consumed the first correction, and the measured 190 shares a second then + // walked down one step per 30 s for 5.5 minutes, saturating the verifier + if now_s - self.last_change_s.max(self.started_s) < 3.0 * self.interval_s { return None; } self.shift = (self.shift + 1).min(cap); @@ -92,8 +95,8 @@ impl Vardiff { } else if per_interval < 0.66 { self.shift = (self.shift + steps).min(cap); } + self.first_done = true; } - self.first_done = true; } else { if since_change < self.min_change_s { return None; @@ -226,4 +229,23 @@ mod tests { } assert_eq!(v.retarget(t, 1 << 20), Some(2), "floor at min_shift"); } + + /// 7 October 2026, pm-1: a member idle for four intervals (its worker compiling) gets eased one step, and the + /// first measured rate must still size the jump (190 shares a second at shift 11 is 11 steps away from one per + /// 10 s; the sized step takes 8 at once), not walk down one step per 30 s. + #[test] + fn an_idle_easing_does_not_consume_the_sized_first_correction() { + let t64 = 1u64 << 36; + let mut v = Vardiff::new(10, 0, 20, 10.0, 0.0); + assert_eq!(v.retarget(31.0, t64), Some(11), "idle three intervals: one easier step"); + assert_eq!(v.retarget(40.0, t64), None, "still idle, nothing within the next three intervals"); + // the worker is ready: 190 shares a second for 10 s + let mut t = 40.0; + for _ in 0..1900 { + t += 10.0 / 1900.0; + v.on_share(t); + } + let s = v.retarget(t + 0.1, t64).expect("the first measured rate sizes the jump"); + assert!(s <= 3, "sized first correction from 11 took at most 8 steps at once, got shift {s}"); + } } From 633e211c9186d8e0d446fa4e90f7cd10ae257b61 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 09:15:10 +0000 Subject: [PATCH 05/10] Pool finished (mission item 11): pool-0's fee published beside the dev fee, TLS 1.3 on the member port with the authorize binding (spec 9.3, O-9.7 closed), the page rows Q68 to Q73, and the open pool: a share sidechain with no operator (spec 9.12) Pool-0: welcome carries software_dev_fee_percent beside the pool fee, the page's Fees row and site/miner.html say pool-0 charges the same 1 percent as the solo dev fee; jobs and seeds carry the latency ladder's rung for 0.3.19. TLS (pool/src/tls.rs): --tls-cert/--tls-key or --tls-self-signed (a P-256 pair under the data dir, the certificate pin printed at start); the binding is the member's BLS signature over this connection's exporter (label EXPORTER-igneum-pool-binding, context the chain id), refused on any other connection; testnet and mainnet refuse the clear without --allow-plain. HiveOS: pools:// and POOL_PIN. Page rows: node state in words (Q68), one formatter set and luck in MiningPoolStats' convention (Q69), samples and check costs persisted (Q70), payments under the lookup (Q71), hourly history with a sparkline, --alert-webhook, /metrics, /health on the node (Q72), the network's finality state beside "payouts follow blue confirmation, not finality" (Q73). The open pool (pool/src/sidechain.rs, open.rs, p2p.rs; igneum-pool --open): the member's own node and daemon, no payout key, no balance; every coinbase carries the member's own address, the share chain's parent (IGNS) and the window's split (IGNP); templates re-stamped on a new chain tip without a node call; shares gossiped and checked by every member (structure, the node's seeds, the PoW); heaviest work wins, a fork's loser is stale; the dev fee as one split entry; shares.log and verify-share for the drop proof; the gate harness pool/tools/open-gate.mjs. The node side (the executor's split from pool_split_activation_daa) is on the fork branch pool-finish-node. Co-Authored-By: Claude Fable 5.1 (cherry picked from commit 3e5271ba6ad2357f6d18d011eb1f50f0d02817f9) --- docs/fud-ledger.md | 4 + docs/plans/pool.md | 108 ++++++ docs/spec/09-pool-protocol.md | 34 +- packaging/hive/h-config.sh | 14 +- packaging/hive/h-run.sh | 4 + pool/Cargo.lock | 36 ++ pool/Cargo.toml | 5 + pool/README.md | 63 +++- pool/src/api.rs | 92 ++++- pool/src/config.rs | 63 +++- pool/src/main.rs | 67 +++- pool/src/node.rs | 133 ++++++- pool/src/open.rs | 300 ++++++++++++++++ pool/src/p2p.rs | 213 ++++++++++++ pool/src/payout.rs | 16 + pool/src/pool.rs | 15 + pool/src/protocol.rs | 28 +- pool/src/server.rs | 94 ++++- pool/src/sidechain.rs | 619 +++++++++++++++++++++++++++++++++ pool/src/state.rs | 109 +++++- pool/src/tls.rs | 184 ++++++++++ pool/tests/fixtures/miner.json | 3 +- pool/tests/fixtures/stats.json | 13 +- pool/tools/open-gate.mjs | 203 +++++++++++ pool/web/index.html | 129 +++++-- site/miner.html | 2 +- 26 files changed, 2454 insertions(+), 97 deletions(-) create mode 100644 pool/src/open.rs create mode 100644 pool/src/p2p.rs create mode 100644 pool/src/sidechain.rs create mode 100644 pool/src/tls.rs create mode 100644 pool/tools/open-gate.mjs diff --git a/docs/fud-ledger.md b/docs/fud-ledger.md index da25983a2..832029966 100644 --- a/docs/fud-ledger.md +++ b/docs/fud-ledger.md @@ -296,6 +296,8 @@ Evidence: design doc Finality v2, Residual risks bullet 3. Fix: overclaims list, Sweep (5 October 2026, evening): stated. `site/litepaper.html`, Finality ends with the pool sentence (overclaim 33); Governance opens with "governed by the hashrate that powers it", pool concentration named as the governance risk with the devnet measurement, top-3 keys 34.5% of 8,090 blocks (ledger sweep, X14 run). Overclaim 43. +Narrowed (7 October 2026, the pool lane, mission item 11): a conforming pool holds no vote. Pool-0 (`pool/`, spec 09) names the member's own vote key in every header it issues and holds no key of its own, so its hashers' weight is theirs (measured on the private fast-time runs of 5 and 7 October: every pool block carried the finding member's key); and the open pool (spec 09 section 9.12, `igneum-pool --open`) has no operator at all: every member builds its own templates from its own node, and a block pays the window from its own coinbase by the executor's split rule. The litepaper's sentence stays true of a custodial pool (`vote_mode: pool`, refused by the member's client by default and visible on chain as one key per payout address); the next litepaper pass should say "a custodial pool" where it says "pools". + ### F11. VDFs are exotic "A class-group VDF with Wesolowski proofs in a consensus-critical path, in a project with no cryptographer. Chia needed years and still got timelord ASICs." @@ -655,6 +657,8 @@ Answer: Correct. Stratum v2 job declaration lets a hasher choose transactions wh Evidence: none in repository. Fix: overclaims list, item 45. +Narrowed (7 October 2026, the pool lane): on Igneum the vote key in a pool's header is the member's, not the pool's (spec 09 section 9.6, built in `pool/` since 5 October and in the open pool since 7 October), so the vote-key caveat no longer applies to a conforming pool. Transaction choice: still the pool's in mode A (mode C, the member's own template, is designed and not built); in the open pool the member's own node chooses the transactions, since the member builds every template, which is Stratum v2's job declaration without an operator to decline it. The litepaper's "pools can be bypassed on transaction choice" stays, with "and in the open pool there is no pool to bypass" when it is next edited. + Cross-reference (external review, 3 October 2026, night): whether members use declared templates is measured under O-9.5; concentration reporting is X14. Sweep (5 October 2026, evening): stated. `site/litepaper.html`, Governance bullet "Pools can be bypassed on transaction choice", a pool may decline, vote keys stay with the pool, label Designed with spec section 9 (overclaim 44). diff --git a/docs/plans/pool.md b/docs/plans/pool.md index 732241fbc..8ff1c005b 100644 --- a/docs/plans/pool.md +++ b/docs/plans/pool.md @@ -326,3 +326,111 @@ its pack is prepared ahead (the pool's `seeds` line carries `next_epoch_seed` fo the next member row); a member on a reconnect now keeps its worker and its compiled pair; a pool operator's verifier sees the sized correction within 10 s of a member's first shares (one share per 10 s per member from then on) instead of 5 minutes of a 190-share-a-second flood; the 10-member load figure is still owed and now has a clean form to run in. + +## 10. 7 October 2026: pool-0 finished, TLS, and the open pool (mission item 11) + +Branch `pool-finish` on `release-0.3.19` 44eee05b with the four daemon commits of `pool-v0-rebase` cherry-picked +(the node fork pinned for the pool work is 0.3.19, whose kaspa-pow needs the latency ladder's igneum-pow; master +b92a5fd4 does not carry it, so the branch sits on the release tree); fork branch `pool-finish-node` on +`release-0.3.19-node` dc141409 with `pool-v0-rebase` merged (the pool-mode miner). Ordered by the project lead on 7 October 2026, +10:1x UK, built in the order of mission 2.11: pool-0, TLS and the page rows, the share sidechain. + +### 10.1 Pool-0 + +What 2.11 asks of pool-0 was in v0 (section 2): the member's vote key in every header, the pool with no key and no +vote, PPLNS, a payout round every 60 s, a 1 percent fee. What was missing was the fee published beside the dev fee +(reinvent 3.5): `welcome.share_scheme` now carries `software_dev_fee_percent` (0 in pool mode: the pool's fee is the only +fee), the page's Fees row says both, and `site/miner.html`'s dev-fee card says pool-0 charges the same 1 percent so solo +and pool cost the same and the choice is about variance alone. The jobs and seeds lines carry the latency ladder's rung +(`shadow_reps`) beside the class and the era, which 0.3.19's program needs; a member checks it against its own node as it +checks the seed, the class and the era (a pool that could choose the rung could choose the program). + +### 10.2 TLS (spec 9.3, O-9.7 closed, Q67) + +`pool/src/tls.rs`: rustls 0.23 with ring, TLS 1.3 only. `--tls-cert`/`--tls-key` (a chain from a trusted root) or +`--tls-self-signed` (a P-256 pair made under the data dir on first start, read back afterwards; the pin +`BLAKE2b("igneum-pool-cert-pin-v1" || DER)` printed at start, shown on the page and in `/api/stats`). The member +(`igneum/miner/src/pool.rs`, fork): `--pool-tls` (the Mozilla roots, the pool's host as the server name) or `--pool-pin +` (a verifier that accepts exactly the pinned certificate). The binding: both sides export 32 bytes with the label +`EXPORTER-igneum-pool-binding` and the chain id (8 bytes LE) as the context; the member signs `igneum-pool-binding-v1/ +|| chain id LE || exporter` under its vote key with its own tag (`DST_BINDING`, `consensus/core/src/finality.rs`); the +pool verifies it against the member's key and refuses the `authorize` with `bye` otherwise. Tests: `finality.rs` +(the signature holds for one exporter and one chain id, is no proof of possession); `tls.rs` (a self-signed pair read +back with one pin, a pinned handshake, both sides' exporters equal, the binding verified on its own connection and +refused on a second, a wrong pin never handshakes). Testnet and mainnet refuse to start in the clear without +`--allow-plain`; the devnet's local daemons stay plain. HiveOS: `pools://` or `POOL_PIN=` (`packaging/hive`). + +### 10.3 The page rows (polish Q68 to Q73) + +| Row | Done | +|---|---| +| Q68 | the page's node line: "node ok, ", "node syncing", "no template", "node unreachable since