From 6c793ced0f1ebf9eefc24111edf072d88d5518f2 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 06:18:53 +0000 Subject: [PATCH] Pool daemon: the confirmation walk on its own gRPC connection, never restarted from the pruning point, bounded per tick 7 October 2026, 06:01Z and 06:12:55Z on pool-1: confirm_loop restarted its chain walk from the pruning point on any failed getVirtualChainFromBlock and then fetched every chain block since (about 120,000) over the one connection the templates used; one timed-out request under four parallel template fetches started it, every template request after it timed out, no job was issued, 105 blocks on stale templates were orphans. Now: a second GrpcClient (pool.walker) for the walk and the network numbers; a failed chain call keeps its cursor (the sink only when the node no longer knows it; cursor_after_failure, node::walk_tests); at most 600 chain blocks per tick with a line saying so; a start with pending blocks walks from the sink and says that older ones resolve by the orphan rule. docs/plans/pool.md section 9.2: the re-run's record (the class question closed on pm-1 and pm-2, 71,240 shares, 0 mismatches), the cause, what stays open. Suite 24 of 24 on igneum-build-1. Protocol and API unchanged. Co-Authored-By: Claude Fable 5.1 (cherry picked from commit 6b05d53162b2d77f9b608fea6824e96d8fb9f87c) --- docs/plans/pool.md | 39 ++++++++++++++++++ pool/src/main.rs | 12 ++++++ pool/src/node.rs | 100 ++++++++++++++++++++++++++++++++++----------- pool/src/pool.rs | 4 ++ 4 files changed, 131 insertions(+), 24 deletions(-) diff --git a/docs/plans/pool.md b/docs/plans/pool.md index 9742307a..4c28d914 100644 --- a/docs/plans/pool.md +++ b/docs/plans/pool.md @@ -250,3 +250,42 @@ retried for `--node-wait-secs` (default 60, each attempt logged), else exit code NO TEMPLATE YET`). Test: `server::bind_tests` (a held port refused with the address in the message, the freed port binds). The 10-member run is still owed: a clean 30-minute window, the daemon bound and its `node ... answers templates` line printed before the first member, on the next pods the weight rule frees (the table must read over 75 percent signed first). + +### 9.2 7 October 2026, 06:00Z to 06:33Z: the re-run (five members), what it settled and what it found + +Settled: the class question. pm-1 and pm-2 (RTX 3070, grpc url `none`) at +7 min: 36,917 and 34,323 shares accepted, 0 +rejected, 0 WORKER MISMATCH, 0 refused, on the class v3 program of the pool's node's epoch (`5530a50d...`, era = the devnet +genesis hash), the genesis day (20729) and dataset size (2^28) installed from the daemon's `seeds` line. The 6 October +failure does not reproduce. + +Found: the daemon stalls on a real chain. From 06:01Z (the first block found) no job was issued for seven minutes, pm-3 +and pm-4 never got one, every block found on the stale templates (55, then 105, 480 DAA behind the virtual) was an +orphan, vardiff could not act (a new share target rides with a job) so the two hashing members sent about 90 shares a +second each and the share check read 9.8 ms on pool-1's cores (1.35 ms on the Mac); after a swap to a 20 s template +timeout (f808b3f3) with the data dir kept, the stall returned within a minute. The node built templates in 0.6 ms all +the while (its prewarm line) and served its own solo miner. + +The cause, from the daemon's code: `confirm_loop` restarted its chain walk from the PRUNING POINT on any failed +`getVirtualChainFromBlock`, and then called `get_block` for every chain block since (about 120,000 on the devnet) over +the ONE gRPC connection the templates used (the client is one request stream per connection, 5 s request timeout); one +timed-out request under four parallel template fetches started the walk, every later request waited behind it and +timed out, which restarted the walk again. The kept state file (105 pending blocks) restarted it after the swap. The +private measurement run (section 5) never saw it: a 600-block chain walks in a moment. + +Fixed (pool commit after bd49c2a9): the walk and the network numbers run on a second gRPC connection (`pool.walker`); a +failed chain call keeps its cursor (moved to the sink only when the node no longer knows it, never to the pruning +point; test `node::walk_tests`); a tick walks at most 600 chain blocks and says how many; a start with pending blocks +walks from the sink and says that older ones resolve by the orphan rule. `--template-timeout-s` stays (default 20). + +Still open from the night, for the next window: the node log showed a new gRPC connection about every 0.8 s with the +count steady at 5 (something opens and closes one each time; the daemon's re-subscribe loop is the suspect, the +subscribe-line count in the daemon log decides it); vardiff's new target should apply to the member's current work +without waiting for a job (a one-line member change, needs a miner rebuild); the 9.8 ms share check on the pool box's +cores against 1.35 ms on the Mac (the verifier's day cache is 1 GiB and random reads on a rented box's memory are the +cost; two verify threads saturate at about 200 shares a second, which vardiff must keep far away); and the 10-member +load figure itself, not taken. + +Consequences per tier: a pool operator needs a box whose memory serves the 1 GiB cache at speed (a rented 2 vCPU box +verifies about 100 shares a second per core at 9.8 ms; at one share per 10 s per member that is 1,000 members per +core, so the verifier is not the limit once vardiff holds); a member on any card is unaffected by any of this; the +chain walk now costs the node at most 600 `get_block` calls per 5 s on its own connection. diff --git a/pool/src/main.rs b/pool/src/main.rs index 206dcec8..43033f1e 100644 --- a/pool/src/main.rs +++ b/pool/src/main.rs @@ -80,6 +80,17 @@ async fn main() { } } let node = clients.remove(0); + let walker = match tokio::time::timeout(Duration::from_secs(10), GrpcClient::connect(cfg.nodes[0].clone())).await { + Ok(Ok(c)) => Arc::new(c), + Ok(Err(e)) => { + eprintln!("node {} (second connection): {e}", cfg.nodes[0]); + std::process::exit(1) + } + Err(_) => { + eprintln!("node {} (second connection): connect timed out", cfg.nodes[0]); + std::process::exit(1) + } + }; // The node must answer a template with its pow_epoch before the pool serves anyone: a pool whose node answers no // template issues no job (the 6 October night: four members connected, "template timed out" per member, no job // for 20 minutes). Retried for --node-wait-secs, saying so; then exit code 3. @@ -129,6 +140,7 @@ async fn main() { next_template_id: AtomicU64::new(0), next_job_id: AtomicU64::new(0), node, + walker, extra_nodes: clients, pool_address, pool_address_hex: pool_address_hex.clone(), diff --git a/pool/src/node.rs b/pool/src/node.rs index a3a92d6a..6f5ebf76 100644 --- a/pool/src/node.rs +++ b/pool/src/node.rs @@ -284,9 +284,22 @@ pub async fn submit_block(pool: Arc, member: Arc, mut raw: RpcRawB pool.want_templates.notify_one(); } +/// Where the confirmation walk continues from after a failed `getVirtualChainFromBlock`: the same cursor, retried +/// next tick. The first version restarted from the pruning point, and on a 120,000-block devnet chain that is a +/// `get_block` per chain block through the connection the templates shared: one RPC timeout under load (7 October 2026, +/// 06:01Z) started a walk that starved every template request for the rest of the window. A cursor the node no +/// longer knows (pruned, or from a state file of another chain) moves to the sink, and the pending blocks older than +/// it resolve by the orphan rule. +pub fn cursor_after_failure(cursor: Hash, sink: Hash, cursor_known: bool) -> Hash { + if cursor_known { cursor } else { sink } +} + +/// At most this many chain blocks (and `get_block` calls) per 5 s tick; the rest continue next tick. +pub const WALK_MAX_PER_TICK: usize = 600; + /// Confirms pending blocks by walking the selected chain: a block is paid when it is a chain block or in the /// mergeset blues of one (the execution layer pays exactly those, executor.rs `b.is_blue`); a pending block more -/// than `orphan_after_daa` behind the virtual with no blue merge is an orphan. +/// than `orphan_after_daa` behind the virtual with no blue merge is an orphan. On its own connection (`pool.walker`). pub async fn confirm_loop(pool: Arc) { let mut last_chain: Option = None; loop { @@ -295,42 +308,64 @@ pub async fn confirm_loop(pool: Arc) { let s = pool.state.lock().unwrap(); s.blocks.iter().filter(|b| b.status == "pending").map(|b| (b.hash.clone(), b.daa_score)).collect() }; - let info = match pool.node.get_block_dag_info().await { + let info = match pool.walker.get_block_dag_info().await { Ok(i) => i, - Err(_) => continue, + Err(e) => { + eprintln!("{} confirm: getBlockDagInfo failed ({e})", now()); + continue; + } }; - if last_chain.is_none() { - last_chain = Some(info.pruning_point_hash); - } if pending.is_empty() { // keep the cursor near the tip so the first pending block costs one short walk last_chain = Some(info.sink); continue; } - let pending_set: HashSet = pending.iter().map(|p| p.0.clone()).collect(); - let low = last_chain.unwrap(); - let chain = match pool.node.get_virtual_chain_from_block(low, false, None).await { + // a first tick with pending blocks (a state file kept across a restart): from the sink, never the pruning + // point; blocks older than the sink that were blue are a payout lost to the restart, said once + let low = match last_chain { + Some(h) => h, + None => { + println!("{} confirm: {} pending block(s) at start; the walk begins at the sink, older ones resolve by the orphan rule", now(), pending.len()); + last_chain = Some(info.sink); + info.sink + } + }; + let t0 = Instant::now(); + let chain = match pool.walker.get_virtual_chain_from_block(low, false, None).await { Ok(c) => c, Err(e) => { - eprintln!("{} confirm: getVirtualChainFromBlock failed ({e}); restarting from the pruning point", now()); - last_chain = Some(info.pruning_point_hash); + let known = pool.walker.get_block(low, false).await.is_ok(); + let next = cursor_after_failure(low, info.sink, known); + eprintln!("{} confirm: getVirtualChainFromBlock from {} failed ({e}); cursor {}", now(), low, if next == low { "kept, retried next tick".to_string() } else { format!("unknown to the node, moved to the sink {next}") }); + last_chain = Some(next); continue; } }; + let added = &chain.added_chain_block_hashes; + let take = added.len().min(WALK_MAX_PER_TICK); let mut blues: HashSet = HashSet::new(); - for h in &chain.added_chain_block_hashes { + let mut fetched = 0usize; + for h in &added[..take] { blues.insert(h.to_string()); - if let Ok(b) = pool.node.get_block(*h, false).await - && let Some(v) = b.verbose_data - { - for m in v.merge_set_blues_hashes { - blues.insert(m.to_string()); + match pool.walker.get_block(*h, false).await { + Ok(b) => { + fetched += 1; + if let Some(v) = b.verbose_data { + for m in v.merge_set_blues_hashes { + blues.insert(m.to_string()); + } + } + } + Err(e) => { + eprintln!("{} confirm: getBlock {h} failed ({e}); the walk stops here and continues next tick", now()); + break; } } - } - if let Some(h) = chain.added_chain_block_hashes.last() { last_chain = Some(*h); } + if added.len() > WALK_MAX_PER_TICK || t0.elapsed() > Duration::from_secs(2) { + println!("{} confirm: walked {fetched} of {} chain blocks in {:.0} ms ({} pending)", now(), added.len(), t0.elapsed().as_secs_f64() * 1e3, pending.len()); + } for (hash, daa) in pending { if blues.contains(&hash) { let r = pool.state.lock().unwrap().confirm_block(&hash, pool.cfg.fee_percent); @@ -348,7 +383,6 @@ pub async fn confirm_loop(pool: Arc) { println!("{} ORPHAN {} (daa {}, virtual {}): not blue within {} DAA", now(), &hash[..16], daa, info.virtual_daa_score, pool.cfg.orphan_after_daa); } } - let _ = pending_set; } } @@ -379,17 +413,17 @@ pub async fn status_loop(pool: Arc) { pub async fn net_loop(pool: Arc) { loop { let mut n = NetInfo { network: pool.cfg.network.clone(), chain_id: pool.cfg.chain_id(), ..Default::default() }; - if let Ok(i) = pool.node.get_block_dag_info().await { + if let Ok(i) = pool.walker.get_block_dag_info().await { n.difficulty = i.difficulty; n.daa_score = i.virtual_daa_score; n.block_count = i.block_count; n.network = i.network.to_string(); } - n.hashrate = pool.node.estimate_network_hashes_per_second(1000, None).await.ok().map(|h| h as f64); - if let Ok(b) = pool.node.get_sink_blue_score().await { + n.hashrate = pool.walker.estimate_network_hashes_per_second(1000, None).await.ok().map(|h| h as f64); + if let Ok(b) = pool.walker.get_sink_blue_score().await { n.blue_score = b; } - if let Ok(i) = pool.node.get_info().await { + if let Ok(i) = pool.walker.get_info().await { n.synced = i.is_synced; n.node_version = i.server_version; } @@ -438,3 +472,21 @@ pub async fn vardiff_loop(pool: Arc) { let _ = pool.next_member_id.load(Ordering::Relaxed); } } + +#[cfg(test)] +mod walk_tests { + use super::*; + + /// The cursor rule after a failed chain call: known cursor kept (the 6 October shape restarted from the pruning + /// point and walked 120,000 blocks), unknown cursor moved to the sink. + #[test] + fn a_failed_chain_call_keeps_a_known_cursor_and_never_goes_to_the_pruning_point() { + let cursor = Hash::from_bytes([1u8; 32]); + let sink = Hash::from_bytes([2u8; 32]); + let pruning = Hash::from_bytes([3u8; 32]); + assert_eq!(cursor_after_failure(cursor, sink, true), cursor); + assert_eq!(cursor_after_failure(cursor, sink, false), sink); + assert_ne!(cursor_after_failure(cursor, sink, false), pruning); + assert!(WALK_MAX_PER_TICK <= 1000, "a tick's walk stays bounded"); + } +} diff --git a/pool/src/pool.rs b/pool/src/pool.rs index e4468258..3e801b74 100644 --- a/pool/src/pool.rs +++ b/pool/src/pool.rs @@ -100,6 +100,10 @@ pub struct Pool { pub next_template_id: AtomicU64, pub next_job_id: AtomicU64, pub node: Arc, + /// A second connection to the same node for the confirmation walk and the network numbers (7 October 2026, + /// 06:01Z and 06:12:55Z: the walk's `get_block` calls shared the template connection and starved every template + /// request; the gRPC client is one request stream per connection) + pub walker: Arc, pub extra_nodes: Vec>, /// The pool's EVM coinbase address, lowercase 0x hex, named in every template's `IGNA` field pub pool_address: [u8; 20],