The hang on a point that does not fit: patch v5 (the server's panic hook names the stage and exits 70, so the proof fails at once), the app's per-shard wall-clock budget and the step-down 2^27 to 2^26 to 2^25 (tests), the GPU fleet's tier rows (24 GB stock sizes, 12 GB 2^26, 10 and 8 GB prove alone at 2^26), the PC 1 hang-case playbook, the bench-log note
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
parent
d2c5692ebe
commit
da4ab9b1be
7 changed files with 377 additions and 26 deletions
|
|
@ -23,10 +23,14 @@
|
|||
//! |---|---|---|
|
||||
//! | 20 GB and more (24 GB, 32 GB) | the server's own sizes (no override) | 16,851 MiB on the v1 shard at upstream's threshold on the 5090, 3.7 GB under the stock server |
|
||||
//! | 14 to 20 GB (16 GB) | 2^27 = 134,217,728 | 12,915 MiB alone, 10.95 GB own plus the miner's 1.7 GB beside it, 17.4 s a v1 shard (the 5090's allocation) |
|
||||
//! | 10 to 14 GB (12 GB) | 2^26 = 67,108,864, mining and proving | the RTX 4070 itself: 9,034 MiB of 12,282 beside its own miner, 24.1 s a v1 shard; 7,553 MiB alone in 7.7 s |
|
||||
//! | under 10 GB (8 GB) | off, with the reason | the compressed v1 shard alone is 7,553 MiB on the 4070; nothing is left for a miner or a bigger shard |
|
||||
//! | 10 to 14 GB (12 GB) | 2^26 = 67,108,864, mining and proving | the RTX 4070 itself: 9,034 MiB of 12,282 beside its own miner on PC 1 (Windows, 24.1 s a v1 shard), 10.1 to 10.2 GB on a headless Linux 4070 and 5070 (the GPU fleet, 27 to 37 s); 7.6 GB alone |
|
||||
//! | 7 to 10 GB (8 GB, 10 GB) | 2^26, PROVES ALONE (off by default: the app proves beside the miner) | a 3080 10 GB proves alone at 7.9 to 8.2 GB (7.0 to 7.2 s) and a 4060 Ti 8 GB at 7.74 GB of 8.19 (9.6 s), the GPU fleet; beside the miner 7.7 + 1.4 GB is over 8 GB; 2^27 never fits |
|
||||
//! | under 7 GB | off, with the reason | the compressed shard alone is 7.7 GB at the smallest threshold that proves it in a minute |
|
||||
//!
|
||||
//! `Settings` overrides the profile (`prove_profile`: auto, 2^25, 2^26, 2^27, stock).
|
||||
//! `Settings` overrides the profile (`prove_profile`: auto, 2^25, 2^26, 2^27, stock). A point that does not fit must
|
||||
//! never sit idle (the fleet's finding: the server hung for 15 minutes at 0%): the server now fails such a shard at
|
||||
//! once (the floor patch's panic hook) and the app gives every shard a wall-clock budget (`shard_budget`) and steps
|
||||
//! the threshold down one notch on a timeout (`step_down`: 2^27 to 2^26 to 2^25) before the next try.
|
||||
|
||||
use crate::state::CardState;
|
||||
|
||||
|
|
@ -50,9 +54,52 @@ fn gb(mb: u64) -> u64 {
|
|||
(mb + 512) / 1024
|
||||
}
|
||||
|
||||
/// The patched server's gate: a 12 GB card (`nvidia-smi` 12,282 for the RTX 4070, 12,288 for a 3060) proves; a
|
||||
/// 10 GB 3080 (10,240) does not.
|
||||
/// The patched server's gate for mining AND proving: a 12 GB card (`nvidia-smi` 12,282 for the RTX 4070, 12,288
|
||||
/// for a 3060). A 10 GB 3080 (10,240) and an 8 GB 4060 Ti (8,188) prove alone (`MIN_VRAM_MB_PROVE_ALONE`).
|
||||
pub const MIN_VRAM_MB_PATCHED: u64 = 11_000;
|
||||
pub const MIN_VRAM_MB_PROVE_ALONE: u64 = 7_000;
|
||||
|
||||
/// The wall-clock budget for one shard: three times the measured time of a v1 shard (22,172 pgas) beside the
|
||||
/// miner on the 12 GB card at the profile's threshold, scaled by the shard's pgas, never under 120 s and never over
|
||||
/// 30 minutes. The point: a shard that does not fit the card is killed and the threshold stepped down instead of
|
||||
/// the card sitting idle (the GPU fleet, 6 October 2026: 15 minutes at 0% on the 8 and 10 GB cards at 2^27).
|
||||
/// Measured bases, the RTX 4070 beside its miner (bench-log "prover floor"): 2^27 17.3 s, 2^26 24.1 s, 2^25
|
||||
/// 40.9 s; the server's own sizes on a 24 GB card about 20 s (the 4090 beside its miner 26.1 s at 2^26, the fleet).
|
||||
pub fn shard_budget(threshold: Option<u64>, pgas: u64) -> std::time::Duration {
|
||||
let base_s: f64 = match threshold {
|
||||
Some(THRESHOLD_2_25) => 40.9,
|
||||
Some(THRESHOLD_2_26) => 24.1,
|
||||
Some(THRESHOLD_2_27) => 17.3,
|
||||
_ => 26.1,
|
||||
};
|
||||
let scale = (pgas as f64 / 22_172.0).max(1.0);
|
||||
let s = (3.0 * base_s * scale).clamp(120.0, 1800.0);
|
||||
std::time::Duration::from_secs(s as u64)
|
||||
}
|
||||
|
||||
/// After a timeout at `threshold`: the next notch down, or None when there is none (2^25 is the last: under it the
|
||||
/// time grows past the deadline, bench-log "route 2" 2^24 at 56.6 s alone). The server's own sizes step to 2^27.
|
||||
pub fn step_down(threshold: Option<u64>) -> Option<Option<u64>> {
|
||||
match threshold {
|
||||
None => Some(Some(THRESHOLD_2_27)),
|
||||
Some(THRESHOLD_2_27) => Some(Some(THRESHOLD_2_26)),
|
||||
Some(THRESHOLD_2_26) => Some(Some(THRESHOLD_2_25)),
|
||||
Some(t) if t > THRESHOLD_2_27 => Some(Some(THRESHOLD_2_27)),
|
||||
Some(t) if t > THRESHOLD_2_26 => Some(Some(THRESHOLD_2_26)),
|
||||
Some(t) if t > THRESHOLD_2_25 => Some(Some(THRESHOLD_2_25)),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn threshold_name(t: Option<u64>) -> String {
|
||||
match t {
|
||||
None => "the server's own sizes".into(),
|
||||
Some(THRESHOLD_2_25) => "2^25".into(),
|
||||
Some(THRESHOLD_2_26) => "2^26".into(),
|
||||
Some(THRESHOLD_2_27) => "2^27".into(),
|
||||
Some(v) => v.to_string(),
|
||||
}
|
||||
}
|
||||
|
||||
/// The per-card profile on the patched server: the environment the host gets, and one line for the tile.
|
||||
#[derive(Clone, Debug, PartialEq, Eq)]
|
||||
|
|
@ -85,9 +132,11 @@ pub fn profile(vram_mb: u64, override_name: &str) -> Profile {
|
|||
} else if vram_mb >= 14_000 {
|
||||
Profile { tier: "16gb", threshold: Some(THRESHOLD_2_27), mine_and_prove: true, line: format!("{} GB card: threshold 2^27 (12.9 GB alone, 10.95 GB plus the miner beside it, measured on the 5090's allocation)", gb(vram_mb)) }
|
||||
} else if vram_mb >= MIN_VRAM_MB_PATCHED {
|
||||
Profile { tier: "12gb", threshold: Some(THRESHOLD_2_26), mine_and_prove: true, line: format!("{} GB card: threshold 2^26, mines and proves (the RTX 4070 measured 9,034 MiB of 12,282 beside its own miner, 24.1 s a shard)", gb(vram_mb)) }
|
||||
Profile { tier: "12gb", threshold: Some(THRESHOLD_2_26), mine_and_prove: true, line: format!("{} GB card: threshold 2^26, mines and proves (the RTX 4070 measured 9,034 MiB of 12,282 beside its own miner on Windows, 24.1 s a shard; 10.1 to 10.2 GB on a headless Linux 4070 and 5070)", gb(vram_mb)) }
|
||||
} else if vram_mb >= MIN_VRAM_MB_PROVE_ALONE {
|
||||
Profile { tier: "8gb", threshold: Some(THRESHOLD_2_26), mine_and_prove: false, line: format!("{} GB card: threshold 2^26, proves alone (a 3080 10 GB at 7.9 to 8.2 GB and a 4060 Ti 8 GB at 7.74 GB of 8.19, the GPU fleet); beside its miner 7.7 + 1.4 GB is over the card, so proving stays off while it mines; 2^27 never fits", gb(vram_mb)) }
|
||||
} else {
|
||||
Profile { tier: "off", threshold: None, mine_and_prove: false, line: format!("{} GB card: proving off, the compressed shard alone needs 7.6 GB (measured on the RTX 4070) and leaves nothing for a miner on this card", gb(vram_mb)) }
|
||||
Profile { tier: "off", threshold: None, mine_and_prove: false, line: format!("{} GB card: proving off, the compressed shard alone needs 7.7 GB at the smallest threshold that proves it in a minute (measured on 8 and 12 GB cards)", gb(vram_mb)) }
|
||||
};
|
||||
match forced {
|
||||
Some(t) if auto.tier != "off" || t.is_some() => Profile {
|
||||
|
|
@ -122,6 +171,8 @@ pub fn decide_with_server(cards: &[CardState], os: &str, wsl_answers: Option<boo
|
|||
let nvidia: Vec<&CardState> = cards.iter().filter(|c| c.vendor == "nvidia").collect();
|
||||
// the stock server: a card needs 24 GB (its floor is 13.9 GB for an empty shard, 20.4 for a full one); the
|
||||
// patched server: 12 GB (the RTX 4070's own measurement)
|
||||
// with the patched server a card that only proves ALONE (8 and 10 GB) is not on by default: the app proves
|
||||
// beside the miner, and 7.7 GB plus the miner's 1.4 GB is over such a card (the GPU fleet); Settings may force it
|
||||
let gate = |c: &CardState| if patched_server { MIN_VRAM_MB_PATCHED } else if c.enabled { MIN_VRAM_MB_MINING } else { MIN_VRAM_MB_PROVE_ONLY };
|
||||
let able: Vec<&CardState> = nvidia.iter().copied().filter(|c| c.vram_mb >= gate(c)).collect();
|
||||
let off = |line: String| Decision { on: false, line };
|
||||
|
|
@ -134,8 +185,10 @@ pub fn decide_with_server(cards: &[CardState], os: &str, wsl_answers: Option<boo
|
|||
} else {
|
||||
nvidia.iter().map(|c| format!("{} {} GB{}", c.name, gb(c.vram_mb), if c.enabled { ", mining" } else { "" })).collect::<Vec<_>>().join(", ")
|
||||
};
|
||||
let why = if patched_server && !nvidia.is_empty() {
|
||||
"the smallest card that proves is 12 GB (the RTX 4070 measured 7.6 GB for a shard alone); this card is under that"
|
||||
let why = if patched_server && nvidia.iter().any(|c| c.vram_mb >= MIN_VRAM_MB_PROVE_ALONE) {
|
||||
"this card proves ALONE on the patched server (7.7 to 8.2 GB a shard at threshold 2^26, measured on a 3080 and a 4060 Ti 8 GB) but not beside its miner, so proving stays off while it mines; Settings switches it on at your own risk"
|
||||
} else if patched_server && !nvidia.is_empty() {
|
||||
"the smallest card that proves is 8 GB (7.7 GB a shard alone at threshold 2^26); this card is under that"
|
||||
} else if nvidia.iter().any(|c| c.vram_mb >= 15_872) {
|
||||
"a full shard needs a 24 GB card (measured 20.4 GB on the adopted shard size, 13.9 GB for an empty one); this card is under that, so Settings would switch proving on at your own risk"
|
||||
} else if nvidia.is_empty() {
|
||||
|
|
@ -181,29 +234,56 @@ mod tests {
|
|||
let p = profile(24_564, "");
|
||||
assert_eq!((p.tier, p.threshold), ("24gb", None));
|
||||
assert_eq!(profile(32_607, "").tier, "32gb");
|
||||
// the GPU fleet's rows (6 October 2026): 10 and 8 GB prove alone at 2^26, never 2^27; under 7 GB off
|
||||
let p = profile(10_240, "");
|
||||
assert_eq!((p.tier, p.threshold, p.mine_and_prove), ("8gb", Some(THRESHOLD_2_26), false));
|
||||
assert!(p.line.contains("proves alone") && p.line.contains("2^27 never fits"), "{}", p.line);
|
||||
let p = profile(8_188, "");
|
||||
assert_eq!((p.tier, p.threshold, p.mine_and_prove), ("8gb", Some(THRESHOLD_2_26), false));
|
||||
let p = profile(6_144, "");
|
||||
assert_eq!((p.tier, p.threshold, p.mine_and_prove), ("off", None, false));
|
||||
assert!(p.line.contains("proving off") && p.line.contains("7.6 GB"), "{}", p.line);
|
||||
assert_eq!(profile(8_192, "").tier, "off");
|
||||
assert!(p.line.contains("proving off") && p.line.contains("7.7 GB"), "{}", p.line);
|
||||
// Settings: a forced threshold keeps the tier and says so; a forced stock profile on a 12 GB card is allowed too
|
||||
let p = profile(12_282, "2^25");
|
||||
assert_eq!((p.tier, p.threshold), ("12gb", Some(THRESHOLD_2_25)));
|
||||
assert!(p.line.contains("Settings forces threshold 2^25") && p.line.contains("would be 2^26"), "{}", p.line);
|
||||
assert_eq!(profile(24_564, "2^27").threshold, Some(THRESHOLD_2_27));
|
||||
assert_eq!(profile(16_303, "stock").threshold, None);
|
||||
// an 8 GB card forced to a threshold proves at the owner's risk; forced to stock it stays off
|
||||
assert_eq!(profile(8_192, "2^25").threshold, Some(THRESHOLD_2_25));
|
||||
assert_eq!(profile(8_192, "stock").tier, "off");
|
||||
// a 6 GB card forced to a threshold proves at the owner's risk; forced to stock it stays off
|
||||
assert_eq!(profile(6_144, "2^25").threshold, Some(THRESHOLD_2_25));
|
||||
assert_eq!(profile(6_144, "stock").tier, "off");
|
||||
assert_eq!(profile(12_282, "nonsense"), profile(12_282, ""));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_shard_budget_scales_with_the_tier_and_the_shard_and_the_step_down_ends_at_2_25() {
|
||||
use std::time::Duration;
|
||||
// the v1 shard: 3x the measured time, floored at 120 s
|
||||
assert_eq!(shard_budget(Some(THRESHOLD_2_26), 22_172), Duration::from_secs(120));
|
||||
assert_eq!(shard_budget(Some(THRESHOLD_2_25), 22_172), Duration::from_secs(122));
|
||||
assert_eq!(shard_budget(None, 0), Duration::from_secs(120), "an unknown size is the v1 shard");
|
||||
// the prototype shard (6.75 M pgas) at 2^25: 3 x 40.9 x 304 would be hours; capped at 30 minutes
|
||||
assert_eq!(shard_budget(Some(THRESHOLD_2_25), 6_751_568), Duration::from_secs(1800));
|
||||
// a 120,000-pgas block's shard at 2^26: 3 x 24.1 x 5.41 = 391 s
|
||||
assert_eq!(shard_budget(Some(THRESHOLD_2_26), 120_000), Duration::from_secs(391));
|
||||
assert_eq!(step_down(None), Some(Some(THRESHOLD_2_27)));
|
||||
assert_eq!(step_down(Some(THRESHOLD_2_27)), Some(Some(THRESHOLD_2_26)));
|
||||
assert_eq!(step_down(Some(THRESHOLD_2_26)), Some(Some(THRESHOLD_2_25)));
|
||||
assert_eq!(step_down(Some(THRESHOLD_2_25)), None, "2^25 is the last notch");
|
||||
assert_eq!(step_down(Some(100_000_000)), Some(Some(THRESHOLD_2_26)), "an odd value steps to the notch below it");
|
||||
assert_eq!(threshold_name(Some(THRESHOLD_2_26)), "2^26");
|
||||
assert_eq!(threshold_name(None), "the server's own sizes");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn with_the_patched_server_a_12_gb_card_is_on_and_a_10_gb_one_off() {
|
||||
let d = decide_with_server(&[card("nvidia", "NVIDIA GeForce RTX 4070", 12_282)], "windows", Some(true), Some(95_902), true);
|
||||
assert!(d.on, "{}", d.line);
|
||||
assert!(d.line.contains("threshold 2^26") && d.line.contains("mines and proves"), "{}", d.line);
|
||||
let d = decide_with_server(&[card("nvidia", "NVIDIA GeForce RTX 3080", 10_240)], "linux", None, None, true);
|
||||
assert!(!d.on && d.line.contains("smallest card that proves is 12 GB"), "{}", d.line);
|
||||
assert!(!d.on && d.line.contains("proves ALONE") && d.line.contains("not beside its miner"), "{}", d.line);
|
||||
let d = decide_with_server(&[card("nvidia", "NVIDIA GeForce GTX 1660", 6_144)], "linux", None, None, true);
|
||||
assert!(!d.on && d.line.contains("smallest card that proves is 8 GB"), "{}", d.line);
|
||||
// the same cards without the patched server: the old 24 GB gate
|
||||
assert!(!decide_with_server(&[card("nvidia", "NVIDIA GeForce RTX 4070", 12_282)], "linux", None, None, false).on);
|
||||
assert!(decide_with_server(&[card("nvidia", "NVIDIA GeForce RTX 5080", 16_303)], "linux", None, None, true).on);
|
||||
|
|
|
|||
|
|
@ -299,17 +299,39 @@ fn run_tool(shared: &Shared, t: &Tools, exe: &Path, args: &[String], env: &[(&st
|
|||
|
||||
/// The threshold the host gets (`SP1_GPU_ELEMENT_THRESHOLD`, as a string) on the patched server: the biggest
|
||||
/// NVIDIA card's profile, with Settings' `prove_profile` override. None on the stock server or at the server's own sizes.
|
||||
fn profile_threshold(shared: &Shared, t: &Tools) -> Option<String> {
|
||||
fn profile_threshold(shared: &Shared, t: &Tools, stepped_down: Option<Option<u64>>) -> Option<u64> {
|
||||
if t.server.is_none() {
|
||||
return None;
|
||||
}
|
||||
let vram = shared.state.lock().unwrap().mining.cards.iter().filter(|c| c.vendor == "nvidia").map(|c| c.vram_mb).max().unwrap_or(0);
|
||||
let override_name = shared.settings.lock().unwrap().prove_profile.clone();
|
||||
let p = crate::provedefault::profile(vram, &override_name);
|
||||
set(shared, |st| st.server_profile = format!("{}: {}", p.tier, p.line));
|
||||
p.threshold.map(|v| v.to_string())
|
||||
match stepped_down {
|
||||
Some(th) => {
|
||||
set(shared, |st| st.server_profile = format!("{}: stepped down to {} after a shard ran past its budget (the measured profile is {})", p.tier, crate::provedefault::threshold_name(th), crate::provedefault::threshold_name(p.threshold)));
|
||||
th
|
||||
}
|
||||
None => {
|
||||
set(shared, |st| st.server_profile = format!("{}: {}", p.tier, p.line));
|
||||
p.threshold
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Kills the GPU server inside Ubuntu-24.04 after a timeout (the host was killed by `run_tool`; the server it
|
||||
/// spawned normally dies with it, this covers a straggler) and unlinks its socket.
|
||||
fn kill_server(t: &Tools) {
|
||||
if !t.wsl {
|
||||
return;
|
||||
}
|
||||
if let Ok(file) = crate::wslhost::write_script("prove-kill", "pkill -f sp1-gpu-server 2>/dev/null; sleep 1; pkill -9 -f sp1-gpu-server 2>/dev/null; rm -f /tmp/sp1-cuda-*.sock; echo killed\n") {
|
||||
let _ = crate::platform::quiet(&mut crate::wslhost::command(&crate::platform::tool("wsl"), crate::wslhost::DISTRO, None, &file.path, true, &[])).output();
|
||||
}
|
||||
}
|
||||
|
||||
/// The prefix of the closure's error when a shard ran past its wall-clock budget (the step-down, not a proving error).
|
||||
const TIMEOUT_MARK: &str = "shard-timeout:";
|
||||
|
||||
/// Windows: puts the shipped, verified server where the SDK looks (`~/.sp1/bin/sp1-gpu-server` in Ubuntu-24.04) when
|
||||
/// the one there differs, or restores the stock one after the fallback; records the server in the state either way.
|
||||
fn install_server(shared: &Shared, t: &mut Tools, fallback_to_stock: &mut bool) {
|
||||
|
|
@ -446,6 +468,9 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
|
|||
// the fallback to the stock server (src/proverserver.rs `fallback`): set after a proof failed because the
|
||||
// patched server did not come up; the next probe restores the stock server and keeps the note
|
||||
let mut fallback_to_stock = false;
|
||||
// a shard that timed out stepped the threshold down (src/provedefault.rs `step_down`); the override holds for
|
||||
// the session and the tile names it (`server_profile`)
|
||||
let mut stepped_down: Option<Option<u64>> = None;
|
||||
// macOS and Linux: the host sits next to the engine, so its pinned ids are read at once, proving on or off
|
||||
// (Windows runs the host inside WSL2, which is probed only once proving is on)
|
||||
if !cfg!(windows) {
|
||||
|
|
@ -736,12 +761,22 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
|
|||
let prover_env = if t.cuda { "cuda" } else { "cpu" };
|
||||
// the per-card profile on the patched server (src/provedefault.rs `profile`): the threshold the server
|
||||
// sizes its buffers by, from the biggest NVIDIA card's VRAM and Settings' override; none on the stock server
|
||||
let threshold = profile_threshold(&shared, t);
|
||||
let threshold = profile_threshold(&shared, t, stepped_down);
|
||||
let threshold_s = threshold.map(|v| v.to_string());
|
||||
let mut env: Vec<(&str, &str)> = vec![("SP1_PROVER", prover_env), ("RUST_LOG", "off")];
|
||||
if let Some(th) = threshold.as_deref() {
|
||||
if let Some(th) = threshold_s.as_deref() {
|
||||
env.push(("SP1_GPU_ELEMENT_THRESHOLD", th));
|
||||
}
|
||||
let (ok, out) = run_tool(&shared, t, &t.host, &[fix_p, "--mode".into(), "compressed".into(), "--shard".into(), w.shard.to_string(), "--prover".into(), payout.clone(), "--out".into(), res_p], &env, Duration::from_secs(3 * 3600), &dir.join(format!("prove-{}-{}.log", w.number, w.shard)));
|
||||
// the wall-clock budget (src/provedefault.rs `shard_budget`): a shard that does not fit the card is
|
||||
// killed and the threshold stepped down, never left at 0% (the GPU fleet's finding, 6 October 2026);
|
||||
// the CPU prover keeps its three hours
|
||||
let budget = if t.cuda && t.server.is_some() { crate::provedefault::shard_budget(threshold, w.pgas) } else { Duration::from_secs(3 * 3600) };
|
||||
let (ok, out) = run_tool(&shared, t, &t.host, &[fix_p, "--mode".into(), "compressed".into(), "--shard".into(), w.shard.to_string(), "--prover".into(), payout.clone(), "--out".into(), res_p], &env, budget, &dir.join(format!("prove-{}-{}.log", w.number, w.shard)));
|
||||
if !ok && t.server.is_some() && out.contains("s limit)") {
|
||||
// the budget ran out: the server is killed with the host (the SDK spawned it kill-on-drop; the
|
||||
// install script's pkill covers a straggler) and the threshold steps down for the next shard
|
||||
return Err(format!("{TIMEOUT_MARK}block {} shard {} ({} pgas) ran past its {} s budget at {}", w.number, w.shard, w.pgas, budget.as_secs(), crate::provedefault::threshold_name(threshold)));
|
||||
}
|
||||
if !ok && t.server.is_some() {
|
||||
if let Some(next) = crate::proverserver::fallback(crate::proverserver::Kind::Patched, &out) {
|
||||
// the closure cannot touch the loop's state: the marker in the error is read after it
|
||||
|
|
@ -781,6 +816,32 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
|
|||
// the fallback (src/proverserver.rs): the patched server did not come up, so the stock server takes over at
|
||||
// the next probe (install_server restores it) and the tile says why
|
||||
if let Err(e) = &outcome {
|
||||
if let Some(what) = e.strip_prefix(TIMEOUT_MARK) {
|
||||
let current = stepped_down.unwrap_or_else(|| profile_threshold(&shared, t, None));
|
||||
kill_server(t);
|
||||
match crate::provedefault::step_down(current) {
|
||||
Some(next) => {
|
||||
stepped_down = Some(next);
|
||||
shared.log(&format!("prover: {what}; the GPU server was killed and the threshold steps down to {} for the next shard", crate::provedefault::threshold_name(next)));
|
||||
set(&shared, |p| {
|
||||
p.failed += 1;
|
||||
p.status = "waiting".into();
|
||||
p.message = format!("a shard ran past its budget at {}; the next shard runs at {}", crate::provedefault::threshold_name(current), crate::provedefault::threshold_name(next));
|
||||
p.current = String::new();
|
||||
});
|
||||
}
|
||||
None => {
|
||||
shared.log(&format!("prover: {what}; 2^25 is the smallest threshold, so this card cannot hold shards of that size (proving stays on for smaller ones)"));
|
||||
set(&shared, |p| {
|
||||
p.failed += 1;
|
||||
p.status = "waiting".into();
|
||||
p.message = format!("a shard ran past its budget at {}, the smallest threshold; this card cannot hold shards of that size", crate::provedefault::threshold_name(current));
|
||||
p.current = String::new();
|
||||
});
|
||||
}
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if let Some(rest) = e.strip_prefix(SERVER_DOWN_MARK) {
|
||||
fallback_to_stock = rest.starts_with("stock");
|
||||
shared.log(&format!("prover: the patched GPU server did not come up ({}); the stock server takes over at the next probe", rest.split_once(": ").map(|(_, w)| w).unwrap_or("")));
|
||||
|
|
|
|||
|
|
@ -2432,6 +2432,28 @@ it and PC 2's stock server came back from the copy (c2642ad1). Consequence: the
|
|||
(2^26), 16 GB and up at 2^27" ships with 0.3.13 on this path; the Windows installer grows about 60 MB zipped; no
|
||||
Mac or Linux user gets the server yet (no CUDA on a Mac; the Linux app proves on the CPU).
|
||||
|
||||
The hang on a point that does not fit (6 October 2026, 13:00 to 13:1xZ; the GPU fleet's finding on rented 8 and 10 GB
|
||||
cards, `docs/analysis/prover-tiers-real-cards.md`: threshold 2^27 on the 3080 and the 4060 Ti left the patched server
|
||||
at the card's memory limit at 0% for 15 minutes until killed). The cause is the one sweep 2 hit on 5 October: an
|
||||
allocation the card cannot meet makes `cudaMallocAsync` fail (`sp1-gpu/crates/cuda/src/stream.rs` 322 to 335, an
|
||||
`AllocError`), the buffer constructor panics inside a tokio worker task, and the request's future never answers; the
|
||||
client waits on its socket. Two fixes on prover-floor: (1) patch v5 (sha256 7fb7f9e886ff6b0f...): a panic hook in
|
||||
`sp1-gpu/crates/server/src/main.rs` names the stage from the panic's location (trace generation, the commit, LogUp
|
||||
GKR, the zerocheck, the jagged sumcheck, the recursion, a device allocation), says "lower SP1_GPU_ELEMENT_THRESHOLD
|
||||
one notch" when the message is an allocation, and exits 70, so the host's proof fails at once with that line; (2) the
|
||||
app (`prover.rs`, `provedefault.rs`): every shard gets a wall-clock budget, three times the measured time of a v1
|
||||
shard beside the miner at the profile's threshold (the 4070: 2^27 17.3 s, 2^26 24.1 s, 2^25 40.9 s) scaled by the
|
||||
shard's pgas, never under 120 s nor over 30 minutes (`shard_budget`); a shard past it is killed with the server and
|
||||
the threshold steps down 2^27 to 2^26 to 2^25 for the next shard (`step_down`), the tile naming the step; tests for
|
||||
both rules (148 in the app). The fleet's rows go into the tiers: 24 GB cards keep the server's own sizes (the stock
|
||||
server proves them, the 4090 at 17.4 GB in 5.6 s); 12 GB 2^26 mines and proves (10.1 to 10.2 GB beside the miner on
|
||||
headless Linux, 9.0 GB on PC 1's Windows); 10 and 8 GB prove ALONE at 2^26 (8,158 MiB own on the 3080 in 7.1 s, 7.74
|
||||
GB of 8.19 on the 4060 Ti in 9.6 s; 2^27 never; beside the miner the 3080 reaches 9,412 of 10,240 on headless Linux,
|
||||
0.8 GB spare, too close for a desktop, so the app leaves them off by default while mining and Settings can force
|
||||
it); the "empty block first" fixture (23.6 M cycles) is out of the tier line. The known-failed cases of the gate: the
|
||||
4070 at 2^27 on the 60 M-cycle prototype shard (PC 1, job `floor-pc1-hangcase`) and the fleet's 3080 at 2^27 on a
|
||||
server rebuilt from v5; their lines follow here when they land.
|
||||
|
||||
Consequences (the rule of 5 October 2026), as they stood after sweep 1 (superseded by the reading above for the 12 and 16 GB tiers): the shipped SP1 GPU server refuses every
|
||||
card under 24 GB before allocating, so 8, 12 and 16 GB NVIDIA cards cannot prove on it whatever the shard; the v1
|
||||
patch takes the shard's term out (20.5 to 12.7 GB on the v1 shard) at a 1.26x time cost (5.3 s against 4.2 s,
|
||||
|
|
|
|||
|
|
@ -297,6 +297,86 @@ index 5dccd9d..574d4fa 100644
|
|||
.await,
|
||||
recursion_verifier,
|
||||
)
|
||||
diff --git a/sp1-gpu/crates/server/src/main.rs b/sp1-gpu/crates/server/src/main.rs
|
||||
index 65e94f5..498357c 100644
|
||||
--- a/sp1-gpu/crates/server/src/main.rs
|
||||
+++ b/sp1-gpu/crates/server/src/main.rs
|
||||
@@ -17,9 +17,55 @@ struct Args {
|
||||
version: bool,
|
||||
}
|
||||
|
||||
+/// Igneum prover-floor patch (6 October 2026, the GPU fleet's finding): a panic inside a prover task (an allocation
|
||||
+/// the card cannot meet, `cudaMallocAsync` failing and `Buffer::with_capacity_in` panicking in a tokio worker) left
|
||||
+/// the request's future waiting for ever, the client on its socket, the card at 0% for the 15 minutes until someone
|
||||
+/// killed it (the 8 GB and 10 GB cards at threshold 2^27). The server must fail the shard instead: this hook names
|
||||
+/// the stage from the panic's location and exits, so the client's proof fails at once with the server's last line.
|
||||
+fn stage_of(location: &str) -> &'static str {
|
||||
+ let l = location.to_ascii_lowercase();
|
||||
+ if l.contains("jagged_tracegen") || l.contains("/tracegen") {
|
||||
+ "trace generation"
|
||||
+ } else if l.contains("commit") || l.contains("basefold") || l.contains("merkle") {
|
||||
+ "the commit (codewords and Merkle trees)"
|
||||
+ } else if l.contains("logup_gkr") {
|
||||
+ "LogUp GKR"
|
||||
+ } else if l.contains("zerocheck") {
|
||||
+ "the zerocheck"
|
||||
+ } else if l.contains("jagged") {
|
||||
+ "the jagged sumcheck"
|
||||
+ } else if l.contains("prover_components") || l.contains("recursion") || l.contains("sp1-prover") || l.contains("sp1_prover") {
|
||||
+ "the recursion (compression)"
|
||||
+ } else if l.contains("cuda") || l.contains("slop") {
|
||||
+ "a device allocation"
|
||||
+ } else {
|
||||
+ "the prover"
|
||||
+ }
|
||||
+}
|
||||
+
|
||||
+fn install_abort_on_panic() {
|
||||
+ std::panic::set_hook(Box::new(|info| {
|
||||
+ let location = info.location().map(|l| format!("{}:{}", l.file(), l.line())).unwrap_or_else(|| "unknown".into());
|
||||
+ let message = info
|
||||
+ .payload()
|
||||
+ .downcast_ref::<&str>()
|
||||
+ .map(|s| s.to_string())
|
||||
+ .or_else(|| info.payload().downcast_ref::<String>().cloned())
|
||||
+ .unwrap_or_default();
|
||||
+ let oom = message.to_ascii_lowercase().contains("alloc") || message.contains("MemoryAllocation") || message.contains("OUT_OF_MEMORY");
|
||||
+ eprintln!(
|
||||
+ "FLOOR abort: {} failed at {location}: {message}{}; the server exits so the client's proof fails instead of waiting",
|
||||
+ stage_of(&location),
|
||||
+ if oom { " (the card's memory could not meet an allocation: lower SP1_GPU_ELEMENT_THRESHOLD one notch)" } else { "" }
|
||||
+ );
|
||||
+ std::process::exit(70);
|
||||
+ }));
|
||||
+}
|
||||
+
|
||||
#[tokio::main]
|
||||
#[allow(clippy::print_stdout)]
|
||||
async fn main() {
|
||||
+ install_abort_on_panic();
|
||||
tracing_subscriber::fmt::init();
|
||||
|
||||
let args = Args::parse();
|
||||
@@ -40,3 +86,19 @@ async fn main() {
|
||||
eprintln!("Error running server: {e}");
|
||||
}
|
||||
}
|
||||
+
|
||||
+#[cfg(test)]
|
||||
+mod floor_tests {
|
||||
+ use super::stage_of;
|
||||
+
|
||||
+ #[test]
|
||||
+ fn the_stage_is_named_from_the_panic_location() {
|
||||
+ assert_eq!(stage_of("sp1-gpu/crates/jagged_tracegen/src/lib.rs:240"), "trace generation");
|
||||
+ assert_eq!(stage_of("sp1-gpu/crates/basefold/src/fri.rs:97"), "the commit (codewords and Merkle trees)");
|
||||
+ assert_eq!(stage_of("sp1-gpu/crates/logup_gkr/src/tracegen.rs:72"), "LogUp GKR");
|
||||
+ assert_eq!(stage_of("sp1-gpu/crates/zerocheck/src/prover.rs:1163"), "the zerocheck");
|
||||
+ assert_eq!(stage_of("sp1-gpu/crates/prover_components/src/builder.rs:70"), "the recursion (compression)");
|
||||
+ assert_eq!(stage_of("sp1-gpu/crates/cuda/src/stream.rs:330"), "a device allocation");
|
||||
+ assert_eq!(stage_of("somewhere/else.rs:1"), "the prover");
|
||||
+ }
|
||||
+}
|
||||
diff --git a/sp1-gpu/crates/server/src/server.rs b/sp1-gpu/crates/server/src/server.rs
|
||||
index 4035f1f..0d0d907 100644
|
||||
--- a/sp1-gpu/crates/server/src/server.rs
|
||||
|
|
|
|||
|
|
@ -39,7 +39,7 @@ cd "\$SRC"
|
|||
git checkout -q -- . && git clean -qfd sp1-gpu/crates >/dev/null 2>&1
|
||||
echo "RESULT source \$(git describe --tags --always) \$(git rev-parse HEAD)"
|
||||
git apply "\$PATCH" || { echo "RESULT build_failed patch does not apply"; exit 2; }
|
||||
touch sp1-gpu/crates/prover_components/src/builder.rs sp1-gpu/crates/jagged_tracegen/src/lib.rs sp1-gpu/crates/server/src/server.rs
|
||||
git diff --name-only | xargs touch
|
||||
echo "RESULT patched \$(git diff --stat | tail -1)"
|
||||
# Go, as the release workflow installs it (the server's native-gnark feature compiles the gnark library with go;
|
||||
# the wrap path is never run by a compressed proof, but the feature is upstream's and stays): a pinned tarball
|
||||
|
|
|
|||
108
tools/prover-floor/pc1-hangcase.ps1
Normal file
108
tools/prover-floor/pc1-hangcase.ps1
Normal file
|
|
@ -0,0 +1,108 @@
|
|||
# Prover floor, route 1 (6 October 2026, the project lead: a real 12 GB card lands today; can it mine AND prove?). The same
|
||||
# fixtures as sweeps 3 and 4 on the card itself: `--mode shard` (the core proof, its verify, then the compressed
|
||||
# proof, in one run; the 1-s sampler split at the core RESULT gives the core-only peak) at thresholds 2^26, 2^27
|
||||
# and 2^25 (the core-only mine-and-prove profile), the v1 shard and an empty block. Two phases, each its own job: FLOOR_PHASE=alone (publish WITH
|
||||
# --stop-miners; measures the card's own idle first) and FLOOR_PHASE=miner (publish WITHOUT --stop-miners; measures
|
||||
# the miner's working set on the card first). Works on PC 1 or PC 2: the card is found by its memory (under
|
||||
# 13,000 MiB; FLOOR_CARD_INDEX overrides) and every proof runs on it through IGNEUM_CUDA_DEVICE (the host passes
|
||||
# the index to the SDK, which starts the server with CUDA_VISIBLE_DEVICES=<index>; CUDA_DEVICE_ORDER=PCI_BUS_ID
|
||||
# keeps nvidia-smi's and CUDA's numbering the same). The kit: /opt/igneum-floor/home/.sp1/bin/sp1-gpu-server (the
|
||||
# v3 build, pc2-build-server.ps1 on this PC) and the host built from the fetched package igneum-prove-wsl2-floor
|
||||
# (jobs\floor-pc1-kit\, the fetch job floor-pc1-kit with --extract); each is tested first and named if missing. Never touches the
|
||||
# app's /opt/igneum host or /root/.sp1 server; leaves the prover ON; the runner restores the miners.
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$urlFile = if ($env:IGNEUM_APP_DIR) { Join-Path $env:IGNEUM_APP_DIR 'app.url' } else { Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
if (-not (Test-Path $urlFile)) { $urlFile = Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
$base = (Get-Content $urlFile -Raw).Trim().TrimEnd('/')
|
||||
function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') }
|
||||
function Prove($on) { try { (Invoke-RestMethod -Method Post -Uri "$base/api/prove" -ContentType 'application/json' -Body (@{on=$on} | ConvertTo-Json -Compress) -TimeoutSec 10) | ConvertTo-Json -Compress } catch { "error: $_" } }
|
||||
$phase = 'hangcase' # the known-failed case: one point that cannot fit, must FAIL in seconds on the v5 server (publish WITH --stop-miners)
|
||||
"RESULT start $(Stamp) phase=$phase machine=$env:COMPUTERNAME"
|
||||
"RESULT cards $((& nvidia-smi --query-gpu=index,name,memory.total,memory.used,pci.bus_id --format=csv,noheader,nounits 2>$null) -join ' | ')"
|
||||
# the 12 GB card: the first index whose memory.total is under 13,000 MiB (a 3060 12 GB reads 12,288; a 4070 12,282)
|
||||
$cards = & nvidia-smi --query-gpu=index,memory.total --format=csv,noheader,nounits 2>$null | ForEach-Object { $p = $_ -split ',\s*'; [pscustomobject]@{ index = [int]$p[0]; total = [int]$p[1] } }
|
||||
$card = if ($env:FLOOR_CARD_INDEX) { [int]$env:FLOOR_CARD_INDEX } else { ($cards | Where-Object { $_.total -lt 13000 } | Select-Object -First 1).index }
|
||||
if ($null -eq $card) { "RESULT measure_failed no card under 13,000 MiB on this PC (set FLOOR_CARD_INDEX to force one)"; "RESULT end $(Stamp)"; exit 2 }
|
||||
"RESULT card index=$card total_mib=$(($cards | Where-Object { $_.index -eq $card }).total)"
|
||||
"RESULT prover_off $(Stamp) $(Prove $false)"
|
||||
Start-Sleep -Seconds 30
|
||||
$job = $env:IGNEUM_JOB_DIR; if (-not $job) { $job = Join-Path $env:TEMP 'igneum-floor-card' }; New-Item -ItemType Directory -Force -Path $job | Out-Null
|
||||
function WslPath($p) { $w = (& wsl.exe -d Ubuntu-24.04 -u root -- wslpath -a ($p -replace '\\', '/') 2>$null); if ($w) { ($w -replace "`0", '').Trim() } else { '/mnt/c' + ($p.Substring(2) -replace '\\', '/') } }
|
||||
$jobW = WslPath $job
|
||||
# the kit (the wiped-jobs-folder class): the fetched package under the jobs folder, tested before use
|
||||
$jobs = Split-Path $env:IGNEUM_JOB_DIR
|
||||
# the app stores a fetch under the FETCH JOB's id (jobs\floor-pc1-kit\, 11:32Z: the --to name became the file's name), so
|
||||
# the kit folder is the fetch id; FLOOR_KIT_DIR names another (a fetch to PC 2 would have its own id)
|
||||
$kit = Join-Path $jobs $(if ($env:FLOOR_KIT_DIR) { $env:FLOOR_KIT_DIR } else { 'floor-pc1-kit' })
|
||||
$kitOk = Test-Path (Join-Path $kit 'igneum-prove-wsl2\package\proving\igneum-prove\Cargo.toml')
|
||||
"RESULT kit $(if ($kitOk) { "present $kit" } else { "MISSING: publish the fetch job of igneum-prove-wsl2-floor.zip to this machine (--dir jobs --extract, the job id names the folder) first" })"
|
||||
$kitW = if ($kitOk) { WslPath (Join-Path $kit 'igneum-prove-wsl2\package') } else { '/nonexistent' }
|
||||
$bash = @'
|
||||
set -uo pipefail
|
||||
export PATH="$HOME/.cargo/bin:$PATH" CUDA_DEVICE_ORDER=PCI_BUS_ID
|
||||
CUDA_DIR="$(ls -d /usr/local/cuda-12.* 2>/dev/null | sort -V | tail -1 || true)"; [ -n "$CUDA_DIR" ] && export PATH="$CUDA_DIR/bin:$PATH" && export LD_LIBRARY_PATH="$CUDA_DIR/lib64:/usr/lib/wsl/lib:${LD_LIBRARY_PATH:-}"
|
||||
stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; }
|
||||
JOB='JOBW_PLACEHOLDER'; KIT='KITW_PLACEHOLDER'; CARD=CARD_PLACEHOLDER; PHASE='PHASE_PLACEHOLDER'
|
||||
FLOORHOME=/opt/igneum-floor/home; SRV=$FLOORHOME/.sp1/bin/sp1-gpu-server; H=/opt/igneum-floor/host/igneum-prove-host
|
||||
echo "RESULT wsl_cards $(nvidia-smi --query-gpu=index,name,memory.total,pci.bus_id --format=csv,noheader,nounits 2>/dev/null | tr '\n' ';')"
|
||||
[ -x "$SRV" ] || { echo "RESULT measure_failed no patched server at $SRV: run the build job (tools/prover-floor/pc2-build-server.ps1) on this PC first (25 to 45 min cold)"; exit 2; }
|
||||
echo "RESULT patched_server sha256=$(sha256sum $SRV | cut -c1-16) version=$($SRV --version 2>/dev/null)"
|
||||
if [ ! -x "$H" ]; then
|
||||
[ -f "$KIT/proving/igneum-prove/Cargo.toml" ] || { echo "RESULT measure_failed no host and no kit: fetch igneum-prove-wsl2-floor.zip first"; exit 2; }
|
||||
echo "STAGE host build $(stamp)"
|
||||
mkdir -p /opt/igneum-floor/host-src && rsync -a "$KIT/" /opt/igneum-floor/host-src/ && find /opt/igneum-floor/host-src -type f -exec touch {} +
|
||||
TD=/root/igneum-prove/proving/igneum-prove/target; [ -d "$TD" ] || TD=/opt/igneum-floor/host-target
|
||||
( cd /opt/igneum-floor/host-src/proving/igneum-prove && CARGO_TARGET_DIR=$TD nice -n 19 cargo build --release -p igneum-prove-host --features igneum-prove-host/cuda > $JOB/host-build.log 2>&1 ) || { echo "RESULT measure_failed host build; tail:"; tail -n 30 $JOB/host-build.log; exit 2; }
|
||||
mkdir -p /opt/igneum-floor/host && cp $TD/release/igneum-prove-host /opt/igneum-floor/host/ && echo "RESULT host built sha256=$(sha256sum $H | cut -c1-16)"
|
||||
fi
|
||||
FX="/opt/igneum-floor/host-src/proving/fixtures"; [ -d "$FX" ] || FX="$KIT/proving/fixtures"
|
||||
V1="$FX/fees-v1-shards2.json"; EMPTY="$FX/block-72854-empty-block-first.json"
|
||||
echo "RESULT host_ids $($H --mode id 2>/dev/null | tr '\n' ' ' | cut -c1-200)"
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock
|
||||
smi() { nvidia-smi -i $CARD --query-gpu=memory.used --format=csv,noheader,nounits | head -1; }
|
||||
echo "RESULT card_before phase=$PHASE used_mib=$(smi) (alone: the card's own idle; miner: idle + the miner's working set on this card)"
|
||||
runshard() { # name fixture env...
|
||||
local name="$1" fx="$2"; shift 2
|
||||
local tag="$name-$(basename $fx .json)"
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock
|
||||
local csv="$JOB/smi-$tag.csv" log="$JOB/log-$tag.txt"
|
||||
nvidia-smi -i $CARD --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null &
|
||||
local SMI=$!
|
||||
local t0=$(date +%s)
|
||||
env HOME=$FLOORHOME SP1_PROVER=cuda IGNEUM_CUDA_DEVICE=$CARD RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode shard --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1
|
||||
local rc=$?
|
||||
local wall=$(( $(date +%s) - t0 ))
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
local n=$(wc -l < "$csv")
|
||||
local core=$(grep -E "^RESULT core shard" "$log" | tail -1); local comp=$(grep -E "^RESULT compressed shard" "$log" | tail -1)
|
||||
local cyc=$(grep -E "^RESULT execute shard" "$log" | tail -1 | sed -E 's/.*: ([0-9]+) cycles.*/\1/')
|
||||
local core_at=$(echo "$core" | sed -E 's/.* at ([0-9T:.-]+Z?)$/\1/'); local core_epoch=$(date -u -d "${core_at}" +%s 2>/dev/null || echo 0)
|
||||
local split=$(( core_epoch - t0 )); [ $split -lt 1 ] && split=$n
|
||||
local corepeak=$(awk -F', *' -v s=$split 'NR<=s+1 { if ($2+0 > m) m=$2+0 } END { print m+0 }' "$csv")
|
||||
local peak=$(awk -F', *' '{ if ($2+0 > m) m=$2+0 } END { print m+0 }' "$csv")
|
||||
local cl=$(echo "$core" | sed -E 's/.*prove ([0-9.]+) s, proof ([0-9]+) bytes, verify ([0-9.]+) s, ([A-Z ]+);.*/core_s=\1 core_bytes=\2 core_verify_s=\3 core_\4/')
|
||||
local pl=$(echo "$comp" | sed -E 's/.*prove ([0-9.]+) s, proof ([0-9]+) bytes, verify ([0-9.]+) s, ([A-Z ]+);.*/compressed_s=\1 compressed_bytes=\2 compressed_verify_s=\3 compressed_\4/')
|
||||
local err=$(grep -iE "error|panick|out of memory|OOM|unsupported" "$log" | grep -v "^FLOOR" | head -1 | cut -c1-200)
|
||||
echo "RESULT card12 phase=$PHASE cfg=$name fixture=$(basename $fx .json) card=$CARD core_peak_mib=$corepeak full_peak_mib=$peak split_s=$split samples=$n wall_s=$wall cycles=${cyc:-na} ${cl:-no_core_result} ${pl:-no_compressed_result} exit=$rc env='$*' ${err:+err=$err}"
|
||||
grep -E "^FLOOR (memory|opts|grow)|^RESULT cuda device" "$log" | sed "s/^/RESULT floorline phase=$PHASE cfg=$name fixture=$(basename $fx .json) /" | head -10
|
||||
}
|
||||
# the known-failed case (6 October 2026, the GPU fleet's finding): threshold 2^27 on the 60 M-cycle prototype shard
|
||||
# took 13,459 MiB on the 5090's allocation, over this card's 12,282, so the old server hung at 0% for 15 minutes; the
|
||||
# v5 server (the panic hook) must fail it within seconds and name the stage. The sampler's wall and the FLOOR line
|
||||
# are the result; a timeout here (the job's cap) is the FAILED verdict for the gate.
|
||||
FULL="$FX/block-338-shard1.json"
|
||||
echo "RESULT server_sha256 $(sha256sum /opt/igneum-floor/home/.sp1/bin/sp1-gpu-server | cut -c1-64) (v5 expected)"
|
||||
t_case=$(date +%s)
|
||||
runshard e27full "$FULL" SP1_GPU_ELEMENT_THRESHOLD=134217728
|
||||
echo "RESULT hangcase seconds_to_close=$(( $(date +%s) - t_case ))"
|
||||
grep -E "FLOOR abort|Error|error|CudaClientError|panicked" "$JOB/log-e27full-block-338-shard1.txt" | head -4 | sed 's/^/RESULT hangline /' | cut -c1-400
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
echo "RESULT card_after phase=$PHASE used_mib=$(smi)"
|
||||
echo "RESULT measure_end $(stamp)"
|
||||
'@
|
||||
$bash = $bash.Replace('JOBW_PLACEHOLDER', $jobW).Replace('KITW_PLACEHOLDER', $kitW).Replace('CARD_PLACEHOLDER', "$card").Replace('PHASE_PLACEHOLDER', $phase)
|
||||
$bashFile = Join-Path $job 'card12.sh'
|
||||
[IO.File]::WriteAllText($bashFile, ($bash -replace "`r`n", "`n"), (New-Object System.Text.UTF8Encoding $false))
|
||||
& wsl.exe -d Ubuntu-24.04 -u root -- bash (WslPath $bashFile) 2>&1 | ForEach-Object { ($_ -replace "`0", '') }
|
||||
"RESULT prover_on $(Stamp) $(Prove $true)"
|
||||
"RESULT end $(Stamp)"
|
||||
File diff suppressed because one or more lines are too long
Loading…
Reference in a new issue