Igneum Miner 0.3.12: the fresh-record rule switch (proving v1), the segment-aligned prover, Ember Tune, the GPU list in performance order
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
commit
017c7db517
72 changed files with 6292 additions and 327 deletions
8
.github/workflows/ci.yml
vendored
8
.github/workflows/ci.yml
vendored
|
|
@ -69,12 +69,16 @@ jobs:
|
|||
run: bash tools/ci/no-conflict-markers.sh
|
||||
- name: copied sources are re-stamped before a build
|
||||
run: bash tools/ci/copied-sources-check.sh
|
||||
- name: second-engine playbooks log to a file and end their tree (C35)
|
||||
run: bash tools/ci/second-engine-check.sh
|
||||
- name: the signer is never piped into head
|
||||
run: bash tools/ci/signer-pipe-check.sh
|
||||
- name: bash bodies in PowerShell job scripts pass bash -n, the lost-quote class (self-test first, then the tree)
|
||||
run: bash tools/ci/bash-body-check.sh --self-test && bash tools/ci/bash-body-check.sh
|
||||
- name: run jobs test their fetched kit before use, the wiped-jobs-folder class (self-test first, then the tree)
|
||||
run: bash tools/ci/kit-path-check.sh --self-test && bash tools/ci/kit-path-check.sh
|
||||
- name: every Windows spawn of the app runs with a hidden console (self-test first, then the tree)
|
||||
run: node tools/ci/windows-spawn-check.mjs --self-test && node tools/ci/windows-spawn-check.mjs
|
||||
- name: pinned guest programs match their manifest and are built only by pin-guests.sh
|
||||
run: bash tools/ci/pinned-guests-check.sh
|
||||
- name: root prover playbooks kill the GPU server and unlink its socket (the root-socket class, 5 October 2026)
|
||||
|
|
@ -91,6 +95,6 @@ jobs:
|
|||
- name: ship tool self-test (version bump, the dl-both and public manifest helpers)
|
||||
run: node tools/ship-app.mjs --self-test
|
||||
- name: relay unit tests (parsers, secret compare, the wake endpoint)
|
||||
run: node --test relay/test/parse.test.mjs relay/test/auth.test.mjs relay/test/wake.test.mjs
|
||||
run: node --test relay/test/parse.test.mjs relay/test/auth.test.mjs relay/test/wake.test.mjs relay/test/ember.test.mjs
|
||||
- name: miner app notice strip and update card (ordering, keys, wording, timers, when the card shows)
|
||||
run: node --test app/igneum-app/ui/notices.test.mjs app/igneum-app/ui/update-card.test.mjs app/igneum-app/ui/view.test.mjs
|
||||
run: node --test app/igneum-app/ui/notices.test.mjs app/igneum-app/ui/update-card.test.mjs app/igneum-app/ui/view.test.mjs app/igneum-app/ui/tune-line.test.mjs
|
||||
|
|
|
|||
2
app/igneum-app/Cargo.lock
generated
2
app/igneum-app/Cargo.lock
generated
|
|
@ -219,7 +219,7 @@ dependencies = [
|
|||
|
||||
[[package]]
|
||||
name = "igneum-app"
|
||||
version = "0.3.11"
|
||||
version = "0.3.12"
|
||||
dependencies = [
|
||||
"ed25519-dalek",
|
||||
"getrandom",
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
[package]
|
||||
name = "igneum-app"
|
||||
version = "0.3.11"
|
||||
version = "0.3.12"
|
||||
edition = "2021"
|
||||
description = "Igneum Miner engine: supervises the node, the miner and the GPU workers, and serves the dashboard on 127.0.0.1"
|
||||
license = "MIT"
|
||||
|
|
|
|||
|
|
@ -6,8 +6,8 @@
|
|||
1 ICON "igneum.ico"
|
||||
|
||||
1 VERSIONINFO
|
||||
FILEVERSION 0,3,11,0
|
||||
PRODUCTVERSION 0,3,11,0
|
||||
FILEVERSION 0,3,12,0
|
||||
PRODUCTVERSION 0,3,12,0
|
||||
FILEFLAGSMASK 0x3fL
|
||||
FILEFLAGS 0x0L
|
||||
FILEOS VOS_NT_WINDOWS32
|
||||
|
|
@ -20,12 +20,12 @@ BEGIN
|
|||
BEGIN
|
||||
VALUE "CompanyName", "Igneum"
|
||||
VALUE "FileDescription", "Igneum Miner engine"
|
||||
VALUE "FileVersion", "0.3.11"
|
||||
VALUE "FileVersion", "0.3.12"
|
||||
VALUE "InternalName", "igneum-app"
|
||||
VALUE "LegalCopyright", "Igneum contributors"
|
||||
VALUE "OriginalFilename", "igneum-app.exe"
|
||||
VALUE "ProductName", "Igneum Miner"
|
||||
VALUE "ProductVersion", "0.3.11"
|
||||
VALUE "ProductVersion", "0.3.12"
|
||||
END
|
||||
END
|
||||
BLOCK "VarFileInfo"
|
||||
|
|
|
|||
|
|
@ -29,6 +29,16 @@ pub struct CardPref {
|
|||
pub sweep_watts: f64,
|
||||
#[serde(default)]
|
||||
pub sweep_mhs: f64,
|
||||
/// Ember Tune (src/ember.rs): the clock cap the last tune chose (0 = unlocked), the driver and program class it
|
||||
/// ran under (a change makes the card due again), and the plan that produced it (full | confirm | baseline)
|
||||
#[serde(default)]
|
||||
pub sweep_clock_mhz: u32,
|
||||
#[serde(default)]
|
||||
pub sweep_driver: String,
|
||||
#[serde(default)]
|
||||
pub sweep_class: String,
|
||||
#[serde(default)]
|
||||
pub sweep_source: String,
|
||||
}
|
||||
|
||||
#[derive(Clone, Serialize, Deserialize)]
|
||||
|
|
@ -70,9 +80,16 @@ pub struct Settings {
|
|||
#[serde(default)]
|
||||
pub prove: bool,
|
||||
/// The efficiency sweep (src/sweep.rs): once after install, then weekly, each NVIDIA card's cap is stepped from
|
||||
/// 100% to 50% on the live program and held at the best MH per watt. Default on. A pinned card is skipped.
|
||||
#[serde(default = "yes")]
|
||||
/// 100% to 50% on the live program and held at the best MH per watt. Default off; implied by `power_control`
|
||||
/// (on when that is switched on, never effective while it is off). A pinned card is skipped.
|
||||
#[serde(default)]
|
||||
pub sweep: bool,
|
||||
/// Power control (the project lead, 5 October 2026: "if we don't have to ask then don't ask"): the NVIDIA power cap and the
|
||||
/// efficiency sweep need administrator rights (one UAC prompt on Windows). Default OFF on every machine; the app
|
||||
/// never raises the prompt on its own. Switching it on asks once, at that moment; a refused, cancelled or
|
||||
/// unanswered prompt switches it back off with a notice, no retries.
|
||||
#[serde(default)]
|
||||
pub power_control: bool,
|
||||
/// When this install first ran (unix s), for the "first hour after install" sweep.
|
||||
#[serde(default)]
|
||||
pub installed_at: u64,
|
||||
|
|
@ -102,16 +119,26 @@ fn yes() -> bool {
|
|||
|
||||
impl Default for Settings {
|
||||
fn default() -> Settings {
|
||||
Settings { setup_done: false, address: String::new(), address_source: String::new(), key_saved: false, identities: 1, cards: HashMap::new(), display_name: String::new(), vote: true, paused: false, accepted_total: 0, auto_update: true, remote_jobs: true, prove: false, sweep: true, installed_at: 0, dev_fee: true, fee_total: 0, proof_verify_trust: false, prove_default_applied: false }
|
||||
Settings { setup_done: false, address: String::new(), address_source: String::new(), key_saved: false, identities: 1, cards: HashMap::new(), display_name: String::new(), vote: true, paused: false, accepted_total: 0, auto_update: true, remote_jobs: true, prove: false, sweep: true, power_control: false, installed_at: 0, dev_fee: true, fee_total: 0, proof_verify_trust: false, prove_default_applied: false }
|
||||
}
|
||||
}
|
||||
|
||||
impl Settings {
|
||||
pub fn load(path: &Path) -> Settings {
|
||||
let mut s: Settings = std::fs::read_to_string(path).ok().and_then(|t| serde_json::from_str(&t).ok()).unwrap_or_default();
|
||||
let mut dirty = false;
|
||||
if s.installed_at == 0 {
|
||||
// an install from before the sweep existed counts as installed now: it gets its first-hour sweep
|
||||
s.installed_at = crate::platform::unix_now();
|
||||
dirty = true;
|
||||
}
|
||||
if s.sweep && !s.power_control {
|
||||
// the sweep is implied by power control (5 October 2026): an install from before that setting carried
|
||||
// sweep = true by default; it no longer prompts on its own
|
||||
s.sweep = false;
|
||||
dirty = true;
|
||||
}
|
||||
if dirty {
|
||||
s.save(path);
|
||||
}
|
||||
s
|
||||
|
|
|
|||
|
|
@ -17,6 +17,8 @@ pub struct Bins {
|
|||
pub metal: Option<std::path::PathBuf>,
|
||||
pub cuda: Option<std::path::PathBuf>,
|
||||
pub opencl: Option<std::path::PathBuf>,
|
||||
/// igneum-gpu-telemetry: AMD power, heat, fans and clocks (proto-opencl/gpu-telemetry.c), 5 October 2026
|
||||
pub telemetry: Option<std::path::PathBuf>,
|
||||
pub dir: std::path::PathBuf,
|
||||
}
|
||||
|
||||
|
|
@ -99,6 +101,7 @@ fn card(index: usize, name: &str, vendor: &str, worker: &str, detail: &str, devi
|
|||
device: device.into(),
|
||||
enabled: true,
|
||||
state: "off".into(),
|
||||
amd_ordinal: -1,
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
|
@ -725,7 +728,7 @@ pub fn find_bins() -> Result<Bins, String> {
|
|||
// the prebuilt CUDA worker needs NVIDIA's nvrtc64_*_0.dll next to it (as igneum-common.ps1 checks)
|
||||
let nvrtc = std::fs::read_dir(dir).ok().map(|rd| rd.flatten().any(|e| { let n = e.file_name().to_string_lossy().to_ascii_lowercase(); n.starts_with("nvrtc64_") && n.ends_with("_0.dll") })).unwrap_or(false);
|
||||
let cuda = opt("igneum-worker-cuda").filter(|_| nvrtc || cfg!(not(windows)));
|
||||
return Ok(Bins { node, miner, metal: opt("igneum-bench"), cuda, opencl: opt("igneum-worker-opencl"), dir: dir.clone() });
|
||||
return Ok(Bins { node, miner, metal: opt("igneum-bench"), cuda, opencl: opt("igneum-worker-opencl"), telemetry: opt("igneum-gpu-telemetry"), dir: dir.clone() });
|
||||
}
|
||||
}
|
||||
Err(format!("igneumd and igneum-miner were not found next to the app (looked in {})", candidates.iter().map(|c| c.display().to_string()).collect::<Vec<_>>().join(", ")))
|
||||
|
|
|
|||
1033
app/igneum-app/src/ember.rs
Normal file
1033
app/igneum-app/src/ember.rs
Normal file
File diff suppressed because it is too large
Load diff
File diff suppressed because it is too large
Load diff
|
|
@ -161,6 +161,13 @@ pub fn apply_pref(c: &mut CardState, p: &CardPref) {
|
|||
c.sweep_watts = p.sweep_watts;
|
||||
c.sweep_mhs = p.sweep_mhs;
|
||||
c.sweep_at = p.sweep_at as f64;
|
||||
// Ember Tune (src/ember.rs): the clock cap the last tune chose, its plan, and the row's Tuned line
|
||||
c.tune_clock_mhz = p.sweep_clock_mhz;
|
||||
c.clock_cap_mhz = if p.pinned || p.sweep_source == "baseline" { 0 } else { p.sweep_clock_mhz };
|
||||
c.tune_source = p.sweep_source.clone();
|
||||
if p.sweep_mhs > 0.0 && p.sweep_watts > 0.0 {
|
||||
c.tune_line = crate::ember::tuned_line(p.sweep_mhs, p.sweep_watts, p.sweep_eff);
|
||||
}
|
||||
}
|
||||
|
||||
/// A card a re-detection added (or brought back): the saved choice if there is one, else the detect defaults it
|
||||
|
|
|
|||
|
|
@ -1115,16 +1115,11 @@ fn run_script(shared: &Arc<Shared>, job: &Job, sink: &Sink, jobs_dir: &Path, dat
|
|||
.map(|(k, v)| format!("$env:{k} = '{}'\r\n", v.replace('\'', "''")))
|
||||
.collect();
|
||||
let wrapper = dir.join("elevated.ps1");
|
||||
let w = format!("{env_lines}& '{}' *>&1 | Out-File -FilePath '{}' -Encoding utf8\r\nexit $LASTEXITCODE\r\n", script.display().to_string().replace('\'', "''"), out_file.display().to_string().replace('\'', "''"));
|
||||
let w = elevated_wrapper(&env_lines, &script.display().to_string(), &out_file.display().to_string());
|
||||
std::fs::write(&wrapper, [b"\xEF\xBB\xBF".as_slice(), w.as_bytes()].concat()).map_err(|e| e.to_string())?;
|
||||
let _ = std::fs::remove_file(&out_file);
|
||||
let inner = format!("-NoProfile -ExecutionPolicy Bypass -File \"{}\"", wrapper.display());
|
||||
// A refused or unanswered UAC prompt makes Start-Process throw (`$p` stays null) and `exit $p.ExitCode`
|
||||
// would exit 0: the 5 October 2026 driver job on PC 1 was reported "done" after Windows cancelled its
|
||||
// prompt at 122 s. The launch failure is exit 251 and says so on stderr.
|
||||
let ps = format!("try {{ $p = Start-Process -FilePath powershell.exe -ArgumentList '{}' -Verb RunAs -Wait -WindowStyle Hidden -PassThru -ErrorAction Stop }} catch {{ Write-Error ('elevated launch failed (UAC refused, cancelled or timed out): ' + $_.Exception.Message); exit 251 }}; if ($null -eq $p) {{ Write-Error 'elevated launch failed: no process'; exit 251 }}; exit $p.ExitCode", inner.replace('\'', "''"));
|
||||
cmd = Command::new(crate::platform::tool("powershell"));
|
||||
cmd.args(["-NoProfile", "-ExecutionPolicy", "Bypass", "-Command", &ps]);
|
||||
cmd = crate::platform::elevated_command("powershell.exe", &inner);
|
||||
} else if shell == "powershell" {
|
||||
cmd = Command::new(crate::platform::tool("powershell"));
|
||||
cmd.args(["-NoProfile", "-ExecutionPolicy", "Bypass", "-File", &script.display().to_string()]);
|
||||
|
|
@ -1134,10 +1129,17 @@ fn run_script(shared: &Arc<Shared>, job: &Job, sink: &Sink, jobs_dir: &Path, dat
|
|||
}
|
||||
cmd.current_dir(&dir);
|
||||
job_env(&mut cmd, shared, job, &dir, data_root);
|
||||
// the elevated script's output reaches this side through a file: follow it while the script runs, so the
|
||||
// 5-minute progress reports carry its lines (6 October 2026: a 35-minute run that never mined showed only
|
||||
// "script running" until it ended; the lines that said why were in the file the whole time)
|
||||
let follow = if elevated { Some(follow_file(sink, out_file.clone())) } else { None };
|
||||
let ran = run_streamed(&mut cmd, sink, ctl, limit, shared, job, started, "script running")?;
|
||||
if elevated {
|
||||
if let Some(f) = follow {
|
||||
f.stop.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||
let seen = f.handle.join().unwrap_or(0);
|
||||
// the tail the follower had not read when the script ended
|
||||
if let Ok(t) = std::fs::read_to_string(&out_file) {
|
||||
for l in t.lines() {
|
||||
for l in t.lines().skip(seen) {
|
||||
sink.line(l);
|
||||
}
|
||||
}
|
||||
|
|
@ -1145,6 +1147,56 @@ fn run_script(shared: &Arc<Shared>, job: &Job, sink: &Sink, jobs_dir: &Path, dat
|
|||
finish_ran(ran, "script")
|
||||
}
|
||||
|
||||
/// Follows a file another process writes (the elevated script's output), feeding each new complete line to the
|
||||
/// sink every 2 s until stopped; returns how many lines it delivered, so the caller can hand over the remainder.
|
||||
struct Follow {
|
||||
stop: Arc<std::sync::atomic::AtomicBool>,
|
||||
handle: std::thread::JoinHandle<usize>,
|
||||
}
|
||||
|
||||
fn follow_file(sink: &Sink, path: PathBuf) -> Follow {
|
||||
let stop = Arc::new(std::sync::atomic::AtomicBool::new(false));
|
||||
let stop2 = stop.clone();
|
||||
let s = Sink { shared: sink.shared.clone(), id: sink.id.clone(), dir: sink.dir.clone(), log_path: sink.log_path.clone(), file: Mutex::new(std::fs::OpenOptions::new().append(true).open(&sink.log_path).ok()), results: Mutex::new(Vec::new()) };
|
||||
let handle = std::thread::spawn(move || {
|
||||
let mut seen = 0usize;
|
||||
loop {
|
||||
if let Ok(t) = std::fs::read_to_string(&path) {
|
||||
let lines: Vec<&str> = t.lines().collect();
|
||||
// only complete lines (the writer may be mid-line): keep the last one for the next pass unless the
|
||||
// text ends with a newline
|
||||
let complete = if t.ends_with('\n') { lines.len() } else { lines.len().saturating_sub(1) };
|
||||
for l in lines.iter().take(complete).skip(seen) {
|
||||
s.line(l);
|
||||
}
|
||||
seen = seen.max(complete);
|
||||
}
|
||||
if stop2.load(std::sync::atomic::Ordering::Relaxed) {
|
||||
break seen;
|
||||
}
|
||||
std::thread::sleep(Duration::from_secs(2));
|
||||
}
|
||||
});
|
||||
Follow { stop, handle }
|
||||
}
|
||||
|
||||
/// The PowerShell wrapper an elevated job runs (its own process, its own environment): the IGNEUM_* values, then one
|
||||
/// line about its console (the elevated process cannot inherit the engine's headless console and gets one of its own;
|
||||
/// `-WindowStyle Hidden` on the launch keeps it hidden, and this line is the running measurement of that on every
|
||||
/// elevated job: "elevated console: hwnd N visible False"), then the script, everything into `out_file` for the engine
|
||||
/// to read back. The console-window class, PC 1, 5 October 2026 (tools/windows/console-watch-elevated.ps1).
|
||||
fn elevated_wrapper(env_lines: &str, script: &str, out_file: &str) -> String {
|
||||
let (script, out) = (crate::platform::ps_quote(script), crate::platform::ps_quote(out_file));
|
||||
format!(
|
||||
"{env_lines}$ErrorActionPreference = 'Continue'\r\n\
|
||||
$igc = ''\r\n\
|
||||
try {{ Add-Type -Name IgCon -Namespace Igneum -MemberDefinition '[DllImport(\"kernel32.dll\")] public static extern System.IntPtr GetConsoleWindow(); [DllImport(\"user32.dll\")] public static extern bool IsWindowVisible(System.IntPtr h);'; $h = [Igneum.IgCon]::GetConsoleWindow(); $igc = \"elevated console: hwnd $h visible $([Igneum.IgCon]::IsWindowVisible($h))\" }} catch {{ $igc = \"elevated console: unknown ($_)\" }}\r\n\
|
||||
$igc | Out-File -FilePath '{out}' -Encoding utf8\r\n\
|
||||
& '{script}' *>&1 | Out-File -FilePath '{out}' -Encoding utf8 -Append\r\n\
|
||||
exit $LASTEXITCODE\r\n"
|
||||
)
|
||||
}
|
||||
|
||||
fn finish_ran(ran: Ran, what: &str) -> Result<Done, String> {
|
||||
match ran.code {
|
||||
Some(0) => Ok(Done { status: "done".into(), exit: 0, summary: format!("{what} finished, exit 0"), extra: json!({}) }),
|
||||
|
|
@ -1271,10 +1323,24 @@ mod tests {
|
|||
assert!(d.summary.contains("administrator prompt"), "{}", d.summary);
|
||||
let d = finish_ran(Ran { code: Some(0), timed_out: false }, "script").unwrap();
|
||||
assert_eq!(d.status, "done");
|
||||
// the launcher string itself: a thrown Start-Process must not fall through to `exit $p.ExitCode`
|
||||
let src = include_str!("jobrun.rs");
|
||||
assert!(src.contains("-Verb RunAs -Wait -WindowStyle Hidden -PassThru -ErrorAction Stop }} catch {{"));
|
||||
assert!(src.contains("if ($null -eq $p) {{ Write-Error 'elevated launch failed: no process'; exit 251 }}"));
|
||||
// the launcher string itself (platform::elevated_ps_line since 13755b9): a thrown Start-Process must not fall
|
||||
// through to `exit $p.ExitCode`
|
||||
let l = crate::platform::elevated_ps_line("powershell.exe", "-NoProfile -File x.ps1");
|
||||
assert!(l.contains("-Verb RunAs -Wait -WindowStyle Hidden -PassThru -ErrorAction Stop } catch {"), "{l}");
|
||||
assert!(l.contains("if ($null -eq $p) { Write-Error 'elevated launch failed: no process'; exit 251 }"), "{l}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn elevated_wrapper_reports_its_console_then_runs_the_script() {
|
||||
let w = elevated_wrapper("$env:IGNEUM_JOB_ID = 'j1'\r\n", r"C:\jobs\j1\script.ps1", r"C:\jobs\it's\elevated-output.log");
|
||||
assert!(w.starts_with("$env:IGNEUM_JOB_ID = 'j1'\r\n$ErrorActionPreference = 'Continue'\r\n"), "{w}");
|
||||
assert!(w.contains("GetConsoleWindow()") && w.contains("IsWindowVisible("), "{w}");
|
||||
assert!(w.contains("$igc | Out-File -FilePath 'C:\\jobs\\it''s\\elevated-output.log' -Encoding utf8\r\n"), "{w}");
|
||||
assert!(w.contains("& 'C:\\jobs\\j1\\script.ps1' *>&1 | Out-File -FilePath 'C:\\jobs\\it''s\\elevated-output.log' -Encoding utf8 -Append\r\n"), "{w}");
|
||||
assert!(w.ends_with("exit $LASTEXITCODE\r\n"), "{w}");
|
||||
// every line ends in CRLF (the file is written for Windows PowerShell): the env line and six of its own
|
||||
assert_eq!(w.matches("\r\n").count(), 7, "{w:?}");
|
||||
assert_eq!(w.matches('\n').count(), 7, "{w:?}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
|
@ -1761,6 +1827,7 @@ fn spawn_relaunch_helper(shared: &Arc<Shared>) -> Result<(), String> {
|
|||
{
|
||||
let dir = std::env::current_exe().ok().and_then(|p| p.parent().map(|d| d.to_path_buf())).ok_or("cannot find the install folder")?;
|
||||
let exe = dir.join("igneum-app.exe");
|
||||
// console: igneum-app.exe is a windows-subsystem program in release builds (main.rs), it never gets a console; SW_HIDE would hide the window host it opens
|
||||
let ps = format!("Start-Sleep 8; Start-Process -FilePath '{}' -ArgumentList '--launch' -WorkingDirectory '{}'", exe.display().to_string().replace('\'', "''"), dir.display().to_string().replace('\'', "''"));
|
||||
c = Command::new(crate::platform::tool("powershell"));
|
||||
c.args(["-NoProfile", "-ExecutionPolicy", "Bypass", "-WindowStyle", "Hidden", "-Command", &ps]);
|
||||
|
|
|
|||
|
|
@ -30,9 +30,11 @@ mod jobrun;
|
|||
mod jobbuild;
|
||||
mod prover;
|
||||
mod provedefault;
|
||||
mod segments;
|
||||
mod verifier;
|
||||
mod wslhost;
|
||||
mod sweep;
|
||||
mod ember;
|
||||
mod watchdog;
|
||||
|
||||
use std::io::{BufRead, Write};
|
||||
|
|
@ -134,7 +136,7 @@ fn main() {
|
|||
let Ok(l) = line else { break };
|
||||
let t = l.trim();
|
||||
match t {
|
||||
"quit" => shared.send(engine::Cmd::Quit),
|
||||
"quit" => shared.send(engine::Cmd::Quit("the window host (quit on stdin: the tray menu or the installer)")),
|
||||
"pause" => shared.send(engine::Cmd::Pause),
|
||||
"resume" => shared.send(engine::Cmd::Resume),
|
||||
// the window host saw WM_DEVICECHANGE (a card plugged in or out): enumerate now, not at the next minute
|
||||
|
|
@ -145,7 +147,7 @@ fn main() {
|
|||
}
|
||||
}
|
||||
if wrapper {
|
||||
shared.send(engine::Cmd::Quit);
|
||||
shared.send(engine::Cmd::Quit("the window host went away (stdin closed)"));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
|
|
|||
|
|
@ -546,3 +546,42 @@ mod tests {
|
|||
assert_eq!(fingerprint("zz"), "");
|
||||
}
|
||||
}
|
||||
|
||||
/// Unix seconds of a manifest `published_at` ("2026-10-04T13:00:00Z", whole seconds, UTC); `None` for any other shape.
|
||||
pub fn unix_from_rfc3339(t: &str) -> Option<u64> {
|
||||
let t = t.trim();
|
||||
let b = t.as_bytes();
|
||||
if b.len() < 20 || b[4] != b'-' || b[7] != b'-' || b[10] != b'T' || b[13] != b':' || b[16] != b':' || !t.ends_with('Z') {
|
||||
return None;
|
||||
}
|
||||
let n = |a: usize, z: usize| t[a..z].parse::<i64>().ok();
|
||||
let (y, m, d, hh, mm, ss) = (n(0, 4)?, n(5, 7)?, n(8, 10)?, n(11, 13)?, n(14, 16)?, n(17, 19)?);
|
||||
if !(1..=12).contains(&m) || !(1..=31).contains(&d) || hh > 23 || mm > 59 || ss > 60 {
|
||||
return None;
|
||||
}
|
||||
// days from civil (Howard Hinnant), valid for every date after 1970
|
||||
let (y2, m2) = if m <= 2 { (y - 1, m + 9) } else { (y, m - 3) };
|
||||
let era = y2.div_euclid(400);
|
||||
let yoe = y2 - era * 400;
|
||||
let doy = (153 * m2 + 2) / 5 + d - 1;
|
||||
let doe = yoe * 365 + yoe / 4 - yoe / 100 + doy;
|
||||
let days = era * 146097 + doe - 719468;
|
||||
if days < 0 {
|
||||
return None;
|
||||
}
|
||||
Some((days as u64) * 86400 + (hh as u64) * 3600 + (mm as u64) * 60 + ss as u64)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod rfc3339_tests {
|
||||
use super::unix_from_rfc3339;
|
||||
#[test]
|
||||
fn a_manifest_publish_time_parses_to_unix_seconds() {
|
||||
assert_eq!(unix_from_rfc3339("1970-01-01T00:00:00Z"), Some(0));
|
||||
assert_eq!(unix_from_rfc3339("2026-10-05T23:57:49Z"), Some(1791244669));
|
||||
assert_eq!(unix_from_rfc3339("2026-10-04T13:00:00Z"), Some(1791118800));
|
||||
assert_eq!(unix_from_rfc3339(""), None);
|
||||
assert_eq!(unix_from_rfc3339("2026-10-05 23:57:49"), None);
|
||||
assert_eq!(unix_from_rfc3339("2026-13-05T23:57:49Z"), None);
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -34,6 +34,8 @@ use std::process::Command;
|
|||
use std::sync::Arc;
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
/// A manifest published this long before the engine started is a catch-up: the hourly rollout slot does not apply.
|
||||
const CATCH_UP_AFTER_S: u64 = 3600;
|
||||
const HEALTHY_AFTER_S: u64 = 90;
|
||||
const CHECK_EVERY_S: u64 = 3600;
|
||||
const RETRY_AFTER_ERROR_S: u64 = 600;
|
||||
|
|
@ -117,6 +119,11 @@ pub struct Updater {
|
|||
deferred_until: Option<Instant>,
|
||||
/// this machine's minute of the hour for applying (manifest::slot_minute of the machine id)
|
||||
slot: u64,
|
||||
/// When this engine started (unix seconds): an update published more than an hour before it is a catch-up, not a
|
||||
/// rollout, and skips the hourly slot (the project lead's morning of 6 October 2026: PC 1 came up after the 0.3.11 publish and
|
||||
/// sat on "installs at the next safe moment" until he pressed Install now).
|
||||
started_unix: u64,
|
||||
catch_up_logged: bool,
|
||||
/// identity counts from /api/live over the last 10 minutes, sampled while an update is ready
|
||||
live_samples: Vec<(Instant, u64)>,
|
||||
live_next: Instant,
|
||||
|
|
@ -163,6 +170,8 @@ impl Updater {
|
|||
apply_launched: None,
|
||||
deferred_until: None,
|
||||
slot: manifest::slot_minute(&shared.runtime.id8()),
|
||||
started_unix: crate::platform::unix_now(),
|
||||
catch_up_logged: false,
|
||||
live_samples: Vec::new(),
|
||||
live_next: now,
|
||||
live_busy: false,
|
||||
|
|
@ -519,7 +528,12 @@ impl Updater {
|
|||
}
|
||||
};
|
||||
let minute = (crate::platform::unix_now() / 60) % 60;
|
||||
let slot_ok = minute == self.slot || std::env::var("IGNEUM_APP_UPDATE_NO_SLOT").map(|v| v == "1").unwrap_or(false);
|
||||
let catch_up = self.manifest.as_ref().and_then(|m| manifest::unix_from_rfc3339(&m.published_at)).map(|p| p + CATCH_UP_AFTER_S <= self.started_unix).unwrap_or(false);
|
||||
if catch_up && !self.catch_up_logged {
|
||||
self.catch_up_logged = true;
|
||||
shared.log(&format!("update: {} was published over an hour before this start, so it installs at the first safe moment (no hourly slot)", self.version()));
|
||||
}
|
||||
let slot_ok = minute == self.slot || catch_up || std::env::var("IGNEUM_APP_UPDATE_NO_SLOT").map(|v| v == "1").unwrap_or(false);
|
||||
let ready_for = self.ready_since.map(|t| now.duration_since(t).as_secs()).unwrap_or(0);
|
||||
let moment = Moment { node_synced: ctx.node_synced, boundary_eta_s: ctx.boundary_eta_s, miner_busy: ctx.miner_busy, ready_for_s: ready_for, urgent: urgent || self.install_asked, slot_ok, network_drop_pct };
|
||||
if !self.auto && !urgent && !self.install_asked {
|
||||
|
|
@ -1264,6 +1278,7 @@ function EngineAlive() { return [bool](Get-Process -Id $EnginePid -ErrorAction S
|
|||
function Relaunch() {
|
||||
if (EngineAlive) { return }
|
||||
$exe = Join-Path $InstallDir 'igneum-app.exe'
|
||||
# console: igneum-app.exe is a windows-subsystem program (no console); -WindowStyle Hidden would hide the window host it opens
|
||||
if (Test-Path $exe) { Log 'engine gone and nothing installed: starting the old app again'; Start-Process -FilePath $exe -ArgumentList '--launch' -WorkingDirectory $InstallDir | Out-Null }
|
||||
}
|
||||
Log "$Mode : engine $EnginePid installer '$Installer' version $Version (the engine keeps mining until the installer runs)"
|
||||
|
|
@ -1278,6 +1293,7 @@ $setupArgs = @('/VERYSILENT', '/SUPPRESSMSGBOXES', '/NORESTART', '/CLOSEAPPLICAT
|
|||
try {
|
||||
# no -Verb RunAs: a per-user installer just runs; an administrator installer makes Windows ask, and a declined or
|
||||
# timed-out prompt comes back here as an exception with the engine still mining
|
||||
# console: the Inno Setup installer is a GUI program (no console), /VERYSILENT shows nothing
|
||||
$p = Start-Process -FilePath $Installer -ArgumentList $setupArgs -Wait -PassThru
|
||||
if ($p.ExitCode -eq 0) {
|
||||
if ($Mode -eq 'rollback') { Done $false "Igneum Miner $Version did not stay up twice; the previous version was reinstalled" $true $false }
|
||||
|
|
|
|||
|
|
@ -396,10 +396,7 @@ pub fn sync_clock() -> Result<String, String> {
|
|||
#[cfg(windows)]
|
||||
{
|
||||
let cmd = tool("cmd").display().to_string();
|
||||
let script = format!("Start-Process -FilePath '{cmd}' -ArgumentList '/c net start w32time & w32tm /resync /force' -Verb RunAs -Wait -WindowStyle Hidden");
|
||||
let mut c = Command::new(tool("powershell"));
|
||||
c.args(["-NoProfile", "-ExecutionPolicy", "Bypass", "-Command", &script]);
|
||||
quiet(&mut c);
|
||||
let mut c = elevated_command(&cmd, "/c net start w32time & w32tm /resync /force");
|
||||
let out = c.output().map_err(|e| e.to_string())?;
|
||||
if out.status.success() {
|
||||
Ok("asked Windows Time to resync (w32tm /resync)".into())
|
||||
|
|
@ -425,18 +422,14 @@ pub fn sync_clock() -> Result<String, String> {
|
|||
pub fn run_elevated(cmdline: &str) -> Result<(), String> {
|
||||
#[cfg(windows)]
|
||||
{
|
||||
let escaped = cmdline.replace('\'', "''");
|
||||
let cmd = tool("cmd").display().to_string();
|
||||
let script = format!("$p = Start-Process -FilePath '{cmd}' -ArgumentList '/c {escaped}' -Verb RunAs -Wait -WindowStyle Hidden -PassThru; exit $p.ExitCode");
|
||||
let mut c = Command::new(tool("powershell"));
|
||||
c.args(["-NoProfile", "-ExecutionPolicy", "Bypass", "-Command", &script]);
|
||||
quiet(&mut c);
|
||||
let mut c = elevated_command(&cmd, &format!("/c {cmdline}"));
|
||||
let out = c.output().map_err(|e| e.to_string())?;
|
||||
if out.status.success() {
|
||||
Ok(())
|
||||
} else {
|
||||
let err = String::from_utf8_lossy(&out.stderr).trim().to_string();
|
||||
Err(if err.contains("canceled") || err.contains("cancelled") || err.is_empty() { "the administrator prompt was cancelled".into() } else { err })
|
||||
Err(elevated_failure(out.status.code(), &err))
|
||||
}
|
||||
}
|
||||
#[cfg(target_os = "linux")]
|
||||
|
|
@ -451,6 +444,52 @@ pub fn run_elevated(cmdline: &str) -> Result<(), String> {
|
|||
}
|
||||
}
|
||||
|
||||
/// The reason an elevated step failed, from the launcher's exit code and stderr: exit 251 (the prompt refused,
|
||||
/// cancelled or timed out, `elevated_ps_line`) and the "canceled" wording name the prompt; any other code is the
|
||||
/// step's own exit (the engine then keeps Power control on: rights were given).
|
||||
pub fn elevated_failure(code: Option<i32>, stderr: &str) -> String {
|
||||
if code == Some(ELEVATED_LAUNCH_FAILED) || stderr.contains("canceled") || stderr.contains("cancelled") {
|
||||
"the administrator prompt was refused, cancelled or timed out".into()
|
||||
} else if stderr.is_empty() {
|
||||
format!("the elevated step exited with code {}", code.map(|c| c.to_string()).unwrap_or_else(|| "?".into()))
|
||||
} else {
|
||||
stderr.to_string()
|
||||
}
|
||||
}
|
||||
|
||||
/// Doubles the single quotes of `s` for a single-quoted PowerShell literal.
|
||||
pub fn ps_quote(s: &str) -> String {
|
||||
s.replace('\'', "''")
|
||||
}
|
||||
|
||||
/// The PowerShell line that starts `file args` as administrator (one UAC prompt), waits, and exits with the child's
|
||||
/// code. Every elevated launch of the app goes through here (the NVIDIA power cap, the sweep helper, the clock sync,
|
||||
/// an elevated remote job) so the console flags live in one place: `-WindowStyle Hidden` is SW_HIDE on the new
|
||||
/// process the AppInfo service creates; the elevated child cannot inherit this process's headless console, so without
|
||||
/// it the child gets a console of its own (5 October 2026, PC 1 watcher, tools/windows/console-watch*.ps1).
|
||||
/// A refused, cancelled or unanswered prompt makes Start-Process throw and `$p` stay null: that is exit 251 with the
|
||||
/// reason on stderr, never `exit $p.ExitCode` = 0 (the 5 October 2026 driver job on PC 1 was reported done after
|
||||
/// Windows cancelled its prompt at 122 s).
|
||||
pub fn elevated_ps_line(file: &str, args: &str) -> String {
|
||||
format!(
|
||||
"try {{ $p = Start-Process -FilePath '{}' -ArgumentList '{}' -Verb RunAs -Wait -WindowStyle Hidden -PassThru -ErrorAction Stop }} catch {{ Write-Error ('elevated launch failed (UAC refused, cancelled or timed out): ' + $_.Exception.Message); exit 251 }}; if ($null -eq $p) {{ Write-Error 'elevated launch failed: no process'; exit 251 }}; exit $p.ExitCode",
|
||||
ps_quote(file),
|
||||
ps_quote(args)
|
||||
)
|
||||
}
|
||||
|
||||
/// The exit code `elevated_ps_line` uses when the elevated process never started (the prompt refused, cancelled or
|
||||
/// timed out).
|
||||
pub const ELEVATED_LAUNCH_FAILED: i32 = 251;
|
||||
|
||||
/// The hidden PowerShell that runs `elevated_ps_line(file, args)`: blocking when run, one UAC prompt on the PC.
|
||||
pub fn elevated_command(file: &str, args: &str) -> Command {
|
||||
let mut c = Command::new(tool("powershell"));
|
||||
c.args(["-NoProfile", "-ExecutionPolicy", "Bypass", "-Command", &elevated_ps_line(file, args)]);
|
||||
quiet(&mut c);
|
||||
c
|
||||
}
|
||||
|
||||
/// Builds a command that runs without a console window on Windows.
|
||||
pub fn quiet(cmd: &mut Command) -> &mut Command {
|
||||
#[cfg(windows)]
|
||||
|
|
@ -463,6 +502,34 @@ pub fn quiet(cmd: &mut Command) -> &mut Command {
|
|||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
#[test]
|
||||
fn elevated_line_is_hidden_and_quoted() {
|
||||
let l = super::elevated_ps_line(r"C:\WINDOWS\system32\cmd.exe", "/c echo it's & exit 3");
|
||||
assert!(l.starts_with("try { $p = "), "{l}");
|
||||
assert!(l.contains("-FilePath 'C:\\WINDOWS\\system32\\cmd.exe' -ArgumentList '/c echo it''s & exit 3' -Verb RunAs -Wait -WindowStyle Hidden -PassThru -ErrorAction Stop } catch {"), "{l}");
|
||||
assert!(l.contains("-Verb RunAs"), "{l}");
|
||||
assert!(l.contains("-WindowStyle Hidden"), "{l}");
|
||||
// a thrown Start-Process (the prompt refused) never falls through to `exit $p.ExitCode`
|
||||
assert!(l.contains("exit 251 }; if ($null -eq $p) { Write-Error 'elevated launch failed: no process'; exit 251 }; exit $p.ExitCode"), "{l}");
|
||||
assert!(l.ends_with("exit $p.ExitCode"), "{l}");
|
||||
assert_eq!(super::ELEVATED_LAUNCH_FAILED, 251);
|
||||
assert_eq!(super::elevated_failure(Some(251), "elevated launch failed (UAC refused, cancelled or timed out): ..."), "the administrator prompt was refused, cancelled or timed out");
|
||||
assert_eq!(super::elevated_failure(Some(1), "The operation was canceled by the user."), "the administrator prompt was refused, cancelled or timed out");
|
||||
assert_eq!(super::elevated_failure(Some(2), ""), "the elevated step exited with code 2");
|
||||
assert_eq!(super::elevated_failure(Some(3), "nvidia-smi: bad"), "nvidia-smi: bad");
|
||||
assert_eq!(super::ps_quote("a'b''c"), "a''b''''c");
|
||||
assert_eq!(super::ps_quote("plain"), "plain");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn elevated_command_is_a_hidden_powershell() {
|
||||
let c = super::elevated_command("powershell.exe", "-NoProfile -File \"C:\\x y\\elevated.ps1\"");
|
||||
let args: Vec<String> = c.get_args().map(|a| a.to_string_lossy().into_owned()).collect();
|
||||
assert_eq!(&args[..4], ["-NoProfile", "-ExecutionPolicy", "Bypass", "-Command"]);
|
||||
assert!(args[4].contains("-ArgumentList '-NoProfile -File \"C:\\x y\\elevated.ps1\"' -Verb RunAs -Wait -WindowStyle Hidden"), "{}", args[4]);
|
||||
assert!(c.get_program().to_string_lossy().contains("powershell"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn token_redaction() {
|
||||
let l = "dashboard at http://127.0.0.1:58776/t/a3a01c537130bceeaa1f6118ba48d63e/ (log x)";
|
||||
|
|
|
|||
|
|
@ -13,6 +13,7 @@ pub enum Source {
|
|||
Watch,
|
||||
Miner(usize), // card index
|
||||
Telemetry, // nvidia-smi -l 5
|
||||
AmdTelemetry, // igneum-gpu-telemetry -l 5 (ADLX or sysfs), 5 October 2026
|
||||
}
|
||||
|
||||
impl Source {
|
||||
|
|
@ -22,6 +23,7 @@ impl Source {
|
|||
Source::Watch => "watch".into(),
|
||||
Source::Miner(i) => format!("miner{}", i + 1),
|
||||
Source::Telemetry => "gpu".into(),
|
||||
Source::AmdTelemetry => "gpu-amd".into(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -47,6 +47,8 @@ pub struct Work {
|
|||
pub shard_wei: u128,
|
||||
/// The first of this machine's keys that is assigned (the label that signs).
|
||||
pub key_hash: String,
|
||||
/// The chain block's DAA score (the deadline clock of spec 7.8).
|
||||
pub daa: u64,
|
||||
}
|
||||
|
||||
/// Parses the node's work list. Newest first, as the node returns it.
|
||||
|
|
@ -67,6 +69,7 @@ pub fn parse_work(v: &Value) -> Vec<Work> {
|
|||
in_pool: w["pool"].as_array().map(|p| !p.is_empty()).unwrap_or(false),
|
||||
shard_wei: hexu(&w["shardWei"]),
|
||||
key_hash: w["assignedKeys"].as_array().and_then(|k| k.first()).and_then(|k| k.as_str()).unwrap_or("").to_string(),
|
||||
daa: hexu(&w["daaScore"]) as u64,
|
||||
})
|
||||
.collect()
|
||||
})
|
||||
|
|
@ -344,6 +347,13 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
|
|||
let ram_mb = crate::detect::total_ram_mb();
|
||||
let mut last_probe = Instant::now() - Duration::from_secs(600);
|
||||
let mut submitted: Vec<(u64, String, u32, u128)> = Vec::new();
|
||||
// proving v1 segment path: (first, last, aggregator wei) of the segment records this machine submitted
|
||||
let mut submitted_segments: Vec<(u64, u64, u128)> = Vec::new();
|
||||
let mut last_segment_secs: Option<f64> = None;
|
||||
// segment records the node refused by the chain rule ("does not chain to ... pending"): held and offered again
|
||||
// every pass until the segment's deadline (the fresh-record window of spec 7.8 is the segment length in DAA on
|
||||
// the rule as shipped, 6 October 2026; from the fresh-rule switch the first retry lands)
|
||||
let mut held_segments: Vec<HeldSegment> = Vec::new();
|
||||
let mut last_verifier_read = Instant::now() - Duration::from_secs(600);
|
||||
let mut asked_restart = false;
|
||||
// macOS and Linux: the host sits next to the engine, so its pinned ids are read at once, proving on or off
|
||||
|
|
@ -450,7 +460,7 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
|
|||
});
|
||||
continue;
|
||||
}
|
||||
let work = match evm_rpc(&shared, "igneum_getAssignedShards", json!([keys.iter().map(|(_, h)| h.clone()).collect::<Vec<_>>(), 60]), Duration::from_secs(10)) {
|
||||
let work = match evm_rpc(&shared, "igneum_getAssignedShards", json!([keys.iter().map(|(_, h)| h.clone()).collect::<Vec<_>>(), crate::segments::WORK_LOOKBACK]), Duration::from_secs(10)) {
|
||||
Ok(v) => parse_work(&v),
|
||||
Err(e) => {
|
||||
set(&shared, |p| {
|
||||
|
|
@ -477,6 +487,52 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
|
|||
});
|
||||
}
|
||||
}
|
||||
// held segment records: offered again, dropped past the deadline
|
||||
if !held_segments.is_empty() {
|
||||
let tip_daa = evm_rpc(&shared, "igneum_getProvingStatus", json!([]), Duration::from_secs(10)).ok().and_then(|st| st["tipDaa"].as_str().and_then(|x| u64::from_str_radix(x.trim_start_matches("0x"), 16).ok())).unwrap_or(0);
|
||||
let mut keep = Vec::new();
|
||||
for h in held_segments.drain(..) {
|
||||
match retry_held(&shared, &h, tip_daa) {
|
||||
Retry::Accepted => {
|
||||
shared.event("proving", &format!("segment {}..{} record accepted on retry {} (held {} s)", h.first, h.last, h.tries + 1, h.since.elapsed().as_secs()));
|
||||
submitted_segments.push((h.first, h.last, h.agg_wei));
|
||||
set(&shared, |p| {
|
||||
p.segments_submitted += 1;
|
||||
p.aggregated += 1;
|
||||
p.segment_note = format!("segment {}..{} accepted on retry", h.first, h.last);
|
||||
});
|
||||
}
|
||||
Retry::Expired(why) => {
|
||||
shared.log(&format!("prover: segment {}..{} record dropped after {} tries: {why}", h.first, h.last, h.tries));
|
||||
}
|
||||
Retry::Again(why) => {
|
||||
let mut h = h;
|
||||
h.tries += 1;
|
||||
if h.tries % 30 == 1 {
|
||||
shared.log(&format!("prover: segment {}..{} record held (try {}): {why}", h.first, h.last, h.tries));
|
||||
}
|
||||
keep.push(h);
|
||||
}
|
||||
}
|
||||
}
|
||||
held_segments = keep;
|
||||
set(&shared, |p| p.segments_held = held_segments.len() as u32);
|
||||
}
|
||||
// paid segments among what we submitted
|
||||
for (first, last, wei) in submitted_segments.clone() {
|
||||
let paid = evm_rpc(&shared, "igneum_getSegmentRecords", json!([format!("{first:#x}")]), Duration::from_secs(10))
|
||||
.ok()
|
||||
.map(|r| !r["paid"].is_null() && r["paid"]["payout"].as_str().map(|a| a.eq_ignore_ascii_case(&payout_address(&shared))).unwrap_or(false))
|
||||
.unwrap_or(false);
|
||||
if paid {
|
||||
submitted_segments.retain(|x| x.0 != first);
|
||||
shared.event("proving", &format!("segment {first}..{last} paid {} IGN to the aggregator", wei as f64 / 1e18));
|
||||
set(&shared, |p| {
|
||||
p.segments_paid += 1;
|
||||
p.segment_paid_wei += wei;
|
||||
});
|
||||
}
|
||||
}
|
||||
let assigned = work.iter().filter(|w| w.assigned).count() as u32;
|
||||
set(&shared, |p| {
|
||||
p.assigned = assigned;
|
||||
|
|
@ -496,10 +552,60 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
|
|||
}
|
||||
}
|
||||
}
|
||||
// proving v1 segment path (src/segments.rs, 6 October 2026): a whole segment first, the newest shard only
|
||||
// when no whole segment qualifies
|
||||
let payout = shared.settings.lock().unwrap().address.clone();
|
||||
if payout.len() == 42 {
|
||||
if let Some((seg, prev_file, expected_pv)) = pick_segment(&shared, &work, &keys[0].1, &mut attempted_segments, last_segment_secs) {
|
||||
attempted_segments.insert(seg.first);
|
||||
let started = Instant::now();
|
||||
match prove_segment(&shared, t, &seg, &keys[0].0, &payout, prev_file.as_deref(), &expected_pv, &mut submitted) {
|
||||
Ok(SegmentOutcome::Held(h)) => {
|
||||
let secs = started.elapsed().as_secs_f64();
|
||||
last_segment_secs = Some(secs);
|
||||
shared.event("proving", &format!("segment {}..{}: {} shards proven and submitted in {secs:.0} s; the segment record is held ({})", seg.first, seg.last, seg.shards.len(), h.why));
|
||||
set(&shared, |p| {
|
||||
p.segment_last_s = secs;
|
||||
p.status = "submitted".into();
|
||||
p.message = format!("segment {}..{}: shards submitted, the segment record waits for the chain rule", seg.first, seg.last);
|
||||
p.segment_note = format!("segment {}..{} proven whole in {secs:.0} s; its record is held: {}", seg.first, seg.last, h.why);
|
||||
p.current = String::new();
|
||||
});
|
||||
held_segments.push(h);
|
||||
set(&shared, |p| p.segments_held = held_segments.len() as u32);
|
||||
}
|
||||
Ok(SegmentOutcome::Submitted(agg_wei)) => {
|
||||
let secs = started.elapsed().as_secs_f64();
|
||||
last_segment_secs = Some(secs);
|
||||
submitted_segments.push((seg.first, seg.last, agg_wei));
|
||||
shared.event("proving", &format!("segment {}..{}: {} shards proven, aggregated and submitted in {secs:.0} s", seg.first, seg.last, seg.shards.len()));
|
||||
set(&shared, |p| {
|
||||
p.segments_submitted += 1;
|
||||
p.segment_last_s = secs;
|
||||
p.status = "submitted".into();
|
||||
p.message = format!("segment {}..{} submitted; paid when a block carries it", seg.first, seg.last);
|
||||
p.segment_note = format!("segment {}..{} proven whole in {secs:.0} s", seg.first, seg.last);
|
||||
p.current = String::new();
|
||||
});
|
||||
}
|
||||
Err(e) => {
|
||||
shared.log(&format!("prover: segment {}..{}: {e}", seg.first, seg.last));
|
||||
set(&shared, |p| {
|
||||
p.failed += 1;
|
||||
p.status = "idle".into();
|
||||
p.message = e.clone();
|
||||
p.segment_note = e;
|
||||
p.current = String::new();
|
||||
});
|
||||
}
|
||||
}
|
||||
continue;
|
||||
}
|
||||
}
|
||||
let Some(w) = choose(&work, &attempted) else {
|
||||
set(&shared, |p| {
|
||||
p.status = if submitted.is_empty() { "idle".into() } else { "submitted".into() };
|
||||
p.message = if assigned == 0 { "no shard assigned to this machine and none open in the last 60 blocks".into() } else { "every assigned and open shard is proven or paid".into() };
|
||||
p.message = if assigned == 0 { format!("no shard assigned to this machine and none open in the last {} blocks", crate::segments::WORK_LOOKBACK) } else { "every assigned and open shard is proven or paid".into() };
|
||||
});
|
||||
continue;
|
||||
};
|
||||
|
|
@ -592,6 +698,218 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
|
|||
}
|
||||
}
|
||||
|
||||
/// A segment record the node refused by the chain rule, kept with its proof for another offer.
|
||||
pub struct HeldSegment {
|
||||
pub first: u64,
|
||||
pub last: u64,
|
||||
pub deadline_daa: u64,
|
||||
pub record: String,
|
||||
pub proof_path: PathBuf,
|
||||
pub agg_wei: u128,
|
||||
pub why: String,
|
||||
pub tries: u32,
|
||||
pub since: Instant,
|
||||
}
|
||||
|
||||
pub enum SegmentOutcome {
|
||||
Submitted(u128),
|
||||
Held(HeldSegment),
|
||||
}
|
||||
|
||||
pub enum Retry {
|
||||
Accepted,
|
||||
Again(String),
|
||||
Expired(String),
|
||||
}
|
||||
|
||||
/// Offers a held segment record again: accepted, held for another pass, or dropped past the segment's deadline.
|
||||
fn retry_held(shared: &Shared, h: &HeldSegment, tip_daa: u64) -> Retry {
|
||||
if tip_daa > 0 && tip_daa + 1 > h.deadline_daa {
|
||||
return Retry::Expired(format!("past the deadline DAA {} at tip DAA {tip_daa}", h.deadline_daa));
|
||||
}
|
||||
let Ok(proof) = std::fs::read(&h.proof_path) else { return Retry::Expired(format!("proof file {} gone", h.proof_path.display())) };
|
||||
let proof_hex = format!("0x{}", proof.iter().map(|b| format!("{b:02x}")).collect::<String>());
|
||||
match evm_rpc(shared, "igneum_submitSegmentRecord", json!([{ "record": h.record, "proof": proof_hex }]), Duration::from_secs(60)) {
|
||||
Ok(r) if r["accepted"].as_bool().unwrap_or(false) => Retry::Accepted,
|
||||
Ok(r) => Retry::Again(r["reason"].as_str().unwrap_or("?").to_string()),
|
||||
Err(e) => Retry::Again(e),
|
||||
}
|
||||
}
|
||||
|
||||
/// Proving v1 segment path, the choice: the node's v1 status (active, the grid start, the segment length, the
|
||||
/// deadline clock), the work list grouped into whole untouched segments (`segments::whole_segments`), the
|
||||
/// candidates inside the deadline ranked for this key, then for the best three the node's segment statement:
|
||||
/// executed and pending; the previous segment either paid with its proof in this node's pool (the chain continues,
|
||||
/// `--prev`) or not paid and with no verified record of it waiting in the pool (fresh). Returns the segment, the
|
||||
/// previous proof's host path when the chain continues, and the public values the node expects.
|
||||
fn pick_segment(shared: &Shared, work: &[Work], key_hash: &str, attempted: &mut HashSet<u64>, last_secs: Option<f64>) -> Option<(crate::segments::SegmentWork, Option<String>, String)> {
|
||||
let hexu = |x: &Value| x.as_str().and_then(|s| u64::from_str_radix(s.trim_start_matches("0x"), 16).ok()).unwrap_or(0);
|
||||
let st = evm_rpc(shared, "igneum_getProvingStatus", json!([]), Duration::from_secs(10)).ok()?;
|
||||
let v1 = &st["v1"];
|
||||
if !v1["active"].as_bool().unwrap_or(false) || v1["start"].is_null() {
|
||||
return None;
|
||||
}
|
||||
let (start, n, unproven, tip_daa) = (hexu(&v1["start"]), hexu(&v1["segmentBlocks"]).max(1), hexu(&v1["unprovenDaa"]), hexu(&st["tipDaa"]));
|
||||
let segs = crate::segments::whole_segments(start, n, unproven, work);
|
||||
let need = crate::segments::need_daa(last_secs);
|
||||
let cands = crate::segments::candidates(&segs, tip_daa, need, key_hash, attempted);
|
||||
if cands.is_empty() {
|
||||
return None;
|
||||
}
|
||||
let dir = shared.runtime.app_dir.join("proving");
|
||||
let _ = std::fs::create_dir_all(&dir);
|
||||
let wsl = cfg!(windows);
|
||||
let as_host_path = |p: &Path| if wsl { wsl_path(p) } else { p.display().to_string() };
|
||||
for seg in cands.into_iter().take(3) {
|
||||
let Ok(stmt) = evm_rpc(shared, "igneum_getSegmentStatement", json!([format!("{:#x}", seg.first)]), Duration::from_secs(10)) else { continue };
|
||||
if !stmt["executed"].as_bool().unwrap_or(false) || stmt["status"]["status"].as_str() != Some("pending") {
|
||||
attempted.insert(seg.first);
|
||||
continue;
|
||||
}
|
||||
let prev = &stmt["previous"];
|
||||
if prev.is_null() {
|
||||
// fresh only when no record of the previous segment is waiting to be carried (the chain rule would
|
||||
// refuse a fresh record once that one pays)
|
||||
if seg.first >= start + n {
|
||||
let p = evm_rpc(shared, "igneum_getSegmentRecords", json!([format!("{:#x}", seg.first - n)]), Duration::from_secs(10)).unwrap_or(Value::Null);
|
||||
let waiting = p["pool"].as_array().map(|a| a.iter().any(|e| e["verified"] == json!(true) && e["includedIn"].is_null())).unwrap_or(false);
|
||||
if waiting || !p["paid"].is_null() {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
return Some((seg, None, stmt["publicValuesFresh"].as_str().unwrap_or("").to_string()));
|
||||
}
|
||||
if prev["proofInPool"] != json!(true) {
|
||||
continue;
|
||||
}
|
||||
let Ok(got) = evm_rpc(shared, "igneum_getSegmentProofBytes", json!([prev["first"], prev["keyHash"]]), Duration::from_secs(60)) else { continue };
|
||||
let hex = got["proof"].as_str().unwrap_or("").trim_start_matches("0x").to_string();
|
||||
if hex.is_empty() {
|
||||
continue;
|
||||
}
|
||||
let bytes: Vec<u8> = (0..hex.len() / 2).map(|k| u8::from_str_radix(&hex[2 * k..2 * k + 2], 16).unwrap_or(0)).collect();
|
||||
let f = dir.join(format!("prev-{}.bin", seg.first));
|
||||
if std::fs::write(&f, bytes).is_err() {
|
||||
continue;
|
||||
}
|
||||
return Some((seg, Some(as_host_path(&f)), stmt["publicValuesContinuing"].as_str().unwrap_or("").to_string()));
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
/// Proving v1 segment path, the work: one export of the chain to the segment's last block, one fixture per block,
|
||||
/// one host run (`--mode chain --save-shards`, `--prev` when the chain continues) that proves every shard and
|
||||
/// aggregates the segment, then every shard record signed and submitted (the shard payouts) and the segment
|
||||
/// record signed and submitted (the aggregator share). Returns the segment's aggregator wei.
|
||||
fn prove_segment(shared: &Shared, t: &Tools, seg: &crate::segments::SegmentWork, label: &str, payout: &str, prev_file: Option<&str>, expected_pv: &str, submitted: &mut Vec<(u64, String, u32, u128)>) -> Result<SegmentOutcome, String> {
|
||||
let (first, last) = (seg.first, seg.last);
|
||||
let dir = shared.runtime.app_dir.join("proving").join(format!("seg-{first}"));
|
||||
let _ = std::fs::create_dir_all(&dir);
|
||||
let as_host_path = |p: &Path| if t.wsl { wsl_path(p) } else { p.display().to_string() };
|
||||
set(shared, |p| {
|
||||
p.status = "proving".into();
|
||||
p.current = format!("segment {first}..{last} ({} shards)", seg.shards.len());
|
||||
p.started_at = crate::platform::unix_now_f();
|
||||
p.message = "exporting the chain and cutting the segment's blocks".into();
|
||||
});
|
||||
shared.log(&format!("prover: segment {first}..{last} claimed ({} shards{}): export, cut, chain ({}), sign, submit", seg.shards.len(), if prev_file.is_some() { ", continuing the previous segment's proof" } else { ", fresh" }, if t.cuda { "CUDA" } else { "CPU" }));
|
||||
// 1. export once, cut every block
|
||||
let seq = dir.join("seq.json");
|
||||
let export = evm_rpc(shared, "igneum_exportSegments", json!(["0x0", format!("{last:#x}")]), Duration::from_secs(300))?;
|
||||
std::fs::write(&seq, export.to_string()).map_err(|e| e.to_string())?;
|
||||
let mut fixtures: Vec<String> = Vec::new();
|
||||
for b in first..=last {
|
||||
let fixture = dir.join(format!("block-{b}.json"));
|
||||
let (ok, out) = run_tool(shared, t, &t.export, &[as_host_path(&seq), b.to_string(), as_host_path(&fixture)], &[], Duration::from_secs(600), &dir.join(format!("export-{b}.log")));
|
||||
if !ok || !fixture.exists() {
|
||||
return Err(format!("exporter, block {b}: {}", out.lines().rev().find(|l| !l.trim().is_empty()).unwrap_or("failed")));
|
||||
}
|
||||
fixtures.push(as_host_path(&fixture));
|
||||
}
|
||||
let _ = std::fs::remove_file(&seq);
|
||||
// 2. the chain: every shard proven, every block aggregated with the previous, in one process
|
||||
set(shared, |p| p.message = format!("proving {} shards and aggregating segment {first}..{last} ({})", seg.shards.len(), if t.cuda { "GPU" } else { "CPU, slow" }));
|
||||
let results = dir.join("chain-results.json");
|
||||
let mut args: Vec<String> = vec!["--mode".into(), "chain".into(), "--chain".into(), fixtures.join(","), "--prover".into(), payout.to_string(), "--save-shards".into(), "--out".into(), as_host_path(&results)];
|
||||
if let Some(pf) = prev_file {
|
||||
args.push("--prev".into());
|
||||
args.push(pf.to_string());
|
||||
}
|
||||
let (ok, out) = run_tool(shared, t, &t.host, &args, &[("SP1_PROVER", if t.cuda { "cuda" } else { "cpu" }), ("RUST_LOG", "off")], Duration::from_secs(3 * 3600), &dir.join("chain.log"));
|
||||
if !ok || !results.exists() {
|
||||
let last_line = out.lines().rev().find(|l| l.contains("RESULT") || l.contains("rror")).unwrap_or("failed").to_string();
|
||||
let hint = if last_line.contains("PermissionDenied") { " (a GPU-server socket /tmp/sp1-cuda-*.sock owned by another user: the root-socket class)" } else { "" };
|
||||
return Err(format!("chain: {last_line}{hint}"));
|
||||
}
|
||||
let res: Value = serde_json::from_str(&std::fs::read_to_string(&results).map_err(|e| e.to_string())?).map_err(|e| e.to_string())?;
|
||||
let to_win = |f: &str| if t.wsl { PathBuf::from(f.replace("/mnt/c/", "C:/")) } else { PathBuf::from(f) };
|
||||
// 3. the shard records
|
||||
let chain = chain_name(shared);
|
||||
let mut shard_ok = 0usize;
|
||||
for b in res["blocks"].as_array().cloned().unwrap_or_default() {
|
||||
for r in b["shard_records"].as_array().cloned().unwrap_or_default() {
|
||||
let (number, hash, shard) = (r["number"].as_u64().unwrap_or(0), r["block_hash"].as_str().unwrap_or("").to_string(), r["shard"].as_u64().unwrap_or(0) as u32);
|
||||
let statement = r["statement"].as_str().unwrap_or("").to_string();
|
||||
let proof_sha = r["proof_sha256"].as_str().unwrap_or("").to_string();
|
||||
let proof_file = r["proof_file"].as_str().unwrap_or("").to_string();
|
||||
let sg = crate::detect::run_timeout(crate::platform::quiet(&mut Command::new(&t.miner)).args(["sign-record", label, &chain, &hash, &number.to_string(), &shard.to_string(), payout, &statement, &proof_sha]), None, Duration::from_secs(20)).ok_or("sign-record did not run")?;
|
||||
let signed: Value = serde_json::from_str(sg.lines().last().unwrap_or("")).map_err(|_| format!("sign-record: {}", sg.trim()))?;
|
||||
let record = signed["record"].as_str().ok_or("sign-record gave no record")?.to_string();
|
||||
let proof = std::fs::read(to_win(&proof_file)).map_err(|e| format!("proof file {proof_file}: {e}"))?;
|
||||
let proof_hex = format!("0x{}", proof.iter().map(|b| format!("{b:02x}")).collect::<String>());
|
||||
let out = evm_rpc(shared, "igneum_submitProofRecord", json!([{ "record": record, "proof": proof_hex }]), Duration::from_secs(60))?;
|
||||
if out["accepted"].as_bool().unwrap_or(false) {
|
||||
shard_ok += 1;
|
||||
let wei = seg.shards.iter().position(|(n, _, s)| *n == number && *s == shard).map(|_| seg.shard_wei / seg.shards.len().max(1) as u128).unwrap_or(0);
|
||||
submitted.push((number, hash.clone(), shard, wei));
|
||||
} else {
|
||||
shared.log(&format!("prover: segment {first}..{last}: block {number} shard {shard} record refused: {}", out["reason"].as_str().unwrap_or("?")));
|
||||
}
|
||||
}
|
||||
}
|
||||
if shard_ok != seg.shards.len() {
|
||||
return Err(format!("{shard_ok} of {} shard records accepted; the segment record is not submitted", seg.shards.len()));
|
||||
}
|
||||
set(shared, |p| {
|
||||
p.proved += shard_ok as u32;
|
||||
p.submitted += shard_ok as u32;
|
||||
});
|
||||
// 4. the segment record: the aggregated statement against the node's native one (every field but provers)
|
||||
let pv = res["segment_public_values"].as_str().ok_or("no public values in the chain results")?.to_string();
|
||||
let proof_sha = res["segment_proof_sha256"].as_str().ok_or("no segment proof hash in the chain results")?.to_string();
|
||||
let proof_file = res["segment_proof_file"].as_str().ok_or("no segment proof file in the chain results")?.to_string();
|
||||
let strip = |h: &str| { let h = h.trim_start_matches("0x"); if h.len() == 680 { format!("{}{}", &h[..472], &h[536..]) } else { h.to_string() } };
|
||||
if strip(&pv) != strip(expected_pv) {
|
||||
return Err(format!("the aggregated statement differs from the node's native statement (it would be vetoed); ours {} node {}", &pv[..66.min(pv.len())], &expected_pv[..66.min(expected_pv.len())]));
|
||||
}
|
||||
let last_hash = seg.shards.iter().rev().find(|(n, _, _)| *n == last).map(|(_, h, _)| h.clone()).ok_or("no last block hash")?;
|
||||
let sg = crate::detect::run_timeout(crate::platform::quiet(&mut Command::new(&t.miner)).args(["sign-segment-record", label, &chain, &first.to_string(), &last.to_string(), &last_hash, payout, &pv, &proof_sha]), None, Duration::from_secs(20)).ok_or("sign-segment-record did not run")?;
|
||||
let signed: Value = serde_json::from_str(sg.lines().last().unwrap_or("")).map_err(|_| format!("sign-segment-record: {}", sg.trim()))?;
|
||||
let record = signed["record"].as_str().ok_or("sign-segment-record gave no record")?.to_string();
|
||||
let proof = std::fs::read(to_win(&proof_file)).map_err(|e| format!("segment proof file {proof_file}: {e}"))?;
|
||||
let proof_hex = format!("0x{}", proof.iter().map(|b| format!("{b:02x}")).collect::<String>());
|
||||
let r = evm_rpc(shared, "igneum_submitSegmentRecord", json!([{ "record": record, "proof": proof_hex }]), Duration::from_secs(60))?;
|
||||
let hexu = |x: &Value| x.as_str().and_then(|s| u128::from_str_radix(s.trim_start_matches("0x"), 16).ok()).unwrap_or(0);
|
||||
let stmt = evm_rpc(shared, "igneum_getSegmentStatement", json!([format!("{first:#x}")]), Duration::from_secs(10)).unwrap_or(Value::Null);
|
||||
let agg_wei = hexu(&stmt["aggregatorWei"]);
|
||||
// the fixtures and the export go; the proofs stay (the next segment's chain link, and a held record's offer)
|
||||
for b in first..=last {
|
||||
let _ = std::fs::remove_file(dir.join(format!("block-{b}.json")));
|
||||
}
|
||||
if !r["accepted"].as_bool().unwrap_or(false) {
|
||||
let why = r["reason"].as_str().unwrap_or("?").to_string();
|
||||
// the chain rule's refusal ("does not chain to ... pending until DAA ..."): held, not failed; anything else
|
||||
// (a bad statement, a late carrier) is an error
|
||||
if why.contains("does not chain") {
|
||||
let deadline = hexu(&stmt["status"]["deadline_daa"]) as u64;
|
||||
return Ok(SegmentOutcome::Held(HeldSegment { first, last, deadline_daa: if deadline > 0 { deadline } else { u64::MAX }, record, proof_path: to_win(&proof_file), agg_wei, why, tries: 0, since: Instant::now() }));
|
||||
}
|
||||
return Err(format!("segment record refused: {why}"));
|
||||
}
|
||||
set(shared, |p| p.aggregated += 1);
|
||||
Ok(SegmentOutcome::Submitted(agg_wei))
|
||||
}
|
||||
|
||||
/// Proving v1 (spec 7.8): one aggregation attempt. When the node reports v1 active, takes the newest executed
|
||||
/// segment that is still pending and not yet attempted here, needs one shard proof per shard of every block in
|
||||
/// this node's pool (`igneum_getProofBytes`, a verified one when there is one) and, when the previous segment is
|
||||
|
|
|
|||
214
app/igneum-app/src/segments.rs
Normal file
214
app/igneum-app/src/segments.rs
Normal file
|
|
@ -0,0 +1,214 @@
|
|||
//! Proving v1 (spec 7.8): segment-aligned work for the prover loop (6 October 2026).
|
||||
//!
|
||||
//! The shipped loop took the newest open shard each pass, so one prover scattered one block in about 45 across
|
||||
//! the segment grid and no segment ever had all its blocks proven (node 1, 04:16Z: pending 55, proven 0). Here a
|
||||
//! free prover claims a whole segment (`proving_v1_segment_blocks` consecutive chain blocks), proves every shard
|
||||
//! of it in order from one export in one host run (`--mode chain --save-shards`), submits the shard records and the
|
||||
//! aggregated segment record, then takes the next. One card completes whole segments at its own rate instead of
|
||||
//! completing none.
|
||||
//!
|
||||
//! The choice is deterministic per prover: among the untouched whole segments still inside their deadline by a
|
||||
//! margin, the lowest FNV-1a of (first block, this prover's key hash) wins, so several provers spread over the
|
||||
//! candidates without a coordinator; the per-block fallback (`prover::choose`) stays for the passes where no whole
|
||||
//! segment qualifies.
|
||||
|
||||
use std::collections::{BTreeMap, HashSet};
|
||||
|
||||
use crate::prover::Work;
|
||||
|
||||
/// The least time a claimed segment is given before its deadline (DAA units, about one a second on devnet): the
|
||||
/// chain of 8 empty blocks took 135.6 s cold beside the miner (bench-log, 5 October 2026), so 240 leaves the
|
||||
/// submission and the carrying block inside the window.
|
||||
pub const SEGMENT_MARGIN_MIN_DAA: u64 = 240;
|
||||
/// The margin grows with what the last segment actually took, times this.
|
||||
pub const SEGMENT_MARGIN_FACTOR: f64 = 1.5;
|
||||
/// How far back the work list reaches (chain blocks): the record window, so every open segment inside the
|
||||
/// deadline is visible.
|
||||
pub const WORK_LOOKBACK: u64 = 600;
|
||||
|
||||
#[derive(Clone, Debug, PartialEq)]
|
||||
pub struct SegmentWork {
|
||||
pub first: u64,
|
||||
pub last: u64,
|
||||
/// the last block's DAA score; the deadline is it plus `proving_v1_unproven_daa`
|
||||
pub last_daa: u64,
|
||||
pub deadline_daa: u64,
|
||||
/// (chain block number, block hash, shard index) in chain order
|
||||
pub shards: Vec<(u64, String, u32)>,
|
||||
/// the shard payouts summed (what the shards earn when carried)
|
||||
pub shard_wei: u128,
|
||||
}
|
||||
|
||||
/// The segment holding chain block `number` on the grid that starts at `start`.
|
||||
pub fn segment_of(start: u64, n: u64, number: u64) -> (u64, u64) {
|
||||
let n = n.max(1);
|
||||
let k = number.saturating_sub(start) / n;
|
||||
(start + k * n, start + k * n + n - 1)
|
||||
}
|
||||
|
||||
/// The DAA margin a segment must have before its deadline: the floor, or 1.5 times the last segment's wall time.
|
||||
pub fn need_daa(last_segment_secs: Option<f64>) -> u64 {
|
||||
let from_last = last_segment_secs.map(|s| (s * SEGMENT_MARGIN_FACTOR).ceil() as u64).unwrap_or(0);
|
||||
from_last.max(SEGMENT_MARGIN_MIN_DAA)
|
||||
}
|
||||
|
||||
/// Groups the node's work list into whole, untouched segments: every block of the segment is in the list, every
|
||||
/// listed shard is open (past its exclusive window), unpaid and not in this node's pool from us. A segment with a
|
||||
/// block missing (inside the exclusive window, or outside the lookback) or a shard already paid is not a candidate.
|
||||
pub fn whole_segments(start: u64, n: u64, unproven_daa: u64, work: &[Work]) -> Vec<SegmentWork> {
|
||||
let n = n.max(1);
|
||||
let mut by_block: BTreeMap<u64, Vec<&Work>> = BTreeMap::new();
|
||||
for w in work.iter().filter(|w| w.number >= start) {
|
||||
by_block.entry(w.number).or_default().push(w);
|
||||
}
|
||||
let mut out = Vec::new();
|
||||
let mut seen = HashSet::new();
|
||||
for number in by_block.keys() {
|
||||
let (first, last) = segment_of(start, n, *number);
|
||||
if !seen.insert(first) {
|
||||
continue;
|
||||
}
|
||||
let mut shards = Vec::new();
|
||||
let mut wei: u128 = 0;
|
||||
let mut last_daa = 0;
|
||||
let mut whole = true;
|
||||
for b in first..=last {
|
||||
let Some(entries) = by_block.get(&b) else {
|
||||
whole = false;
|
||||
break;
|
||||
};
|
||||
let mut e: Vec<&&Work> = entries.iter().collect();
|
||||
e.sort_by_key(|w| w.shard);
|
||||
e.dedup_by_key(|w| w.shard);
|
||||
if e.iter().any(|w| !w.open || w.paid || w.in_pool) {
|
||||
whole = false;
|
||||
break;
|
||||
}
|
||||
for w in e {
|
||||
shards.push((w.number, w.hash.clone(), w.shard));
|
||||
wei = wei.saturating_add(w.shard_wei);
|
||||
if b == last {
|
||||
last_daa = w.daa;
|
||||
}
|
||||
}
|
||||
}
|
||||
if whole && !shards.is_empty() {
|
||||
out.push(SegmentWork { first, last, last_daa, deadline_daa: last_daa.saturating_add(unproven_daa), shards, shard_wei: wei });
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// FNV-1a 64 of the segment's first block and this prover's key hash: the per-prover rank.
|
||||
pub fn rank(first: u64, key_hash: &str) -> u64 {
|
||||
let mut h: u64 = 0xcbf29ce484222325;
|
||||
for b in first.to_be_bytes().iter().chain(key_hash.as_bytes()) {
|
||||
h ^= *b as u64;
|
||||
h = h.wrapping_mul(0x100000001b3);
|
||||
}
|
||||
h
|
||||
}
|
||||
|
||||
/// The segments to try, best first: inside the deadline by `need` DAA at `tip_daa`, not attempted, ranked by
|
||||
/// `rank(first, key)` (ties by the older first block).
|
||||
pub fn candidates(segs: &[SegmentWork], tip_daa: u64, need: u64, key_hash: &str, attempted: &HashSet<u64>) -> Vec<SegmentWork> {
|
||||
let mut c: Vec<SegmentWork> = segs.iter().filter(|s| !attempted.contains(&s.first) && s.deadline_daa >= tip_daa.saturating_add(1).saturating_add(need)).cloned().collect();
|
||||
c.sort_by(|a, b| rank(a.first, key_hash).cmp(&rank(b.first, key_hash)).then(a.first.cmp(&b.first)));
|
||||
c
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn w(number: u64, shard: u32, daa: u64, open: bool, paid: bool, in_pool: bool) -> Work {
|
||||
Work { number, hash: format!("0x{number:064x}"), shard, pgas: 0, tx_count: 0, assigned: false, open, paid, in_pool, shard_wei: 10, key_hash: String::new(), daa }
|
||||
}
|
||||
|
||||
fn grid(start: u64, n: u64, segments: u64, daa0: u64) -> Vec<Work> {
|
||||
(0..segments * n).map(|i| w(start + i, 0, daa0 + i, true, false, false)).collect()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_grid_is_counted_from_the_first_v1_block() {
|
||||
assert_eq!(segment_of(100, 8, 100), (100, 107));
|
||||
assert_eq!(segment_of(100, 8, 107), (100, 107));
|
||||
assert_eq!(segment_of(100, 8, 108), (108, 115));
|
||||
assert_eq!(segment_of(100, 8, 123), (116, 123));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn only_whole_open_unpaid_untouched_segments_qualify() {
|
||||
let mut work = grid(100, 4, 3, 1000); // 100..111, three segments
|
||||
work.retain(|x| x.number != 105); // 104..107 has a block missing (inside its exclusive window, say)
|
||||
work.iter_mut().find(|x| x.number == 110).unwrap().paid = true; // 108..111 has a paid shard
|
||||
let segs = whole_segments(100, 4, 600, &work);
|
||||
assert_eq!(segs.len(), 1);
|
||||
assert_eq!((segs[0].first, segs[0].last), (100, 103));
|
||||
assert_eq!(segs[0].shards.len(), 4);
|
||||
assert_eq!(segs[0].last_daa, 1003);
|
||||
assert_eq!(segs[0].deadline_daa, 1603);
|
||||
assert_eq!(segs[0].shard_wei, 40);
|
||||
// a shard of ours already in the pool, or one still exclusive, also disqualifies
|
||||
let mut work = grid(100, 4, 1, 1000);
|
||||
work[1].in_pool = true;
|
||||
assert!(whole_segments(100, 4, 600, &work).is_empty());
|
||||
let mut work = grid(100, 4, 1, 1000);
|
||||
work[3].open = false;
|
||||
assert!(whole_segments(100, 4, 600, &work).is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_block_with_several_shards_lists_them_in_order() {
|
||||
let mut work = grid(100, 2, 1, 1000);
|
||||
work.push(w(101, 1, 1001, true, false, false));
|
||||
work.push(w(100, 1, 1000, true, false, false));
|
||||
let segs = whole_segments(100, 2, 600, &work);
|
||||
assert_eq!(segs[0].shards.iter().map(|(n, _, s)| (*n, *s)).collect::<Vec<_>>(), vec![(100, 0), (100, 1), (101, 0), (101, 1)]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_deadline_margin_and_the_attempted_set_filter_the_candidates() {
|
||||
let work = grid(100, 8, 4, 1000); // 100..131, deadlines 1607, 1615, 1623, 1631
|
||||
let segs = whole_segments(100, 8, 600, &work);
|
||||
assert_eq!(segs.len(), 4);
|
||||
// at tip DAA 1380 with a 240 margin only the segments with a deadline at or past 1621 remain
|
||||
let c = candidates(&segs, 1380, 240, "0xkey", &HashSet::new());
|
||||
let firsts: Vec<u64> = c.iter().map(|s| s.first).collect();
|
||||
assert_eq!(firsts.len(), 2);
|
||||
assert!(firsts.contains(&116) && firsts.contains(&124));
|
||||
let mut attempted = HashSet::new();
|
||||
attempted.insert(firsts[0]);
|
||||
let c2 = candidates(&segs, 1380, 240, "0xkey", &attempted);
|
||||
assert_eq!(c2.len(), 1);
|
||||
assert_eq!(c2[0].first, firsts[1]);
|
||||
// past every deadline: nothing
|
||||
assert!(candidates(&segs, 1700, 240, "0xkey", &HashSet::new()).is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_order_is_deterministic_per_key_and_differs_between_keys() {
|
||||
let work = grid(100, 8, 6, 1000);
|
||||
let segs = whole_segments(100, 8, 600, &work);
|
||||
let a = candidates(&segs, 1000, 240, "0xaaaa", &HashSet::new());
|
||||
let a2 = candidates(&segs, 1000, 240, "0xaaaa", &HashSet::new());
|
||||
assert_eq!(a, a2);
|
||||
assert_eq!(a.len(), 6);
|
||||
// two provers rank the six candidates differently (the spread); the sets are the same
|
||||
let b = candidates(&segs, 1000, 240, "0xbbbb", &HashSet::new());
|
||||
let (fa, fb): (Vec<u64>, Vec<u64>) = (a.iter().map(|s| s.first).collect(), b.iter().map(|s| s.first).collect());
|
||||
let mut sa = fa.clone();
|
||||
let mut sb = fb.clone();
|
||||
sa.sort();
|
||||
sb.sort();
|
||||
assert_eq!(sa, sb);
|
||||
assert_ne!(fa, fb, "two keys should not rank six segments identically");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_margin_follows_the_last_segment_time() {
|
||||
assert_eq!(need_daa(None), 240);
|
||||
assert_eq!(need_daa(Some(100.0)), 240);
|
||||
assert_eq!(need_daa(Some(190.0)), 285);
|
||||
}
|
||||
}
|
||||
|
|
@ -319,6 +319,12 @@ fn api_post(shared: &Arc<Shared>, path: &str, body: Value) -> Result<Value, Stri
|
|||
shared.send(Cmd::SweepPin(key, pinned));
|
||||
Ok(json!({ "ok": true }))
|
||||
}
|
||||
// Power control (config.rs power_control): on = one administrator prompt now for the cap, off = nothing asks
|
||||
"/api/power/control" => {
|
||||
let on = body.get("on").and_then(|v| v.as_bool()).ok_or("on missing")?;
|
||||
shared.send(Cmd::PowerControl(on));
|
||||
Ok(json!({ "ok": true }))
|
||||
}
|
||||
"/api/sweep/enable" => {
|
||||
let on = body.get("on").and_then(|v| v.as_bool()).ok_or("on missing")?;
|
||||
shared.send(Cmd::SweepEnable(on));
|
||||
|
|
@ -333,7 +339,8 @@ fn api_post(shared: &Arc<Shared>, path: &str, body: Value) -> Result<Value, Stri
|
|||
Ok(json!({ "ok": true }))
|
||||
}
|
||||
"/api/quit" => {
|
||||
shared.send(Cmd::Quit);
|
||||
// the caller is on 127.0.0.1 and holds the token: the installer, the OTA apply, a script that read app.url
|
||||
shared.send(Cmd::Quit("POST /api/quit (a local caller with the token: the installer, the OTA apply, or a script that read app.url)"));
|
||||
Ok(json!({ "ok": true }))
|
||||
}
|
||||
_ => Err("unknown api".into()),
|
||||
|
|
|
|||
|
|
@ -97,6 +97,11 @@ pub struct CardState {
|
|||
pub temp_gpu: f64,
|
||||
pub temp_mem: f64,
|
||||
pub telemetry_at: f64,
|
||||
// AMD through igneum-gpu-telemetry (ADLX on Windows, amdgpu sysfs on Linux), 5 October 2026; 0 = unknown
|
||||
pub fan_pct: f64,
|
||||
pub fan_rpm: f64,
|
||||
pub mclk_mhz: f64,
|
||||
pub util_pct: f64,
|
||||
// hash per watt (src/sweep.rs)
|
||||
pub eff_mhw: f64, // live: hash_now over power_w, MH per watt; 0 = unknown
|
||||
pub sweep_supported: bool, // NVIDIA with readable limits; the note says why not otherwise
|
||||
|
|
@ -108,6 +113,18 @@ pub struct CardState {
|
|||
pub sweep_mhs: f64,
|
||||
pub sweep_at: f64, // unix s of the last sweep
|
||||
pub pinned: bool, // the user set the cap by hand; the sweep records but does not change it
|
||||
// Ember Tune (src/ember.rs): the two-knob tune, 5 October 2026
|
||||
pub clock_max_mhz: u32, // the vendor's maximum core clock (0 = unknown)
|
||||
pub clock_min_mhz: u32, // the vendor's floor for a cap (0 = 60% of the maximum)
|
||||
pub gclk_mhz: f64, // core clock now
|
||||
pub clock_cap_mhz: u32, // the cap in force (0 = unlocked)
|
||||
pub amd_ordinal: i64, // the `amd N` ordinal of igneum-gpu-telemetry (-1 = unknown)
|
||||
pub driver: String, // the driver version (nvidia-smi, or the worker's race line)
|
||||
pub program_class: String, // the program class of the race line (loads and wide loads per hash); "" = unknown
|
||||
pub tune_control: bool, // both knobs reach the card (else measure only; sweep_note says why)
|
||||
pub tune_clock_mhz: u32, // the clock cap the last tune chose (0 = unlocked)
|
||||
pub tune_source: String, // full | confirm | baseline
|
||||
pub tune_line: String, // "Tuned: 122.3 MH/s at 290 W (0.422 MH/W)" once tuned
|
||||
// the kernel variant race (docs/design/miner-tuning.md): what the worker's last race chose
|
||||
pub variant: String,
|
||||
pub race_mhs: f64,
|
||||
|
|
@ -172,6 +189,9 @@ pub struct ProvingState {
|
|||
pub submitted: u32,
|
||||
pub paid: u32,
|
||||
pub failed: u32,
|
||||
/// wei, serialised as a decimal string: serde_json's `to_value` refuses a u128 over u64::MAX (about 18.45 IGN,
|
||||
/// 15 paid shards at 1.23 IGN), and that refusal emptied the whole `/api/state` reply to "{}" (6 October 2026)
|
||||
#[serde(serialize_with = "u128_string")]
|
||||
pub paid_wei: u128,
|
||||
pub current: String,
|
||||
pub started_at: f64,
|
||||
|
|
@ -201,6 +221,15 @@ pub struct ProvingState {
|
|||
/// proving v1: segment records this machine aggregated and submitted, and the aggregator's last line
|
||||
pub aggregated: u32,
|
||||
pub segment_note: String,
|
||||
/// proving v1 segment path (6 October 2026): whole segments this machine proved and submitted, paid, and what
|
||||
/// they paid (wei as a decimal string, see `paid_wei`); the last segment's wall time
|
||||
pub segments_submitted: u32,
|
||||
pub segments_paid: u32,
|
||||
#[serde(serialize_with = "u128_string")]
|
||||
pub segment_paid_wei: u128,
|
||||
pub segment_last_s: f64,
|
||||
/// segment records the node refused by the chain rule and this machine offers again each pass
|
||||
pub segments_held: u32,
|
||||
}
|
||||
|
||||
#[derive(Clone, Serialize, Default)]
|
||||
|
|
@ -244,8 +273,16 @@ pub struct SettingsState {
|
|||
pub remote_jobs: bool,
|
||||
/// the prover service (src/prover.rs)
|
||||
pub prove: bool,
|
||||
/// the efficiency sweep (src/sweep.rs): once after install, then weekly
|
||||
/// the efficiency sweep (src/sweep.rs): once after install, then weekly; effective only with `power_control`
|
||||
pub sweep: bool,
|
||||
/// the NVIDIA power cap and the sweep may ask for administrator rights (config.rs: default off, one prompt when
|
||||
/// switched on)
|
||||
pub power_control: bool,
|
||||
/// the line beside the Power control switch: why it is off, or that the rights were given
|
||||
pub power_note: String,
|
||||
/// Ember Tune is paused fleet-wide by the signed manifest's kill switch (tuning.ember.enabled = false)
|
||||
pub tuning_off: bool,
|
||||
pub tuning_note: String,
|
||||
/// the miner software's dev fee switch (settings; `--dev-fee 0` when off)
|
||||
pub dev_fee: bool,
|
||||
/// devnet only: the node trusts proof records without a verifier (`IGNEUM_PROOF_VERIFY=trust`)
|
||||
|
|
@ -401,3 +438,22 @@ impl Rings {
|
|||
out
|
||||
}
|
||||
}
|
||||
|
||||
/// A u128 as a decimal JSON string (the dashboard reads it with `Number()`).
|
||||
pub fn u128_string<S: serde::Serializer>(v: &u128, s: S) -> Result<S::Ok, S::Error> {
|
||||
s.serialize_str(&v.to_string())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod paid_wei_tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn a_paid_total_over_u64_max_still_serialises_the_whole_state() {
|
||||
let mut st = State::default();
|
||||
st.proving.paid_wei = u64::MAX as u128 + 1;
|
||||
let v = serde_json::to_value(&st).expect("the state serialises");
|
||||
assert_eq!(v["proving"]["paid_wei"], serde_json::Value::String("18446744073709551616".into()));
|
||||
assert!(v["mining"].is_object());
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -340,10 +340,12 @@ impl Run {
|
|||
}
|
||||
}
|
||||
|
||||
/// The elevated helper that sets caps for a sweep (one administrator prompt per sweep, not one per step). It polls
|
||||
/// `<dir>/cmd.txt` twice a second: a line `<seq> <watts>` runs `nvidia-smi -i <device> -pl <watts>`, `quit` ends it.
|
||||
/// After 20 minutes without a new command it restores `<restore watts>` and exits by itself, so an engine that died
|
||||
/// mid-sweep leaves the card on its old limit. It writes what it ran to `<dir>/helper.log`.
|
||||
/// The elevated helper that sets limits for a tune (one administrator prompt per tune, not one per step). It polls
|
||||
/// `<dir>/cmd.txt` twice a second; each line is `<seq> pl <watts>` (`nvidia-smi -i <device> -pl <watts>`; the
|
||||
/// 0.3.9 form `<seq> <watts>` still works), `<seq> lgc <mhz>` (`-lgc 0,<mhz>`, the core clock cap; the memory clock
|
||||
/// is never touched) or `<seq> rgc` (`-rgc`, unlocked); `quit` ends it. After 20 minutes without a new command it
|
||||
/// restores `<restore watts>`, resets the clocks and exits by itself, so an engine that died mid-tune leaves the
|
||||
/// card on its old limits. It writes what it ran to `<dir>/helper.log`.
|
||||
pub fn helper_script_windows() -> &'static str {
|
||||
r#"param([string]$Dir, [string]$Smi, [string]$Device, [string]$Restore)
|
||||
$ErrorActionPreference = 'Continue'
|
||||
|
|
@ -359,15 +361,27 @@ while ($true) {
|
|||
$last = $c
|
||||
$idle = Get-Date
|
||||
if ($c -eq 'quit') { "$(Get-Date -Format o) quit" | Out-File -FilePath $log -Append -Encoding utf8; break }
|
||||
$w = ($c -split ' ')[-1]
|
||||
if ($w -match '^\d+$') {
|
||||
$out = (& $Smi -i $Device -pl $w 2>&1 | Out-String).Trim()
|
||||
"$(Get-Date -Format o) -pl $w : $out" | Out-File -FilePath $log -Append -Encoding utf8
|
||||
foreach ($line in ($c -split "`n")) {
|
||||
$p = ($line.Trim() -split ' ')
|
||||
if ($p.Count -lt 2) { continue }
|
||||
$op = $p[1]; $v = $p[-1]
|
||||
if ($p.Count -eq 2 -and $v -match '^\d+$') { $op = 'pl' }
|
||||
if ($op -eq 'pl' -and $v -match '^\d+$') {
|
||||
$out = (& $Smi -i $Device -pl $v 2>&1 | Out-String).Trim()
|
||||
"$(Get-Date -Format o) $($p[0]) -pl $v : $out" | Out-File -FilePath $log -Append -Encoding utf8
|
||||
} elseif ($op -eq 'lgc' -and $v -match '^\d+$') {
|
||||
$out = (& $Smi -i $Device -lgc "0,$v" 2>&1 | Out-String).Trim()
|
||||
"$(Get-Date -Format o) $($p[0]) -lgc 0,$v : $out" | Out-File -FilePath $log -Append -Encoding utf8
|
||||
} elseif ($op -eq 'rgc') {
|
||||
$out = (& $Smi -i $Device -rgc 2>&1 | Out-String).Trim()
|
||||
"$(Get-Date -Format o) $($p[0]) -rgc : $out" | Out-File -FilePath $log -Append -Encoding utf8
|
||||
}
|
||||
}
|
||||
}
|
||||
if (((Get-Date) - $idle).TotalMinutes -gt 20) {
|
||||
$out = (& $Smi -i $Device -pl $Restore 2>&1 | Out-String).Trim()
|
||||
"$(Get-Date -Format o) idle 20 min: restored $Restore W and quit: $out" | Out-File -FilePath $log -Append -Encoding utf8
|
||||
$out2 = (& $Smi -i $Device -rgc 2>&1 | Out-String).Trim()
|
||||
"$(Get-Date -Format o) idle 20 min: restored $Restore W, clocks reset, and quit: $out / $out2" | Out-File -FilePath $log -Append -Encoding utf8
|
||||
break
|
||||
}
|
||||
Start-Sleep -Milliseconds 500
|
||||
|
|
@ -388,11 +402,20 @@ while true; do
|
|||
if [ -n "$c" ] && [ "$c" != "$last" ]; then
|
||||
last="$c"; idle=$(date +%s)
|
||||
if [ "$c" = "quit" ]; then echo "$(date -u +%FT%TZ) quit" >> "$dir/helper.log"; break; fi
|
||||
w="${c##* }"
|
||||
case "$w" in ''|*[!0-9]*) ;; *) echo "$(date -u +%FT%TZ) -pl $w : $("$smi" -i "$dev" -pl "$w" 2>&1)" >> "$dir/helper.log";; esac
|
||||
printf '%s\n' "$c" | while IFS= read -r line; do
|
||||
set -- $line
|
||||
[ $# -ge 2 ] || continue
|
||||
op="$2"; v="${line##* }"
|
||||
[ $# -eq 2 ] && op=pl
|
||||
case "$op" in
|
||||
pl) case "$v" in ''|*[!0-9]*) ;; *) echo "$(date -u +%FT%TZ) $1 -pl $v : $("$smi" -i "$dev" -pl "$v" 2>&1)" >> "$dir/helper.log";; esac ;;
|
||||
lgc) case "$v" in ''|*[!0-9]*) ;; *) echo "$(date -u +%FT%TZ) $1 -lgc 0,$v : $("$smi" -i "$dev" -lgc "0,$v" 2>&1)" >> "$dir/helper.log";; esac ;;
|
||||
rgc) echo "$(date -u +%FT%TZ) $1 -rgc : $("$smi" -i "$dev" -rgc 2>&1)" >> "$dir/helper.log" ;;
|
||||
esac
|
||||
done
|
||||
fi
|
||||
if [ $(( $(date +%s) - idle )) -gt 1200 ]; then
|
||||
echo "$(date -u +%FT%TZ) idle 20 min: restored $restore W: $("$smi" -i "$dev" -pl "$restore" 2>&1)" >> "$dir/helper.log"; break
|
||||
echo "$(date -u +%FT%TZ) idle 20 min: restored $restore W, clocks reset: $("$smi" -i "$dev" -pl "$restore" 2>&1) / $("$smi" -i "$dev" -rgc 2>&1)" >> "$dir/helper.log"; break
|
||||
fi
|
||||
sleep 0.5
|
||||
done
|
||||
|
|
@ -405,8 +428,9 @@ pub fn unsupported_reason(vendor: &str, power_default_w: f64, device: &str) -> O
|
|||
match vendor {
|
||||
"nvidia" if power_default_w > 0.0 && !device.is_empty() => None,
|
||||
"nvidia" => Some("not available: nvidia-smi did not report this card's power limits"),
|
||||
"apple" => Some("not available on Apple silicon: there is no power cap to set, and powermetrics needs administrator rights for the draw"),
|
||||
"amd" => Some("not available for AMD in this version: the app has no power reading or cap for AMD cards (nothing like nvidia-smi ships with the driver)"),
|
||||
// Ember Tune (src/ember.rs, 5 October 2026): AMD is tuned through igneum-gpu-telemetry, Apple measures only;
|
||||
// the tune itself says which at its start (the card row's note)
|
||||
"apple" | "amd" => None,
|
||||
_ => Some("not available: no power reading or cap for this card"),
|
||||
}
|
||||
}
|
||||
|
|
@ -580,14 +604,16 @@ mod tests {
|
|||
fn unsupported_reasons() {
|
||||
assert!(unsupported_reason("nvidia", 575.0, "0").is_none());
|
||||
assert!(unsupported_reason("nvidia", 0.0, "0").unwrap().contains("power limits"));
|
||||
assert!(unsupported_reason("apple", 0.0, "").unwrap().contains("powermetrics"));
|
||||
assert!(unsupported_reason("amd", 0.0, "1").unwrap().contains("AMD"));
|
||||
assert!(unsupported_reason("apple", 0.0, "").is_none(), "measure only, said by the tune");
|
||||
assert!(unsupported_reason("amd", 0.0, "1").is_none(), "tuned through igneum-gpu-telemetry");
|
||||
assert!(unsupported_reason("other", 0.0, "1").is_some());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn helper_scripts_carry_the_protocol() {
|
||||
for s in [helper_script_windows(), helper_script_unix()] {
|
||||
assert!(s.contains("cmd.txt") && s.contains("quit") && s.contains("-pl") && s.contains("20 min"));
|
||||
assert!(s.contains("-lgc") && s.contains("-rgc"), "the clock cap and its reset");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -189,6 +189,7 @@ pub fn bash_line(file: &Path, login: bool, args: &[&str]) -> String {
|
|||
/// command line exactly as `bash_line` wrote it; elsewhere the words are ordinary arguments (nothing runs wsl there).
|
||||
/// The caller adds stdio, the hidden-window flag and the timeout.
|
||||
pub fn command(wsl_exe: &Path, distro: &str, user: Option<&str>, file: &Path, login: bool, args: &[&str]) -> Command {
|
||||
// console: a builder; every caller runs it through run_capture, run_streamed or platform::quiet (tools/ci/windows-spawn-check.mjs)
|
||||
let mut c = Command::new(wsl_exe);
|
||||
c.args(["-d", distro]);
|
||||
if let Some(u) = user.filter(|u| !u.is_empty()) {
|
||||
|
|
|
|||
|
|
@ -258,7 +258,21 @@ var View = (function () {
|
|||
function shortHex(h, head, tail) { h = String(h || ''); head = head || 8; tail = tail || 6; return h.length > head + tail + 2 ? h.slice(0, head) + '…' + h.slice(-tail) : h; }
|
||||
var kindWord = Notices.kindWord;
|
||||
// hot-plug (src/hotplug.rs): a removed card's row hides after five minutes (gone); a faulty one has no switch
|
||||
function shownCards(cards) { return (cards || []).filter(function (c) { return !c.gone; }); }
|
||||
// The list is ordered by performance (the project lead, 6 October 2026): usable cards first, then by the measured rate since the
|
||||
// start (5 MH/s buckets so the order does not flicker), then discrete, external and Apple before integrated, then
|
||||
// memory; removed and unusable cards last. Ties keep the detection order.
|
||||
function perfRank(c) {
|
||||
var usable = !(c.removed_at > 0) && !c.problem ? 0 : 1;
|
||||
var integrated = c.kind === 'integrated' ? 1 : 0;
|
||||
var bucket = -Math.floor(((c.enabled ? (c.hash_avg || c.hash_now || 0) : 0)) / 5);
|
||||
return [usable, integrated, bucket, -(c.vram_mb || 0)];
|
||||
}
|
||||
function shownCards(cards) {
|
||||
return (cards || []).map(function (c, i) { return { c: c, i: i, r: perfRank(c) }; }).sort(function (a, b) {
|
||||
for (var k = 0; k < a.r.length; k++) { if (a.r[k] !== b.r[k]) return a.r[k] - b.r[k]; }
|
||||
return a.i - b.i;
|
||||
}).map(function (x) { return x.c; }).filter(function (c) { return !c.gone; });
|
||||
}
|
||||
// listed, driver fine, not unplugged: a worker can run on it
|
||||
function present(c) { return !(c.removed_at > 0) && !c.problem; }
|
||||
// the tooltip on a card's name: what the tool calls it (gfx1201), its device index, the OpenCL platform, the PCI address
|
||||
|
|
@ -372,7 +386,35 @@ var View = (function () {
|
|||
}
|
||||
return { PAGES: PAGES, page: page, withCommas: withCommas, compact: compact, rel: rel, shortHex: shortHex, kindWord: kindWord, shownCards: shownCards, present: present, cardTitle: cardTitle, cardRow: cardRow, toggle: toggle, nodeWords: nodeWords, skewWord: skewWord, peersLine: peersLine, heightLine: heightLine, nextSwitch: nextSwitch, switchLine: switchLine, proveWords: proveWords, verifierWords: verifierWords, devFeeText: devFeeText, devFeeLine: devFeeLine, jobsNote: jobsNote };
|
||||
})();
|
||||
if (typeof module === 'object' && module && module.exports) { module.exports = Notices; module.exports.UpdateCard = UpdateCard; module.exports.View = View; }
|
||||
/* ---------- Ember Tune: the card row's tuning line (pure; tune-line.test.mjs loads this block) ----------
|
||||
One line per card from the card state (src/state.rs, src/ember.rs): running (the phase and the step), tuned
|
||||
("Tuned: 122.3 MH/s at 290 W (0.422 MH/W)" plus the point and when), measure only (Apple, NVIDIA without Power
|
||||
control: the measured line and why nothing is set), stopped (the reason), or not run yet. */
|
||||
var TuneLine = (function () {
|
||||
'use strict';
|
||||
function point(cd) {
|
||||
var pct = cd.sweep_pct || cd.power_pct || 0;
|
||||
if (cd.tune_clock_mhz > 0) return cd.tune_clock_mhz + ' MHz at ' + pct + '%';
|
||||
return pct ? pct + '%, clock unlocked' : '';
|
||||
}
|
||||
// the model: {kind: running|tuned|measured|stopped|idle|off, text, note}
|
||||
function model(cd, now) {
|
||||
cd = cd || {};
|
||||
if (cd.vendor !== 'nvidia' && cd.vendor !== 'amd' && cd.vendor !== 'apple') return { kind: 'off', text: cd.sweep_note || 'tuning: no power or clock control for this card' };
|
||||
if (cd.sweep_state === 'running') return { kind: 'running', text: cd.sweep_note || 'tuning: running' };
|
||||
var when = cd.sweep_at && now ? ', ' + rel(now - cd.sweep_at) + ' ago' : '';
|
||||
if (cd.tune_line) {
|
||||
if (cd.tune_source === 'baseline' || !cd.tune_control) return { kind: 'measured', text: cd.tune_line + ' (measured as it runs' + when + ')', note: cd.tune_control ? '' : (cd.sweep_note || '') };
|
||||
var src = cd.tune_source === 'confirm' ? 'from the fleet prior, confirmed' : 'full tune';
|
||||
return { kind: 'tuned', text: cd.tune_line, note: point(cd) + ', ' + src + when + (cd.pinned ? '; your setting stays pinned' : '') + (cd.sweep_note && cd.sweep_note.indexOf('stopped') === 0 ? '; ' + cd.sweep_note : '') };
|
||||
}
|
||||
if (cd.sweep_note && cd.sweep_note.indexOf('tuning stopped') === 0) return { kind: 'stopped', text: cd.sweep_note };
|
||||
return { kind: 'idle', text: cd.sweep_note || 'tuning: not run yet (starts after 120 s of steady mining)' };
|
||||
}
|
||||
function rel(s) { s = Math.max(0, Math.floor(s)); return s < 60 ? s + ' s' : s < 3600 ? Math.floor(s / 60) + ' min' : s < 86400 ? Math.floor(s / 3600) + ' h' : Math.floor(s / 86400) + ' d'; }
|
||||
return { model: model, point: point, rel: rel };
|
||||
})();
|
||||
if (typeof module === 'object' && module && module.exports) { module.exports = Notices; module.exports.UpdateCard = UpdateCard; module.exports.View = View; module.exports.TuneLine = TuneLine; }
|
||||
|
||||
if (typeof document !== 'undefined') (function () {
|
||||
'use strict';
|
||||
|
|
@ -558,7 +600,8 @@ if (typeof document !== 'undefined') (function () {
|
|||
$('s-devfee').addEventListener('change', function () { api('api/settings', { dev_fee: this.checked }).then(function (r) { if (r.ok) toast(r.restart ? 'Applied; the miner restarts' : 'Applied'); }); });
|
||||
$('s-login').addEventListener('change', function () { var on = this.checked; api('api/settings', { start_at_login: on }).then(function (r) { if (!r.ok) { toast(r.error || 'could not change'); $('s-login').checked = !on; } }); });
|
||||
$('s-jobs-allow').addEventListener('change', function () { api('api/jobs/allow', { on: this.checked }); });
|
||||
$('s-sweep').addEventListener('change', function () { api('api/sweep/enable', { on: this.checked }).then(function (r) { if (r.ok) toast($('s-sweep').checked ? 'Sweep on: once after install, then weekly' : 'Sweep off'); }); });
|
||||
$('s-sweep').addEventListener('change', function () { api('api/sweep/enable', { on: this.checked }).then(function (r) { if (r.ok) toast($('s-sweep').checked ? 'Ember Tune on: once after install, then weekly' : 'Ember Tune off'); }); });
|
||||
$('s-power-control').addEventListener('change', function () { var on = this.checked; api('api/power/control', { on: on }).then(function (r) { if (r.ok) toast(on ? 'Power control on: Windows asks for administrator rights once' : 'Power control off; nothing asks'); }); });
|
||||
$('s-trust').addEventListener('change', function () { var on = this.checked; api('api/settings', { proof_verify_trust: on }).then(function (r) { if (r.ok) toast(on ? 'Trust mode on (devnet only); the node restarts' : 'Trust mode off; the node restarts'); else { toast(r.error || 'could not change'); $('s-trust').checked = !on; } }); });
|
||||
$('s-live').addEventListener('click', function () { if (state && state.live_page) api('api/open', { url: state.live_page }); });
|
||||
$('s-log-open').addEventListener('click', function () { setDrawer(true); });
|
||||
|
|
@ -591,8 +634,8 @@ if (typeof document !== 'undefined') (function () {
|
|||
$('s-cards').addEventListener('click', function (e) {
|
||||
var b = e.target.closest('button'); if (!b) return;
|
||||
if (b.dataset.d) { var inp = b.parentNode.querySelector('input'), v = parseInt(inp.value, 10) || 1; inp.value = Math.max(1, Math.min(64, v + parseInt(b.dataset.d, 10))); sendCards('Identities set; that card’s worker restarts'); }
|
||||
else if (b.dataset.sweepStart) api('api/sweep/start', { key: b.dataset.sweepStart }).then(function () { toast('Sweep queued: 100% down to 50%, 75 s a step'); });
|
||||
else if (b.dataset.sweepStop) api('api/sweep/stop', {}).then(function () { toast('Sweep stopped; cap restored'); });
|
||||
else if (b.dataset.sweepStart) api('api/sweep/start', { key: b.dataset.sweepStart }).then(function () { toast('Tune queued: the power limit and the core clock, 75 s a step'); });
|
||||
else if (b.dataset.sweepStop) api('api/sweep/stop', {}).then(function () { toast('Tune stopped; the card is back where it was'); });
|
||||
else if (b.dataset.sweepPin) api('api/sweep/pin', { key: b.dataset.sweepPin, pinned: b.dataset.pinned === '1' }).then(function () { toast(b.dataset.pinned === '1' ? 'Cap pinned' : 'The sweep chooses the cap again'); });
|
||||
else if (b.dataset.powerRetry) api('api/power/apply', {}).then(function () { toast('Administrator prompt: allow it to set the cap'); });
|
||||
});
|
||||
|
|
@ -919,7 +962,7 @@ if (typeof document !== 'undefined') (function () {
|
|||
if (s.detecting) { det.hidden = false; $('detect-text').textContent = 'asking the graphics cards to report in'; $('btn-cards-next').disabled = true; $('cards-sub').textContent = 'Asking the graphics cards to report in.'; cardsRendered = ''; return; }
|
||||
det.hidden = true;
|
||||
var cards = shownCards(s.mining.cards);
|
||||
var sig = cards.map(function (c) { return c.key + ':' + (c.problem || '') + ':' + (c.removed_at > 0 ? 'r' : '') + ':' + (c.enabled ? 'on' : 'off'); }).join('|');
|
||||
var sig = cards.map(function (c) { return c.key + ':' + (c.problem || '') + ':' + (c.removed_at > 0 ? 'r' : '') + ':' + (c.enabled ? 'on' : 'off'); }).join('|'); // cards is already in performance order, so a reorder changes the signature
|
||||
if (sig !== cardsRendered) { cardsRendered = sig; renderCardRows(list, cards); }
|
||||
if (!cards.length) {
|
||||
$('cards-sub').textContent = 'No GPU this app can drive was found.';
|
||||
|
|
@ -1011,6 +1054,11 @@ if (typeof document !== 'undefined') (function () {
|
|||
setText('pv-submitted', String(pv.submitted || 0));
|
||||
setText('pv-paid', String(pv.paid || 0));
|
||||
setText('pv-paid-sub', pv.paid_wei ? (Number(pv.paid_wei) / 1e18).toFixed(4) + ' IGN earned' : 'shards paid out');
|
||||
// proving v1 segments (6 October 2026): whole segments this machine proved, and the segment path's last line
|
||||
var segLine = '';
|
||||
if (pv.segments_submitted) segLine = 'Segments: ' + pv.segments_submitted + ' proven whole, ' + (pv.segments_paid || 0) + ' paid' + (pv.segment_paid_wei && Number(pv.segment_paid_wei) ? ' (' + (Number(pv.segment_paid_wei) / 1e18).toFixed(4) + ' IGN to the aggregator)' : '') + (pv.segment_last_s ? ', the last in ' + Math.round(pv.segment_last_s) + ' s' : '') + (pv.segments_held ? ', ' + pv.segments_held + ' record' + (pv.segments_held > 1 ? 's' : '') + ' held for the chain rule' : '') + '.';
|
||||
if (pv.segment_note && enabled) segLine += (segLine ? ' ' : '') + pv.segment_note + '.';
|
||||
$('pv-seg-note').hidden = !segLine; setText('pv-seg-note', segLine);
|
||||
var v = View.verifierWords(pv);
|
||||
setText('pv-verifier', v.word); $('pv-verifier').className = 'big-word ' + v.tone;
|
||||
$('pv-verifier-note').hidden = !pv.verifier_note; setText('pv-verifier-note', pv.verifier_note || '');
|
||||
|
|
@ -1097,7 +1145,8 @@ if (typeof document !== 'undefined') (function () {
|
|||
var s = state;
|
||||
var sw = function (id, on) { if (document.activeElement !== $(id)) $(id).checked = !!on; };
|
||||
sw('s-vote', s.settings.vote); sw('s-devfee', s.settings.dev_fee); sw('s-login', s.settings.start_at_login);
|
||||
sw('s-jobs-allow', s.jobs && s.jobs.allowed); sw('s-sweep', s.settings.sweep); sw('s-trust', s.settings.proof_verify_trust);
|
||||
sw('s-jobs-allow', s.jobs && s.jobs.allowed); sw('s-sweep', s.settings.sweep); sw('s-trust', s.settings.proof_verify_trust); sw('s-power-control', s.settings.power_control);
|
||||
setText('s-power-note', s.settings.tuning_off ? s.settings.tuning_note : (s.settings.power_note || (s.settings.power_control ? '' : 'Off: NVIDIA cards measure only; AMD cards need no rights.')));
|
||||
setText('s-devfee-text', View.devFeeText(s));
|
||||
if (document.activeElement !== $('s-name')) $('s-name').value = s.display_name || '';
|
||||
setText('s-mid', (s.machine_id || '').slice(0, 8));
|
||||
|
|
@ -1113,27 +1162,30 @@ if (typeof document !== 'undefined') (function () {
|
|||
if (unusable) return h + '<p class="help">' + esc(cd.reason || 'No worker can drive this card.') + '</p></div>';
|
||||
if (nv) {
|
||||
h += '<div class="ctl"><span class="k">Power cap</span><input type="range" min="50" max="100" step="5" value="' + pct + '" aria-label="Power cap for ' + esc(cd.name) + '"><span class="pv">' + pct + '% · ' + Math.round(cd.power_default_w * pct / 100) + ' W</span>' +
|
||||
'<p class="help">Lower is cooler and quieter; most cards hash nearly as fast at 70%. ' + (cd.pinned ? 'Set by hand, so the sweep leaves it.' : 'The sweep may move it.') + '</p></div>';
|
||||
'<p class="help">Lower is cooler and quieter; most cards hash nearly as fast at 70%. ' + (cd.pinned ? 'Set by hand, so the tune leaves it.' : 'Ember Tune may move it.') + '</p></div>';
|
||||
var want = Math.round(cd.power_default_w * pct / 100);
|
||||
h += cd.power_applied ? '<div class="line ok">cap applied: <b>' + want + ' W</b>' + (cd.pinned ? ' <span class="pin">pinned</span>' : cd.sweep_pct === pct && cd.sweep_pct ? ' <span class="pin">chosen by the sweep</span>' : '') + '</div>' : '<div class="line hot">cap not applied (needs the administrator prompt) <button class="btn tiny" data-power-retry="1">Retry</button></div>';
|
||||
if (cd.telemetry_at > 0) h += '<div class="line">draw <b>' + (cd.power_w ? Math.round(cd.power_w) + ' W' : 'n/a') + '</b><span>GPU <b>' + (cd.temp_gpu ? Math.round(cd.temp_gpu) + ' °C' : 'n/a') + '</b></span><span>memory <b>' + (cd.temp_mem ? Math.round(cd.temp_mem) + ' °C' : 'n/a') + '</b></span><span>efficiency <b>' + (cd.eff_mhw ? cd.eff_mhw.toFixed(3) + ' MH/W' : 'n/a') + '</b></span></div>';
|
||||
var running = cd.sweep_state === 'running', t;
|
||||
if (!cd.sweep_supported) t = esc(cd.sweep_note || 'sweep not available on this card');
|
||||
else if (running) t = esc(cd.sweep_note || 'sweep running');
|
||||
else if (cd.sweep_pct) t = 'sweep: best <b>' + cd.sweep_pct + '%</b> · <b>' + cd.sweep_eff.toFixed(3) + ' MH/W</b> (' + cd.sweep_mhs.toFixed(1) + ' MH/s at ' + Math.round(cd.sweep_watts) + ' W' + (cd.sweep_at ? ', ' + rel(state.now - cd.sweep_at) : '') + ')';
|
||||
else t = esc(cd.sweep_note || 'sweep: not run yet');
|
||||
var btn = !cd.sweep_supported ? '' : running ? '<button class="btn tiny" data-sweep-stop="1">Stop</button>' : '<button class="btn tiny" data-sweep-start="' + esc(cd.key) + '">Sweep now</button>' + (cd.pinned ? ' <button class="btn tiny" data-sweep-pin="' + esc(cd.key) + '" data-pinned="0">Unpin</button>' : '');
|
||||
h += '<div class="line' + (running ? ' on' : '') + '">' + t + ' ' + btn + '</div>';
|
||||
} else if (cd.vendor === 'apple') {
|
||||
h += '<p class="help">Apple silicon manages its own power; there is no cap to set.</p>';
|
||||
h += '<p class="help">Apple silicon manages its own power; there is no cap to set. Ember Tune measures the card as it runs.</p>';
|
||||
}
|
||||
// Ember Tune's line (TuneLine.model): running, tuned, measured, stopped or not run yet, for every card
|
||||
h += tuneLineHtml(cd);
|
||||
h += '<div class="ctl"><span class="k">Identities</span><p class="help" style="grid-column:2">Each identity votes and is paid on its own. 8 suits a big card, 2 a small one, 1 an integrated GPU. Changing it restarts that card’s worker.</p><div class="ids"><button type="button" data-d="-1" aria-label="fewer identities">−</button><input type="number" min="1" max="64" value="' + (cd.identities || 1) + '" aria-label="Identities on ' + esc(cd.name) + '"><button type="button" data-d="1" aria-label="more identities">+</button></div></div>';
|
||||
return h + '</div>';
|
||||
}
|
||||
function tuneLineHtml(cd) {
|
||||
var m = TuneLine.model(cd, state ? state.now : 0);
|
||||
if (m.kind === 'off') return m.text ? '<div class="line dim">' + esc(m.text) + '</div>' : '';
|
||||
var running = m.kind === 'running';
|
||||
var t = m.kind === 'tuned' || m.kind === 'measured' ? '<b>' + esc(m.text) + '</b>' + (m.note ? ' <span class="dim">' + esc(m.note) + '</span>' : '') : esc(m.text);
|
||||
var btn = running ? '<button class="btn tiny" data-sweep-stop="1">Stop</button>' : '<button class="btn tiny" data-sweep-start="' + esc(cd.key) + '">Tune now</button>' + (cd.pinned ? ' <button class="btn tiny" data-sweep-pin="' + esc(cd.key) + '" data-pinned="0">Unpin</button>' : '');
|
||||
return '<div class="line' + (running ? ' on' : '') + '">' + t + ' ' + btn + '</div>';
|
||||
}
|
||||
function renderSetCards(s) {
|
||||
var cards = shownCards(s.mining.cards), box = $('s-cards');
|
||||
if (box.querySelector('input[type=range]:active') || box.contains(document.activeElement)) return;
|
||||
var sig = JSON.stringify(cards.map(function (c) { return [c.key, c.enabled, c.identities, c.power_pct, c.pinned, c.power_applied, c.sweep_state, c.sweep_pct, c.sweep_note, Math.round(c.power_w || 0), Math.round(c.temp_gpu || 0), Math.round(c.temp_mem || 0), c.problem || '', c.removed_at > 0]; }));
|
||||
var sig = JSON.stringify(cards.map(function (c) { return [c.key, c.enabled, c.identities, c.power_pct, c.pinned, c.power_applied, c.sweep_state, c.sweep_pct, c.sweep_note, c.tune_line, c.tune_source, c.tune_control, Math.round(c.power_w || 0), Math.round(c.temp_gpu || 0), Math.round(c.temp_mem || 0), c.problem || '', c.removed_at > 0]; }));
|
||||
if (sig === setCardsSig) return;
|
||||
setCardsSig = sig;
|
||||
setText('s-cards-eyebrow', cards.length ? cards.length + ' card' + (cards.length === 1 ? '' : 's') : '');
|
||||
|
|
|
|||
|
|
@ -219,6 +219,7 @@
|
|||
<div class="cell"><div class="k">proven</div><div class="v" id="pv-submitted">0</div><div class="s">proofs sent to the node</div></div>
|
||||
<div class="cell"><div class="k">paid</div><div class="v" id="pv-paid">0</div><div class="s" id="pv-paid-sub">shards paid out</div></div>
|
||||
</div>
|
||||
<p class="note" id="pv-seg-note" hidden></p>
|
||||
<div class="grid2">
|
||||
<div class="card">
|
||||
<div class="card-head"><h3>Verifier</h3><div class="eyebrow">the node's check</div></div>
|
||||
|
|
@ -366,8 +367,10 @@
|
|||
<div class="card">
|
||||
<div class="card-head"><h3>Graphics cards</h3><div class="eyebrow" id="s-cards-eyebrow"></div></div>
|
||||
<div class="set-cards" id="s-cards"><div class="empty">No card yet.</div></div>
|
||||
<label class="switch"><input type="checkbox" id="s-sweep"><span class="track"></span><span>Find each NVIDIA card's best efficiency</span></label>
|
||||
<p class="help">Once after install, then weekly, the power cap steps from 100% down to 50% and holds the step with the most hashes per watt. A cap you set by hand is left alone.</p>
|
||||
<label class="switch"><input type="checkbox" id="s-sweep"><span class="track"></span><span>Ember Tune: tune every card for hashes per watt</span></label>
|
||||
<p class="help">Once after install, then weekly and after a driver or program change: the power limit steps from 100% down to 50%, then the core clock from its maximum down to 60%, 75 s a step on the live program; the memory clock is never touched. The card keeps the point with the most hashes per watt within 1% of its top rate. A step with a rejected hash, a hot GPU or a dragged memory clock is reverted. A card whose model the fleet already knows starts at that point and confirms it in two steps. Every result goes back to the fleet without anything that identifies you. A cap you set by hand is left alone.</p>
|
||||
<label class="switch"><input type="checkbox" id="s-power-control"><span class="track"></span><span>Power control: let the app set NVIDIA limits</span></label>
|
||||
<p class="help">Windows asks for administrator rights once; the NVIDIA cap and the tune need them. Off, the app never asks and NVIDIA cards measure only. AMD cards need no rights. <span id="s-power-note"></span></p>
|
||||
</div>
|
||||
<div class="card">
|
||||
<div class="card-head"><h3>This machine</h3></div>
|
||||
|
|
|
|||
44
app/igneum-app/ui/tune-line.test.mjs
Normal file
44
app/igneum-app/ui/tune-line.test.mjs
Normal file
|
|
@ -0,0 +1,44 @@
|
|||
// node --test app/igneum-app/ui/tune-line.test.mjs (no dependencies; CI runs it in the site job)
|
||||
// The card row's Ember Tune line (app.js TuneLine): what a user sees per state, from the card state fields.
|
||||
import { test } from 'node:test';
|
||||
import assert from 'node:assert/strict';
|
||||
import { readFileSync } from 'node:fs';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
import { dirname, join } from 'node:path';
|
||||
|
||||
const src = readFileSync(join(dirname(fileURLToPath(import.meta.url)), 'app.js'), 'utf8');
|
||||
const mod = { exports: {} };
|
||||
new Function('module', src)(mod);
|
||||
const { model, point } = mod.exports.TuneLine;
|
||||
const NOW = 1_800_000_000;
|
||||
const card = over => ({ vendor: 'nvidia', sweep_state: 'idle', sweep_note: '', sweep_pct: 0, power_pct: 80, sweep_at: 0, tune_line: '', tune_source: '', tune_clock_mhz: 0, tune_control: true, pinned: false, ...over });
|
||||
|
||||
test('tuned: the line the brief asks for, with the point, the source and when', () => {
|
||||
const m = model(card({ tune_line: 'Tuned: 122.3 MH/s at 290 W (0.422 MH/W)', tune_source: 'full', tune_clock_mhz: 2470, sweep_pct: 100, sweep_at: NOW - 3600 }), NOW);
|
||||
assert.equal(m.kind, 'tuned');
|
||||
assert.equal(m.text, 'Tuned: 122.3 MH/s at 290 W (0.422 MH/W)');
|
||||
assert.equal(m.note, '2470 MHz at 100%, full tune, 1 h ago');
|
||||
const c = model(card({ tune_line: 'Tuned: 17.7 MH/s at 177 W (0.100 MH/W)', tune_source: 'confirm', sweep_pct: 90, sweep_at: NOW - 120, vendor: 'amd' }), NOW);
|
||||
assert.equal(c.note, '90%, clock unlocked, from the fleet prior, confirmed, 2 min ago');
|
||||
const p = model(card({ tune_line: 'Tuned: 1 MH/s at 1 W (1.000 MH/W)', tune_source: 'full', sweep_pct: 70, pinned: true }), NOW);
|
||||
assert.match(p.note, /your setting stays pinned$/);
|
||||
});
|
||||
|
||||
test('measure only: Apple and NVIDIA without Power control say so beside the measured line', () => {
|
||||
const a = model(card({ vendor: 'apple', tune_control: false, tune_line: 'Tuned: 26.7 MH/s at 38 W (0.703 MH/W)', tune_source: 'baseline', sweep_note: 'measure only on Apple silicon: the system sets the clocks and the power; no control exposed', sweep_at: NOW - 60 }), NOW);
|
||||
assert.equal(a.kind, 'measured');
|
||||
assert.equal(a.text, 'Tuned: 26.7 MH/s at 38 W (0.703 MH/W) (measured as it runs, 1 min ago)');
|
||||
assert.match(a.note, /^measure only on Apple silicon/);
|
||||
const n = model(card({ tune_control: false, tune_line: 'Tuned: 122.3 MH/s at 290 W (0.422 MH/W)', tune_source: 'baseline', sweep_note: 'measure only until Power control is on in Settings (Windows asks for administrator rights once)' }), NOW);
|
||||
assert.equal(n.kind, 'measured');
|
||||
assert.match(n.note, /Power control/);
|
||||
});
|
||||
|
||||
test('running, stopped, idle and off', () => {
|
||||
assert.deepEqual(model(card({ sweep_state: 'running', sweep_note: 'tuning: holding 2472 MHz · 100% · 41 s (step 7 of 9)' }), NOW), { kind: 'running', text: 'tuning: holding 2472 MHz · 100% · 41 s (step 7 of 9)' });
|
||||
assert.equal(model(card({ sweep_note: 'tuning stopped: a remote job took the GPU' }), NOW).kind, 'stopped');
|
||||
assert.equal(model(card({}), NOW).text, 'tuning: not run yet (starts after 120 s of steady mining)');
|
||||
assert.equal(model(card({ sweep_note: 'tuning: waits for 120 s of steady mining' }), NOW).text, 'tuning: waits for 120 s of steady mining');
|
||||
assert.equal(model(card({ vendor: 'other' }), NOW).kind, 'off');
|
||||
assert.equal(point({ tune_clock_mhz: 0, sweep_pct: 0, power_pct: 0 }), '');
|
||||
});
|
||||
|
|
@ -172,6 +172,24 @@ test('hot-plug (src/hotplug.rs): a removed card and a faulty card are shown as s
|
|||
// a row that is gone (five minutes after removal) is not shown at all; the big button counts only present cards
|
||||
assert.deepEqual(V.shownCards([card(), card({ key: 'x', gone: true })]).map((c) => c.key), ['nvidia:0:RTX 5090']);
|
||||
assert.equal(V.present(card()), true);
|
||||
// performance order (6 October 2026): PC 1 detects 5090, integrated AMD, 9070 XT; the list shows the 5090, the 9070 XT,
|
||||
// then the integrated card, whatever the detection order and whatever the integrated card's rate
|
||||
const pc1 = [
|
||||
card({ key: 'nvidia:0:RTX 5090', hash_avg: 122 }),
|
||||
card({ key: 'amd:gfx1036', name: 'AMD Radeon(TM) Graphics', vendor: 'amd', kind: 'integrated', vram_mb: 512, hash_avg: 2.1, enabled: true }),
|
||||
card({ key: 'amd:gfx1201', name: 'AMD Radeon RX 9070 XT', vendor: 'amd', kind: 'discrete', vram_mb: 16384, hash_avg: 18.2 }),
|
||||
];
|
||||
assert.deepEqual(V.shownCards(pc1).map((c) => c.key), ['nvidia:0:RTX 5090', 'amd:gfx1201', 'amd:gfx1036']);
|
||||
// before any rate (first start): discrete by memory, integrated last; a removed card last of all; ties keep detection order
|
||||
const fresh = [
|
||||
card({ key: 'amd:gfx1036', kind: 'integrated', vram_mb: 512, hash_avg: 0, hash_now: 0, state: 'off' }),
|
||||
card({ key: 'amd:gfx1201', kind: 'discrete', vram_mb: 16384, hash_avg: 0, hash_now: 0, state: 'waiting' }),
|
||||
card({ key: 'nvidia:0:RTX 5090', hash_avg: 0, hash_now: 0, state: 'waiting' }),
|
||||
card({ key: 'nvidia:1:RTX 5090', hash_avg: 0, hash_now: 0, state: 'waiting', removed_at: 5 }),
|
||||
];
|
||||
assert.deepEqual(V.shownCards(fresh).map((c) => c.key), ['nvidia:0:RTX 5090', 'amd:gfx1201', 'amd:gfx1036', 'nvidia:1:RTX 5090']);
|
||||
// a small rate change does not reorder (5 MH/s buckets): 122 and 124 sort as equal and keep detection order
|
||||
assert.deepEqual(V.shownCards([card({ key: 'a', hash_avg: 122 }), card({ key: 'b', hash_avg: 124 })]).map((c) => c.key), ['a', 'b']);
|
||||
assert.equal(V.present(gone), false);
|
||||
assert.equal(V.present(bad), false);
|
||||
const t = V.toggle({ state: 'mining', paused: false, cards: [gone, bad] }, { synced: true }, {});
|
||||
|
|
|
|||
|
|
@ -3,6 +3,6 @@
|
|||
// packaging/windows/Igneum-Miner.iss when the app version moves. Include guards, not #pragma once: rc.exe reads it too.
|
||||
#ifndef IGNEUM_HOST_VERSION_H
|
||||
#define IGNEUM_HOST_VERSION_H
|
||||
#define IGNEUM_HOST_VERSION_STR "0.3.11"
|
||||
#define IGNEUM_HOST_VERSION_RC 0,3,11,0
|
||||
#define IGNEUM_HOST_VERSION_STR "0.3.12"
|
||||
#define IGNEUM_HOST_VERSION_RC 0,3,12,0
|
||||
#endif
|
||||
|
|
|
|||
|
|
@ -1551,6 +1551,116 @@ What is measured: one BLS12-381 aggregate signature over 16 summed G1 keys plus
|
|||
|
||||
Reading (the NEW finding, ledger C4). With the module off GHOSTDAG alone converges on the heavier chain and the losing side's records re-determine (F24 works when the chain moves). With the module on the overlay holds during the split (A, with 30% of the frozen table, locks nothing; B locks 7 and 8) and then fails at the heal in the shipped node: B's certificates for blocks off n0's chain are "kept pending until the chain decides (no lock at this index)", n0's chain never decides because GHOSTDAG keeps its heavier tip and nothing turns the certificate into a fork-choice constraint, and once n0's last lock (index 7, DAA 209) is one window old (DAA 329) the frozen table stops applying on A's chain ("no frozen table (no lock on this chain inside the window)"), A's two keys are 100% of A's own window (B's post-cut blocks are red there) and n0 locks 10, 11, 12 alone; B's certificates for 10 and 11 then log CONFLICTING on n0 (n0 log, 17:27:04 to 17:29:54 BST). A finality fork from a 96-s honest partition, no attacker, table intact at the heal; the 150-s run and the v2 control end the same way. The spec's fork choice ("GHOSTDAG among tips through all certified checkpoints", 3.5) is therefore implemented only for certificates over blocks already on the node's chain. Fix named in the ledger entry: verify an off-chain certificate against the table at its own block and let it constrain fork choice (a certificate-driven reorg), then re-determine. Raw: `scratchpad fud-a/c4-results-*.md`, node logs `c4-on90-tmp/`, `c4-v2-control-tmp/`.
|
||||
|
||||
## 5 October 2026 (evening), the 9070 XT on the eGPU: why 17.9 MH/s, and what moved
|
||||
|
||||
PC 1 (ae432dc7, Windows 11, Ryzen 7 9800X3D with its gfx1036, RTX 5090 on CUDA), an AMD Radeon RX 9070 XT (gfx1201, RDNA 4) in a Sonnet Breakaway Box 850T5 over USB4, Adrenalin 26.9.2 (OpenCL driver string `3683.0 (PAL,LC)`, platform `OpenCL 2.1 AMD-APP (3683.0)`). Branch `opencl-rdna4`. the project lead: "the hashrate is low" (17.9 MH/s with one worker; two workers on the card earlier gave 8.9 and 9.4).
|
||||
|
||||
**Before, from PC 1's own app log** (`node tools/logs.mjs win-ae432dc7-20261005-181046`, the miner's STATUS line for the card `amd:1:gfx1201`, 2^21-nonce jobs): `hash=17.82 MH/s wall (17.83 MH/s inside jobs) ... idle=0.3%`. Wall equals inside, so the host loop (template fetch, job line, read-back, scan) costs nothing measurable; the dispatch itself is slow. The worker's `ready` line: `exchange 0` (local memory: AMD lists `cl_khr_subgroups` and no shuffle extension), `batch 4194304`, `dataset-log2 28` (1 GiB), device `[1] gfx1201` on the 3683.0 platform, `AMD wavefront width 32`. The same card was listed again as `[3] gfx1201` on the older platform `3652.0` (the 32.0.21042 driver's OpenCL registration is still present after the update): that is the two-worker run.
|
||||
|
||||
**Hypotheses, each with its number** (the measurement job `rdna4-bench-1`, 18:39:25 to 18:41:17 UTC, the card switched off in the app through `POST /api/cards` for key `amd:1:gfx1201` only, the 5090 untouched; worker exe sha256 `53c7e8c9…5403e10` built from this branch by `proto-cuda/nvrtc/build-windows.sh`; read back with `node tools/jobs.mjs rdna4-bench-1`):
|
||||
|
||||
| # | Hypothesis | Measured | Verdict |
|
||||
|---|---|---|---|
|
||||
| 1 | The dataset or program is re-sent over the eGPU link per job | Nothing is re-sent: the dataset (1 GiB) and cache (256 MiB) are built on the device once per pair (`info first pack ... cache 11 dataset 51 ms` on the Mac check); per 2^21-nonce job the old path sent 32 B up and read 16 MiB down; the serve A/B below puts a number on that read-back | Not the cause |
|
||||
| 2 | Work-group, occupancy, wave width, the exchange | `clGetKernelSubGroupInfoKHR`: sub-group 32 for a 32-item work-group (wave32), private memory 0 (no spills), preferred multiple 32; `--group-warps 1, 2, 4, 8` = 18.024, 18.063, 18.070, 18.039 MH/s (`--batches 3`, 2^24, device event time); `--batch-log2 21` (the app's job size) = 18.108 | Not the cause: the shape does not move the number |
|
||||
| 3 | The wrong AMD platform | The app's worker runs on `[1]`, the 3683.0 platform (ready line). The old platform's `[3]` gives 18.049 MH/s: the same. The duplicate listing is real and is the two-worker halving | Not the cause of 17.9; fixed anyway (below) |
|
||||
| 4 | The card's own random-read rate | `--memprobe`: dependent random 4-byte loads over 1024 MiB top out at 2.42 to 2.68 G loads/s from 4,096 lanes up (table below); 128 loads per hash gives a ceiling of 18.9 to 20.9 MH/s; the hash runs at 18.0 to 18.1 | THE CAUSE: the hash is at 87 to 95% of what this card does for this access pattern |
|
||||
|
||||
**The memprobe on the 9070 XT** (`igneum-worker-opencl.exe --device 1 --memprobe`, device event time, best of 3, 256 dependent steps per lane; `chase` = one dependent random 4-byte load per step, `indep x8` = eight independent chains per lane):
|
||||
|
||||
| Buffer | Work-group | Lanes in flight | chase G loads/s | ns per dependent load | indep x8 G loads/s |
|
||||
|---|---|---|---|---|---|
|
||||
| 4 MiB (inside the 8 MB L2, approximate size) | 256 | 4,096 | 34.95 | 117 | |
|
||||
| 4 MiB | 256 | 262,144 | 64.63 | 4,056 | 63.8 (262k lanes) |
|
||||
| 64 MiB (the 64 MB Infinity Cache, approximate size) | 256 | 4,096 | 9.17 | 447 | |
|
||||
| 64 MiB | 256 | 262,144 | 9.18 | 28,561 | 8.8 (262k lanes) |
|
||||
| 1024 MiB (GDDR6) | 32 | 4,096 | 2.64 | 1,552 | |
|
||||
| 1024 MiB | 32 | 65,536 | 2.60 | 25,181 | |
|
||||
| 1024 MiB | 32 | 4,194,304 | 2.43 | 1,729,136 | |
|
||||
| 1024 MiB | 256 | 4,096 | 2.64 | 1,552 | |
|
||||
| 1024 MiB | 256 | 262,144 | 2.45 | 106,831 | 2.46 (262k lanes) |
|
||||
| 1024 MiB | 256 | 4,194,304 | 2.42 | 1,732,023 | 2.42 (4M lanes) |
|
||||
| ALU chain, 1,048,576 lanes x 4,096 steps | 256 | | 6,219 G int ops/s (5 ops per step counted, approximate) | | |
|
||||
|
||||
Reading: at the dataset size the card delivers about 2.5 G random 4-byte reads per second whatever the parallelism (4,096 lanes already saturate it; more lanes only queue, the ns column is Little's law on a fixed throughput). Eight independent loads per lane give the same 2.4 G/s, so it is not a latency-hiding problem in the kernel. Inside the Infinity Cache the same chain runs 3.7x faster and inside L2 26x faster, so the cap is the path to GDDR6 for random reads. The ALU chain says the shader clock is not parked (approximate: 6.2 T int ops/s is of the order of 64 CUs x 64 lanes x 2.46 GHz with quarter-rate multiplies).
|
||||
|
||||
**Against the other two cards** (same probe; the 5090 through NVIDIA's OpenCL `[4]` WHILE its CUDA worker was mining, so a lower bound; the Mac through Apple OpenCL, wall time, a Mac at high load, approximate):
|
||||
|
||||
| Card | 1024 MiB chase at 4,096 lanes | 1024 MiB chase ceiling | indep x8 ceiling | ceiling / 128 = hash ceiling | measured hash rate |
|
||||
|---|---|---|---|---|---|
|
||||
| RX 9070 XT, eGPU over USB4 | 2.64 G/s, 1,552 ns | 2.42 to 2.68 G/s | 2.42 G/s | 18.9 to 20.9 MH/s | 18.0 to 18.1 MH/s (bench), 17.8 (app) |
|
||||
| RTX 5090, PCIe 5 x16, contended | 9.09 G/s, 451 ns | 16.4 to 18.0 G/s | 16.2 to 16.7 G/s | 128 to 141 MH/s | 127 MH/s (app, the project lead), 139.7 alone (M11) |
|
||||
| Apple M5 Max, Apple OpenCL | 2.10 G/s, 1,949 ns | 3.41 to 3.49 G/s | 3.45 to 3.47 G/s | 26.6 to 27.3 MH/s | 27.9 Mhash/s (README, Apple OpenCL) |
|
||||
|
||||
Reading: on all three cards the hash runs within a few percent of 1/128 of the card's dependent random-read ceiling, which is what a 128-load program should do; the probe is a good model of the hash. The 5090 does 6.6x the random reads of the 9070 XT for 2.8x the rated bandwidth (1,792 against 640 GB/s, vendor figures): the rest is access granularity and DRAM behaviour on random 4-byte reads, which the kernel cannot change.
|
||||
|
||||
**Power, heat, fans and clocks, measured** (branch `opencl-rdna4-telemetry`; the project lead watched the 9070 XT at 90% usage with its fans barely turning and the app had no AMD reading, the MH/W line came from nvidia-smi only; a new helper `proto-opencl/gpu-telemetry.c` reads ADLX on Windows and the amdgpu sysfs on Linux. Job `tele-measure-1`, 20:27:45 to 20:29:41 UTC, both cards mining in the app, nothing touched: `igneum-gpu-telemetry -l 5` (sha256 `703cf69c…a9c69b`) and `nvidia-smi --query-gpu=index,name,power.draw,temperature.gpu,fan.speed,clocks.mem,clocks.gr,utilization.gpu -l 5` side by side, the app's `hash_now` every 5 s; `node tools/jobs.mjs tele-measure-1`):
|
||||
|
||||
| Card | Samples | Watts (mean, min to max) | Temperature | Fan | Memory clock | Shader clock | Busy | Hash (mean of 24) | MH/W, measured |
|
||||
|---|---|---|---|---|---|---|---|---|---|
|
||||
| RX 9070 XT, bus 98, ADLX `GPUPower` | 12 (the helper's buffered tail was lost at the kill; fixed, `fflush` per sample) | 198.9 (193 to 212) | 64 C | 657 rpm (ADLX gives rpm; no percent) | 2,505 MHz | 3,290 MHz | 100% | 17.73 MH/s | 0.089 |
|
||||
| RTX 5090, nvidia-smi, 450 W cap | 24 | 307.6 (306.3 to 308.7) | 69 C | 44% | 13,801 MHz | 2,850 MHz | 94% | 122.30 MH/s | 0.398 |
|
||||
| gfx1036 (integrated, idle) | 12 | 42.7 (32 to 56; the package, not the GPU alone) | 62 C | none | 2,800 MHz | 600 MHz | 0% | off | |
|
||||
|
||||
Reading: the 9070 XT draws 199 W of its 304 W board rating (vendor figure) at 100% busy with the shader clock at its top, so the die is waiting on memory, which is the ceiling finding again; the fans at 657 rpm and 64 C are the card's own curve at that load, not a fault. Per watt the 5090 is 4.5x the 9070 XT on this program class (0.398 against 0.089 MH/W). The earlier per-watt claim from the board rating (304 W) would have read 0.058 MH/W; the measured number is 1.5x that.
|
||||
|
||||
**Is it the eGPU link?** No. 2.42 G loads/s x 64 B lines = 155 GB/s of DRAM traffic, forty times what a USB4 PCIe tunnel carries (about 4 GB/s, approximate); the 1 GiB buffer sits in the card's own memory (the 4 and 64 MiB cases show the card's caches at work above it, and a buffer in host memory would run below 0.1 G/s). A PCIe slot would move the per-job read-back (16 MiB per 2^21-nonce job on the old path, now gone) and nothing else; the random-read ceiling is the card's. What a PCIe slot would give: the same 18 MH/s.
|
||||
|
||||
**What changed on `opencl-rdna4`** (`proto-opencl/host.c`, `app/igneum-app/src/detect.rs`):
|
||||
|
||||
| Change | Before | After |
|
||||
|---|---|---|
|
||||
| Duplicate platform | `--list` showed the card twice ([1] 3683.0 and [3] 3652.0); the app made two cards and ran two workers (8.9 + 9.4 MH/s) | the older platform's entry prints as ` dup [3] ... hidden, use [1]`, the default pick skips it, the app's parser (`parse_opencl_list`, 3 tests) never makes a card of it; `--device 3` still works for comparison. Verified on PC 1: `platforms: 2 device(s) hidden ...`, cards `amd:0:gfx1036` and `amd:1:gfx1201` only |
|
||||
| Kernel report | work-group and local memory | plus preferred multiple, private memory (spills), sub-group size on every exchange path (`info kernel:` in serve mode) |
|
||||
| Read-back per dispatch | 8 B per nonce (16 MiB per job) and a host scan of 2^21 words | a GPU select pass: the hits (index, hash) behind an atomic counter plus 34 sentinel words; 276 B per chunk plus 16 B per hit; found lines in nonce order; `--readback full` / `IGNEUM_READBACK=full` keeps the old path; a chunk with over 256 hits falls back to the full read |
|
||||
| Transfer accounting | none | bytes up and down per chunk and the mean device time of kernel, select, read-back and scan in the stats line every 200 jobs and at quit |
|
||||
| `--memprobe` | none | the tables above, no pack needed |
|
||||
|
||||
Correctness: `proto-opencl/test-generic.sh` on the Mac (Apple OpenCL) PASS on both paths: "15 sampled hashes (both packs, both sides of the 32-bit nonce boundary) equal igneum-pow hash-bound"; select path transfers `5 chunks, up 180 B, down 4452 B`, full path `up 160 B, down 1536 B` (the check's jobs are 32 to 64 nonces with every nonce a hit). The bench on the 9070 XT: cache check PASS, dataset self-test PASS, 6 of 6 vector warps PASS, batch fingerprint `3cc4fbf90fa6366c` at 2^24 for the devnet pack (the Apple OpenCL value in the README), at every `--group-warps`.
|
||||
|
||||
**The serve-mode A/B on the card** (job `rdna4-serve-4`, 19:11 UTC, card off in the app, worker exe sha256 `324a6d9b…2bfdfff`; 200 real `job` lines of 2,097,152 nonces each, the app's `--job-nonces`, against the emulator test pack `pack-a` (epoch `edc4fa84…`, self-test PASS, 96 of 96 vector lanes), target `0000100000000000` so that 408 hits fall in 200 jobs on both paths; `done` ms over jobs 11 to 200; `node tools/jobs.mjs rdna4-serve-4`):
|
||||
|
||||
| Read-back | Bytes down per job | Kernel (device, mean) | Select pass | Read-back (wall) | Host scan | Mean job | Inside-job rate |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| full (before) | 16,777,216 | 116.12 ms | 0 | 7.28 ms | 0.55 ms | 124.22 ms | 16.88 MH/s |
|
||||
| select (after) | 309 | 116.00 ms | 0.039 ms | 0.78 ms | 0.00 ms | 117.38 ms | 17.87 MH/s |
|
||||
| select (repeat) | 309 | 115.96 ms | 0.038 ms | 0.76 ms | 0.00 ms | 117.33 ms | 17.87 MH/s |
|
||||
|
||||
Reading: the kernel is the same 116.0 ms on both paths (18.08 MH/s pure kernel, the bench's number). The old path paid 7.8 ms per job for 16 MiB over the eGPU link (2.3 GB/s, the USB4 tunnel's rate; a PCIe slot would read it in about 1 ms, approximate) and the host scan. The select pass removes it: +5.9% per job on this link, nothing on the kernel. Both paths found the same 408 hits. The `--group-warps` and exchange levers were already shown flat above, so this is the whole host-side gain available on the 9070 XT.
|
||||
|
||||
**Probes with a fresh seed per repetition** (the first probe round replayed the same addresses on repeats, so its low-lane rows were cache hits; fixed in `probeLaunch`, job `rdna4-serve-4`): 1024 MiB chase at 256 lanes 276 ns per dependent load, at 1,024 lanes 422 ns, at 4,096 lanes 1,560 ns (2.63 G/s, the cap). Random 64-byte lines (four `uint4` loads per step) at 1024 MiB: 2.46 to 2.88 G lines/s = 158 to 184 GB/s in lines, the same count per second as the 4-byte chase: every random 4-byte read costs this card a 64-byte line fetch. Coalesced stream over the whole 1024 MiB: 635.2 GB/s against the vendor's 640 GB/s, so the memory clock is in its full state and the card is not parked. Inside the 64 MiB buffer the line probe reaches 8.3 to 14.0 G lines/s (533 to 894 GB/s in lines: the Infinity Cache, approximate).
|
||||
|
||||
**A second defect found on the way: the pack export race.** PC 1's app log since its 19:02 UTC restart (`node tools/logs.mjs win-ae432dc7-20261005-190232`): `worker error: error 0 pack packs\devnet: the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT` at 19:07:03, 19:07:19 and 19:08:07, so the 9070 XT was not mining at all in the app while this entry was written (my job `rdna4-serve-1` at 18:43 hit the same folder in the same state). Cause, from `app/igneum-app/src/engine.rs` `prepare_worker`: one thread per card, each running `igneum-miner export-pack` into the one folder `packs\devnet`; across an epoch change the two exports interleave and the folder keeps one epoch's `program.h` with the other's `seeds.txt` until the next export. Fix on this branch: a process-wide mutex around both export sites (`EXPORT_LOCK`); the second export rewrites the same pack. Not measured in the app yet: it ships with the branch.
|
||||
|
||||
**Answer to the project lead.** The 9070 XT does 2.5 G random 4-byte reads per second from its memory for this access pattern, and the hash needs 128 of them, so about 19 MH/s is this card's ceiling for the current program class, on any slot; it was running at 92% of that. The eGPU link cost 6% per job through the read-back, now removed (17.87 against 16.88 MH/s inside jobs standalone). The duplicate platform that halved it to 8.9 + 9.4 is folded away. The pack race that stopped it is serialised. Nothing else in the worker's control moves the number: the next step for this card is the program class itself (fewer, wider loads per hash would favour AMD's 64-byte lines), which is a consensus question, not a worker one.
|
||||
|
||||
## 5 October 2026 (night), Ember Tune: the two-knob efficiency tune, the fleet prior, and what PC 1 could measure tonight (miner-community-lead)
|
||||
|
||||
Branch `ember-tune` (54ff1bc), docs/plans/ember-tune.md. Every card tuned for MH per watt out of the box: the power limit and the core clock cap stepped on the live kernel (memory clock never touched), the point with the best MH per watt within 1% of the top rate kept and pinned, every result uploaded as a `TUNE {json}` record (a hash of the install id, no address) and folded per (card model, driver major, program class) into a prior the signed manifest carries back, so a new card of a known model starts there and confirms it in two steps.
|
||||
|
||||
**What was measured tonight (PC 1, machine ae432dc7, from its own uploads to the intake):**
|
||||
|
||||
| Fact | Where it was read | Consequence |
|
||||
|---|---|---|
|
||||
| The installed 0.3.9 app runs as `DESKTOP-KMCV30N\Admin` with `elevated=False` (account line, 19:02:33 UTC) | app log `win-ae432dc7-20261005-190232` | `nvidia-smi -pl` and `-lgc` need administrator rights; the one prompt is the Power control switch (3562f26), which the app never raises by itself |
|
||||
| Two in-app sweep attempts aborted at 20:09 UTC: `the_elevated_helper_did_not_run_(the_administrator_prompt_was_cancelled)` | the same log | no stored sweep result from today exists; the 5090's two-knob tune is owed to the morning (one click on Power control, then it runs by itself within 2 minutes of steady mining) |
|
||||
| The RX 9070 XT left PC 1's bus at about 20:40 UTC, was back at 21:09 and gone again at 21:22:59 UTC (the eGPU link, third drop today) | the telemetry agent and the PC 1 scheduler | the AMD path (ADLX, no prompt) is unit-tested on the helper's captured line shapes; its end-to-end run waits for the card |
|
||||
|
||||
**The pipeline, verified without a card:** 9 `ember` unit tests (plans, clamps, the choice rule, the five marks, a faulted step reverted inside a fake-clock run, the confirm verdicts, the baseline plan, the record and prior shapes, the vendor reasons), the AMD `tune` line and the 0.3.10 sample line parsed (`engine::amd_telemetry_tests`), the helper protocol (`sweep::tests`), 6 relay aggregation tests (five samples converge on 2,470 MHz at 100%; an outlier at 0.908 MH/W moves the median by nothing; baseline records make no prior; de-duplication; the manifest merge keeps lever 2's cards; the canonical round trip), 3 UI line tests. A test manifest was signed on this Mac with `packaging/ota/publish-manifest.sh --tuning` from fixture priors: `tuning.priors["NVIDIA_GeForce_RTX_5090|581|l128w16"]` = 2,470 MHz at 100%, 5 samples, beside the kernel-variant `cards` entry and `tuning.ember {enabled: true, min_samples: 5, rate_tolerance_pct: 1}`, signature verified by the signer, 21:25 UTC.
|
||||
|
||||
**Tier consequences** (docs/plans/ember-tune.md section 7): a 9-step full tune costs about 12 minutes once and 3 minutes a week per card, under 1% of the hour, the worker never stops; a rig tunes one card at a time and every card of a known model after the first takes the 3-minute confirm; a pool user gives up the same 1% of shares at most; Apple silicon and AMD on Linux measure only and the row says so.
|
||||
|
||||
**The PC 1 run, 22:30 UTC (job ember-tune-pc1-1, engine aeea3228..., PC 1 on 0.3.10):** the job published at 22:29:40Z, the installed app stopped its miners and started the second engine at 22:30:21Z, and at 22:31:06Z the installed app quit (its log: `quit: stopping the miners, then the node`, then `job ember-tune-pc1-1: aborted (the app is quitting)`), 46 s in, before any step. Nothing was set. Corrected the same night (C35), then named the next morning from the second engine's own log (collect ember-c35-collect-1, 06:59Z): the second engine, reporting 0.3.9 (the branch's Cargo version) under the manifest's `min_supported_version`, took the 0.3.10 update as urgent (the "urgent" rule beats the copied `auto_update = false`), downloaded it at 22:31:02Z and started `ota-apply.ps1` with the per-user installer at 22:31:05Z; the installer's PrepareToInstall sent `POST /api/quit` to the installed app, which logged `quit:` at 22:31:06Z. So the source was my own second engine's updater, through the installer, one second before: a second install of 0.3.10 over the 0.3.10 PC 1 had taken through the shipper's update-now at 21:40:41Z (release-0.3.10.md section 8), whose only effect was the quit and the hang. The first reading (the 0.3.11 rollout) was wrong in the cause and right in the class: an installer. What else is established: the engine's quit then HUNG for 24 minutes in the jobs runner's abort, waiting for EOF on the script's stdout pipe whose write end the second engine and its miners had inherited, and those miners (2 igneum-miner, 2 CUDA workers, 1 OpenCL worker) mined on, orphaned, until the relay lane killed them at about 23:00Z; the second engine also raised one administrator prompt at about 22:30:25Z (`apply_power_limits` at start counted `--sweep` as Power control), 41 s before the quit; PC 2's unexplained quit at 20:01:09Z came 20 s after a cancelled prompt of the same class, so the prompt is the common factor and the morning's test (one prompt raised beside the mining app on PC 2, the stamped quit line read). Fixed on the branch: b671c8b (quit sources, Power control alone decides, no cap at start under `--sweep`), 8ab9068 (no pipe into a second engine, its tree ended, the CI check), and the third close: a second engine never runs the updater (`IGNEUM_APP_NO_OTA=1`, implied by `--sweep`; the playbooks set it; the CI check demands it). What the run did record, the "before" snapshots with the miners stopped:
|
||||
|
||||
| Card | Read back at 22:30:20Z | Meaning |
|
||||
|---|---|---|
|
||||
| RTX 5090 (driver 617.14) | limit 450 W of 575 W default (min 400, max 600), draw 259.9 W idle-after-stop, core 2,850 MHz, `clocks.max.gr` 3,090 MHz, memory 14,001 MHz | the two-knob plan for this card is 5 power steps (575, 518, 460, 403, 400 W) and 4 clock steps (2,781, 2,472, 2,163, 1,854 MHz); it needs the one administrator prompt (Power control) |
|
||||
| RX 9070 XT (bus 98, present again) | `tune 1 ... gmax 0 gmax_range -500 1000 plimit 0 plimit_range -30 10 factory 1 ok` | the helper's clock range is an OFFSET from stock in MHz, not a ceiling: a probe reading it as a 1,000 MHz maximum would have asked for `--set-gmax 900`, an overclock. Fixed at 054e041: an offset range closes the clock knob (until the stock clock is known) and the power ladder runs on the percent scale bounded by the range, so the 9070 XT's plan is 100, 90, 80, 70% (the -30 floor), 4 steps |
|
||||
| Radeon(TM) Graphics (integrated) | `tune 0 ... gmax - ... factory 0 ok` | no manual tuning: measure only, and it is off by default anyway |
|
||||
|
||||
**Run 2, 6 October 2026, 07:21 to 07:56Z (job ember-tune-pc1-2, elevated on the project lead's word, engine 25113f52..., PC 1 on 0.3.11):** the project lead answered the one prompt; the installed app stopped its miners at 07:21:16Z; the second engine ran for the whole 35-minute budget at "waiting, 0.00 MH/s" and no step ran. Cause: the playbook wrote the engine's copy of settings.json with PowerShell 5.1's `Set-Content -Encoding utf8`, which adds a UTF-8 BOM; the engine's JSON parser refuses it, `Settings::load` fell back to defaults (no payout address, no cards), the engine logged `[error] no payout address` and never started a miner. Run 1's scratch log carried the same line the night before. Readbacks, idle both times: the 5090 at 90.6 W before and 69.9 W after (2,505 then 2,407 MHz core, 14,001 MHz memory, limit 450 W of 575), the 9070 XT at factory (`gmax 0`, `plimit 0`). Nothing set on either card. The installed app's runner released the miners-stopped hold by itself on the failed exit (`job finished; the miners restart` at 07:56:50Z, both miners up by 07:57:04Z, `mining` at 07:57:29Z): mining paused 36 min 13 s. Fix 8273494: the copy is written without a BOM, the address is read back and the job fails within seconds if it is empty (`RESULT TUNE scratch settings: address ..., cards N, first bytes ...`), and the CI check fails any playbook writing JSON with `Set-Content -Encoding utf8`. The re-run needs one more click on the prompt.
|
||||
|
||||
Consequence for the tiers: an AMD card is tuned on its power limit alone until its stock core clock is read (a 9070 XT at -30% is the floor the driver allows, 4 steps, 5 minutes); every NVIDIA card's two-knob plan waits on the user's one click on Power control; the re-run on PC 1 is held until the quit's source is named (the event-log collect) and follows the 0.3.11 rollout (the update clears the jobs folder, so the engine and the helper are fetched again), with the scheduler's slot.
|
||||
## 5 October 2026 (night), read width of the lottery hash: 4, 16 and 64-byte loads, a per-load mix, a written scratch; three cards (gate 1 experiment, cryptographer)
|
||||
|
||||
Branch `readwidth` (commits 019b014, b970dda, 4badcee, a9e002c, d0018cf and the entry commit); plan and recommendation in `docs/plans/read-width.md`. Nothing here changes consensus: every class sits behind `igneum-pow --class` and the default class is generator version 2 byte for byte (`igneum-pow/tests/packs.rs` passes on the four pinned packs after every commit). Question (the project lead, after "the 9070 XT on the eGPU" above): would wider reads keep the latency-bound random-access property while closing the vendor gap. Additions from the coordinator: a per-load width drawn from an era-fixed mix, and a written per-warp scratch (measurement only, no soundness claim).
|
||||
|
|
@ -1961,3 +2071,155 @@ Ten distinct programs (seed strings `igneum-devnet-v4-epoch0`, `/epoch1` .. `/ep
|
|||
Cache fill 1.95 ms GPU (192.4 ms one core), dataset build 20.8 ms GPU for 1 GiB. The devnet pack three times through `packbench --pack ../proto-cuda/packs/igneum-devnet-v4-epoch0 --batches 1 --batch-log2 20 --group 256` (the pack's two libraries, `memhard.metal` and `program.metal`): compile 79 ms, 1 ms, 1 ms (the system shader cache answers the identical source from the second run); cache fill 0.6 to 0.7 ms GPU, dataset build 20.7 to 20.8 ms GPU.
|
||||
|
||||
Reading: a fresh program compiles in about 18 ms on this card with the Metal compiler service warm, 79 ms for a pack with its dataset kernels, up to 1.8 s cold (the variant-racing entry's first seed), 0 to 444 ms at the fleet's live boundaries (M11). The hot table fill of layer 5 is 0.07 to 0.22 ms (ca2-cache). So the Mac's per-epoch compile-ahead is under 2 s without the race and about 38 s with it (M11: 34.0 / 34.9 / 37.8 s), and the race is the only item visible against the 600-s window in which the program is known (lead 1,200 s minus the 600-s VDF, fixed at every epoch length). PC cards, cited in the plan: RTX 5090 NVRTC 151 to 180 ms, prepare 0.5 to 1.0 s without the dataset (M11), race one round about 37 s; RX 9070 XT OpenCL compile NOT MEASURED at the current worker (owed: `host.c` times `clBuildProgram` only in the `prepare` path and no `prepared` line from gfx1201 is in any upload); Intel UHD build 3.0 to 6.4 s (M11). Floor by the rule (slowest compile-ahead under 10% of the epoch and inside the window, dataset excluded): 600 DAA s, carried by the race at 6.3% of 600 s; with the race off (M11 found base wins on both the 5090 and the Mac) the slowest measured row is the Intel iGPU at 1.1%. Consequences per tier and the difficulty-settle constraint (24% of a 600-s epoch in settle at the measured 144 s) are in the plan.
|
||||
|
||||
### 6 October 2026, 00:4xZ, the empty `/api/state` reply (proving v1 branch)
|
||||
|
||||
Reported by the aggregation-cost agent: PC 2's `/api/state` answered `{}` (2 bytes) at 22:22Z, 22:41Z and 00:18Z. Not measured on PC 2 (no job); derived from the app source and node 1's RPC, read-only on the Mac:
|
||||
|
||||
| Figure | Value | Source |
|
||||
|---|---|---|
|
||||
| Paid shards, devnet, all provers | 663 | `curl -s 127.0.0.1:26790 -d '{"jsonrpc":"2.0","id":1,"method":"igneum_getProvingStatus","params":[]}'` at tip DAA 0x22caf |
|
||||
| Paid wei, all provers | 0x2c2961a69990745400 = 814.64 IGN | same call |
|
||||
| Average per paid shard | 1.23 IGN (approximate: the mean over 663) | 814.64 / 663 |
|
||||
| u64::MAX in IGN | 18.45 | 2^64 - 1 over 1e18 |
|
||||
| Paid shards per app start before the reply empties | 15 (approximate: at the mean payout) | 18.45 / 1.23 |
|
||||
|
||||
Cause: `ProvingState.paid_wei: u128` and serde_json `to_value` (1.0.151, `value/ser.rs` `serialize_u128`: u64 range or an error); the error became `json!({})`. Fix: the field serialises as a decimal string; `state_json` logs the error once. Test `a_paid_total_over_u64_max_still_serialises_the_whole_state` (`cargo test --offline -q paid_wei`, 1 passed).
|
||||
|
||||
| The fast-time 3-node harness (`tools/proving-v1/net.mjs`, 29950+, suffix 956, every node in trust mode, three vmine voters, v0 at DAA 60, v1 at DAA 120, 4 blocks a segment, unproven after 60 DAA, a tenth to the aggregator; fork b177718e built on this Mac) | run 2, 19:13:01Z to 19:16:19Z, under the run lock: PASSED, 21 checks in 197.3 s (`tools/proving-v1/report-2026-10-05.json`). v1 start = chain block 119 on all three nodes; the native statement identical on all three. Known-finished: segment 119..122's fresh-chain record submitted to n1 at t=131.1 s, relayed, verified (trust) and PAID on n0 1.0 s later at chain block 129, 253,611,648,000,000,000 wei = a tenth of the four credits, the same on every node, the payout address holding it. Chain rule: segment 123..126's fresh-chain record refused ("does not chain to segment 119..122 ... proven (record paid at chain block 129)"), the continuing one (chain_len 8) accepted and paid. Known-failed: segment 127..130 left without a record: a fresh-chain record for 131..134 refused while 127..130 was pending ("pending until DAA 191"); at DAA 192 the status read unproven, a late record for 127..130 refused ("unproven: carried after the deadline"), the fresh-chain record for 131..134 accepted and paid with chain_len 4; `segmentsInWindow` proven 3, unproven 1. The shard side: a v1 shard's `shardWei` = 90% of its block's credit. Run 1 (19:10Z) failed in its own tooling (the signer's argument order), fixed |
|
||||
|
||||
## 5 October 2026 (night), aggregation cost on the RTX 5090: what a per-block aggregation spends and what each lever gives (proving engineer, agg-cost)
|
||||
|
||||
the project lead, 5 October 2026: "fix everything else in the numbers tonight". The number under test: the chained segment aggregation cost 9.6 to 9.7 s a block on PC 2's 5090 while the card mined (`chain-pc2-pv1c`, the entry above), 2.2 s on 4 October with the card to itself. Target: under 3 s a block, the miner's slowdown of the prover under 1.5x, the proof statement unchanged. Branch `agg-cost` (worktree `igneum-wt-agg-cost`, from `proving-v1` 219517f). Host changes (statement untouched, `elf/` untouched): the aggregation's stdin build timed apart from the prove call, the deferred-proof count and the SP1 knobs in the RESULT lines, `--mode chain --save-shards` (every shard's compressed proof written next to the results, so `--mode aggregate` re-runs the same proofs under other settings). Jobs: `agg-cost-pc2-1` (21:01:20Z to 21:25:11Z, `tools/proving-v1/pc2-agg-cost.ps1`, the package `igneum-prove-wsl2-aggcost.zip` fetched by `fetch-prove-aggcost` 20:55:39Z, built in WSL2 against the live target dir in 5 s, installed to `/opt/igneum-aggcost`, the live `/opt/igneum` untouched, `--mode id` the pinned pair) and `agg-cost-pc2-2` (21:34:00Z, the same script). The live prover was switched OFF for the runs (its sp1-gpu-server would otherwise be shared through `/tmp/sp1-cuda-0.sock` and carry its own environment; `gpu_server_before running=0`) and ON again at the end. Fixtures: four consecutive live blocks cut from PC 2's own node (86165..86168 at tip 86195, one empty shard each, every one MATCHES natively), the same four for every phase of job 1. App 0.3.9 on PC 2 throughout.
|
||||
|
||||
Known-finished case of the host changes before the GPU (this Mac, CPU, run lock, 20:41Z to 20:44Z): `--mode chain` over `fixtures/chain/block-81046.json` with `--save-shards` (shard 38.5 s, aggregate 43.4 s, the proof file written), then `--mode aggregate` over that saved shard proof with `SP1_WORKER_VERIFY_INTERMEDIATES=false` (46.6 s, the same statement `0x3dedb8ea...`), `--mode verify-segment` VERIFIED in 0.027 s; known-failed: a wrong statement NOT VERIFIED in 0.027 s. Unit tests: `cargo test --release -p igneum-prove-core -p igneum-prove-host`: core 8 passed, host 9 passed and 1 ignored (build lock, 20:53Z).
|
||||
|
||||
### Lever 1, the profile: where a per-block aggregation goes
|
||||
|
||||
| What | Measured (job `agg-cost-pc2-1`) |
|
||||
|---|---|
|
||||
| The host's own share of an aggregation (the stdin build: the AggInput, the proof clones into the request) | 0.000 s on every block, mining or idle (the `stdin` field of every `RESULT chain block` line): everything is inside the one `prove().compressed()` call to the GPU server |
|
||||
| The GPU server's log at `RUST_LOG=info` (phase A0, the same chain of 1, stderr captured) | 1 line: sp1-gpu-server 6.8.1 prints no spans and no timings, so the step costs below are read from the deferred-proof count, not from a profiler |
|
||||
| Aggregation with 1 deferred proof (the first block, no previous proof) against 2 (every chained block), the card mining | 7.9 s against 9.6, 9.6, 9.8 s: the second deferred proof costs 1.7 to 1.9 s under the miner |
|
||||
| The same, the miners paused (phase C, the same fixtures, 21:05:51Z) | 1.7 s against 2.1, 2.1, 2.2 s: the second deferred proof costs 0.4 to 0.5 s alone |
|
||||
| The shard proof of an empty shard | 7.4 to 7.8 s mining, 1.9 to 2.2 s alone |
|
||||
| A whole block (one empty shard plus its aggregation) | 17.1 to 17.3 s mining (end to end 67.4 s for 4 blocks), 4.1 s alone (16.4 s for 4) |
|
||||
| GPU utilisation over the phase (1-s `nvidia-smi` samples) | 93.9% mining (80 samples, the miner's), 15.8% alone (32 samples): the prover alone keeps the card busy a sixth of the time. Its work is short GPU bursts between CPU phases (the executor, the witness and recursion-program generation run on the CPU inside the server), and the miner's kernels fill the gaps |
|
||||
| GPU memory peak | 16,195 MiB mining (the miner's 3.4 GB resident), 14,483 MiB alone |
|
||||
| The slowdown by the miner, same fixtures, same host, 2 min apart | shards 3.6x, the first aggregation 4.6x, a chained aggregation 4.5x, a block 4.2x |
|
||||
| Setup per host process (client plus two key setups) | 13.0 to 15.7 s, mining or not |
|
||||
|
||||
Reading. An aggregation is three or four recursion steps on the card (the aggregator guest's one core shard, its lift, one deferred program per verified proof, the compose), each a burst of under half a second when the card is free. The chained aggregation's extra deferred proof is the only part that grows with the chain rule, 0.4 to 0.5 s alone. Everything else the 9.7 s holds is the miner: with the card at 94% from the lottery kernels, every prover burst waits for a time slice, and a 2.1-s aggregation becomes 9.7 s. The 4 October 2.2 s (two shards, no previous proof, the card to itself) and tonight's 1.7 s (one shard) and 2.1 s (one shard plus the previous proof) agree within the deferred count.
|
||||
|
||||
### Lever 2, batch and tree folds (estimate from the measured step costs; the statement is pinned, no guest was changed tonight)
|
||||
|
||||
A fold of K blocks' shard proofs plus the previous segment proof in ONE aggregator call would cost one core shard, one lift, K + 1 deferred programs and the compose tree in place of K chained aggregations. From the measured rows (alone: a 1-deferred aggregation 1.7 s, each further deferred proof 0.45 s; mining: 7.9 s and 1.8 s):
|
||||
|
||||
| Fold | Deferred proofs per call | Per block, card alone (estimate) | Per block, card mining (estimate) | Rule |
|
||||
|---|---|---|---|---|
|
||||
| chained, as pinned (measured) | 2 | 2.1 s | 9.7 s | one call per block |
|
||||
| batch of 4 | 5 | (1.7 + 4 x 0.45) / 4 = 0.9 s | (7.9 + 4 x 1.8) / 4 = 3.8 s | one call per 4 blocks |
|
||||
| batch of 8 | 9 | (1.7 + 8 x 0.45) / 8 = 0.7 s | (7.9 + 8 x 1.8) / 8 = 2.8 s | one call per 8 blocks |
|
||||
| tree of 4 (2 + 2, then the pair) | 3 per call, 3 calls | 3 x (1.7 + 2 x 0.45) / 4 = 1.9 s | 3 x (7.9 + 2 x 1.8) / 4 = 8.6 s | no gain over the chain: every call pays the fixed part |
|
||||
|
||||
Reading. A batch fold halves to quarters the per-block aggregation but changes the aggregator's statement (`AggInput` carries one block's shards and the guest asserts one block hash), so it is a new pinned guest and a new program id: a provers-off drain and a rollout (proving/README.md, pinned guests). It does not reach 3 s on a mining card by itself (2.8 s at K = 8 is on the line), and the shard proof beside it stays 7.4 s a block on a mining card. The lever that moves both is the card's other job, lever 4. A tree fold gains nothing here because the fixed part of a call (the core shard and the lift) dominates the per-proof part 4 to 1.
|
||||
|
||||
### Levers 3 and 4, two streams and the miner's kernels (job `agg-cost-pc2-2` and the re-run)
|
||||
|
||||
Job `agg-cost-pc2-2` (21:34:00Z to 21:49:22Z) ran with the 5090 idle throughout: job 1's `/api/resume` had left the worker off (below), so the rows that needed the miner (the batch-log2 curve, the two streams beside the miner, the time-slice policy, the chosen combination) are void and wait for a re-run; the idle rows are measured.
|
||||
|
||||
| What | Measured (job `agg-cost-pc2-2`, card idle) |
|
||||
|---|---|
|
||||
| Aggregate-only over job 1's four saved shard proofs (`--mode aggregate --proofs b1;b2;b3;b4 --parent ...`, one process, the same statement `0x3a995f24...` as the chain run), default knobs (phase B0, then C1) | 1.7, 2.0, 2.0, 2.0 s (1, 2, 2, 2 deferred proofs), 8.1 s for four; C1: 1.8, 2.1, 2.1, 2.1 s, 8.3 s |
|
||||
| The same with `SP1_WORKER_VERIFY_INTERMEDIATES=false` (phase B; the server inherits the host's environment, the knob printed in the `sp1 knobs` line) | 1.7, 2.0, 2.0, 2.0 s, 7.8 s for four: no gain (0.3 s over four, inside the run-to-run spread of 0.2 s). The knobs that change the recursion shape (`SP1_WORKER_MAX_COMPOSE_ARITY`, `MAX_REDUCE_ARITY`) were not tried: a different shape is a different recursion key set and the pinned verifier would refuse the proof |
|
||||
| A 4-deferred aggregation (block-344-shards4, four prototype shards of 6.75 M pgas, phase C2) | shards 42.8 s (10.7 s each, the 4 October 10.2 to 10.7 s), aggregation 2.4 s with 4 deferred proofs; GPU peak 28,402 MiB (the prototype shard's 28.3 GB), utilisation 27.7% over the phase. With 1.7 s at one deferred proof and 2.0 to 2.1 s at two: 0.25 s per further deferred proof alone, so a batch of 8 would cost about 3.5 s a call, 0.45 s a block (estimate, the pinned statement forbids it) |
|
||||
| Two host processes at once on the one card (phase G0: chains of 2 on disjoint blocks, started 2 s apart) | both connected to ONE sp1-gpu-server (the first process's child; the socket is per device, `/tmp/sp1-cuda-0.sock`): process 1 shard 2.2 and 3.5 s, aggregation 3.0 and 4.0 s (12.9 s for 2 blocks against 8.2 s alone); process 2 shard 3.3 s, aggregation 3.6 s, then its second block died with `CudaClientError: Failed to read the response: early eof` when process 1 finished and its server exited. GPU 24,911 MiB, utilisation 12.4% and 13.1%. Two streams through SP1 6.8.1's server are serialised on one socket and the second dies with the first: no throughput gain (3 blocks in 33 s against 4 in 16.4 s) and a failure mode; lever 3 is closed on this SP1 version |
|
||||
| Job 3 (`agg-cost-pc2-3`, 22:41:15Z, app 0.3.10, the same script with the socket rule and a card switch): phase A, the app's 5090 miner at 117.0 MH/s mean (n 3, STATUS lines 22:44:45Z to 22:46:11Z), four fresh live blocks 90896..90899 | shards 8.0, 7.8, 7.6, 7.8 s; aggregations 8.0 s (1 deferred), 10.0, 10.0, 10.0 s (2 deferred); 69.5 s for four, 17.8 s a block; GPU 93.8%, peak 16,245 MiB: the job-1 baseline reproduced 100 min later on other blocks |
|
||||
| Job 3's own-miner phases | void: the state reads came back empty (the class below), the card switch did nothing, phase D launched my miner beside the app's (the app's dropped to 62.2 MH/s, mine read 60.6 MH/s), then PC 2's app restarted at 23:03:30Z and the job died with it; no curve point |
|
||||
| The GPU time-slice policy (`nvidia-smi compute-policy --set-timeslice`, the restore job `agg-cost-restore-1`, 23:16:53Z) | "Not Supported" on PC 2 (RTX 5090, driver 13.3, the Windows nvidia-smi, not elevated): the lever is closed on this driver; an elevated try is not worth a slot, the error is the driver's, not a permission's |
|
||||
| The own-miner phases of job 2 | void: no 5090 miner was running to copy the command line from (the worker off since 21:25Z) |
|
||||
|
||||
The curve, job `agg-cost-pc2-6` (01:12:09Z to 01:24:14Z, app 0.3.11, PC 2 to itself; every phase closed before the next job landed on PC 2 at 01:24:21Z). The app's 5090 miner switched off through `/api/cards` (the keys from `settings.json`; the worker was still alive after 120 s, `/api/pause` as the fallback stopped it in 5 s), then the job's OWN miner on the 5090 with the app's command line (`igneum-miner mine ... --worker igneum-worker-cuda.exe --identities 8 --worker-args "--device 0 --pack packs\devnet --race off [--batch-log2 B]"`, the base variant, its STATUS line every 10 s), the same four live blocks 96556..96559 (one empty shard each) proven by `--mode chain` under it, the miner's rate from its own `now=` field (the first two lines skipped). `--batch-log2 B` sets the worker's nonces per kernel launch (2^B; 22 is the worker's default, 4,194,304 nonces, about 35 ms a launch at 120 MH/s; `proto-cuda/nvrtc/worker.cpp`).
|
||||
|
||||
| batch-log2 | Shard proof (4, s) | Aggregation (1 deferred, then 2) (s) | A block (s) | GPU util. (%) | GPU peak (MiB) | Own miner (MH/s wall, n) | Against the card alone (4.1 s a block) |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| 22 (the default), phase D | 8.1, 7.8, 7.9, 7.8 | 8.4; 10.3, 10.0, 10.4 | 18.1 | 95.5 | 16,580 | 103.9 (9) | 4.4x |
|
||||
| 20, E20 | 8.1, 7.8, 7.8, 7.8 | 8.4; 10.3, 10.1, 10.1 | 18.0 | 94.9 | 16,461 | 103.7 (8) | 4.4x |
|
||||
| 18, E18 | 7.0, 6.7, 6.7, 6.7 | 7.2; 8.9, 8.8, 8.8 | 15.6 | 91.5 | 16,487 | 99.3 (7), minus 4.4% | 3.8x |
|
||||
| 16, E16 | 5.1, 4.9, 4.9, 4.9 | 5.0; 6.1, 6.2, 6.2 | 11.1 | 85.3 | 16,519 | 83.8 (6), minus 19% | 2.7x |
|
||||
| 16 again, phase H (the job's own choice: the shortest chain) | 5.0, 4.9, 4.8, 4.9 | 4.9; 6.1, 6.2, 6.2 | 11.1 | 85.7 | 16,487 | 84.0 (6) | 2.7x |
|
||||
|
||||
Reading. Between 2^22 and 2^20 nothing moves: the card's time-slice scheduler alternates the two contexts whatever the kernel length above a few milliseconds. From 2^18 down the miner's launches get short enough (about 2 ms at 2^18, 0.5 ms at 2^16) that the prover's bursts find the card sooner, and the miner pays in launch overhead and idle gaps: at 2^16 the prover runs 1.6x faster (18.1 to 11.1 s a block, the chained aggregation 10.2 to 6.2 s) for a fifth of the hash rate, and it is still 2.7x slower than on a card to itself. The trade is about 1 MH/s per 0.37 s of block time at the 2^16 point, and the 3-s aggregation and the 1.5x slowdown are not reachable on a mining card by the kernel length; a 2^14 point (approximate, extrapolated) would be about 8 s a block at about 65 MH/s. The phase E0 (a 4-deferred aggregation under the miner) failed in 0.1 s: its proof paths pointed at `/` where job 1 had left its shard proofs, but job 2's block-344 proofs sit in job 2's own folder (`$JOB` was exported from job 2 on); the 4-deferred cost under the miner stays an estimate (lever 2 above). The app's own 5090 miner ran at 117 MH/s (job 3, 22:44Z) and 110 to 129 MH/s (its STATUS lines at 01:10Z) with the prover beside it, against my miner's 104 MH/s at the default batch: my miner runs the base variant with `--race off` (no tuning file on PC 2), so the curve's rates are relative to each other, not to the app's.
|
||||
|
||||
### Lever 5, the host side under WSL2 (what the chain-mode numbers leave out)
|
||||
|
||||
| What | Measured |
|
||||
|---|---|
|
||||
| The export (`igneum_exportSegments` 0..tip, 75 to 77 MB over curl.exe to a file on `C:`) | 1.1 to 1.5 s |
|
||||
| The cut (`igneum-prove-export` replaying from genesis, then `--mode native`), four blocks | 18 s for four including the native checks (21:01:28Z to 21:01:46Z), about 4 s a block; the export's file sits on `/mnt/c` |
|
||||
| The key setup per host process | 13.0 to 15.7 s on PC 2 (8.0 to 8.5 s on the Mac CPU): `--mode chain` and `--mode aggregate` pay it once per process, the app's loop pays it per shard |
|
||||
| The proof file write through the WSL2 bridge | the 4 October entry ("shard proving on the RTX 5090"): 24 min of unbuffered `save` across `/mnt/c`, fixed by the 4 MB buffer; tonight `--save-shards` wrote the four 1.27 MB proofs inside the chain phase with no visible gap (the A phase's 80.4 s wall against 67.4 s of proving plus 13.0 s of setup) |
|
||||
| Native Linux | not measured: no native Linux machine with an NVIDIA card exists in the project tonight, and the 4 October numbers were also WSL2 (Ubuntu 24.04 under PC 2's Windows). The WSL2 cost inside a `prove()` call is not separable from here; the host-side pieces above are what a native box would also skip or keep |
|
||||
|
||||
### What went wrong, measured
|
||||
|
||||
| What | Fixed |
|
||||
|---|---|
|
||||
| Job 1's per-phase command ran with `$JOB` empty (the bash variables of `vars.sh` were set, not exported, and the command runs in a child bash): `--out /results-A.json`, the saved shard proofs in `/` on the WSL root, so the aggregate-only phases B0, B, C1 and the prototype-shard phase C2 failed in 0.0 s ("No such file") | `export` in `vars.sh`; job 2 reads the proofs from `/` |
|
||||
| Job 1's own-miner phases launched the iGPU miner (the first `igneum-miner mine` process matched; the 5090's is the second) and `if (StartMiner ...)` was always true (PowerShell: a function's emitted RESULT strings are part of its output), so D and E ran with the 5090 idle and the AMD iGPU at 3.4 MH/s: three more idle replicates of the chain (2.0 to 2.2 s shards, 1.8 and 2.2 s aggregations), no curve | the miner matched on `igneum-worker-cuda`, the outcome in a script-scope flag, `--race off` for the own miner (no tuning file on PC 2; a race costs up to 120 s a start) |
|
||||
| Job 1's `/api/resume` at 21:25:11Z answered ok and the 5090 miner stayed off (card state `off`, hash 0.0, 1,760 MiB on the card) until the 0.3.10 restart; job 2 waited its full 600 s for a hash rate and ran its mining phases void | the restore job `tools/proving-v1/pc2-agg-cost-restore.ps1` also posts `/api/start`; the Counter ASIC coordinator opened a task chip for the resume defect |
|
||||
| Jobs 3 and 4 (`agg-cost-pc2-3` 22:41Z on app 0.3.10, `agg-cost-pc2-4` 00:18Z on 0.3.11): every `/api/state` read came back as the two bytes `{}` (job 4's raw-body print: `raw_len=2`; the same reads gave the full state on 0.3.9 at 21:01Z and the AMD agent saw the empty reply at 22:22Z), so the card switch found no card, the app's 5090 miner kept mining, and job 3 ran a second miner beside it (two miners at about 60 MH/s each) while job 4's double-mining guard voided its own-miner phases. The class is the app's, not the reader's: `state_json()` (engine.rs:180) does `serde_json::to_value(st).unwrap_or(json!({}))`, and the value that fails is `ProvingState.paid_wei: u128` (serde_json 1.0.151 refuses a u128 over u64::MAX, 18.45 IGN; the proving-v1 agent's diagnosis): a paid shard averages 1.23 IGN, so the reply empties about 15 paid shards after every app start and comes back at the next restart, which matches the times (full at 21:01Z with paid_wei 0, empty from 22:22Z after the prover had paid from 22:02Z). Fixed on the app branch proving-v1 at 6714a45 (paid_wei as a decimal string, the error logged, an `{"error":...}` reply on any future failure) | job 5 reads the card keys from the app's `settings.json` (`cards`: key to enabled and identities), restores the 5090's 8 identities first (the restore job of 23:16:53Z had set 2: its parser read the next card's value), refuses before any pause when it cannot name the card, waits on the CUDA worker process count for the card to stop, and checks the worker is back at the end |
|
||||
| Job 5 (`agg-cost-pc2-5`, 01:10:44Z) failed at PowerShell's parse in 1 s: `$RestoreIdentities:` inside a double-quoted string (a drive-qualified variable); no card or miner touched | `${RestoreIdentities}:`; the other `$name:` shapes are inside single-quoted bash here-strings |
|
||||
| Job 6's identities step found `settings.json` already at 8 identities under the active key `nvidia:0:NVIDIA GeForce RTX 5090` (a stale key `nvidia:NVIDIA GeForce RTX 5090` carries 2), so no change was sent; job 6's `/api/cards` with the 5090 disabled answered ok but the worker ran on for 120 s, `/api/pause` stopped it in 5 s, and at the end `/api/resume` brought it back in 5 s on 0.3.11 | the card switch keeps the pause as its fallback; the resume path works on 0.3.11 |
|
||||
| PC 2 ran three jobs at once from 01:24Z (`run-prover-on-pc2-20261006` at 01:24:21Z, the ledger suites build at 01:26:15Z, while agg-cost-pc2-6's closing report was still being uploaded): the app does not serialise jobs, "one job per machine at a time" holds only by the coordinator's word; job 6 had closed at 01:24:14Z, so its rows are clean | nothing of mine to fix; a rule for the job runner |
|
||||
| The make-package gate ran the exporter's side files (`block-N.json.node-plan.json`) as fixtures and failed; its execute step took the exclusive `measure` lock for a cycle count and queued 25 min behind a packbench run | the glob skips `.node-plan.json`; the execute step runs under the `run` lock (a count, not a time) |
|
||||
|
||||
### 6 October 2026, 07:12Z to 07:17Z, the host's chain mode with --save-shards records and --prev, on the Mac's CPU
|
||||
|
||||
`tools/lock/with-lock.sh run`, `SP1_PROVER=cpu igneum-prove-host --mode chain --chain proving/fixtures/chain/block-81046.json,block-81047.json --save-shards --out chain-a.json`, then `--chain block-81048.json --save-shards --prev segment-81047-aggregated.bin --out chain-b.json` (the app branch at ce8f34a, Apple M5 Max, CPU prover). The flags the app's segment path needs, before PC 2 (approximate figures: a CPU run, one sample each):
|
||||
|
||||
| Step | Value |
|
||||
|---|---|
|
||||
| Shard proof, CPU, empty block | 34.7 s and 36.3 s |
|
||||
| Aggregation, CPU, 1 then 2 deferred proofs | 39.1 s, 50.8 s |
|
||||
| Chain of 2, end to end | 160.9 s |
|
||||
| Per-shard records written | 2 (number, block_hash, shard, statement, proof_sha256, proof_bytes 1,272,897, proof_file, prove_seconds) |
|
||||
| `--prev` run: base_chain_len, final chain_len | 2, 3 (the chain continued; a wrong previous proof is refused by number and parent hash) |
|
||||
|
||||
### 6 October 2026, 07:52Z to 08:24Z, the segment-aligned prover beside the miner on PC 2's RTX 5090 (job `segments-pc2-pv1c`)
|
||||
|
||||
`tools/proving-v1/pc2-segments.ps1` (app branch 330207d; the host from the package `igneum-prove-wsl2-segal`, built on PC 2 in 7 s warm to `/opt/igneum-segal`, pinned guests unchanged); the app's own prover OFF for the run through `/api/prove`, ON again at the end; the app's miner running (8 identities, batch-log2 22); `SP1_PROVER=cuda`, the stock 6.8.1 GPU server; a 1-s nvidia-smi sampler under every chain. Payouts read on node 1 (read-only, `igneum_getProofRecords` per block at 08:30Z). The miner's rate from the app's uploaded log (`status: ... MH/s` every 30 s, run win-1ccfe586-20261005-235130).
|
||||
|
||||
| Figure | Value | Note |
|
||||
|---|---|---|
|
||||
| Segments claimed in 30 min | 9 (114470, 114654, 114862, 115022, 115198, 115366, 115542, 115710, 115870) | one every 210 s; 32.1 min of loop |
|
||||
| Candidates per pass | 32 to 38 whole segments inside the margin | margin 580 to 589 DAA at claim |
|
||||
| Export (the chain to the segment's last block) | 99.6 to 100.6 MB in 1.4 to 1.6 s | once per segment |
|
||||
| Cut (8 fixtures, the exporter) | 45.1 to 46.0 s | the exporter replays from genesis per block; the next lever |
|
||||
| Chain run wall (8 shards, 8 aggregations, one key setup) | 159.7 to 160.6 s | host `--mode chain --save-shards` |
|
||||
| Shard proofs, 8 per segment | 63.0 to 63.5 s (7.9 s a shard) | empty blocks |
|
||||
| Aggregation, 8 chained | 80.4 to 81.1 s (10.1 s a block) | the fixed cost per block beside the miner |
|
||||
| End to end per segment (export, cut, chain, sign, submit) | 210.0 to 211.2 s | |
|
||||
| GPU memory peak during a chain | 16,484 to 17,573 MiB (miner resident) | the 24 GB tier's gate holds |
|
||||
| GPU utilisation during a chain | 94.9 to 95.3% | |
|
||||
| Shard records accepted | 72 of 72 | 8 per segment |
|
||||
| Shard records paid on chain | 72 of 72 | 0.905 to 2.719 IGN a shard (90% of the credit); carried 180 to 226 blocks after the block |
|
||||
| Segment records accepted | 0 of 9 | every one refused: "does not chain to segment N-8..N-1 (chain_len 8), which is pending until DAA ..." |
|
||||
| Miner alone (the app's prover off), 07:25 to 07:51Z | 117.86 MH/s mean (n=52) | min 46.37 is the switch-off dip at 07:22Z |
|
||||
| Miner beside the segment prover, 07:55 to 08:24Z | 104.90 MH/s mean (n=58, min 98.39, max 119.24) | 12.96 MH/s = 11.0% of the miner, at 95% GPU utilisation from the prover |
|
||||
| The 0.3.11 prover as shipped beside the miner (5 October row) | 5.0 MH/s = 4.0% | one shard per 46 s; this run proves 8 shards per 210 s, 2.8x the shards |
|
||||
| Node 1's v1 window at 08:24Z | pending 59, proven 0, unproven 16, paid segments 0 | unchanged by the run: the chain rule |
|
||||
|
||||
What the refusal is (the fork, `igneum/exec/src/proving.rs` `check_segment_record`): a fresh record (chain_len = N) is valid only when the previous segment is UNPROVEN at the carrier, and the record's own deadline is the previous segment's deadline plus one segment length in DAA, so a fresh record is valid for 8 DAA (about 8 s) per segment and must be carried inside them. With one prover every previous segment is pending at proof time. Fixed on the fork branch behind `proving_v1_fresh_rule_daa` (0f0dda95): from the switch a fresh record is valid whenever the previous segment is not proven; the app holds a refused record and offers it again every pass until the deadline (272b025).
|
||||
|
||||
Run b (`segments-pc2-pv1b`, 07:20Z to 07:51Z) claimed nothing in 88 passes: the driver's segment keys were doubles against int64 hashtable keys (fixed in 330207d); its 30 minutes are the miner-alone baseline above.
|
||||
|
||||
### 6 October 2026, 08:26Z to 08:35Z, the fast-time harness on the fresh-record rule (Mac, `tools/lock/with-lock.sh run`)
|
||||
|
||||
`IGNEUM_PV1_BIN=vendor/igneum-node/target-pv1/release node tools/proving-v1/net.mjs --segment 8 --unproven 10 [--fresh-rule 0]` (fork 0f0dda95, 3 nodes at 60x, ports 29950+):
|
||||
|
||||
| Case | Checks | Time |
|
||||
|---|---|---|
|
||||
| The rule as shipped (no switch): fresh refused while the previous segment is pending (known-failed), accepted after it is unproven | 22 passed | 166.2 s |
|
||||
| `--fresh-rule 0`: fresh accepted while the previous segment is pending, `freshAdmissible` true, still refused after a proven one, the second offer a duplicate ("segment already paid") | 23 passed | 139.9 s |
|
||||
|
||||
|
|
|
|||
|
|
@ -7,6 +7,7 @@ shard run reported as exit 0, 7a7e873).
|
|||
| Date | Symptom | Cause | Fix | Proven by |
|
||||
|---|---|---|---|---|
|
||||
| 4 Oct 2026 | Every `ci` run on master red since 67bf226 (eleven pushes), unnoticed | `sim/difficulty/records/testnet-v2-2026-10-04.schedule.log` carried a home path; `.log` was outside the identity scrub's extension list in `tools/ci/identity-check.sh` (and in the mirror's `tools/sync.sh`) | 2996cca: `.log` scrubbed like the other text files; the record rewritten with `~`; the same list in igneum-public `tools/sync.sh` (local commit e18256d, not pushed) | `bash tools/ci/identity-check.sh` 0 hits locally; run 37226816xxx on master green |
|
||||
| 5 Oct 2026 | PC 1 (Windows 11 Pro 26200, default terminal Windows Terminal 1.24): "Windows Command Processor" windows whenever a remote job runs (the project lead) | measured, not guessed: `tools/windows/console-watch.ps1` (job run-20261005-182528) started every candidate child from the app's job runner, whose console is headless (`conhost.exe 0x4`, hwnd 0), with a user32 EnumWindows sampler every 30 ms: powershell, cmd, query, curl, nvidia-smi, wsl --status, a distro, interop cmd and powershell, `powershell -WindowStyle Hidden`, `Start-Process -WindowStyle Hidden`: 0 windows each; `Start-Process cmd` in a new console: a Terminal window and a cmd PseudoConsoleWindow (the known-failed case fires). The 25-minute background watcher (console-watch-bg.ps1, run-20261005-184330, 18:44 to 19:09 UTC, every 200 ms) across an app restart, a build job, two run jobs, two collect jobs and the sweep helper's elevated launch at 19:04:43: 0 console or Terminal windows, 69 conhost starts (every one `conhost.exe 0x4`, headless, under curl, wsl, wslhost, powershell), 1 cmd.exe (under wslhost, WSL interop, no window). The one road that creates a console of its own is the elevated launch (`Start-Process -Verb RunAs`, the AppInfo service: the power cap, the sweep helper, the clock sync, an elevated job); it carried `-WindowStyle Hidden` in four copies, and "Windows Command Processor" is also the name on the UAC prompt the engine raises for cmd.exe (the sweep helper prompted at 17:00, 17:30 and 18:12 UTC, the power cap at every start; the elevated watcher's own prompt, run-20261005-184610, timed out unanswered at 122 s) | `platform::elevated_ps_line` + `elevated_command`: one builder for every elevated launch, hidden by construction, exit 251 when the prompt is refused; the elevated job wrapper reports its own console (`elevated console: hwnd N visible False`) on every elevated job; `tools/ci/windows-spawn-check.mjs` fails CI on a Command::new without the quiet flag, a creation_flags other than CREATE_NO_WINDOW, a Start-Process without -WindowStyle Hidden/-NoNewWindow, or a host.cpp spawn without CREATE_NO_WINDOW / SW_HIDE | the watcher's known-failed case (2 windows) and known-finished case (0); the CI check's self-test (9 cases) and the tree (0 hits); the igneum-app test suite on PC 1 |
|
||||
| 4 Oct 2026 | `collect-pc1-board3` printed PowerShell parse errors (`.Name`, `.AdapterRAM`) | the publishing shell expanded `$_` inside double quotes to nothing before the command reached the jobs file; nothing to do with Format-List or Out-String (board2 and board4 printed their values) | publish-jobs.sh refuses a collect command that pipes into a script block without `$_` or `$PSItem` | the eaten form refused with the reason, the single-quoted form published to a test folder |
|
||||
| 4 Oct 2026 | the same job reported `done (exit 0)` over `command exit Some(1)` | `run_collect` in `app/igneum-app/src/jobrun.rs` builds `Done` from the upload count only; the command's exit code is logged and dropped | branch `bugfix-collect-exit`, 35ccdc8 rebased on c257444 (app engine; merge by the main session) | `cargo test --bin igneum-app`: all 28 tests pass on the rebased branch; the new one covers the board3 shape (`Some(1)` is failed exit 1), `Some(0)` done, the cap as timeout, failed uploads still failing |
|
||||
| 4 Oct 2026 | `publish-jobs.sh --deploy` said "not reachable, differs from the local one, or does not verify yet" after a deploy that had succeeded | one check the instant the CLI returned, while the edge still served the previous file; the deploy's own exit status was hidden by `\|\| true` | `verify_live`: up to `--tries` (12) checks 5 s apart, each failure names its condition; `publish-jobs.sh verify` re-checks on its own; a failed deploy stops before the check | finished: `verify --tries 2` against the live file (try 1 of 2); failed: a local server with an older file ("differs", both publish stamps named) and a closed port ("is not reachable") |
|
||||
|
|
|
|||
181
docs/plans/ember-tune.md
Normal file
181
docs/plans/ember-tune.md
Normal file
|
|
@ -0,0 +1,181 @@
|
|||
# Ember Tune: every card tuned for MH per watt, out of the box
|
||||
|
||||
5 October 2026, night. the project lead: "make sure we have ember tuning every single card for efficiency out of the box, the
|
||||
more data = the better the tune, make an awesome system." Branch `ember-tune`, worktree `../igneum-wt-ember-tune`,
|
||||
on top of the AMD telemetry commit (7adcd4c, branch `opencl-rdna4-telemetry`) and the Power control commit (3562f26,
|
||||
branch `job-console`), both cherry-picked. Lever 3 of docs/plans/miner-eff.md grows two knobs and a fleet memory;
|
||||
lever 2 (docs/design/miner-tuning.md) carries the priors in the same signed `tuning` section.
|
||||
|
||||
## 1. What a user sees
|
||||
|
||||
| Moment | The card row says | What happened |
|
||||
|---|---|---|
|
||||
| First 2 minutes of mining | `tuning: waits for 120 s of steady mining` | The worker warms up; nothing is touched. |
|
||||
| Tuning | `tuning: holding 2472 MHz · 100% · 41 s (step 7 of 9)` with a Stop button | One card at a time, on the live kernel, never restarting the worker. |
|
||||
| Tuned | **Tuned: 122.3 MH/s at 290 W (0.422 MH/W)**, then `2470 MHz at 100%, full tune, 1 h ago` | The point is pinned on the card; the result went to the fleet. |
|
||||
| Known model | the same line, `from the fleet prior, confirmed, 2 min ago` | The card started at its model's prior and confirmed it in two steps instead of nine. |
|
||||
| Apple silicon | **Tuned: 26.7 MH/s at 38 W (0.703 MH/W)** `(measured as it runs)`, and `measure only on Apple silicon: the system sets the clocks and the power; no control exposed` | Nothing can be set; the number is still reported so the row and the fleet know what the card does. |
|
||||
| NVIDIA, Power control off | the measured line and `measure only until Power control is on in Settings (Windows asks for administrator rights once)` | The app never raises the prompt by itself (5 October 2026). One switch, one prompt, and the full tune runs. |
|
||||
| Slider moved | `your setting stays pinned` | A manual point is never overridden; the tune still measures and reports. |
|
||||
| Stopped | `tuning stopped: a remote job took the GPU` and the card back where it was | Any fault reverts the step and the run. |
|
||||
| Fleet pause | Settings: `tuning paused fleet-wide by the signed manifest` | The kill switch. |
|
||||
|
||||
Settings: one switch, "Ember Tune: tune every card for MH per watt out of the box (once after install, then weekly,
|
||||
and after a driver or program change)". AMD needs no rights. NVIDIA needs the Power control switch (one administrator
|
||||
prompt) for both knobs; off, it measures only.
|
||||
|
||||
## 2. The knobs, per vendor
|
||||
|
||||
| Vendor | Power limit | Core clock cap | Memory clock | How | Rights |
|
||||
|---|---|---|---|---|---|
|
||||
| NVIDIA | `nvidia-smi -pl <W>`, percent of the default, inside `power.min_limit` and `power.max_limit` | `nvidia-smi -lgc 0,<MHz>`, percent of `clocks.max.gr`; `-rgc` = unlocked | never touched (`-lmc` is not used); read back as `clocks.mem` | directly when the engine is elevated, else the one-prompt helper (`<seq> pl <W>`, `<seq> lgc <MHz>`, `<seq> rgc` in `sweep/cmd.txt`) | administrator, so only with Power control on |
|
||||
| AMD | `igneum-gpu-telemetry --card N --set-plimit <offset>` (0 = default, -20 = 80%), inside the `tune` line's `plimit_range` (PC 1's 9070 XT: -30 to 10, so 70% is the floor) | `--set-gmax` only when the `tune` line's `gmax_range` is absolute MHz (floor 0 or above); on RDNA 4 the range is an offset from stock (-500 to 1000 on PC 1) and the clock knob stays closed until the stock clock is known; `--reset` for the default point | not settable through ADLX on RDNA 4; read back as `mclk_mhz`, and a step whose mean memory clock falls under 95% of the baseline's is marked and cannot win | the helper, one process per request, exit 0 and a `tune ... ok` line | none on Windows (ADLX manual tuning); root on Linux, so measure only there |
|
||||
| Apple | none | none | none | measure only | none |
|
||||
|
||||
Vendor limits are never exceeded and the floor is never undercut: the plan clamps every point (`Limits::clamp_clock`,
|
||||
`Limits::watts_for`), and a clock floor the vendor does not report is 60% of the maximum.
|
||||
|
||||
## 3. The plan and the choice
|
||||
|
||||
Full plan (a new model, or a prior that lost its confirm check): the power ladder 100, 90, 80, 70, 60, 50% at the
|
||||
unlocked clock (duplicate watts dropped where the card's floor clamps them), then the clock ladder 90, 80, 70, 60%
|
||||
of the maximum at the power point the power ladder chose. 60 s hold after 15 s settle per step; 9 steps on an
|
||||
RTX 5090 (five power, four clock), about 12 minutes.
|
||||
|
||||
Confirm plan (the model's prior has 5 or more reports): the prior's point, then one neighbour (the next clock step up
|
||||
when the prior caps the clock, else one power step down). If the neighbour beats the prior by over 1% MH/W, the full
|
||||
plan is queued; else the prior stands. Two steps, about 3 minutes.
|
||||
|
||||
Baseline plan (measure only): one step at the card's current point. The "before" number for the row and the fleet.
|
||||
|
||||
The choice (`ember::choose`): among the usable steps whose rate is within the tolerance (1%, settable from the
|
||||
manifest) of the fastest step, the best MH per watt; within 1% on efficiency the higher rate; within 1% on both the
|
||||
lower draw. A card never gives up more than the tolerance in blocks for the saving. A step is unusable when it is
|
||||
marked: `faulted` (a rejected or mismatched hash during the hold: the step is reverted and marked), `hot` (the GPU
|
||||
reached 85 C; the run aborts at 90), `memory_clock_dropped`, `unapplied` (the readback disagreed with the request),
|
||||
`no_readings` (under three draw samples or no STATUS line).
|
||||
|
||||
## 4. The data flow
|
||||
|
||||
```
|
||||
card mines 120 s ──> probe (limits, driver, how to set) ──> plan ──> steps ──> choice ──> point pinned
|
||||
│
|
||||
app log: TUNE start / TUNE card=.. step=.. / TUNE chosen / TUNE {json} (and stdout under --sweep)
|
||||
│
|
||||
log upload (every minute, the existing intake, site/api/log.mjs) ──> Neon miner_logs
|
||||
│
|
||||
relay/lib/ember.mjs aggregate: per (card model | driver major | program class)
|
||||
median clock cap (10 MHz), median power %, median MH/W, MH/s, W, spread (MAD %), samples, machines
|
||||
│ │
|
||||
console: /r/<token>/c/tuning, `node tools/console.mjs tuning` site: tools/tuning.mjs --priors --site
|
||||
│ -> site/miner-priors.json -> /miners#priors
|
||||
tools/tuning.mjs --priors --write tuning.json (priors + ember settings beside the kernel-variant cards)
|
||||
│
|
||||
packaging/ota/publish-manifest.sh --tuning tuning.json --deploy (signed; carried over when not given)
|
||||
│
|
||||
every app: <app data>/tuning.json ──> ember::settings_of (kill switch, min samples, tolerance, period)
|
||||
──> ember::prior_of(key) ──> a new card's confirm plan
|
||||
```
|
||||
|
||||
The record (`ember::record_json`): `ts`, `machine` (the first 8 hex of SHA-256 over the install id; the id itself
|
||||
is random per install and never sent), `app`, `os`, `card`, `vendor`, `driver`, `driver_major`, `class`, `key`,
|
||||
`plan`, `steps` (the full table: clock, power %, limit, watts, MH/s, MH/W, core and memory clock, hottest reading,
|
||||
faults, mark), `chosen`, `before` (the full plan's 100% step), `eff`, `mhs`, `watts`. The key: `<card model with
|
||||
underscores>|<driver major>|<program class>`, the class from the worker's race line (`l128w16` today; `v2` before a
|
||||
race has run).
|
||||
|
||||
## 5. Scheduling and safety
|
||||
|
||||
| Rule | Where |
|
||||
|---|---|
|
||||
| One card at a time; the card must have mined 120 s and have a STATUS line | `tick_sweep` |
|
||||
| Never under a remote job hold, a pause, inside 600 s of the hour boundary, or while the app quits | `tick_sweep`, `sweep_drive` |
|
||||
| Due once after install, every 7 days (manifest `ember.period_s`), and when the driver major or the program class changed since the last tune | `tick_sweep` (`CardPref.sweep_driver`, `sweep_class`) |
|
||||
| A pinned card (the slider) is measured, never changed | `sweep_finish` |
|
||||
| Kill switch: `tuning.ember.enabled = false` in the signed manifest stops every tune fleet-wide; the Settings line says so | `ember::settings_of`, `tick_sweep` |
|
||||
| Faults: a rejected or mismatched hash marks the step; the card leaving `mining`, a worker error, a job, a pause or 90 C aborts the run and restores the point from before | `Run::sample_fault`, `sweep_drive`, `sweep_abort` |
|
||||
| Memory clock held: never set; a step that drags it under 95% of the baseline's cannot win | `Row::from_samples` |
|
||||
| Vendor limits: every point clamped to the reported range; the clock floor 60% when none is reported | `Limits` |
|
||||
| A signed prior is only ever a starting point inside the card's OWN reported limits (`power.min_limit` to `power.max_limit`, the clock floor to `clocks.max.gr` or the ADLX `gmax_range`), never a memory clock, never a value the card did not report; the confirm step measures it and the full plan replaces it when a neighbour beats it, so a bad prior costs the fleet one confirm step per card, not a setting. The signing key (K1, docs/security/keys.md) therefore cannot push a card past its vendor ceiling or under its floor | `Plan::confirm` clamps through `Limits::clamp_clock` and `power_pct.clamp(50, 100)`; proven by `ember::tests::the_confirm_plan_checks_the_prior_and_its_neighbour` (a prior of 9,000 MHz at 30% becomes 3,090 MHz at 50%) and `limits_never_exceed_the_vendor_or_undercut_the_floor` |
|
||||
| No prompt the user did not ask for: the NVIDIA helper starts only with Power control on; the `--sweep` job never counts as permission | `sweep_probe_known`, `sweep_helper_start` |
|
||||
| The elevated helper restores the limit and resets the clocks by itself after 20 idle minutes | `sweep::helper_script_*` |
|
||||
| A playbook that starts a second engine beside the installed app (the PC measurement jobs) gives it NO pipe (its output goes to a file the script tails: a pipe's write end is inherited by the engine's miners, and the installed app's jobs runner then waits forever for EOF after an abort; C35, PC 1 22:31 UTC, a 24-minute hang and orphaned miners), ends the engine's whole process tree at the end and on the budget (`taskkill /T /F`), and lets the installed app's miners come back only after that | `relay/playbooks/ember-tune-pc1.ps1`, `sweep-5090.ps1`; CI `tools/ci/second-engine-check.sh` fails any playbook without both |
|
||||
| Every `quit:` line in the app log names its source (the window host's stdin, the host gone, `POST /api/quit`, the `--sweep` run's end) | `Cmd::Quit(&'static str)` (b671c8b) |
|
||||
| A second engine never runs the updater: `IGNEUM_APP_NO_OTA=1` (implied by `--sweep`) skips the OTA tick and refuses Check now, whatever the manifest's `min_supported_version` says (the installer it would launch quits the installed app: PC 1, 22:31 UTC) | `Engine.no_ota`; the playbooks set the variable; `tools/ci/second-engine-check.sh` demands it |
|
||||
|
||||
## 6. Tests
|
||||
|
||||
| Test | What it fixes |
|
||||
|---|---|
|
||||
| `ember::tests::the_full_plan_is_the_power_ladder_then_the_clock_ladder_at_the_chosen_power` | 5 + 4 steps on the 5090's limits, the clamps, the dynamic second half, the 1% and 5% choices |
|
||||
| `limits_never_exceed_the_vendor_or_undercut_the_floor` | clamps |
|
||||
| `the_choice_keeps_the_best_mh_per_watt_within_the_rate_tolerance` | the rule, the ties, marked rows never win |
|
||||
| `the_guards_mark_a_step_so_it_cannot_win` | faulted, hot, memory clock, unapplied, no readings, the line |
|
||||
| `a_fault_during_a_step_reverts_it_and_the_run_goes_on` | the state machine with a fake clock: the faulted 70% step is marked and never chosen |
|
||||
| `the_confirm_plan_checks_the_prior_and_its_neighbour` | the two steps, Keep against FullDue, a prior outside the range clamped |
|
||||
| `a_baseline_plan_measures_the_card_as_it_runs` | no control, still a number and the Tuned line |
|
||||
| `the_record_and_the_prior_round_trip_through_the_manifest_shape` | record fields (no address, no host), `priors` and `ember` beside `cards`, the sample floor, the kill switch |
|
||||
| `control_reasons_per_vendor` | who measures only and why |
|
||||
| `sweep::tests::helper_scripts_carry_the_protocol` | the helper's `pl`, `lgc`, `rgc` |
|
||||
| `relay/test/ember.test.mjs` | five samples converge (2,470 MHz at 100%), an outlier (0.908 MH/W at 1,854 MHz) moves nothing, baseline records make no prior, de-duplication, the manifest merge keeps lever 2's cards, the canonical round trip, AMD keys |
|
||||
| `app/igneum-app/ui/tune-line.test.mjs` | the row line per state |
|
||||
|
||||
Run: `cargo test -p igneum-app ember sweep` (on a PC through the build job, or on the Mac under the build lock),
|
||||
`node --test relay/test/ember.test.mjs app/igneum-app/ui/tune-line.test.mjs`.
|
||||
|
||||
## 7. The tier consequences
|
||||
|
||||
| Tier | What Ember Tune does for it | What it costs |
|
||||
|---|---|---|
|
||||
| A laptop GPU (NVIDIA, 60 to 115 W) | the power ladder usually finds the vendor floor binding; the clock ladder is where a memory-bound program saves watts; the thermal mark keeps a hot chassis from winning a step it cannot hold | about 12 minutes once, then 3 minutes a week; under 1% of the hour during the tune (the worker never stops) |
|
||||
| One 8 GB card | the same two knobs; the 8 GB card is identities-limited (2 by default), the tune does not change that | the same |
|
||||
| One 12 or 16 GB card | the same | the same |
|
||||
| One 24 or 32 GB card (the 5090) | the draw sits far under the cap (290 W under 460 W on PC 1), so the power ladder is flat and the clock ladder is the lever; expected saving from the 4 October stability line: tens of watts at under 1% rate, to be measured | the same |
|
||||
| A rig (several cards) | one card at a time, so a six-card rig takes about 70 minutes to tune once; every card of one model after the first starts at the prior (3 minutes); the tune never touches a card a remote job holds | linear in cards once, then the confirm plan |
|
||||
| A pool user | the same per card; a pool submits the same hashes, so the 1% rate tolerance is the same 1% of shares | the same |
|
||||
| AMD on Linux | measure only (sysfs needs root); the row says so | 60 s a week |
|
||||
| Apple silicon | measure only; the row says so | 60 s a week |
|
||||
|
||||
Privacy line: what is uploaded is the record in section 4 and nothing else: a hash of the random install id, the
|
||||
card model, the driver version, the OS, the program class, the step table and the chosen point. No address, no
|
||||
hostname, no raw machine id, no user name. The public priors table carries only the aggregate per model.
|
||||
|
||||
## 8. Measurements
|
||||
|
||||
### PC 1, 5 October 2026 (night)
|
||||
|
||||
Tonight's constraints, read from PC 1's own uploads: the installed app runs as `DESKTOP-KMCV30N\Admin` with
|
||||
`elevated=False` (the account line at 19:02:33 UTC), the two in-app sweep attempts at 20:09 UTC aborted on the
|
||||
cancelled administrator prompt (`SWEEP aborted ... the_elevated_helper_did_not_run_(the_administrator_prompt_was_cancelled)`),
|
||||
so no stored sweep result exists from today, and the RX 9070 XT left the PCI bus at about 20:40 UTC (eGPU link,
|
||||
not restarted tonight). NVIDIA's `-pl` and `-lgc` need administrator rights, the project lead is asleep, and the app never raises
|
||||
the prompt by itself, so tonight's run on PC 1 is the baseline plan on the 5090 through the whole pipeline (probe,
|
||||
measure, TUNE record, upload, aggregation, prior shape in a test manifest). The two-knob tune on the 5090 and the
|
||||
9070 XT run are owed: the 5090 the moment Power control is switched on (one prompt, then the tune runs by itself
|
||||
within 2 minutes of steady mining), the 9070 XT when the card is back on the bus.
|
||||
|
||||
Run 1 (ember-tune-pc1-1, 22:30 UTC): aborted 46 s in by the installed app quitting, named the next morning: the
|
||||
second engine's own updater (0.3.9 under min_supported_version = urgent) ran the per-user installer, whose
|
||||
PrepareToInstall quit the installed app through its api/quit (C35 in the bench log); before any step; nothing set; the "before" snapshots are in the bench log (5090: 450 W of 575, 2,850 MHz core, 3,090 MHz
|
||||
maximum, 14,001 MHz memory; 9070 XT present on bus 98 with OFFSET ranges `gmax_range -500 1000`, `plimit_range -30
|
||||
10`). The offset finding changed the AMD mapping (054e041): an offset clock range closes the clock knob and the power
|
||||
ladder runs on a percent scale bounded by `plimit_range`. The re-run follows the 0.3.11 rollout.
|
||||
|
||||
## 8a. Next-cut notes (for the 0.3.12 shipper)
|
||||
|
||||
| Commit | What | Where |
|
||||
|---|---|---|
|
||||
| b671c8b | every `quit:` names its source; Power control alone decides; no cap at start under `--sweep` | main.rs, server.rs, engine.rs (separable) |
|
||||
| e600e63 | a second engine never runs the updater (`IGNEUM_APP_NO_OTA`, implied by `--sweep`) | engine.rs (6 lines, separable) |
|
||||
| 1e9550e (this commit, amended) | the elevated job path's output file is followed while the script runs, so the 5-minute progress reports carry its lines (a 35-minute run that never mined showed only "script running" on 6 October 2026); the tune playbook's watchdog fails a run that mines nothing within 120 s of its first status line, with the engine's last log line in the RESULT | jobrun.rs `follow_file`, relay/playbooks/ember-tune-pc1.ps1 |
|
||||
|
||||
## 9. Open
|
||||
|
||||
- The NVIDIA clock readback: `nvidia-smi -lgc` is confirmed only through the core clock during the hold (a mean over
|
||||
the cap by 5% marks the step `unapplied`); the first run with Power control on tells whether the driver honours
|
||||
the lock on the 5090 under this kernel.
|
||||
- ADLX on RDNA 4 exposes no memory-clock setter; the memory-clock mark is the guard. The telemetry agent's 9070 XT
|
||||
sweep tells whether a core cap drags the memory clock on that card.
|
||||
- The confirm plan's neighbour is one step; a second neighbour (the other knob) would cost 75 s more and catch a
|
||||
prior that is wrong on both knobs.
|
||||
- Intel: no knob yet; the row says measure only.
|
||||
|
|
@ -114,6 +114,8 @@ Reading. Nobody pays an aggregator as a separate role: Aztec's 30% goes to whoev
|
|||
|
||||
The resume path (5 October 2026, the 0.3.11 app): `POST /api/resume` on 0.3.9 re-armed only FAULTED cards (`stop_miners("paused")` clears every slot's `restart_at`), so a healthy paused card stayed "off" at 0 MH/s until the app was relaunched: PC 2 at 21:25:11Z (the aggregation-cost job's pause and resume; `[ok] mining resumed` then `0.00 MH/s, waiting` for 20 minutes), the Mac that afternoon. Now every slot without a live worker is re-armed and its pack exported again before the start, and 90 s later `resume_check` logs `resume: <card> is not mining 90 s after resume (state ..., pid ...)` for every enabled card without a hash rate (`engine.rs`, three unit tests: the state machine, the 21:25:11Z case against the old rule, the check).
|
||||
|
||||
The prover-floor agent's first sweep (job `floor-sweep-1`, 22:34 to 22:38Z, PC 2's 5090, the miners stopped, this plan's per-point recipe, its patched `sp1-gpu-server` 5568108b built for sm_86, sm_89 and sm_120, every proof VERIFIED by the unpatched pv1 host): the control at upstream's sizes reproduces the curve above (empty shard 13,892 MiB and 2.2 s; the v1 shard 20,516 MiB and 4.2 s); with the core element threshold at 2^26 the v1 shard proves as four core shards in 5.3 s at **12,708 MiB** and the empty shard at 12,772 MiB; 2^25 gives 12,836 MiB at 8.5 s; 2^27 gives 15,396 MiB. The 12.7 GB left is the server's Setup (five recursion keys pre-built at a fixed 2^27 capacity plus the shrink and core keys: 9.7 GB before the first shard), which its patch v2 sizes to the need. Decided for the 12 GB profile: the split that lands under 11 GB wins (5.3 s a shard is inside the loop's own 25 to 30 s of carriage and 100x inside T); 2^27 is the second profile only if v2 leaves it under 11 GB with the miner's 1.8 GB beside it. The 12 GB row stays OPEN until the final pair (alone and beside the miner) lands and the on-order 3060 runs it.
|
||||
|
||||
### A self-built CUDA server (the 12 GB path), before 0.3.12 (consequences C26)
|
||||
|
||||
If the prover-floor agent's rebuilt `sp1-gpu-server` (the Setup sizes cut, built on PC 2 under WSL2) proves a shard under 11 GB, it becomes a shipped artefact and needs its own row of rules before 0.3.12: it is built from a pinned SP1 source tag with `CUDA_ARCHS` covering sm_86, sm_89 and sm_120 (the 12 and 16 GB tiers are Ampere and Ada, not only the 5090's Blackwell; one card family per measured row), by the packaging path that builds the Windows payload (PC 1's build job for the Linux binary, the Mac signs the manifest as it does the DMG), lands in the DMG and the WSL2 package beside the host as `wsl2/bin/sp1-gpu-server` with its sha256 in `payload-inputs.json`, is named in `evidence.md` beside the prover rows ("prover built from SP1 <tag> at <sha>"), is rebuilt and re-measured at every SP1 upgrade, and ships only after `--mode verify-segment` and `--mode verify` on proofs it made show the pinned verifying keys unchanged (the server changes allocation, not the circuit; the ids `0x2b1a81cb...` and `0x474678f3...` must still verify them). The 12 GB claim itself waits for the on-order RTX 3060 to run that server on the same fixtures and recipe as the curve; until then the public line stays at 24 GB.
|
||||
|
|
@ -130,8 +132,8 @@ The GPU server of SP1 6.8.1 sets the memory, not the shard: a floor of 13.9 GB f
|
|||
|---|---|---|---|
|
||||
| 32 GB (RTX 5090) | the prototype shard, 28.3 GB, 10.8 s; the v1 shard 20.4 GB, 4.3 s | the prototype shard 30.1 GB, 33 s; the v1 shard 22.2 GB, 13.2 s | on, mine and prove, today |
|
||||
| 24 GB (RTX 4090, 3090) | the v1 shard 20.4 GB; the prototype shard does NOT fit (28.3 GB) | the v1 shard 22.2 GB measured on the 5090's allocation (2.3 GB spare on a 24 GB card; approximate for the card itself) | on, mine and prove, with the line "until the devnet's fee switch its shards are the prototype size, which needs 32 GB, so this card proves from the switch on" |
|
||||
| 16 GB (RTX 5080, 4080) | an empty shard only (13.9 GB) | nothing (15.7 GB for an empty shard, no room for the display) | off, with the line |
|
||||
| 12 GB (RTX 3060, 4070) | nothing: the floor is 13.9 GB, and the shipped server refuses the card outright | nothing | off; the project lead's "make sure we can prove on 12 GB cards" is OPEN and in work: the prover-floor agent (branch prover-floor, 5 October night) read SP1 v6.8.1's GPU server source (`sp1-gpu/crates/prover_components/src/builder.rs` lines 35 to 39): it reads the card's memory, adds 4 and panics under 24 ("Unsupported GPU memory ... must be at least 24GB"), and builds its core (ELEMENT_THRESHOLD 2^28 + 2^27 elements + 2^21), recursion (2^27), shrink (2^25) and wrap (85 M element) provers at Setup whatever the mode, which is the 13.9 GB floor; no knob reaches them, so the fix is a server rebuilt from source on PC 2 (WSL2, nvcc 12.8, CUDA_ARCHS=120) with those sizes cut, measured on the same fixtures and recipe as the curve above (D2 carries the curve) |
|
||||
| 16 GB (RTX 5080, 4080) | the shipped server: an empty shard only (13.9 GB); the patched server v3 b37defef at threshold 2^27: the v1 shard 12,915 MiB and 4.3 s, the prototype shard 13,459 MiB and 16.8 s (measured by the prover-floor agent on the 5090's allocation, job `floor-sweep-3`, 00:13 to 00:17Z 6 October; 2^27 + 2^26 gives 16,115 MiB, over the card) | the shipped server: nothing (15.7 GB for an empty shard); the patched server at 2^27 beside the miner (the 5090 mining at 95%, 338 W, same card; `floor-sweep-4`, 00:35 to 00:38Z): the v1 shard 14,786 MiB total with the miner's 3,833 MiB resident inside it, the server's own 10,953 MiB, 17.4 s a shard; on a 16 GB card that is 10.95 GB server + 1.7 GB miner = 12.7 GB plus the display, under the 15.0 GB line | off on the shipped server, with the line; on (mines and proves, 17 s a v1 shard, 4.3x the alone time) once the patched server ships (the packaging row below) |
|
||||
| 12 GB (RTX 3060, 4070) | the shipped server: nothing (the floor is 13.9 GB, and the server refuses the card outright); the patched server v3 b37defef at threshold 2^26 (`SP1_GPU_ELEMENT_THRESHOLD=67108864`, the 12 GB profile): **the v1 shard 10,291 MiB and 5.7 s, an empty shard 9,971 MiB and 3.3 s**, the card's 2,089 MiB idle inside the peak and the server's own working set about 8.2 GB (6,535 MiB after Setup), so a 12 GB card proves alone with about 3 GB over it (measured by the prover-floor agent on the 5090's allocation, `floor-sweep-3`; the on-order RTX 3060 run is pending) | measured beside the miner (`floor-sweep-4`): at 2^26 the v1 shard 12,066 MiB total with the miner's 3,833 MiB inside, the server's own 8,233 MiB, 24.4 s; the empty shard 11,586 MiB, 13.0 s; 2^25 gains nothing (12,066 MiB, 43.5 s). On a 12 GB card that is 8.2 GB server + 1.7 GB miner = 9.9 GB before the display, over the 9.0 GB line the project lead set, so mine-and-prove on 12 GB is NOT claimed | off on the shipped server; "proves alone" on the patched one once it ships (the packaging row below); mine-and-prove stays off on 12 GB (9.9 GB plus the display, over the 9.0 GB line). The public gate stays "12 GB proves; 16 GB mines and proves", both on the patched server, both pending a run on the card itself. the project lead's "make sure we can prove on 12 GB cards" is answered on the 5090's allocation and OPEN on the card itself: the prover-floor agent (branch prover-floor, 5 October night) read SP1 v6.8.1's GPU server source (`sp1-gpu/crates/prover_components/src/builder.rs` lines 35 to 39): it reads the card's memory, adds 4 and panics under 24 ("Unsupported GPU memory ... must be at least 24GB"), and builds its core (ELEMENT_THRESHOLD 2^28 + 2^27 elements + 2^21), recursion (2^27), shrink (2^25) and wrap (85 M element) provers at Setup whatever the mode, which is the 13.9 GB floor; no knob reaches them, so the fix is a server rebuilt from source on PC 2 (WSL2, nvcc 12.8, CUDA_ARCHS=120) with those sizes cut, measured on the same fixtures and recipe as the curve above (D2 carries the curve) |
|
||||
| under 12 GB | nothing | nothing | off, mine only |
|
||||
| AMD-only and Apple machines | nothing on the GPU: no zkVM proves on an AMD GPU today (`docs/analysis/amd-proving.md`, branch amd-prove); the CPU prover is about 5 minutes a shard at a 30 GB RSS whatever the shard size (PC 1, bench-log "the SP1 CPU prover on PC 1") | | off, "mines and does not prove"; the only non-NVIDIA path with a shipped backend is RISC Zero's Metal prover behind the `ProofSystem` seam (a second guest and pinned id, a verifier for both formats, no shared aggregation): an open item, not 0.3.11 |
|
||||
|
||||
|
|
@ -141,3 +143,81 @@ The aggregation-cost agent's first rows (branch agg-cost, 5 October 2026 night,
|
|||
|
||||
The re-plans of block 344 at 2.25 M and 4.5 M pgas peak at 28.3 to 28.4 GB alone (the server's buffers step up between 4.7 M and 20 M cycles and are flat to 60 M), so no shard size between the v1 budget and the prototype one changes a tier; with the miner the adopted shard proves 3.1x slower (13.2 s against 4.2 s) and the chained aggregation 9.7 s against 2.5 s: a mining 24 GB card delivers one adopted-size shard plus one aggregation in about 23 s, inside T by 25x.
|
||||
|
||||
## Segment-aligned proving (6 October 2026, the project lead: "find a way to solve this")
|
||||
|
||||
**The fault was the work order, not the rules.** The shipped loop took the newest open shard each pass (`prover::choose`), so one prover scattered one block in about 45 across the segment grid and no segment ever had all its blocks proven: pending 56, proven 0, unproven 20 at 06:28Z. No consensus parameter moves.
|
||||
|
||||
**The change (app branch, commit ce8f34a, app and host):**
|
||||
|
||||
| Part | What it does | Where |
|
||||
|---|---|---|
|
||||
| Whole-segment claiming | the work list (lookback 600, the record window) grouped into whole untouched segments: every block present, every shard open (past its 10-DAA exclusive window), unpaid, not in our pool | `app/igneum-app/src/segments.rs` `whole_segments` |
|
||||
| The choice | candidates inside their deadline by a margin (240 DAA, or 1.5x the last segment's wall time), ranked by FNV-1a of (first block, this prover's key hash): deterministic per prover, different between provers, so several provers spread over the candidates with no coordinator; an attempted segment is not retried | `segments::candidates`, `rank`, `need_daa` |
|
||||
| The statement check | for the best three: executed and pending; fresh only when the previous segment is not paid and no verified record of it waits in the pool (the chain rule would refuse a fresh record once that one pays); chained (`--prev`) when the previous is paid and its proof is in this node's pool | `prover.rs` `pick_segment` |
|
||||
| The work | one export to the segment's last block, one fixture per block, one host run `--mode chain --save-shards [--prev]` that proves every shard and aggregates the segment in one process (one key setup), then every shard record signed and submitted (the shard payouts, 90% of the credit) and the segment record signed and submitted (the aggregator share, 10%) | `prover.rs` `prove_segment` |
|
||||
| The fallback | when no whole segment qualifies, the per-block path as shipped | `prover::choose` |
|
||||
| The host | `--mode chain` takes `--prev <aggregated.bin>` (chain_len continues; a previous proof that is not the parent block's is refused by number and parent hash) and with `--save-shards` writes per-shard records (statement, proof sha256, file, time) into the results | `proving/igneum-prove/host/src/main.rs` `run_chain` |
|
||||
| The tile | "Segments: N proven whole, M paid (x IGN to the aggregator), the last in T s" and the segment path's last line; the state carries `segments_submitted`, `segments_paid`, `segment_paid_wei`, `segment_last_s` | `ui/app.js`, `state.rs` |
|
||||
|
||||
Tests: the grid, the grouping (missing block, paid shard, our shard in the pool, exclusive shard), the margin and the attempted set, the per-key order (deterministic, different between two keys), the margin from the last time: 6 unit tests in `segments.rs`; 120 app tests, 8 core and 9 host tests pass. The host flags were run on the Mac's CPU first (bench-log, 07:12Z to 07:17Z): per-shard records written for a chain of 2, then a chain of 1 continued from it (`base_chain_len` 2, final `chain_len` 3).
|
||||
|
||||
**Item 2, "own pool only", answered from the source:** a relayed proof record carries its proof bytes (`protocol/flows/src/v10/proving.rs`: `IgneumProofRecordMessage { record, proof }`, 8 MB bound, handed to the pool with `local = false`), and so does a segment record (message 75). So "this node's pool" is every record relayed to it, and any aggregator can already fold any 8 proven blocks it has received; no node or consensus change is needed for that. What does limit carriage: a block template carries only entries the node's own verifier marked verified (`template_segment_section`, `template_section`), so a node with the verifier `Off` (node 1) never carries a record and a node with `Trust` carries unverified ones; PC 2's node runs the host and verifies. With one prover the carrier is PC 2's own next block.
|
||||
|
||||
**What one 5090 completes (arithmetic from the 5 October rows, the measurement below replaces it):** a chain of 8 empty blocks took 135.6 s cold beside the miner; one export and one key setup per segment instead of eight; so about one segment every 150 to 200 s, 9 to 12 segments an hour out of 450 (2 to 3%), against none. The aggregator share of a segment is 8 x 0.088 IGN = 0.70 IGN plus the 8 shards' 90% share; the forfeited share of the other 97% stays in the escrow until the fleet grows (47 mining cards or 6 dedicated provers for 100% at 1 block/s).
|
||||
|
||||
**PC 2 measurement (job `segments-pc2-pv1c`, 07:52Z to 08:24Z, `tools/proving-v1/pc2-segments.ps1`; bench-log "the segment-aligned prover beside the miner"):**
|
||||
|
||||
| Figure | Value |
|
||||
|---|---|
|
||||
| Whole segments proven in 30 min, one 5090 beside its miner | 9, one every 210 s (export 1.5 s, cut 45 s, chain 160 s: 8 shards 63 s, 8 aggregations 80 s) |
|
||||
| Shard records accepted and paid | 72 of 72, 0.91 to 2.72 IGN a shard (90% of the credit), carried 180 to 226 blocks after the block |
|
||||
| Segment records accepted | 0 of 9: refused by the chain rule as shipped (below) |
|
||||
| Miner's cost | 117.86 to 104.90 MH/s, 13.0 MH/s = 11.0% (the shipped prover: 5.0 MH/s, 4.0%, for 2.8x fewer shards) |
|
||||
| GPU memory peak | 16.5 to 17.6 GB with the miner resident; 24 GB tier unchanged |
|
||||
|
||||
**The second fault, found by the measurement: the chain rule as shipped.** `check_segment_record` accepts a fresh record (chain_len = N) only when the previous segment is UNPROVEN at the carrier. A segment's own deadline is its last block's DAA plus 600, the previous segment's deadline is 8 DAA earlier, so a fresh record is valid for 8 DAA (8 s on devnet) and must be proven before and carried inside them. The refusal on the live node, 9 times: "segment 114470..114477 does not chain to segment 114462..114469 (chain_len 8), which is pending until DAA 169681". The fast-time harness passed on 5 October because its chain continued from proven segments (case 2) and its fresh case ran exactly in that window (case 3, 4 DAA at N = 4). No prover-side move escapes it: the record must be proven, submitted and carried inside the window, which the 210-s proof cannot meet.
|
||||
|
||||
**The fix (fork branch 0f0dda95, behind a switch, never by default):** `proving_v1_fresh_rule_daa`; from it a fresh record is valid whenever the previous segment is not proven (pending or unproven) at the carrier; after a proven one a record must still chain. Two records that do not chain each attest their own blocks against the native statement, so nothing is lost but the longer proof chain, which restarts. In the consensus digest only once set (the v1 pattern), so a 0.3.12 node on the live devnet keeps the 0.3.11 digest until the override file sets it; `igneum_getProvingStatus.v1.freshRuleDaa/freshRuleActive` and `igneum_getSegmentStatement.freshAdmissible` report it. Tests: the digest moves once set; fresh after a pending segment passes from the switch, is refused before it, never after a proven one; the harness `--fresh-rule 0` inverts case 3. This is a rule change in the execution layer, not a parameter tuning, and the code proved it unavoidable: the project lead's "no consensus parameter change unless the code proves it is unavoidable" is met by the 8-DAA window above and the nine refusals. The 0.3.12 coordinator sets the height (the same tip + 14,400 rule) in the override object with the release.
|
||||
|
||||
**Until the switch:** the app (272b025) holds a refused segment record and offers it again every pass until the segment's deadline, so on the shipped rule it lands only if a carrier falls inside the 8-DAA window, and from the switch it lands on the first retry; the shard records (90% of the credit) land either way, 8 per segment.
|
||||
|
||||
**Per tier, with this change and the switch:**
|
||||
|
||||
| Card | What it does | Per 30 min, empty blocks (measured on the 5090, approximate elsewhere) |
|
||||
|---|---|---|
|
||||
| 32 GB mining and proving (5090) | 9 whole segments, 72 shards paid, 9 aggregator shares once the switch is set | miner 11.0% down; 72 x 0.9 IGN = 65 IGN of shard payouts measured, plus 9 x 0.70 IGN aggregator share from the switch |
|
||||
| 24 GB mining and proving | the same path at the 16.5 to 17.6 GB peak measured; the fee-switch shard not yet measured on a 24 GB card | approximate: the 5090's numbers |
|
||||
| 16 GB | mines and proves on the patched server only (prover-floor rows); segment path untested there | pending the prover-floor agent's build |
|
||||
| 12 GB prove-only | proves alone on the patched server (10.3 GB); a dedicated prover takes 4 s a block alone (agg-cost rows), so about one segment every 40 s | approximate: 45 segments per 30 min, 6 such cards for 100% |
|
||||
| A rig (several cards) | one prover loop per machine today; the segment path claims one segment at a time on the aggregation card | the per-card loop is the next item |
|
||||
|
||||
## v1 live on devnet (6 October 2026, C47)
|
||||
|
||||
v1 active at 154,800 (crossed at DAA 154,814, 03:51:42Z); first segment record: none, because on a one-prover devnet no segment can be proven. The app's aggregator (`aggregate_once`, 0.3.11) needs a shard proof of every shard of every block of the segment in its node's pool, and PC 2 alone proves 13 shards per 10 minutes of about 600 blocks (2.2% coverage), so a run of 8 consecutive proven blocks never occurs: node 1 at 04:16Z reads segmentsInWindow pending 55, proven 0, unproven 20, paidSegments 0, and PC 2's app log (run win-1ccfe586-20261005-235130, 04:11Z to 04:16Z) reads every 42 s "aggregator: segment N..N+7: waiting for shard proofs N/0 ... N+7/0 in this node's pool", all 8 missing, each segment then past its 600-DAA deadline unproven. No fault in the node, the app or the record path; the fast-time harness passed because its shards ran at 90% coverage. Meanwhile the aggregator share (a tenth of every block's pool credit) stays in the escrow; shard payouts continue; miners and block watchers see nothing. What ends it: coverage at 8 consecutive blocks, 47 mining 5090-class cards with the shard loop as shipped in 0.3.11 (13 shards per 10 minutes a card), 18 mining cards through the chain mode, or 6 (approximate) proving-only cards, at 1 block/s on empty blocks (the fleet table above), or the segment length lowered on a small devnet (`proving_v1_segment_blocks`, a consensus param, so a digest change). No PC 2 job and no 0.3.12 item follow from this; the open item is the fleet, not the code.
|
||||
|
||||
## The empty `/api/state` reply (6 October 2026)
|
||||
|
||||
The aggregation-cost agent's jobs read the two bytes `{}` from `/api/state` on PC 2 at 22:22Z, 22:41Z and 00:18Z (0.3.10 and 0.3.11); the 21:01Z reply was full. Cause, from the app source and node 1's RPC: `ProvingState.paid_wei` is a `u128`, and serde_json's `to_value` refuses a u128 over u64::MAX (18,446,744,073,709,551,615 wei, 18.45 IGN); `state_json()` turned that refusal into `json!({})` with no log line. A paid shard is 1.23 IGN on average (node 1, `igneum_getProvingStatus`: 814.64 IGN over 663 shards at 00:3xZ), so the fifteenth paid shard after an app start empties the reply. PC 2's prover was blind to the root-owned socket from 20:00:56Z to 22:01Z (paid_wei stayed 0, hence the full reply at 21:01Z), proved from 22:02:13Z, and crossed 18.45 IGN inside its first 15 paid shards, before 22:22Z. Every app restart resets the counter, so the reply comes back for about 15 shards and goes again.
|
||||
|
||||
What it means: the dashboard on a proving machine shows nothing within about 12 minutes of its prover's first payout; every PC playbook that reads a card from `/api/state` fails the same way (the agent's job 5 reads settings.json instead). Mining, proving and payouts are untouched; it is the status page only.
|
||||
|
||||
Fix on the app branch: `paid_wei` serialises as a decimal string (the dashboard already reads it with `Number()`), `state_json` logs `[error] state_json: ...` once instead of answering `{}`, and the reply on any future serialisation error carries `error` and `version` rather than nothing; unit test `a_paid_total_over_u64_max_still_serialises_the_whole_state`. Not in 0.3.11 (that tree closed at 22c2363, master 630da6b, published); 6714a45 heads 0.3.12, the morning's first cut, app only, before PC 1's relaunch (coordinator, counter-asic-2-rollout.md 8a); until then the workaround is settings.json for the card keys.
|
||||
|
||||
## Aggregation cost (5 October, night)
|
||||
|
||||
the project lead, 5 October 2026: "fix everything else in the numbers tonight". Branch `agg-cost`; every number in `docs/bench-log.md`, "aggregation cost on the RTX 5090", with its job id. The proof statement and the pinned guests are unchanged: every existing fixture proof still verifies (`verify-segment` 0.027 s on the Mac, 0.036 to 0.041 s on PC 2).
|
||||
|
||||
| What | Before (5 October evening, `chain-pc2-pv1c`) | After (5 October night) |
|
||||
|---|---|---|
|
||||
| Chained aggregation, the card mining | 9.6 to 9.7 s a block | 9.6 to 9.8 s a block, the same (job `agg-cost-pc2-1`, phase A); the miner's presence is the whole cost: 2.1 to 2.2 s a block with the card to itself, 1.7 s unchained |
|
||||
| Shard proof (empty shard), the card mining | 7.3 to 7.7 s | 7.4 to 7.8 s; 1.9 to 2.2 s with the card to itself |
|
||||
| The miner's slowdown of the prover | 3 to 4x (against 4 October) | measured on the same fixtures 2 min apart: shards 3.6x, chained aggregation 4.5x, a block 4.2x |
|
||||
| Where the time goes | not profiled | the host's share 0.000 s (the prove call is everything); the GPU server prints no timings; the second deferred proof (the chain rule) costs 0.4 to 0.5 s alone and 1.7 to 1.9 s under the miner; the prover alone keeps the card busy 15.8% of the time, the miner 93.9% |
|
||||
| SP1 knobs (`SP1_WORKER_VERIFY_INTERMEDIATES=false`) | not tried | no gain: 7.8 s against 8.1 s over four aggregations, inside the spread; the shape knobs would change the recursion keys the pinned verifier accepts |
|
||||
| Batch fold (K blocks in one aggregator call) | not estimated | estimate from the measured step costs: 0.9 s a block alone and 3.8 s mining at K = 4, 0.7 and 2.8 s at K = 8 (0.25 s per further deferred proof alone, 1.8 s mining); a tree fold gains nothing. A new pinned guest and program id either way, so not tonight |
|
||||
| Two prover processes on one card | not tried | closed on SP1 6.8.1: both share one GPU server socket, run slower together (6.9 s a block against 4.1) and the second dies with the first (`early eof`) |
|
||||
| The miner's kernel length (`--batch-log2` of the CUDA worker, 2^B nonces a launch; job `agg-cost-pc2-6`, the 5090 alone with the job's own miner) | not tried | 2^22 (the default) and 2^20: 18.1 and 18.0 s a block, 10.0 to 10.4 s a chained aggregation, 104 MH/s; 2^18: 15.6 s, 8.8 s, 99 MH/s (minus 4%); 2^16: 11.1 s, 6.1 to 6.2 s, 84 MH/s (minus 19%), reproduced |
|
||||
| The GPU time-slice policy (`nvidia-smi compute-policy --set-timeslice`) | not tried | "Not Supported" on PC 2 (driver 13.3, Windows): closed |
|
||||
| The chosen combination | the defaults | the defaults stay: batch-log2 22 and SP1's default knobs. The one knob that moves the prover (2^16) costs a fifth of the hash rate all the time for a prover that is busy a few seconds a minute on the devnet; it is the project lead's trade, not a default (below) |
|
||||
|
||||
Reading. The per-block aggregation is 2.1 s and a block 4.1 s on a 5090 that only proves, 9.7 and 17.5 s on one that also mines; no knob, fold or stream on tonight's SP1 changes the first pair, and only the miner's kernel length changes the second, at 1 MH/s per 0.37 s of block time. So "under 3 s a block" and "under 1.5x" are met on a card that is not mining and are not reachable on one that is. What that means per tier: a 5090 that mines and proves delivers a proven empty block every 17.5 s (6 cards for 1 block/s), the same card proving only every 4.1 s (2 cards, plus the shard work of full blocks: the fleet table above), and a batch fold of the aggregator (a new pinned guest) would bring the proving-only card to about 2.7 s a block and the mining one to about 12 s. What is being done: the app and host defaults are left as measured; the plan's open decision for the project lead is whether a card that holds a shard assignment should drop to 2^16 for the proof's minute (1.6x faster proof, 19% of its hash rate for that minute) or whether proving-only cards carry the aggregation (the clean 2.1 s), and the batch fold goes on the next pin's list. The state class found on the way (`/api/state` answering `{}` once `paid_wei` passes u64::MAX, fixed on the app branch at 6714a45) is in the bench log with the rest.
|
||||
|
|
|
|||
193
docs/plans/release-0.3.12.md
Normal file
193
docs/plans/release-0.3.12.md
Normal file
|
|
@ -0,0 +1,193 @@
|
|||
# Igneum Miner 0.3.12: the fresh-record rule switch (proving v1) and the app cut, prepared to the publish gate, 6 October 2026
|
||||
|
||||
Release engineer, from 08:05 UTC, on the coordinator's instruction ("prepare 0.3.12, APP ONLY, up to the publish gate and STOP there; the project lead
|
||||
gives the go"), widened at 08:55Z on its clock: "no longer app only", the proving agent proved the segment record rule needs a consensus
|
||||
switch, so the node fork `proving-v1` 0f0dda95 (`proving_v1_fresh_rule_daa`: never by default, in the digest only once set; from it a fresh
|
||||
segment record is valid whenever the previous segment is not proven) and the app's segment-aligned prover (272b025, docs aea2f6a) ride in it.
|
||||
Worktree `/Users/joshm/Projects/igneum-wt-ship0312`, branch `release-0.3.12` from master ddfcdac; `vendor/` symlinked to the main checkout's (46
|
||||
entries); the fork worktree `vendor/igneum-node-0312`, branch `release-0.3.12-node` = 0f0dda95 cherry-picked onto 89dfcb95 (its parent ece42979
|
||||
is inside 89dfcb95, so the rebase is the one commit: `params.rs`, `exec/proving.rs`, `exec/rpc.rs`, `daemon.rs`) = **83089544**. The 0.3.11 recipe (`release-0.3.11.md`) throughout; every Mac build under the main checkout's lock;
|
||||
igneum-labs commits. Times are UTC.
|
||||
|
||||
## 1. What 0.3.12 carries
|
||||
|
||||
| Change | Where | State |
|
||||
|---|---|---|
|
||||
| `/api/state` never answers `{}` again: `paid_wei` (u128) is a decimal string, the error reply is logged; test | `proving-v1` app 6714a45 (the first item) | merged 9bcf4cd (the `docs/bench-log.md` conflict: both sides kept, the log is append-only) |
|
||||
| An update published over an hour before the engine started skips the hourly rollout slot; `manifest::unix_from_rfc3339` + tests | `update-catchup` 2207cd7 | merged b984c17 |
|
||||
| The GPU list ordered by performance (usable, discrete before integrated, rate in 5 MH/s buckets, memory); 3 UI tests | `card-order` ffb2bfa | merged c6608c1 |
|
||||
| HiveOS local mode carries the override (`OVERRIDE=` in the Flight Sheet's extra config, written to `data/override-params.json` by `h-run.sh`, C41), rigs mine only until a Linux prover ships, `IDENTITIES=auto` by VRAM, the per-card README table | `hive-words` 98271ff (packaging/hive only) | merged b0a6231 |
|
||||
| Ember Tune: two-knob plans + priors + the UI line; every quit names its source (b671c8b); a second engine never runs the updater (e600e63, C35); no pipe into a second engine (8ab9068); `jobrun.rs` elevated `follow_file` (1e9550e); the BOM fix + CI check (8273494); Power control switch (49bbe14 = 3562f26); `igneum-gpu-telemetry.exe` (ADLX) built by `build-windows.sh` and carried in the Windows inputs | `ember-tune` 9a6469f | NOT MERGED: conflicts in seven files against the 0.3.10/0.3.11 app (its base ca8d9f3 predates both): `ci.yml`, `config.rs`, `engine.rs` (the detect path, the power-cap plan, the test module), `ui/app.js` (four hunks against miner-ui-2's View), `ui/index.html` (the settings panel 0.3.10 removed), `proto-opencl/README.md`, `bench-log.md`. Its agent is rebasing it onto release-0.3.12 (section 2) |
|
||||
| The hidden-console builder for every elevated launch, `windows-spawn-check.mjs`, the PC 1 console-watch scripts | `job-console` 13755b9 (+ 3562f26 Power control) | NOT MERGED: conflicts in six files (`ci.yml`, `config.rs`, `engine.rs`, `jobrun.rs`, `app.js`, `index.html`), base a93199a; carried by the ember-tune rebase (it already holds 3562f26) |
|
||||
| The miner's gRPC resubscribe after a node restart (C43, the 0.3.11 finding) | no commit exists (the ledger entries b19fe5f, 0751dde, c8c831c only) | OWED, listed in section 10 |
|
||||
| The fresh-record rule switch: `proving_v1_fresh_rule_daa` (Option, never by default; a node with the field set prints it and carries it in the digest; a fresh segment record is valid from it whenever the previous segment is not proven) | fork `proving-v1` 0f0dda95 on ece42979 | cherry-picked onto 89dfcb95 as 83089544 (`release-0.3.12-node`) |
|
||||
| The segment-aligned prover: a segment record the chain rule refuses is held and offered again every pass until the segment closes; the fast-time harness on the fresh-record rule (both cases); the prover host and export; the WSL2 prover package script; `infra/fast-time/override-60x.json` (measured by its agent: 9 segments per 30 min on PC 2, 72 of 72 shards paid, 11% hash cost, 17.6 GB peak) | `proving-v1` app 272b025 + docs aea2f6a (on 6714a45) | merged 49e0e2c (clean) |
|
||||
| The packaged line (C34): the ten-field object of section 4 | 7dd3ff7 | `packaged-config.sh --test` passes |
|
||||
| The six version files | 81e4ecb (`--check`: 0.3.12 in all 6) | |
|
||||
|
||||
Left out on the coordinator's word: prover-floor's server (its packaging row is 0.3.13), explorer d7e797c, pool-v0, rig-install, ota-k2, the
|
||||
ledger forks.
|
||||
|
||||
Changelog line (draft, for the manifest notes at the go): "Igneum Miner 0.3.12: the fresh-record rule for proving v1 from DAA 192,000 (a fresh
|
||||
segment record is valid whenever the previous segment is not proven) and the segment-aligned prover; the GPU list in performance order; an
|
||||
old update no longer waits for the hour; Ember Tune (every card tuned for MH per watt, Power control off by default, the app never asks for
|
||||
administrator rights on its own); a second engine never installs over the app; /api/state always answers; HiveOS rigs carry the override.
|
||||
Node 83089544."
|
||||
|
||||
## 2. The branch
|
||||
|
||||
| Commit | What |
|
||||
|---|---|
|
||||
| 9bcf4cd, b984c17, c6608c1, b0a6231 | the four merges above, in the coordinator's order (proving-v1 first) |
|
||||
| 81e4ecb | `Igneum Miner 0.3.12: the six version files` |
|
||||
| ebea8b6 | the ember-tune rebase tip 7f6c4e6 (with job-console 13755b9 inside), merged as one branch (section 3) |
|
||||
| 11e8ca6, ab01f48, 01abcc2 | the plan |
|
||||
| 062c3f8 | `node-source.pin` 83089544 with the second inputs push (the Windows-build commit of 0.3.12) |
|
||||
| 37b6a7f | `tools/proving-v1/pc2-agg-cost.ps1`: `pkill -f sp1-gpu-server` (the CI root-socket check) |
|
||||
| 88df58e | master d3b64cb merged (docs only): the release tip, CI green |
|
||||
| 6532adf | `make-payload.sh`: on CI the AMD telemetry helper is taken from the unpacked inputs (the worker glob `igneum-worker-*.exe` missed `igneum-gpu-telemetry.exe`, so the first payload, run 37435975425, shipped without it: the inputs had it, the zip did not). The CI commit of 0.3.12 |
|
||||
|
||||
Checks on 81e4ecb before the rebase landed: the app `cargo test --release -p igneum-app` under the lock: ok 115 (lib) + 28 (ota-sign) + 8
|
||||
(prove-verify), 0 failed (08:19:22 to 08:19:28Z, warm target cloned from the 0.3.11 worktree); the UI tests `notices`, `update-card`, `view`:
|
||||
23 of 23.
|
||||
|
||||
## 3. Builds and artefacts
|
||||
|
||||
| What | Command | Result |
|
||||
|---|---|---|
|
||||
| The HiveOS package (first build, app-only cut) | the 0.3.11 node and workers with hive-words' scripts | 08:19:45Z: b0a20917... (24,501,272); superseded below once the node changed |
|
||||
| The merged tree | ember-tune 7f6c4e6 (release-0.3.12 b0a6231 merged INTO ember-tune as c5918c7, job-console 13755b9 cherry-picked on top; 0.3.11's six-section View and card order kept whole, Ember Tune's TuneLine block and the Power control switch added in 0.3.11's markup, `engine.rs` keeps the detect arm with the tune fields in `hotplug::apply_pref`, both test modules, `jobrun.rs` the hidden-console builder plus `follow_file`) merged as one branch | **ebea8b6**, 08:22Z (the CI commit is 062c3f8); the version files still 0.3.12 in all 6; packaged line, `node-source.pin` and `vendor/` untouched against master |
|
||||
| The app | `cargo test --release -p igneum-app` under the lock, then `cargo build --release` | 08:22:56 to 08:23:06Z: ok 133 + 28 + 8, 0 failed; `igneum-app 0.3.12` (2,273,664) |
|
||||
| The UI and relay tests | `node --test` notices, update-card, view, tune-line; `relay/test/*.test.mjs` | 26 of 26; 23 of 23 |
|
||||
| The CI checks on the Mac | identity, no-conflict-markers, copied-sources, signer-pipe, prover-socket, second-engine, bash-body (self-test + tree), kit-path (self-test + tree), windows-spawn (self-test + tree), pinned-guests, no-secrets, check-workflow-shell | all ok; `link-check` passes after `node site/build.mjs` (as ci.yml runs it: the committed `litepaper.html` points at `/bench#counter-asic-2-0-the-numbers`, an id the site build creates from `bench-log.md`) |
|
||||
| The Windows workers and the AMD telemetry helper | `proto-cuda/nvrtc/build-windows.sh` (mingw) under the lock, 08:23Z | the worker SOURCES are unchanged against master (`git diff master HEAD -- proto-cuda proto-opencl proto-metal igneum-pow`: only `build-windows.sh`, the new `gpu-telemetry.c` and its `.rc`), so the inputs carry the 0.3.11-verified workers that mined all night, igneum-worker-cuda.exe 2b3b8c92885442179f6bf2907c6f3eb453dc4a19908d90fd05981a09b7c2674c (1,536,512) and igneum-worker-opencl.exe edc4a75da3b93d814caa69fd635010780d63d5b622ec24c3741d433c584f91e3 (478,208), not this morning's rebuild of the same sources (12bfaa27..., e0fd7042...: mingw PE builds are not byte-reproducible); NEW igneum-gpu-telemetry.exe 8d679b52b19af3cbd6bf4fd6f77d337b2fb78af13d02627e7aa7e92507993459 (387,584; ADLX, SetupAPI, PDH; the Igneum resources, version 0.3.0 as the workers carry) |
|
||||
| The Windows inputs | `IGNEUM_WIN_RELEASE=<fork>/target-integration/x86_64-pc-windows-gnu/release IGNEUM_NODE_SRC=vendor/igneum-node-0312 packaging/windows/push-inputs.sh`, 08:24:29Z (a deploy of the downloads folder only; the manifest untouched) | node 89dfcb95 (igneumd.exe be8e83c0..., igneum-miner.exe 1ba1a249..., PC 2's 0.3.11 build), the two workers above, the telemetry helper, the mingw DLLs and nvrtc; signed, verified, live (HTTP 200); `node-source.pin` unchanged 89dfcb95 |
|
||||
| The DMG (first build, app-only cut) | the 0.3.11 node | 08:24:33Z: ff630e9d... (41,630,620); superseded below once the node changed |
|
||||
| The fork's Mac node, 83089544 | `CARGO_TARGET_DIR=vendor/igneum-node/target-0312 cargo build --release -j 4 -p kaspad -p igneum-miner --features kaspad/igneum-pow` from `vendor/igneum-node-0312`, under the lock; the target dir cloned by APFS from `target-0311`; `target-integration` in the fork worktree links to it | 08:38:13 to 08:41:31Z: igneumd **746a931fde9b840ca444a03cd757854e2e2ce7ebc4782ddd00ed161644d705f9** (41,386,160), igneum-miner 5381683e5717d91416c5a97456e0d050dd9645b1e27bce7d55808ce621cc1a26 (8,763,936); `igneumd/2.1.0-83089544` |
|
||||
| The seed's Linux node (glibc 2.36 target, zig) | `NODE_SRC=<abs fork> TARGET_DIR=vendor/igneum-node/target-0312-linux OUT_DIR=<scratch>/cross infra/cross/build-linux.sh` under the lock | 08:38:21 to 08:41:18Z (175 s): igneumd **4f142d5148f218f1286e24e7c4a167aa2f54262336f96e7cf281f520c714fc6f** (47,919,144), igneum-miner 38397ae66c265b63db8e5458b46e7feb942121a7dc5625919df0a8d35e7a1ba1 (9,861,072); version.txt names 83089544 |
|
||||
| The prover host and export (the pinned guests unchanged) | `cargo build --release -j 4 -p igneum-prove-export -p igneum-prove-host` in `proving/igneum-prove` under the lock (the target cloned from the 0.3.11 worktree) | 08:39:07Z: igneum-prove-host b90d58d0529ce29f0e7ca8ae780a6442f92edcf1c71752fc60fbc72bc5c11fd8 (58,626,560), igneum-prove-export b60056127d32bda363c0e305e73e9f699a5774c0988aee5bfd28a0fc61a56b9a (2,808,160); `--mode id`: shard program id 0x2b1a81cb413236cf063077b46ed3111628f6c41036bcf6e23ee4cbbf5679ef7a, the pin of 0.3.9 to 0.3.11; `pinned-guests-check` ok, `proving/igneum-prove/elf/` untouched |
|
||||
| The HiveOS package | `NODE_OUT=<scratch>/cross WORKERS_OUT=<the 0.3.11 Linux workers> VERSION=0.3.12 OUT=<scratch>/hive packaging/hive/make-hive-package.sh` (the Linux workers 4aaff27f.../82d90890... unchanged: their sources are) | 08:41:54Z: `igneum-hive-0.3.12.tar.gz` **7972af92e7cd9a032303eca4d95b533f53e0e68d1b9cae5bfe406a5b7c30a454** (24,506,282); `h-run.sh` writes `data/override-params.json` from `OVERRIDE` and starts the node with `--override-params-file` (the 0.3.11 open item closed); the node inside is 83089544 |
|
||||
| The DMG | `NODE=<fork igneumd> MINER=<fork igneum-miner> PROVE_HOST/PROVE_EXPORT=<this tree's build> packaging/mac/build-dmg.sh` under the lock | 08:42:49Z: `Igneum-Miner-0.3.12.dmg` **7a4a5f5f772956e983127280a5ec62a4fcfaf903b3afa38fe3a89a37cee23520** (41,702,535), engine 0.3.12, node 83089544 (igneumd 41,163,744 inside, stripped by the DMG build), the new prover host and export, `igneum-bench` from `proto-metal/main.swift` (unchanged, 66ec0e78...), `packaged-config` carries the ten-field object, hdiutil checksum valid |
|
||||
| The node suites with the igneum-pow feature (the coordinator's ask; the PC runner's test units carry no features field, so this is the Mac's run; the PC 2 run is owed to the Counter ASIC 3.0 coordinator's window, section 3a) | `CARGO_TARGET_DIR=vendor/igneum-node/target-0312 cargo test --release -j 4 -p kaspa-consensus -p kaspa-consensus-core -p igneum-exec -p kaspa-pow -p igneum-miner -p kaspa-p2p-flows --features kaspa-consensus/igneum-pow,kaspa-pow/igneum-pow` from the fork, under the lock, 08:39:31 to 08:45:01Z | igneum-exec 17 of 17, igneum-miner 18 of 18, kaspa-consensus 97 passed, 2 failed, 4 ignored. The two: (1) `pruning_proof::igneum_m20_tests::witnesses_are_checked_in_epoch_order_under_their_own_seeds` (`igneum_m20_tests.rs:122`: the expected `EpochSeeds.era` is all zeros, the code draws `515e...`: the test predates the era draw of class v3) FAILS THE SAME on 89dfcb95 (run 08:45:48Z on the 0.3.11 fork): the known M20 era fail, NOT fixed by 0f0dda95, still owed; (2) `finality::tests::ban_is_decided_by_the_carrying_block_so_nodes_agree_on_every_voter_list` (`finality.rs:1883`) failed inside the full crate run and PASSES alone on both 83089544 and 89dfcb95: order-dependent, not a regression of this cut, owed as flaky. kaspa-consensus-core 107 passed, 1 failed (`config::params::tests::fast_time_60x_file_is_the_devnet_at_60x`: `infra/fast-time/override-60x.json` does not parse into `OverrideParams`, "duplicate field `proving_v1_activation_daa`" at line 64: the file has carried a second proving-v1 block since c2544be on 5 October, so the test fails on master's file and on 89dfcb95 alike; not a 0.3.12 regression, the file is owed a dedupe), `db_compat` 7 of 7, kaspa-pow 14 of 14, kaspa-p2p-flows 33 of 33 (08:47:13Z, no fail-fast). Net: 3 failures, each present on 89dfcb95, none from 0f0dda95 |
|
||||
| PC 2 build-and-suite job | `IGNEUM_WIN_RELEASE=<scratch>/pc2-out node tools/build-job.mjs run --node vendor/igneum-node-0312 --target 1ccfe586 --targets linux,windows --node-tests "kaspa-consensus kaspa-consensus-core igneum-exec kaspa-pow igneum-miner kaspa-p2p-flows" --app-tests igneum-app` from this worktree, published 08:54:56Z on the prover-floor agent's "PC 2 is yours" (its floor-core-alone and floor-core-miner closed 08:48:37Z and 08:53:13Z, after the Counter ASIC 3.0 coordinator's release at 08:43:24Z); CPU only, the prover on, nothing else touched. The coordinator's later order (after the prover-floor agent's SECOND pair) arrived once the job had run; the app's queue serialised them anyway: this job ended 09:02:04Z and floor-build-6 started 09:02:05Z, then floor-core2-miner closed 09:13:22Z with the prover on and the miners never stopped, so nothing ran beside a GPU row | `build-20261006-085456`: started 08:55:24Z, done 09:02:04Z (400 s), every stage ok: Linux node 135 s (igneumd 34877b86..., igneum-miner c779777f...), Windows node 165 s (igneumd.exe 5bbcbd59f592fa31bf31c18516cef81cc0e7e537d398382fc8063e3402d80917, igneum-miner.exe af973318...; PC-built, NOT shipped: the inputs carry the Mac cross-build f580b4aa..., placed under the scratchpad), the app both targets; `RESULT test node [the six crates] exit 0 44 s` (without the igneum-pow feature, the runner's shape: the M20 era test and the fast-time file test are outside its reach there) and `RESULT test app/igneum-app exit 0 6 s` |
|
||||
| The Windows node exes | `CARGO_TARGET_DIR=vendor/igneum-node/target-0312-win proto-cuda/windows-node/cross-build.sh <fork> 4` on the Mac (mingw, the 0.3.5/0.3.6/0.3.9 path; PC 2 is the Counter ASIC 3.0 coordinator's this morning), the target cloned from `target-release-win`, under the lock | 08:40:43 to 08:48:05Z (6 min 42 s): igneumd.exe **f580b4aad1e19a47742d0d836a56dad36b9380d3890ca115b1babced4d83a8db** (52,177,920), igneum-miner.exe 06c17d4c23c8b1327793bebcd9b2cba115045ea91b68baea8cb92b232a94678b (11,040,768); static (KERNEL32, advapi32, api-ms-win-core only) |
|
||||
| The Windows inputs, second push | `IGNEUM_WIN_RELEASE=<target-0312-win>/x86_64-pc-windows-gnu/release IGNEUM_NODE_SRC=vendor/igneum-node-0312 packaging/windows/push-inputs.sh`, 08:48:28Z | node 83089544 (the two exes above), the 0.3.11-verified workers 2b3b8c92.../edc4a75d..., the telemetry helper 8d679b52..., the mingw DLLs and nvrtc; `payload-inputs.zip` f8e567bd164b382d32a33fb488df658fa68555092ac6bdac85387d0a8bd5d547 (65,259,161), signed and verified, live (HTTP 200); `node-source.pin` 83089544 committed as **062c3f8**, the CI commit of 0.3.12 |
|
||||
|
||||
| The Windows installer and zip, first runs | runs 37435975425 (ebea8b6, no telemetry helper) and 37436904041 (6532adf, the 0.3.11 node) | superseded |
|
||||
| The Windows installer and zip | `windows.yml` run 37438673235 on 062c3f8 (dispatched 08:49:12Z after the second inputs push) | `Igneum-Miner-Setup-0.3.12.exe` **f11a296acf1ea3efa8a6151efa357cc5c222e3b2ffec5701ea4bcefe29307810** (45,270,093); `igneum-windows-app.zip` **a6f33ef21bb1d1682f48ca22332d50c0302f4b4f552a99157581f3249c42ea3b** (65,507,597): igneumd.exe f580b4aa... (the cross-build, 52,177,920), igneum-miner.exe, igneum-app.exe 0.3.12 (3,700,736), the two workers, `igneum-gpu-telemetry.exe` (387,584) this time, the mingw DLLs, nvrtc64_120_0.dll and nvrtc-builtins64_128.dll |
|
||||
|
||||
## 4. The override objects and the digests (the 0.3.12 Mac node 746a931f..., ports 60975/60976, 22 s each, under `run`)
|
||||
|
||||
| Override file | Lines | Digest |
|
||||
|---|---|---|
|
||||
| none | `igneumd/2.1.0-83089544`, no activation line | c562d70e1428c9789823cc40067623b4767f7c555ce7ff4ea11c1498f013ef6c, EQUAL to 0.3.11's no-file digest: the new field is never by default and leaves the digest alone until set |
|
||||
| the fleet's live nine-field object | the six activation lines of 0.3.11, no fresh-rule line | **0139ab9dc2992d449ec787d8f021974933631eb55740ab4b6ce9d5c226e72888**, EQUAL to the fleet's digest today: publish 1 (the binary) changes no handshake, a 0.3.12 node and a 0.3.11 node on the nine-field file accept each other |
|
||||
| the ten-field object at the FIRST pin (`proving_v1_fresh_rule_daa` 192000, void: the floor failed at the go) | the six lines plus the fresh-rule line at 192000 | bd786a4b521e87c05bce3da4c46b4f4696deb16dfbdc913f181c980a8eb51688 (never published) |
|
||||
| the ten-field object as SHIPPED (`proving_v1_fresh_rule_daa` 198000) | the six lines plus `Proving v1 fresh-record rule from the override file: from DAA score 198000 a fresh segment record is valid whenever the previous segment is not proven` | **7bd98cc4118616455709d5e32a30b799e6e67caa42d2b5d09875cd49848a7ed7** (read 11:22:01Z on 746a931f...) |
|
||||
|
||||
The ten-field object (the packaged line 7dd3ff7, the manifest of publish 2, the hand nodes' and the seed's files at step 2):
|
||||
|
||||
```
|
||||
{"difficulty_v2_activation_daa":33000,"proving_v0_activation_daa":84100,"fees_v1_activation_daa":210000,"finality_v3_activation_daa":135200,"program_class_v3_activation_daa":154800,"proving_v1_activation_daa":154800,"proving_v1_segment_blocks":8,"proving_v1_unproven_daa":600,"proving_v1_aggregator_share_bps":1000,"proving_v1_fresh_rule_daa":198000}
|
||||
```
|
||||
|
||||
N was first pinned at 192,000 (tip + 14,400 for a publish near 10:15Z; the floor would hold while the tip was at or under 181,200, about
|
||||
11:15Z). the project lead's go came at 11:20Z with the tip at 181,582: the floor read 10,418, under 10,800, so N was RE-PINNED to 198,000 (tip + 14,400
|
||||
for publish 2 near 11:55Z; the floor holds until tip 187,200, about 12:55Z): the packaged line 8a6b133, the DMG rebuilt 11:22:08Z
|
||||
(7bcbb8a94038ea7a87ebfab514b6771f93b8fce2d991340bd5610713dc9548e5, 41,702,608), the installer rebuilt on CI (windows-ci 37455874734 on 8a6b133,
|
||||
green 11:27:33Z: `Igneum-Miner-Setup-0.3.12.exe` a6b3ea275373411e9f988ecd4ecc79f2cda5f68d54d681662dcbf7590689aef0, 45,275,988; `igneum-windows-app.zip`
|
||||
7ab28e772670dc58428cda4c9ff584b35f70507057d525a85d04391ee145dc53, 65,507,595), the digest re-read, publish 1 delayed by 8 minutes. The lesson for
|
||||
the next cut: pin N at the go, not at the forecast, or pin with a 3,600 margin over tip + 14,400 when the go is more than an hour out. A 0.3.11 node given the ten-field file dies on the unknown field
|
||||
(`deny_unknown_fields`), which is why publish 2 comes only after every node runs the 0.3.12 binary (the reviewer's C39, the 0.3.11 order).
|
||||
|
||||
## 5. The publish gate: what runs at the project lead's go, in which order (the 0.3.11 two-publish shape)
|
||||
|
||||
This was the plan at the gate; sections 6 and 7 record what ran. Runbook: the session scratchpad's `r0312/rollout-0312.sh` (every step a function; `step_floor` before each publish).
|
||||
|
||||
| Step | What | Gate |
|
||||
|---|---|---|
|
||||
| 0 the floor | `step_floor`: 192,000 minus the tip's DAA at least 10,800 | read before publish 1 and again before publish 2 |
|
||||
| 1a the hand nodes, the seed | the observer and node 1 on the 0.3.12 binary with the NINE-field file (`IGNEUMD=<fork>/target-integration/release/igneumd IGNEUMD_COMMIT=83089544 infra/devnet/restart-hand-nodes.sh '<nine>'`), then the seed (`IGNEUMD_LINUX=<scratch>/cross/igneumd IGNEUMD_LINUX_SHA256=4f142d51... infra/devnet/restart-seed.sh '<nine>'`); every one prints 0139ab9d..., nobody is refused; then `step_mac_miners` (the Mac's miner does not reconnect to a restarted node 1 by itself, C43) | the project lead's go |
|
||||
| 1b publish 1 | `node tools/ship-app.mjs 0.3.12 --node vendor/igneum-node-0312 --branch release-0.3.12 --public --activation-height 154800 --deadline-note "program class v3 + proving v1" --notes '<section 1>' --from ci`: ci "already" (the green Windows run), fetch, dmg "already", copy, manifest with `consensus` CARRIED OVER (the nine-field object; the digest stays 0139ab9d...), deploy, verify (`--from console` after the public index settles at the edge), the console item; `--public` carries the HiveOS package 7972af92... | after 1a |
|
||||
| 1c update-now | the Mac (d937c69d) first; the laptop (37ba0461) with it if it is on the air; PC 2 (1ccfe586) on the Counter ASIC 3.0 coordinator's word (PC 2 is its this morning; the proving agent's constraints: the app's prover stays on, no quit or restart of anything but the update's own); PC 1 (ae432dc7) last, once the project lead has relaunched its app (down since 22:31:06Z yesterday, on 0.3.10: it takes the nine-field object and 0.3.12 at its relaunch through the manifest; Power control is off by default so nothing asks for administrator rights) | each machine's STATUS line back on 0.3.12 with 0139ab9d... |
|
||||
| 2a the floor again | `step_floor` | >= 10,800 or re-pin |
|
||||
| 2b the hand nodes, the seed | the same two scripts with the TEN-field file; each prints bd786a4b... and refuses the nine-field side until it switches; `step_mac_miners` again | every app node on the 0.3.12 binary (1c) |
|
||||
| 2b publish 2 | `publish-manifest.sh --version 0.3.12 --override '<ten>' --activation-height 192000 --deadline-note "proving v1 fresh-record rule" --notes '<section 1>' --public --deploy` | after the hand nodes |
|
||||
| 2b update-now (switch) | the Mac and the laptop, then PC 2 on the 3.0 coordinator's word, then PC 1: each app writes the ten-field file at the manifest take and restarts its node at a safe moment (the Mac's node is node 1, already switched: nothing to restart) | |
|
||||
| the sweep | every node prints bd786a4b521e87c05bce3da4c46b4f4696deb16dfbdc913f181c980a8eb51688; the fresh-record rule arms at DAA 192,000 | |
|
||||
|
||||
One line for the project lead, per machine, when he says go: the Mac and PC 2 each restart their engine once for 0.3.12 (under a minute, the miner back on
|
||||
the next template) and their node once more for the fresh-record switch (a few seconds, mining resumes on the same chain); PC 1 does the
|
||||
same at its relaunch and, with Power control off by default, never asks for administrator rights again (its 5090 runs uncapped until he
|
||||
switches Power control on in Settings); the observer, node 1 and the seed are restarted by hand twice; from DAA 198,000 (about 15:55Z) a
|
||||
prover may file a fresh segment record whenever the previous segment is not proven, so paid segments stop stalling behind an unproven one;
|
||||
until the switch nothing changes in consensus (digest 0139ab9d... through publish 1).
|
||||
|
||||
## 6. The rollout (the project lead's go 11:20Z through the coordinator; two publishes)
|
||||
|
||||
Baseline 11:20:22Z: tip 181,582; the observer, node 1 and the seed on 89dfcb95 at 0139ab9d; the Mac app 0.3.11 (its miner PAUSED since
|
||||
07:10Z on the project lead's order "stop mining on the Mac", not the C42 class: I resumed it once at 11:26:11Z before the order reached me and the
|
||||
coordinator re-paused it; it stays paused, no restart-miners after the hand restarts); PC 2 0.3.11 at 113 MH/s; PC 1 0.3.11, relaunched by
|
||||
the project lead at 11:17Z with the 5090 and a 4070 in the enclosure; the laptop and Sam's Mac off the air.
|
||||
|
||||
| Step | Time | Result |
|
||||
|---|---|---|
|
||||
| 0 the floor | 11:20:22Z | 10,418 < 10,800: FAILED at 192,000; re-pinned to 198,000 (section 4), publish 1 delayed to the installer rebuild |
|
||||
| 1a the observer, node 1 | 11:22:12Z (pid 92464), 11:22:24Z (pid 92624) | `igneumd/2.1.0-83089544` on the nine-field file, digest 0139ab9d... (unchanged, nobody refused) |
|
||||
| 1a the seed | 11:22:45Z (MainPID 136418) | the same binary 4f142d51..., the same digest |
|
||||
| 1a restart-miners, the Mac | 11:22:56Z | ran; nothing to restart, the miner is paused on the project lead's order (above) |
|
||||
| 1b publish 1 | the ship 11:28:30 to 11:31:55Z from cf1ad2b (master f11b02e merged first: the preflight refuses a tree behind origin/master) | ci "already" (37455874734), fetch "already" (the re-pinned installer), dmg "already", copy ok, manifest 0.3.12 with `consensus` CARRIED OVER (the nine-field object, activation 154800), deploy ok, verify refused the public index at the edge (every cut); `--from console` 11:44:41Z: item #368 |
|
||||
| 1c update-now, the Mac | 11:32:18Z | engine restart 11:32:56Z (run `mac-d937c69d-20261006-113256`), "updated to Igneum Miner 0.3.12 from 0.3.11", STATUS "0.00 MH/s, paused, node 5 peers, synced" (node 1) |
|
||||
| 1c update-now, PC 2 | 11:32:46Z (the 3.0 coordinator's mkdir lock `/tmp/igneum-devnet/pc2-ca3.lock` absent; `pc2-ca3.clear` is a note, not a lock) | engine restart 11:33:41Z (run `win-1ccfe586-20261006-113341`), igneumd 83089544 started 11:33:45Z on the nine-field file (0139ab9d), worker ready 11:34:39Z, mining 11:34:42Z, 0 faults |
|
||||
| 1c update-now, PC 1 | 11:33:28Z (on the prover-floor agent's "PC 1 build closed" 11:31:34Z and the coordinator's "PC 1 back") | the installer downloaded and verified 11:34:06Z, "per-user install, no administrator prompt", engine restart 11:34:12Z (run `win-ae432dc7-20261006-113412`), cards "RTX 5090, RTX 4070 [discrete], AMD integrated [off]", STATUS mining 11:35:13Z, the 5090's race base 140.2 MH/s, digest 0139ab9d |
|
||||
| 2a the floor | 11:36:53Z | tip 182,570; 15,430 >= 10,800 at 198,000 |
|
||||
| 2b the observer, node 1 | 11:36:55Z (pid 13642), 11:37:07Z (pid 13777) | the ten-field file, digest **7bd98cc4118616455709d5e32a30b799e6e67caa42d2b5d09875cd49848a7ed7** |
|
||||
| 2b the seed | 11:37:25Z (MainPID 136590) | 7bd98cc4... |
|
||||
| 2b publish 2 | 11:37:36Z | `publish-manifest.sh --version 0.3.12 --override '<ten>' --activation-height 198000 --deadline-note "proving v1 fresh-record rule" --public --deploy`; the HiveOS package 7972af92... served at `/public/igneum-miner-hive.tar.gz` and `dl/public/igneum-hive-0.3.12.tar.gz` (HTTP 200, 24,506,282), the 0.3.11 package removed |
|
||||
| 2b switch, the Mac | 11:40:30Z | ran 11:40:58Z: "consensus override changed; the node restarts with it at a safe moment"; its node is node 1 (external), already on 7bd98cc4, nothing to restart |
|
||||
| 2b switch, PC 2 | 11:40:56Z | ran 11:41:23Z, "restarting the node with the new consensus parameters", igneumd started 11:41:25Z (pid 18732) on 7bd98cc4..., mining again 11:43:42Z, 113.0 MH/s at 11:44:42Z |
|
||||
| 2b switch, PC 1 (last) | 11:42:54Z | ran 11:43:28Z, node restarted 11:43:29Z (pid 5556) on 7bd98cc4..., "waiting" 11:43:44 to 11:44:14Z, mining 11:44:44Z, 100.95 MH/s ramping at 11:45:14Z: its miners read 0 MH/s for about a minute after the node restart before coming back (the C43 class: the miner waits out the restarted node instead of resubscribing at once; the coordinator's note); the Ember Tune run 3 on PC 1 (ember-tune-pc1-3, 11:44:58Z) then took the box, after this restart, not under it |
|
||||
| the laptop, Sam's Mac | off the air | they take 0.3.12 and the ten-field object through the manifest when they return; no 0.3.11 app was on the air to take the ten-field file before its binary (C39) |
|
||||
|
||||
## 7. The digest sweep (closed 11:45:20Z)
|
||||
|
||||
| Node | Binary | Digest | Since |
|
||||
|---|---|---|---|
|
||||
| the observer | `igneumd/2.1.0-83089544` (746a931f...) | 7bd98cc4... | 11:36:55Z |
|
||||
| node 1 | the same | 7bd98cc4... | 11:37:07Z |
|
||||
| the seed | 83089544 (4f142d51..., glibc 2.36 target) | 7bd98cc4... | 11:37:25Z |
|
||||
| PC 2 | the installer's igneumd.exe f580b4aa... (the Mac cross-build) | 7bd98cc4... | 11:41:25Z |
|
||||
| PC 1 | the same | 7bd98cc4... | 11:43:29Z |
|
||||
| the Mac | attached to node 1 | node 1's | 11:37:07Z |
|
||||
| the laptop, Sam's Mac | 0.3.10 / 0.3.9 | pending | off the air |
|
||||
|
||||
Tip 183,154 at 11:45:20Z, no refusals on the hand nodes after the switch; the fresh-record rule arms at DAA 198,000 (about 15:55Z at 0.98 DAA/s).
|
||||
The fleet during the window: PC 2 and PC 1 mined on 0139ab9d while the hand nodes and the seed were on 7bd98cc4 (11:37 to 11:41Z); each
|
||||
rejoined at its switch; the Mac's miner paused throughout on the project lead's order.
|
||||
|
||||
## 8. CI
|
||||
|
||||
| Run | On | Result |
|
||||
|---|---|---|
|
||||
| `ci` 37435705568 | ebea8b6 (the branch push, 08:22:40Z) | green (pow tests and census build, simulators, site build + link check + identity, the PowerShell 5.1 parse job) |
|
||||
| `windows-ci` 37435975425 | ebea8b6 | green 08:30:58Z (the parse job 08:25:16 to 08:25:56Z; engine, window host, payload, installer, smoke run 08:26:00 to 08:30:58Z); superseded by the run below (no telemetry helper in its payload) |
|
||||
| `ci` 37436904569 | 6532adf (the branch push, 08:33Z) | (pending) |
|
||||
| `windows-ci` 37436904041 | 6532adf | green 08:39:20Z; superseded by the run below (the node changed) |
|
||||
| `ci` 37436904569 | 6532adf | green |
|
||||
| `windows-ci` 37438673235 | 062c3f8 (`gh workflow run windows.yml --ref release-0.3.12`, 08:49:12Z, after the second inputs push) | green 08:54:27Z (the parse job 08:49:18 to 08:49:59Z; engine, window host, payload, installer, smoke run 08:50:02 to 08:54:27Z against the 83089544 inputs); fetched 09:03:17Z with `OTA_SKIP=1 CONSOLE_SKIP=1 packaging/windows/fetch-ci-artifacts.sh 37438673235` into the downloads folder, NOT deployed |
|
||||
| `ci` 37438674529 | 062c3f8 | FAILED in one step, `prover-socket-check.sh`: `tools/proving-v1/pc2-agg-cost.ps1` (in through the proving-v1 merge) ends its root prover with `pkill -x sp1-gpu-server`, and the check wants `pkill -f`; the line now reads `pkill -f ... ; rm -f /tmp/sp1-cuda-*.sock` (one playbook line, nothing the app or the packaging reads; `git diff 062c3f8 <fix> -- app packaging proto-cuda proto-opencl proto-metal vendor` is empty, so the Windows artefacts of 37438673235 stand, as 0.3.11's did across 3b0262f and 2a62735); the rerun is the row below |
|
||||
| `ci` 37440456687 | 37b6a7f (the one-line playbook fix) | green 09:07Z |
|
||||
| `ci` 37440559793 | 88df58e (master d3b64cb merged in: the morning summary, docs only; the release tip) | green 09:08:08Z. The 0.3.12 CI verdict is therefore run 37440559793 on 88df58e; the Windows build is run 37438673235 on 062c3f8, the same app, packaging and node sources (`git diff 062c3f8 88df58e -- app packaging proto-cuda proto-opencl proto-metal vendor igneum-pow` is empty) |
|
||||
|
||||
The ship state file `~/.cache/igneum/ship/0.3.12.json` carries `sha` = 062c3f8 (the Windows-build commit, which the ci step looks up by commit; the tree is 88df58e, docs and one playbook line later, as 0.3.11's was two docs commits past its build commit),
|
||||
`forkCommit` 89dfcb95, `bumpedAt` before the DMG's mtime (so the dmg step reads "already"), and `runId` once the Windows run is green.
|
||||
|
||||
## 9. Owed to the next cut (0.3.13)
|
||||
|
||||
| Item | What |
|
||||
|---|---|
|
||||
| C43, the miner's dead gRPC channel | no commit exists; `igneum-miner mine grpc://` must re-subscribe after its node restarts (the 0.3.11 finding: 11 min of 26 MH/s burned on the Mac); until then every hand restart of node 1 is followed by `restart --what miners` |
|
||||
| prover-floor's server | its packaging row is 0.3.13 (the coordinator's word) |
|
||||
| explorer d7e797c, pool-v0, rig-install, ota-k2, the ledger forks | left out on the coordinator's word |
|
||||
| the Windows workers' reproducibility | mingw PE builds differ byte-for-byte between builds of the same sources (12bfaa27 vs 2b3b8c92 today); a `-Wl,--no-insert-timestamp` (or `SOURCE_DATE_EPOCH`) in `build-windows.sh` would make G5's one-commit rule checkable by hash |
|
||||
| the site build's non-idempotence and the pre-push flip | unchanged from 0.3.10 and 0.3.11 |
|
||||
|
|
@ -64,6 +64,7 @@
|
|||
"proving_v1_activation_daa": 18446744073709551615,
|
||||
"proving_v1_segment_blocks": 8,
|
||||
"proving_v1_unproven_daa": 10,
|
||||
"proving_v1_fresh_rule_daa": 0,
|
||||
"proving_v1_aggregator_share_bps": 1000,
|
||||
"fees": {"pgas": {"version": 1, "cycles_per_pgas": 1000, "intrinsic_pgas_per_tx": 300, "modexp_base": 10, "modexp_per_byte_numer": 1, "modexp_per_byte_denom": 10}, "block_proving_gas_limit": 120000, "shard_proving_gas_budget": 30000, "min_execution_base_fee_wei": 100000000000, "min_proving_base_fee_wei": 10000000000000, "initial_execution_base_fee_wei": 100000000000, "initial_proving_base_fee_wei": 10000000000000, "base_fee_change_denominator": 8}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -6,6 +6,23 @@ the two GPU workers (`igneum-worker-cuda` for NVIDIA, `igneum-worker-opencl` for
|
|||
scripts were self-tested with stub binaries (`selftest.sh`) and the binaries were cross-compiled on a Mac; the first real
|
||||
run on a Hive rig is still to come. Report what breaks.
|
||||
|
||||
**A HiveOS rig mines only.** The package (`make-hive-package.sh`) carries `igneumd`, `igneum-miner` and the two
|
||||
workers; no `igneum-prove-host`, no `igneum-prove-export`, no SP1 GPU server. So a rig on this package earns from the
|
||||
80% lottery share and nothing from the 20% proving share until a Linux prover build is published. The Ubuntu rig
|
||||
installer (`packaging/linux/README.md`, branch `rig-install`, commit dd632c1) carries a prover unit that idles in state
|
||||
`setup` for the same reason; its per-card table is the fuller version of the one below. What each card could do once
|
||||
the prover ships, from the app's default (`app/igneum-app/src/provedefault.rs` at 440fd59 on `proving-v1`: proving on
|
||||
at 24 GB or more, off below) and from the proving agent's S_p curve of 5 October 2026 (`docs/bench-log.md`, "proving
|
||||
v1": jobs `memsweep-pc2-pv1`, `memminer-pc2-pv1` and the S_p curve, one RTX 5090, SP1 6.8.1's GPU server):
|
||||
|
||||
| Card | Once a Linux prover ships | Measured on 5 October 2026 |
|
||||
|---|---|---|
|
||||
| 8 GB | mines only | the prover's floor is 13.9 GB for an empty shard |
|
||||
| 12 GB | mines only; proves nothing on this SP1 build | the same 13.9 GB floor |
|
||||
| 16 GB | in practice mines only: proves empty shards alone, nothing beside the miner | 13.9 GB alone; 15.7 GB beside the miner leaves nothing for the display |
|
||||
| 24 GB | proves the adopted v1 shard (30,000 pgas) from the fee switch at DAA 210,000, nothing before it | 20.4 GB alone, about 22 GB beside the miner (approximate, not measured on a 24 GB card); the prototype shard at 28.3 GB does not fit |
|
||||
| 32 GB | mines and proves, prototype shard included | 28.3 GB alone, 30.0 GB beside the miner, 2.5 GB spare |
|
||||
|
||||
## Flight Sheet
|
||||
|
||||
| Field | Value |
|
||||
|
|
@ -16,7 +33,9 @@ run on a Hive rig is still to come. Report what breaks.
|
|||
| Wallet and worker template | `0x<40 hex>.%WORKER_NAME%`: the payout address is an EVM address you hold the key for; the part after the dot labels this rig's keys |
|
||||
| Pool URL | `grpc://<your node>:26610` (your own igneumd, solo mining), or `local` to run the bundled node on the rig |
|
||||
| Pass | empty |
|
||||
| Extra config arguments | `DEV_FEE=1 IDENTITIES=8 WORKER=auto VOTE=1` (one per line also works); `PEERS=a:26611,b:26611` for the bundled node; `EXTRA="..."` for more miner flags |
|
||||
| OVERRIDE | the devnet's consensus override as one JSON object on its own line, needed with `local`: `OVERRIDE={"difficulty_v2_activation_daa":33000,"proving_v0_activation_daa":84100,"fees_v1_activation_daa":210000,"finality_v3_activation_daa":135200}` is the four-field object 0.3.10 shipped; after the 0.3.11 switch it has nine fields, and the downloads page carries the live one. Set OVERRIDE from the downloads page when it changes. **A rig without it is refused**: its node runs on genesis parameters, prints another digest, and every devnet peer drops it (`h-run.sh` warns in the main log) |
|
||||
| Extra config arguments | `DEV_FEE=1 IDENTITIES=auto WORKER=auto VOTE=1` (one per line also works); `PEERS=a:26611,b:26611` for the bundled node; `EXTRA="..."` for more miner flags |
|
||||
| IDENTITIES | vote keys per card: 8 for a card with 8 GB or more, else 2 (`IDENTITIES=auto` applies that rule, per card, from `nvidia-smi` or the amdgpu sysfs; 8 when neither answers; a number overrides it for every card). The rule is the app's (`app/igneum-app/src/detect.rs`) |
|
||||
|
||||
Solo mining: there is no pool. Each card mines block templates from the node and the block reward pays the wallet in
|
||||
the template (80% of each block to its finder, 20% to the proving pool; the protocol takes no fee for anyone).
|
||||
|
|
@ -37,8 +56,8 @@ The protocol carries no fee: this is the software's, and any other miner client
|
|||
|
||||
| Hook | What |
|
||||
|---|---|
|
||||
| `h-config.sh` | writes `igneum.conf` from the Flight Sheet (node URL, wallet, label, DEV_FEE, IDENTITIES, WORKER, VOTE, PEERS, EXTRA); refuses a wallet that is not 0x + 40 hex |
|
||||
| `h-run.sh` | starts the bundled node when the URL is `local`, waits for the node, exports the hourly program pack (`igneum-miner export-pack`), then one `igneum-miner` per GPU with its worker; restarts a miner that exits (exit 42 = program change without prepare support: the pack is re-exported first); per-GPU logs `<log>.gpu<N>.log`, merged into the main log |
|
||||
| `h-config.sh` | writes `igneum.conf` from the Flight Sheet (node URL, wallet, label, DEV_FEE, IDENTITIES, WORKER, VOTE, PEERS, EXTRA, OVERRIDE); refuses an OVERRIDE that is not `{...}`; resolves `IDENTITIES=auto` to `IDENTITIES_GPU<N>` keys by VRAM; refuses a wallet that is not 0x + 40 hex |
|
||||
| `h-run.sh` | starts the bundled node when the URL is `local` (writes `data/override-params.json` from OVERRIDE and passes `--override-params-file`; warns when OVERRIDE is empty; copies the node's switch lines and its `Consensus params digest` line into the main log), waits for the node, exports the hourly program pack (`igneum-miner export-pack`), then one `igneum-miner` per GPU with its worker (`--identities` from `IDENTITIES_GPU<N>`, else `IDENTITIES`); restarts a miner that exits (exit 42 = program change without prepare support: the pack is re-exported first); per-GPU logs `<log>.gpu<N>.log`, merged into the main log |
|
||||
| `h-stats.sh` | per-GPU hash rate from each miner's last `STATUS` line (`now=<MH/s>`), accepted and rejected totals, dev-fee block count, temperatures and fans from Hive's `gpu-stats` (else `nvidia-smi`), uptime, version |
|
||||
|
||||
Stats JSON (what Hive reads from `$stats`): `hs` (kH/s per GPU), `hs_units` (`khs`), `temp`, `fan`, `uptime` (s),
|
||||
|
|
@ -50,7 +69,9 @@ Stats JSON (what Hive reads from `$stats`): `hs` (kH/s per GPU), `hs_units` (`kh
|
|||
on the library path). The worker compiles the hourly program at run time with NVRTC; without the library it says so
|
||||
and the miner retries. Hive images ship the driver; whether `libnvrtc.so.12` is present depends on the image, untested.
|
||||
- AMD: an OpenCL ICD (`libOpenCL.so.1` from ROCm or amdgpu-pro). The OpenCL worker compiles the program through the ICD.
|
||||
- The bundled node (`local`) keeps its chain data under the miner folder (`data/`); a devnet chain is small today.
|
||||
- The bundled node (`local`) keeps its chain data under the miner folder (`data/`); a devnet chain is small today. It
|
||||
needs OVERRIDE (the Flight Sheet table): compare the `node: Consensus params digest:` line in the main log with the
|
||||
digest on the downloads page; a different one means the override is stale and the node is refused.
|
||||
- Ports: the bundled node listens on 26611 (p2p) and answers RPC on 127.0.0.1:26610 only.
|
||||
|
||||
## Building the package
|
||||
|
|
@ -67,3 +88,7 @@ Stats JSON (what Hive reads from `$stats`): `hs` (kH/s per GPU), `hs_units` (`kh
|
|||
- GPU order: the CUDA device index is assumed to follow `nvidia-smi` order and Hive's `gpu-stats` arrays (NVIDIA first);
|
||||
a mixed NVIDIA and AMD rig may show temperatures against the wrong card.
|
||||
- Each card runs its own `igneum-miner` and node connection; the node's template RPC serves them all.
|
||||
- Before this change no package carried the override: `h-run.sh` started the node without `--override-params-file`, so
|
||||
a `local` rig was refused by every devnet peer (release 0.3.11, section 5, C41). Still untested on a rig.
|
||||
- No prover in the package (the second paragraph above): the rig earns nothing from the proving share until a Linux
|
||||
prover build ships.
|
||||
|
|
|
|||
|
|
@ -8,11 +8,19 @@
|
|||
# CUSTOM_USER_CONFIG extra lines, KEY=VALUE, one per line or separated by spaces:
|
||||
# DEV_FEE=1 the miner software's dev fee in whole percent (1 block template in 100 to the dev
|
||||
# address); DEV_FEE=0 turns it off. The protocol itself takes no fee.
|
||||
# IDENTITIES=8 vote keys per card (8 for a big card, 2 for a small one)
|
||||
# IDENTITIES=auto vote keys per card: 8 for a card with 8 GB or more, else 2 (IDENTITIES=auto
|
||||
# applies that rule per card from nvidia-smi or the amdgpu sysfs, 8 when neither
|
||||
# answers; the app's rule, app/igneum-app/src/detect.rs). A number overrides it
|
||||
# for every card.
|
||||
# WORKER=auto auto | cuda | opencl (the GPU worker; auto = cuda on NVIDIA, opencl on AMD)
|
||||
# VOTE=1 sign finality checkpoints (0 = mine without voting)
|
||||
# PEERS=a:26611,b:26611 peers for the bundled node when CUSTOM_URL=local
|
||||
# EXTRA="..." appended to every igneum-miner command line
|
||||
# OVERRIDE={...} the devnet's consensus override, one JSON object on its own line (single or
|
||||
# double quotes around it are fine in the Hive UI); the bundled node (CUSTOM_URL=local)
|
||||
# starts with --override-params-file from it. Without it the node runs on genesis
|
||||
# parameters and every devnet peer refuses it. Copy the live object from the
|
||||
# downloads page whenever it changes.
|
||||
# Hive sources h-manifest.conf before this hook; the fallback is for a run outside Hive (selftest.sh)
|
||||
[[ -z "$CUSTOM_CONFIG_FILENAME" ]] && . "$(dirname "${BASH_SOURCE[0]}")/h-manifest.conf"
|
||||
|
||||
|
|
@ -29,22 +37,67 @@ if ! [[ "$wallet" =~ ^0x[0-9a-fA-F]{40}$ ]]; then
|
|||
fi
|
||||
|
||||
# defaults, then the user's KEY=VALUE lines
|
||||
DEV_FEE=1; IDENTITIES=8; WORKER=auto; VOTE=1; PEERS=""; EXTRA=""
|
||||
while read -r kv; do
|
||||
[[ -z "$kv" || "$kv" == \#* ]] && continue
|
||||
key="${kv%%=*}"; val="${kv#*=}"
|
||||
case "$key" in
|
||||
DEV_FEE|IDENTITIES|WORKER|VOTE|PEERS|EXTRA) printf -v "$key" '%s' "$val" ;;
|
||||
*) echo "Igneum: unknown setting '$key' ignored" ;;
|
||||
esac
|
||||
done < <(printf '%s\n' "$CUSTOM_USER_CONFIG" | tr ' ' '\n' | sed 's/^"//; s/"$//')
|
||||
DEV_FEE=1; IDENTITIES=auto; WORKER=auto; VOTE=1; PEERS=""; EXTRA=""; OVERRIDE=""
|
||||
while IFS= read -r line; do
|
||||
[[ -z "$line" || "$line" == \#* ]] && continue
|
||||
if [[ "$line" == OVERRIDE=* ]]; then # the whole line: JSON may carry spaces; quotes around it are stripped
|
||||
val="${line#OVERRIDE=}"; val="${val#\'}"; val="${val%\'}"; val="${val#\"}"; val="${val%\"}"
|
||||
OVERRIDE="$val"; continue
|
||||
fi
|
||||
while read -r kv; do
|
||||
[[ -z "$kv" ]] && continue
|
||||
key="${kv%%=*}"; val="${kv#*=}"
|
||||
case "$key" in
|
||||
DEV_FEE|IDENTITIES|WORKER|VOTE|PEERS|EXTRA) printf -v "$key" '%s' "$val" ;;
|
||||
*) echo "Igneum: unknown setting '$key' ignored" ;;
|
||||
esac
|
||||
done < <(printf '%s\n' "$line" | tr ' ' '\n' | sed 's/^"//; s/"$//')
|
||||
done < <(printf '%s\n' "$CUSTOM_USER_CONFIG")
|
||||
if [[ -n "$OVERRIDE" && ! "$OVERRIDE" =~ ^\{.*\}$ ]]; then
|
||||
echo -e "${YELLOW:-}Igneum: OVERRIDE must be one JSON object, {\"...\":N,...} from the downloads page; got '${OVERRIDE:0:40}'${NOCOLOR:-}"
|
||||
return 1
|
||||
fi
|
||||
[[ "$DEV_FEE" =~ ^[0-9]+$ ]] || DEV_FEE=1
|
||||
[[ "$IDENTITIES" =~ ^[0-9]+$ ]] || IDENTITIES=8
|
||||
[[ "$IDENTITIES" =~ ^[0-9]+$ || "$IDENTITIES" == "auto" ]] || IDENTITIES=auto
|
||||
|
||||
# IDENTITIES=auto: 8 for a card with 8 GiB (8192 MiB) or more, else 2, per card, in the order h-run.sh numbers them
|
||||
# (NVIDIA first unless WORKER=opencl, then AMD unless WORKER=cuda). VRAM from nvidia-smi (MiB per line) and from
|
||||
# /sys/class/drm/card<N>/device/mem_info_vram_total (bytes, amdgpu). A card whose VRAM cannot be read gets 8.
|
||||
ident_block="" # IDENTITIES_GPU<N>=... lines, one per card, newline-terminated
|
||||
ident_summary=""
|
||||
if [[ "$IDENTITIES" == "auto" ]]; then
|
||||
vrams=()
|
||||
if [[ "$WORKER" != "opencl" ]] && command -v nvidia-smi >/dev/null 2>&1; then
|
||||
nvq=(nvidia-smi --query-gpu=memory.total --format=csv,noheader,nounits)
|
||||
command -v timeout >/dev/null 2>&1 && nvq=(timeout 20 "${nvq[@]}")
|
||||
while read -r mib; do
|
||||
mib="${mib//[[:space:]]/}"
|
||||
[[ "$mib" =~ ^[0-9]+$ ]] && vrams+=("$mib") || vrams+=("")
|
||||
done < <("${nvq[@]}" 2>/dev/null)
|
||||
fi
|
||||
if [[ "$WORKER" != "cuda" ]]; then
|
||||
for d in "${IGNEUM_DRM_ROOT:-/sys/class/drm}"/card*; do
|
||||
[[ "$(basename "$d")" =~ ^card[0-9]+$ && -r "$d/device/mem_info_vram_total" ]] || continue
|
||||
bytes="$(cat "$d/device/mem_info_vram_total" 2>/dev/null)"
|
||||
[[ "$bytes" =~ ^[0-9]+$ ]] && vrams+=("$((bytes / 1048576))") || vrams+=("")
|
||||
done
|
||||
fi
|
||||
n=0
|
||||
for mib in ${vrams[@]+"${vrams[@]}"}; do
|
||||
if [[ -z "$mib" || "$mib" -ge 8192 ]]; then ident=8; else ident=2; fi
|
||||
ident_block+="IDENTITIES_GPU$n=$ident"$'\n'
|
||||
ident_summary+="gpu$n ${mib:-?}MiB:$ident "
|
||||
n=$((n + 1))
|
||||
done
|
||||
IDENTITIES=8 # the fallback h-run.sh uses for a card h-config.sh could not see
|
||||
[[ $n == 0 ]] && ident_summary="no VRAM readable, 8 per card"
|
||||
fi
|
||||
|
||||
url="$CUSTOM_URL"
|
||||
[[ "$url" == "local" ]] && url="local"
|
||||
[[ "$url" != "local" && "$url" != grpc://* ]] && url="grpc://$url"
|
||||
|
||||
sq() { local v="${1//\'/\'\\\'\'}"; printf "'%s'" "$v"; } # single-quoted for sourcing: JSON and flags carry shell characters
|
||||
mkdir -p "$(dirname "$CUSTOM_CONFIG_FILENAME")"
|
||||
cat > "$CUSTOM_CONFIG_FILENAME" <<CONF
|
||||
# written by h-config.sh from the Flight Sheet; edit the Flight Sheet, not this file
|
||||
|
|
@ -53,9 +106,10 @@ WALLET=$(printf '%s' "$wallet" | tr 'A-F' 'a-f')
|
|||
LABEL=$label
|
||||
DEV_FEE=$DEV_FEE
|
||||
IDENTITIES=$IDENTITIES
|
||||
WORKER=$WORKER
|
||||
${ident_block}WORKER=$WORKER
|
||||
VOTE=$VOTE
|
||||
PEERS=$PEERS
|
||||
EXTRA=$EXTRA
|
||||
EXTRA=$(sq "$EXTRA")
|
||||
OVERRIDE=$(sq "$OVERRIDE")
|
||||
CONF
|
||||
echo "Igneum: config written to $CUSTOM_CONFIG_FILENAME (node $url, wallet ${wallet:0:8}..., dev fee ${DEV_FEE}%, $IDENTITIES identities per card)"
|
||||
echo "Igneum: config written to $CUSTOM_CONFIG_FILENAME (node $url, wallet ${wallet:0:8}..., dev fee ${DEV_FEE}%, identities ${ident_summary:-$IDENTITIES per card}, override ${OVERRIDE:+set}${OVERRIDE:-NONE: a local node will be refused by devnet peers})"
|
||||
|
|
|
|||
|
|
@ -27,9 +27,20 @@ if [[ "$NODE_URL" == "local" ]]; then
|
|||
peers=()
|
||||
if [[ -n "$PEERS" ]]; then IFS=',' read -r -a plist <<< "$PEERS"; for p in "${plist[@]}"; do peers+=("--addpeer=$p"); done
|
||||
else peers+=("--addpeer=188.245.5.161:26611"); fi # the public devnet seed (app/igneum-app/src/config.rs)
|
||||
say "starting the bundled node (devnet v4, data $HERE/data, peers ${peers[*]#--addpeer=})"
|
||||
# the consensus override (OVERRIDE in the Flight Sheet's extra config, written to igneum.conf by h-config.sh): the
|
||||
# devnet's activation heights. A node without it runs on genesis parameters, prints another digest in the p2p
|
||||
# handshake and is refused by every devnet peer (release 0.3.11, section 5, C41).
|
||||
override=()
|
||||
if [[ -n "${OVERRIDE:-}" ]]; then
|
||||
printf '%s\n' "$OVERRIDE" > "$HERE/data/override-params.json"
|
||||
override=("--override-params-file=$HERE/data/override-params.json")
|
||||
say "consensus override from the Flight Sheet: $OVERRIDE"
|
||||
else
|
||||
say "WARNING: no OVERRIDE in the Flight Sheet; the node runs on genesis parameters and every devnet peer will refuse it. Set OVERRIDE from the downloads page."
|
||||
fi
|
||||
say "starting the bundled node (devnet, data $HERE/data, peers ${peers[*]#--addpeer=})"
|
||||
"$BIN/igneumd" --devnet "--appdir=$HERE/data" --rpclisten=127.0.0.1:26610 --listen=0.0.0.0:26611 "${peers[@]}" \
|
||||
--nodnsseed --disable-upnp --nologfiles --yes >> "$CUSTOM_LOG_BASENAME.node.log" 2>&1 &
|
||||
${override[@]+"${override[@]}"} --nodnsseed --disable-upnp --nologfiles --yes >> "$CUSTOM_LOG_BASENAME.node.log" 2>&1 &
|
||||
pids+=($!)
|
||||
fi
|
||||
say "node $NODE_URL; waiting for it to answer"
|
||||
|
|
@ -37,6 +48,13 @@ for _ in $(seq 1 60); do
|
|||
"$BIN/igneum-miner" watch 1 "$NODE_URL" >/dev/null 2>&1 && break
|
||||
sleep 2
|
||||
done
|
||||
if [[ -f "$CUSTOM_LOG_BASENAME.node.log" ]]; then
|
||||
# the switch lines and the digest the node printed at start, so the operator can compare the digest with the one
|
||||
# on the downloads page (another digest = refused by every peer)
|
||||
n=0
|
||||
while IFS= read -r l; do say "node: $l"; n=$((n + 1)); done < <(grep -E 'override params file|from the override file|Consensus params digest' "$CUSTOM_LOG_BASENAME.node.log" | head -12)
|
||||
[[ $n == 0 ]] && say "node: no digest line yet in $CUSTOM_LOG_BASENAME.node.log (an older node, or not started); check it by hand"
|
||||
fi
|
||||
|
||||
# 2. the GPUs: Hive's gpu-detect when present, else the driver tools
|
||||
nv=0; amd=0
|
||||
|
|
@ -59,7 +77,8 @@ run_gpu() {
|
|||
local log="$CUSTOM_LOG_BASENAME.gpu$idx.log"
|
||||
local args=(mine "$NODE_URL" 1 100000000 "$LABEL-gpu$idx" --worker "$BIN/$worker" --worker-args "--device $dev --pack packs/devnet"
|
||||
--prepare-packs packs/prepare --exit-on-seed-change --evm-address "$WALLET" --payout-label "$LABEL-gpu$idx" --status-secs 30)
|
||||
[[ "$IDENTITIES" -gt 1 ]] && args+=(--identities "$IDENTITIES")
|
||||
local identv="IDENTITIES_GPU$idx" ident="$IDENTITIES"; [[ -n "${!identv:-}" ]] && ident="${!identv}" # per-card from h-config.sh, else the fallback
|
||||
[[ "$ident" -gt 1 ]] && args+=(--identities "$ident")
|
||||
[[ "$VOTE" == "0" ]] && args+=(--no-vote)
|
||||
[[ "$DEV_FEE" != "1" ]] && args+=(--dev-fee "$DEV_FEE")
|
||||
[[ -n "$EXTRA" ]] && args+=($EXTRA)
|
||||
|
|
|
|||
|
|
@ -35,6 +35,14 @@ case "$1" in
|
|||
esac
|
||||
FAKE
|
||||
chmod +x "$M/bin/igneum-miner"
|
||||
cat > "$M/bin/igneumd" <<'FAKE'
|
||||
#!/usr/bin/env bash
|
||||
echo "fake igneumd $*"
|
||||
for a in "$@"; do case "$a" in --override-params-file=*) echo "Finality rule v3 from the override file: active from checkpoint DAA score 135200" ;; esac; done
|
||||
echo "Consensus params digest: 0139ab9dc2992d449ec787d8f021974933631eb55740ab4b6ce9d5c226e72888 (exchanged in the p2p handshake; a peer with another digest is refused)"
|
||||
sleep 600
|
||||
FAKE
|
||||
chmod +x "$M/bin/igneumd"
|
||||
mkdir -p "$T/bin"
|
||||
printf '#!/bin/sh\n[ "$1" = NVIDIA ] && echo 2 || echo 0\n' > "$T/bin/gpu-detect"
|
||||
cat > "$T/bin/gpu-stats" <<'GS'
|
||||
|
|
@ -49,6 +57,27 @@ export CUSTOM_CONFIG_FILENAME="$M/igneum.conf" CUSTOM_LOG_BASENAME="$T/log/igneu
|
|||
( . "$M/h-config.sh" ) || { echo "h-config.sh failed"; exit 1; }
|
||||
grep -q '^WALLET=0xabcd000000000000000000000000000000000001$' "$M/igneum.conf" && grep -q '^DEV_FEE=0$' "$M/igneum.conf" && grep -q '^LABEL=rig7$' "$M/igneum.conf" && grep -q '^IDENTITIES=4$' "$M/igneum.conf" && echo " conf ok: $(tr '\n' ' ' < "$M/igneum.conf" | cut -c1-160)"
|
||||
( CUSTOM_TEMPLATE="notanaddress" . "$M/h-config.sh" >/dev/null 2>&1 ) && { echo "h-config.sh accepted a bad wallet"; exit 1; } || echo " bad wallet refused ok"
|
||||
echo "== h-config.sh IDENTITIES=auto"
|
||||
# no GPU tool on the PATH (gpu-detect is not a VRAM source) and no amdgpu sysfs: every card falls back to 8
|
||||
( CUSTOM_USER_CONFIG="IDENTITIES=auto" IGNEUM_DRM_ROOT="$T/no-drm" CUSTOM_CONFIG_FILENAME="$T/auto-none.conf" . "$M/h-config.sh" >/dev/null ) || { echo "h-config.sh failed on IDENTITIES=auto"; exit 1; }
|
||||
grep -q '^IDENTITIES=8$' "$T/auto-none.conf" && ! grep -q '^IDENTITIES_GPU' "$T/auto-none.conf" && echo " auto with no GPU tools: IDENTITIES=8, no per-card keys ok" || { echo "FAIL: auto without tools"; cat "$T/auto-none.conf"; exit 1; }
|
||||
# the default is auto: no IDENTITIES line at all gives the same
|
||||
( CUSTOM_USER_CONFIG="" IGNEUM_DRM_ROOT="$T/no-drm" CUSTOM_CONFIG_FILENAME="$T/auto-default.conf" . "$M/h-config.sh" >/dev/null ) && grep -q '^IDENTITIES=8$' "$T/auto-default.conf" && echo " default (no IDENTITIES line) is auto ok" || { echo "FAIL: default not auto"; exit 1; }
|
||||
# a stubbed nvidia-smi: 6 GB and 24 GB cards resolve to 2 and 8, in nvidia-smi order
|
||||
printf '#!/bin/sh\nprintf "6144\\n24576\\n"\n' > "$T/bin/nvidia-smi"; chmod +x "$T/bin/nvidia-smi"
|
||||
( CUSTOM_USER_CONFIG="IDENTITIES=auto" IGNEUM_DRM_ROOT="$T/no-drm" CUSTOM_CONFIG_FILENAME="$T/auto-nv.conf" . "$M/h-config.sh" >/dev/null ) || { echo "h-config.sh failed on stubbed nvidia-smi"; exit 1; }
|
||||
grep -q '^IDENTITIES_GPU0=2$' "$T/auto-nv.conf" && grep -q '^IDENTITIES_GPU1=8$' "$T/auto-nv.conf" && grep -q '^IDENTITIES=8$' "$T/auto-nv.conf" && grep -q '^WORKER=auto$' "$T/auto-nv.conf" && echo " auto with nvidia-smi 6144/24576: gpu0=2, gpu1=8 ok" || { echo "FAIL: auto by VRAM"; cat "$T/auto-nv.conf"; exit 1; }
|
||||
# a number still overrides every card
|
||||
( CUSTOM_USER_CONFIG="IDENTITIES=4" CUSTOM_CONFIG_FILENAME="$T/auto-num.conf" . "$M/h-config.sh" >/dev/null ) && grep -q '^IDENTITIES=4$' "$T/auto-num.conf" && ! grep -q '^IDENTITIES_GPU' "$T/auto-num.conf" && echo " numeric override keeps IDENTITIES=4 ok" || { echo "FAIL: numeric override"; exit 1; }
|
||||
rm -f "$T/bin/nvidia-smi"
|
||||
echo "== h-config.sh OVERRIDE"
|
||||
OV='{"difficulty_v2_activation_daa":33000,"proving_v0_activation_daa":84100,"fees_v1_activation_daa":210000,"finality_v3_activation_daa":135200}'
|
||||
( CUSTOM_USER_CONFIG="DEV_FEE=0 VOTE=1"$'\n'"OVERRIDE='$OV'"$'\n'"IDENTITIES=4" CUSTOM_CONFIG_FILENAME="$T/ov.conf" . "$M/h-config.sh" >/dev/null ) || { echo "h-config.sh failed on OVERRIDE"; exit 1; }
|
||||
( . "$T/ov.conf"; [[ "$OVERRIDE" == "$OV" && "$DEV_FEE" == 0 && "$IDENTITIES" == 4 ]] ) && echo " OVERRIDE (single-quoted, own line) sourced back intact beside the other keys ok" || { echo "FAIL: OVERRIDE round trip"; cat "$T/ov.conf"; exit 1; }
|
||||
( CUSTOM_USER_CONFIG="OVERRIDE=notjson" CUSTOM_CONFIG_FILENAME="$T/ov-bad.conf" . "$M/h-config.sh" >/dev/null 2>&1 ) && { echo "FAIL: OVERRIDE=notjson accepted"; exit 1; } || echo " OVERRIDE that is not {...} refused ok"
|
||||
( CUSTOM_USER_CONFIG="" CUSTOM_CONFIG_FILENAME="$T/ov-none.conf" . "$M/h-config.sh" | grep -q "override NONE" ) && grep -q "^OVERRIDE=''$" "$T/ov-none.conf" && echo " no OVERRIDE: empty in the conf and named in the summary ok" || { echo "FAIL: empty OVERRIDE"; exit 1; }
|
||||
# h-run.sh reads IDENTITIES_GPU<N> first: write a conf with gpu1=3 and check the second miner's argv
|
||||
( CUSTOM_USER_CONFIG=$'DEV_FEE=0\nIDENTITIES=auto\n'"OVERRIDE=$OV" IGNEUM_DRM_ROOT="$T/no-drm" CUSTOM_CONFIG_FILENAME="$M/igneum.conf" . "$M/h-config.sh" >/dev/null ) && printf 'IDENTITIES_GPU1=3\n' >> "$M/igneum.conf"
|
||||
echo "== h-run.sh (fake GPUs: 2 NVIDIA, fake node, fake miner)"
|
||||
( cd "$M" && CUSTOM_CONFIG_FILENAME="$M/igneum.conf" CUSTOM_LOG_BASENAME="$T/log/igneum" ./h-run.sh > "$T/log/run.out" 2>&1 ) &
|
||||
disown
|
||||
|
|
@ -56,9 +85,21 @@ sleep 8
|
|||
ls "$T/log" | sed 's/^/ /'
|
||||
_lines=$(grep -c "dev fee off (--dev-fee 0)" "$T/log/igneum.log" || true); echo " dev fee lines in the main log: $_lines"; [[ "$_lines" -ge 2 ]] || { echo "FAIL: expected the two miners' dev fee lines in the main log"; cat "$T/log/run.out"; exit 1; }
|
||||
grep -q -- "--dev-fee 0" "$M/argv.log" && echo " DEV_FEE=0 reached the miner as --dev-fee 0 ok" || { echo "FAIL: --dev-fee 0 missing"; exit 1; }
|
||||
grep -q -- "--identities 4" "$M/argv.log" && echo " IDENTITIES=4 reached the miner ok" || { echo "FAIL: --identities 4 missing"; exit 1; }
|
||||
grep -q -- "rig7-gpu0 .*--identities 8" "$M/argv.log" && echo " gpu0 took the IDENTITIES=8 fallback ok" || { echo "FAIL: --identities 8 missing on gpu0"; cat "$M/argv.log"; exit 1; }
|
||||
grep -q -- "rig7-gpu1 .*--identities 3" "$M/argv.log" && echo " gpu1 took IDENTITIES_GPU1=3 over the fallback ok" || { echo "FAIL: --identities 3 missing on gpu1"; cat "$M/argv.log"; exit 1; }
|
||||
[[ "$(cat "$M/data/override-params.json")" == "$OV" ]] && echo " data/override-params.json written from OVERRIDE ok" || { echo "FAIL: override file"; exit 1; }
|
||||
grep -q -- "--override-params-file=$M/data/override-params.json" "$T/log/igneum.node.log" && echo " the node got --override-params-file ok" || { echo "FAIL: node flag"; cat "$T/log/igneum.node.log"; exit 1; }
|
||||
grep -q "node: Consensus params digest: 0139ab9d" "$T/log/igneum.log" && grep -q "node: Finality rule v3 from the override file" "$T/log/igneum.log" && echo " the node's digest and switch lines reached the main log ok" || { echo "FAIL: digest relay"; cat "$T/log/igneum.log"; exit 1; }
|
||||
grep -q "igneum-worker-cuda" "$M/argv.log" && echo " NVIDIA cards got the cuda worker ok" || { echo "FAIL: cuda worker missing"; exit 1; }
|
||||
[[ -f "$M/exited42" && $(grep -c "^mine" "$M/argv.log") -ge 3 ]] && echo " exit 42 restarted the miner and re-exported the pack ok" || { echo "FAIL: no restart after exit 42"; exit 1; }
|
||||
echo "== h-stats.sh (sourced)"
|
||||
( . "$M/h-stats.sh"; echo " khs=$khs"; echo " stats=$stats"; python3 -c "import json,sys; s=json.loads(sys.argv[1]); assert s['hs_units']=='khs' and len(s['hs'])==2 and s['ar'][0]>0 and s['algo']=='igneum' and len(s['temp'])==2, s; print(' stats JSON ok: hs', s['hs'], 'temp', s['temp'], 'ar', s['ar'], 'bus', s['bus_numbers'])" "$stats" )
|
||||
echo "== h-run.sh without OVERRIDE (the warning)"
|
||||
( CUSTOM_USER_CONFIG=$'DEV_FEE=0' IGNEUM_DRM_ROOT="$T/no-drm" CUSTOM_CONFIG_FILENAME="$T/noov.conf" . "$M/h-config.sh" >/dev/null )
|
||||
mkdir -p "$T/log2"
|
||||
( cd "$M" && CUSTOM_CONFIG_FILENAME="$T/noov.conf" CUSTOM_LOG_BASENAME="$T/log2/igneum" ./h-run.sh > "$T/log2/run.out" 2>&1 ) &
|
||||
disown
|
||||
sleep 4
|
||||
grep -q "WARNING: no OVERRIDE in the Flight Sheet" "$T/log2/igneum.log" && ! grep -q -- "--override-params-file" "$T/log2/igneum.node.log" && echo " no OVERRIDE: warning in the main log, no flag on the node ok" || { echo "FAIL: missing warning"; cat "$T/log2/igneum.log"; exit 1; }
|
||||
_p2="$(cat "$T/log2/igneum.pid" 2>/dev/null)"; [[ -n "$_p2" ]] && kill -TERM "$_p2" 2>/dev/null; sleep 1
|
||||
echo "== self-test passed (scripts and stats shape; Hive itself is untested)"
|
||||
|
|
|
|||
|
|
@ -11,7 +11,7 @@
|
|||
<key>CFBundleVersion</key>
|
||||
<string>VERSION_STAMP</string>
|
||||
<key>CFBundleShortVersionString</key>
|
||||
<string>0.3.11</string>
|
||||
<string>0.3.12</string>
|
||||
<key>CFBundlePackageType</key>
|
||||
<string>APPL</string>
|
||||
<key>CFBundleExecutable</key>
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ DL_HOST="https://dl.igneum.network"
|
|||
# carry the devnet's activation height here, the same N as every other devnet node, before it is cut (Mac and CI alike:
|
||||
# make-payload.sh sources this file). Rule and order: docs/plans/difficulty-v2-rollout-devnet.md.
|
||||
# Example: NODE_OVERRIDE_PARAMS='{"difficulty_v2_activation_daa": 123456}'
|
||||
NODE_OVERRIDE_PARAMS='{"difficulty_v2_activation_daa":33000,"proving_v0_activation_daa":84100,"fees_v1_activation_daa":210000,"finality_v3_activation_daa":135200,"program_class_v3_activation_daa":154800,"proving_v1_activation_daa":154800,"proving_v1_segment_blocks":8,"proving_v1_unproven_daa":600,"proving_v1_aggregator_share_bps":1000}'
|
||||
NODE_OVERRIDE_PARAMS='{"difficulty_v2_activation_daa":33000,"proving_v0_activation_daa":84100,"fees_v1_activation_daa":210000,"finality_v3_activation_daa":135200,"program_class_v3_activation_daa":154800,"proving_v1_activation_daa":154800,"proving_v1_segment_blocks":8,"proving_v1_unproven_daa":600,"proving_v1_aggregator_share_bps":1000,"proving_v1_fresh_rule_daa":198000}'
|
||||
|
||||
# igneum_secret_file <env var name> <base name> -> the file to read: the variable when set, else <base>.next when it
|
||||
# exists, else <base>; IGNEUM_CONFIG_DIR (tests) replaces ~/.config/igneum
|
||||
|
|
|
|||
|
|
@ -9,7 +9,7 @@
|
|||
#define ArtDir "..\..\brand\icons"
|
||||
#endif
|
||||
#ifndef AppVersion
|
||||
#define AppVersion "0.3.11"
|
||||
#define AppVersion "0.3.12"
|
||||
#endif
|
||||
#define AppName "Igneum Miner"
|
||||
#define Publisher "Igneum"
|
||||
|
|
|
|||
|
|
@ -62,7 +62,8 @@ done
|
|||
NVRTC_DIR="$ROOT/proto-cuda/nvrtc"
|
||||
found_workers=0
|
||||
if [ -n "${IGNEUM_WORKERS_DIR:-}" ] && [ -d "$IGNEUM_WORKERS_DIR" ]; then
|
||||
for f in "$IGNEUM_WORKERS_DIR"/igneum-worker-*.exe "$IGNEUM_WORKERS_DIR"/nvrtc*.dll "$IGNEUM_WORKERS_DIR"/LICENSE-*.txt "$IGNEUM_WORKERS_DIR"/THIRD-PARTY.md; do
|
||||
# igneum-gpu-telemetry.exe rides in the inputs too (push-inputs.sh); the worker glob does not match its name (6 October 2026, 0.3.12)
|
||||
for f in "$IGNEUM_WORKERS_DIR"/igneum-worker-*.exe "$IGNEUM_WORKERS_DIR"/igneum-gpu-telemetry.exe "$IGNEUM_WORKERS_DIR"/nvrtc*.dll "$IGNEUM_WORKERS_DIR"/LICENSE-*.txt "$IGNEUM_WORKERS_DIR"/THIRD-PARTY.md; do
|
||||
[ -f "$f" ] && { cp "$f" "$STAGE/"; found_workers=1; }
|
||||
done
|
||||
else
|
||||
|
|
@ -72,6 +73,7 @@ else
|
|||
ls "$STAGE"/nvrtc64_*_0.dll >/dev/null 2>&1 || echo "warning: igneum-worker-cuda.exe without nvrtc64_*_0.dll (run $NVRTC_DIR/fetch-redist.sh); the engine will not use it"
|
||||
fi
|
||||
if [ -f "$ROOT/proto-opencl/igneum-worker-opencl.exe" ]; then cp "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$STAGE/"; found_workers=1; fi
|
||||
if [ -f "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" ]; then cp "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" "$STAGE/"; fi # AMD power, heat, fans, clocks (5 October 2026)
|
||||
fi
|
||||
if [ "$found_workers" = 1 ]; then echo "workers: $(cd "$STAGE" && ls igneum-worker-*.exe nvrtc*.dll 2>/dev/null | tr '\n' ' ')"
|
||||
else echo "note: no prebuilt igneum-worker-cuda.exe / igneum-worker-opencl.exe found; the engine builds the CUDA worker from proto-cuda\\ on the PC (CUDA Toolkit and MSVC needed)"; fi
|
||||
|
|
|
|||
|
|
@ -1 +1 @@
|
|||
89dfcb95be5ace1a2b7a4fb18d88aeb22023cab4
|
||||
830895447497bdea5e4779c9aa9d4118eac3cbbc
|
||||
|
|
|
|||
|
|
@ -28,6 +28,7 @@ REL="${IGNEUM_WIN_RELEASE:-$ROOT/vendor/igneum-node/target-integration/x86_64-pc
|
|||
MINGW=/opt/homebrew/opt/mingw-w64/toolchain-x86_64/x86_64-w64-mingw32
|
||||
NVRTC_DIR="$ROOT/proto-cuda/nvrtc"
|
||||
CL_WORKER="$ROOT/proto-opencl/igneum-worker-opencl.exe"
|
||||
TELEMETRY="$ROOT/proto-opencl/igneum-gpu-telemetry.exe" # AMD power, heat, fans, clocks (5 October 2026)
|
||||
TOKEN_FILE="$HOME/.config/igneum/dl-token"
|
||||
DLSITE="${IGNEUM_DLSITE:-}"
|
||||
[ -n "$DLSITE" ] || { [ -f "$HOME/.config/igneum/dlsite-dir" ] && DLSITE="$(tr -d '[:space:]' < "$HOME/.config/igneum/dlsite-dir")"; } || true
|
||||
|
|
@ -59,6 +60,7 @@ if [ -f "$NVRTC_DIR/igneum-worker-cuda.exe" ]; then
|
|||
ls "$STAGE"/nvrtc64_*_0.dll >/dev/null 2>&1 || echo "warning: igneum-worker-cuda.exe without nvrtc64_*_0.dll (run $NVRTC_DIR/fetch-redist.sh)"
|
||||
else echo "warning: no $NVRTC_DIR/igneum-worker-cuda.exe (run $NVRTC_DIR/build-windows.sh); the app will build the CUDA worker on the PC"; fi
|
||||
[ -f "$CL_WORKER" ] && cp "$CL_WORKER" "$STAGE/" || echo "warning: no $CL_WORKER"
|
||||
[ -f "$TELEMETRY" ] && cp "$TELEMETRY" "$STAGE/" || echo "warning: no $TELEMETRY (AMD cards show no draw or temperature)"
|
||||
|
||||
# the signer, built from the app crate (it includes src/manifest.rs and src/inputs.rs, so it signs what the runner verifies)
|
||||
KEY="$HOME/.config/igneum/ota-signing-key"
|
||||
|
|
|
|||
|
|
@ -25,7 +25,7 @@ VERIFY="$ROOT/packaging/windows/resources/verify-exe.py"
|
|||
# The coin icon and the version blocks, as COFF objects the linker takes like any other input
|
||||
[ -f "$ICONS/igneum.ico" ] || { echo "== no $ICONS/igneum.ico, making the icons"; python3 "$ICONS/make-icons.py"; }
|
||||
RES="$(mktemp -d)"
|
||||
for w in cuda opencl; do
|
||||
for w in cuda opencl gpu-telemetry; do
|
||||
"$WINDRES" -I "$ICONS" -i "$HERE/igneum-worker-$w.rc" -O coff -o "$RES/igneum-worker-$w.res.o"
|
||||
done
|
||||
|
||||
|
|
@ -37,11 +37,16 @@ echo "== igneum-worker-opencl.exe"
|
|||
-I "$RED/include" -I "$PLACEHOLDER" -DIGNEUM_KERNEL_PATH='"kernel_bound.cl"' \
|
||||
-o "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$ROOT/proto-opencl/host.c" "$RES/igneum-worker-opencl.res.o"
|
||||
"$STRIP" "$ROOT/proto-opencl/igneum-worker-opencl.exe"
|
||||
echo "== igneum-gpu-telemetry.exe (ADLX, SetupAPI, PDH; vendor/adlx is the SDK clone)"
|
||||
[ -f "$ROOT/vendor/adlx/SDK/Include/ADLX.h" ] || { echo "no vendor/adlx: git clone --depth 1 https://github.com/GPUOpen-LibrariesAndSDKs/ADLX.git $ROOT/vendor/adlx" >&2; exit 1; }
|
||||
"$CC" -std=gnu99 -O2 -Wall -Wno-unused-parameter -Wno-unused-function -static -I "$ROOT/vendor/adlx/SDK/Include" \
|
||||
-o "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" "$ROOT/proto-opencl/gpu-telemetry.c" "$ROOT/vendor/adlx/SDK/ADLXHelper/Windows/C/ADLXHelper.c" "$RES/igneum-worker-gpu-telemetry.res.o" -lsetupapi -lpdh
|
||||
"$STRIP" "$ROOT/proto-opencl/igneum-gpu-telemetry.exe"
|
||||
rm -rf "$RES"
|
||||
for exe in "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe"; do
|
||||
for exe in "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$ROOT/proto-opencl/igneum-gpu-telemetry.exe"; do
|
||||
printf '%s: %d bytes, imports:' "$(basename "$exe")" "$(stat -f %z "$exe")"
|
||||
x86_64-w64-mingw32-objdump -p "$exe" | sed -n 's/^[[:space:]]*DLL Name: //p' | tr '\n' ' '
|
||||
echo
|
||||
done
|
||||
# the icon and version block survived the strip (strip keeps .rsrc; this proves it)
|
||||
python3 "$VERIFY" --version 0.3.0 "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe"
|
||||
python3 "$VERIFY" --version 0.3.0 "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$ROOT/proto-opencl/igneum-gpu-telemetry.exe"
|
||||
|
|
|
|||
35
proto-cuda/nvrtc/igneum-worker-gpu-telemetry.rc
Normal file
35
proto-cuda/nvrtc/igneum-worker-gpu-telemetry.rc
Normal file
|
|
@ -0,0 +1,35 @@
|
|||
// Windows resources for igneum-gpu-telemetry.exe: the coin icon Explorer shows and the version block under Properties > Details.
|
||||
// Compiled with x86_64-w64-mingw32-windres (the icon path is relative to brand/icons, passed with -I).
|
||||
// the project lead's rule (4 October 2026): every shipped exe carries the coin icon and a version block, like the Mac app and DMG.
|
||||
#include <winver.h>
|
||||
|
||||
1 ICON "igneum.ico"
|
||||
|
||||
1 VERSIONINFO
|
||||
FILEVERSION 0,3,0,0
|
||||
PRODUCTVERSION 0,3,0,0
|
||||
FILEFLAGSMASK 0x3fL
|
||||
FILEFLAGS 0x0L
|
||||
FILEOS VOS_NT_WINDOWS32
|
||||
FILETYPE VFT_APP
|
||||
FILESUBTYPE VFT2_UNKNOWN
|
||||
BEGIN
|
||||
BLOCK "StringFileInfo"
|
||||
BEGIN
|
||||
BLOCK "040904B0"
|
||||
BEGIN
|
||||
VALUE "CompanyName", "Igneum"
|
||||
VALUE "FileDescription", "Igneum Miner GPU telemetry (AMD power, heat, fans, clocks)"
|
||||
VALUE "FileVersion", "0.3.0"
|
||||
VALUE "InternalName", "igneum-gpu-telemetry"
|
||||
VALUE "LegalCopyright", "Igneum contributors"
|
||||
VALUE "OriginalFilename", "igneum-gpu-telemetry.exe"
|
||||
VALUE "ProductName", "Igneum Miner"
|
||||
VALUE "ProductVersion", "0.3.0"
|
||||
END
|
||||
END
|
||||
BLOCK "VarFileInfo"
|
||||
BEGIN
|
||||
VALUE "Translation", 0x409, 1200
|
||||
END
|
||||
END
|
||||
|
|
@ -59,6 +59,8 @@ proto-opencl/
|
|||
cl_dynamic.h Windows one-click build: OpenCL.dll loaded at run time (IGNEUM_CL_DYNAMIC)
|
||||
test-generic.sh the --pack mode checked here through Apple OpenCL (needs proto-cuda/nvrtc/emu/test.sh's packs)
|
||||
test_host.c device-free unit tests of host.c's rules (the duplicate-platform fold); run with test-host.sh
|
||||
gpu-telemetry.c igneum-gpu-telemetry: AMD power, temperature, fan, clocks and busy per card (ADLX on Windows, amdgpu sysfs on Linux),
|
||||
one line per card per sample; the app's AMD card row reads it (engine.rs amd_telemetry_line)
|
||||
build.sh macOS (-framework OpenCL, or the Khronos ICD loader) and Linux (-lOpenCL)
|
||||
build.bat Windows (MSVC cl.exe + OpenCL.lib)
|
||||
WAVEFRONT.md wave32 vs wave64 on AMD, and why the kernel cannot tell the difference
|
||||
|
|
|
|||
274
proto-opencl/gpu-telemetry.c
Normal file
274
proto-opencl/gpu-telemetry.c
Normal file
|
|
@ -0,0 +1,274 @@
|
|||
// igneum-gpu-telemetry: power, temperature, fan and clocks of every AMD GPU, one line per card per sample.
|
||||
// 5 October 2026, after the project lead watched a 9070 XT at 90% usage with its fans barely turning and the app could not say
|
||||
// what it drew (the app's draw, temperature and MH per watt line came from nvidia-smi only).
|
||||
//
|
||||
// igneum-gpu-telemetry [-l SECONDS] one sample (default), or one every SECONDS until stdin closes or SIGTERM
|
||||
//
|
||||
// Windows: ADLX (the AMD Device Library eXtra, amdadlx64.dll, shipped with Adrenalin; vendor/adlx is the SDK clone,
|
||||
// MIT) for the metrics, keyed by the card's PCI bus from SetupAPI (the display class, matched by the same name ADLX
|
||||
// reports). Without ADLX (no AMD driver, an old one, or the DLL missing) only the utilisation is read, from the
|
||||
// GPU Engine performance counters through PDH, keyed by the adapter LUID that Windows uses there.
|
||||
// Linux: the amdgpu sysfs (/sys/class/drm/card*/device: hwmon power1_average, temp1_input, fan1_input, pwm1,
|
||||
// pp_dpm_mclk, gpu_busy_percent), keyed by the PCI address of the device link.
|
||||
//
|
||||
// Line format (space separated, every field present, a value the source cannot give prints as -):
|
||||
// amd <ordinal> bus <pci bus or address> kind integrated|discrete name "<name>" watts <W> temp_c <C> fan_rpm <rpm>
|
||||
// fan_pct <%> mclk_mhz <MHz> gclk_mhz <MHz> util_pct <%> source adlx|sysfs|perfcounter
|
||||
// then one `end <ms>` line per sample. The app (engine.rs amd_telemetry_line) parses it; parsers are unit-tested
|
||||
// against lines captured on PC 1.
|
||||
#define _CRT_SECURE_NO_WARNINGS
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <signal.h>
|
||||
|
||||
static volatile int gStop = 0;
|
||||
static void onSignal(int s) { (void)s; gStop = 1; }
|
||||
|
||||
typedef struct {
|
||||
char bus[64];
|
||||
char kind[16];
|
||||
char name[128];
|
||||
double watts, tempC, fanRpm, fanPct, mclk, gclk, util; /* -1 = not available */
|
||||
const char* source;
|
||||
} Sample;
|
||||
|
||||
static void sampleInit(Sample* s) { memset(s, 0, sizeof(*s)); strcpy(s->bus, "-"); strcpy(s->kind, "-"); strcpy(s->name, "-"); s->watts = s->tempC = s->fanRpm = s->fanPct = s->mclk = s->gclk = s->util = -1.0; s->source = "-"; }
|
||||
static void printNum(double v, const char* fmt) { if (v < 0) printf(" -"); else printf(fmt, v); }
|
||||
static void printSample(int ordinal, const Sample* s) {
|
||||
printf("amd %d bus %s kind %s name \"%s\" watts", ordinal, s->bus, s->kind, s->name);
|
||||
printNum(s->watts, " %.1f"); printf(" temp_c"); printNum(s->tempC, " %.1f"); printf(" fan_rpm"); printNum(s->fanRpm, " %.0f");
|
||||
printf(" fan_pct"); printNum(s->fanPct, " %.0f"); printf(" mclk_mhz"); printNum(s->mclk, " %.0f"); printf(" gclk_mhz"); printNum(s->gclk, " %.0f");
|
||||
printf(" util_pct"); printNum(s->util, " %.0f"); printf(" source %s\n", s->source);
|
||||
}
|
||||
|
||||
#ifdef _WIN32
|
||||
#define WIN32_LEAN_AND_MEAN
|
||||
#include <windows.h>
|
||||
#include <setupapi.h>
|
||||
#include <pdh.h>
|
||||
#include "../vendor/adlx/SDK/ADLXHelper/Windows/C/ADLXHelper.h"
|
||||
#include "../vendor/adlx/SDK/Include/IPerformanceMonitoring.h"
|
||||
|
||||
/* The SDK declares these three and leaves them to the platform file of each sample. */
|
||||
adlx_handle ADLX_CDECL_CALL adlx_load_library(const TCHAR* filename) { return (adlx_handle)LoadLibrary(filename); }
|
||||
int ADLX_CDECL_CALL adlx_free_library(adlx_handle module) { return FreeLibrary((HMODULE)module) ? 1 : 0; }
|
||||
void* ADLX_CDECL_CALL adlx_get_proc_address(adlx_handle module, const char* procName) { return (void*)GetProcAddress((HMODULE)module, procName); }
|
||||
|
||||
static double nowMs(void) { LARGE_INTEGER f, c; QueryPerformanceFrequency(&f); QueryPerformanceCounter(&c); return (double)c.QuadPart * 1000.0 / (double)f.QuadPart; }
|
||||
|
||||
/* The PCI bus of every display-class device, by its name (SetupAPI; the names are the ones ADLX reports). */
|
||||
typedef struct { char name[128]; int bus; } BusEntry;
|
||||
static int listBuses(BusEntry* out, int cap) {
|
||||
static const GUID DISPLAY = { 0x4d36e968, 0xe325, 0x11ce, { 0xbf, 0xc1, 0x08, 0x00, 0x2b, 0xe1, 0x03, 0x18 } };
|
||||
HDEVINFO set = SetupDiGetClassDevsA(&DISPLAY, NULL, NULL, DIGCF_PRESENT);
|
||||
SP_DEVINFO_DATA d;
|
||||
DWORD i;
|
||||
int n = 0;
|
||||
if (set == INVALID_HANDLE_VALUE) return 0;
|
||||
d.cbSize = sizeof(d);
|
||||
for (i = 0; SetupDiEnumDeviceInfo(set, i, &d) && n < cap; ++i) {
|
||||
char name[128] = { 0 };
|
||||
DWORD bus = 0, type = 0, got = 0;
|
||||
if (!SetupDiGetDeviceRegistryPropertyA(set, &d, SPDRP_DEVICEDESC, &type, (BYTE*)name, sizeof(name) - 1, &got)) continue;
|
||||
if (!SetupDiGetDeviceRegistryPropertyA(set, &d, SPDRP_BUSNUMBER, &type, (BYTE*)&bus, sizeof(bus), &got)) continue;
|
||||
snprintf(out[n].name, sizeof(out[n].name), "%s", name); out[n].bus = (int)bus; ++n;
|
||||
}
|
||||
SetupDiDestroyDeviceInfoList(set);
|
||||
return n;
|
||||
}
|
||||
static int busOf(const BusEntry* b, int n, const char* name, int* taken) {
|
||||
int i;
|
||||
for (i = 0; i < n; ++i) if (!taken[i] && strcmp(b[i].name, name) == 0) { taken[i] = 1; return b[i].bus; }
|
||||
return -1;
|
||||
}
|
||||
|
||||
/* ADLX: one sample of every GPU. Returns the number of lines printed, -1 when ADLX is not usable (reason printed). */
|
||||
static IADLXSystem* gSys = NULL;
|
||||
static IADLXPerformanceMonitoringServices* gPerf = NULL;
|
||||
static int adlxOpen(void) {
|
||||
ADLX_RESULT r = ADLXHelper_Initialize();
|
||||
if (!ADLX_SUCCEEDED(r)) { printf("info adlx: ADLXHelper_Initialize returned %d (no AMD driver with ADLX; amdadlx64.dll missing or too old)\n", (int)r); return 0; }
|
||||
gSys = ADLXHelper_GetSystemServices();
|
||||
if (!gSys) { printf("info adlx: no system services\n"); return 0; }
|
||||
r = gSys->pVtbl->GetPerformanceMonitoringServices(gSys, &gPerf);
|
||||
if (!ADLX_SUCCEEDED(r) || !gPerf) { printf("info adlx: GetPerformanceMonitoringServices returned %d\n", (int)r); return 0; }
|
||||
return 1;
|
||||
}
|
||||
static int adlxSample(const BusEntry* buses, int nBuses) {
|
||||
IADLXGPUList* gpus = NULL;
|
||||
adlx_uint it;
|
||||
int ordinal = 0;
|
||||
int taken[32] = { 0 };
|
||||
ADLX_RESULT r = gSys->pVtbl->GetGPUs(gSys, &gpus);
|
||||
if (!ADLX_SUCCEEDED(r) || !gpus) { printf("info adlx: GetGPUs returned %d\n", (int)r); return 0; }
|
||||
for (it = gpus->pVtbl->Begin(gpus); it != gpus->pVtbl->End(gpus); ++it) {
|
||||
IADLXGPU* gpu = NULL;
|
||||
IADLXGPUMetrics* m = NULL;
|
||||
Sample s;
|
||||
const char* name = NULL;
|
||||
ADLX_GPU_TYPE type = GPUTYPE_UNDEFINED;
|
||||
adlx_double dv = 0; adlx_int iv = 0;
|
||||
if (!ADLX_SUCCEEDED(gpus->pVtbl->At_GPUList(gpus, it, &gpu)) || !gpu) continue;
|
||||
sampleInit(&s);
|
||||
s.source = "adlx";
|
||||
if (ADLX_SUCCEEDED(gpu->pVtbl->Name(gpu, &name)) && name) snprintf(s.name, sizeof(s.name), "%s", name);
|
||||
if (ADLX_SUCCEEDED(gpu->pVtbl->Type(gpu, &type))) strcpy(s.kind, type == GPUTYPE_INTEGRATED ? "integrated" : type == GPUTYPE_DISCRETE ? "discrete" : "-");
|
||||
{ int b = busOf(buses, nBuses, s.name, taken); if (b >= 0) snprintf(s.bus, sizeof(s.bus), "%d", b); }
|
||||
r = gPerf->pVtbl->GetCurrentGPUMetrics(gPerf, gpu, &m);
|
||||
if (ADLX_SUCCEEDED(r) && m) {
|
||||
if (ADLX_SUCCEEDED(m->pVtbl->GPUPower(m, &dv))) s.watts = dv;
|
||||
if (s.watts < 0 && ADLX_SUCCEEDED(m->pVtbl->GPUTotalBoardPower(m, &dv))) s.watts = dv;
|
||||
if (ADLX_SUCCEEDED(m->pVtbl->GPUTemperature(m, &dv))) s.tempC = dv;
|
||||
if (ADLX_SUCCEEDED(m->pVtbl->GPUFanSpeed(m, &iv))) s.fanRpm = iv;
|
||||
if (ADLX_SUCCEEDED(m->pVtbl->GPUVRAMClockSpeed(m, &iv))) s.mclk = iv;
|
||||
if (ADLX_SUCCEEDED(m->pVtbl->GPUClockSpeed(m, &iv))) s.gclk = iv;
|
||||
if (ADLX_SUCCEEDED(m->pVtbl->GPUUsage(m, &dv))) s.util = dv;
|
||||
m->pVtbl->Release(m);
|
||||
} else {
|
||||
printf("info adlx: GetCurrentGPUMetrics for \"%s\" returned %d\n", s.name, (int)r);
|
||||
}
|
||||
/* fan percent: ADLX gives rpm only here; the tuning interface has the range, the app shows rpm when pct is - */
|
||||
printSample(ordinal++, &s);
|
||||
gpu->pVtbl->Release(gpu);
|
||||
}
|
||||
gpus->pVtbl->Release(gpus);
|
||||
return ordinal;
|
||||
}
|
||||
|
||||
/* PDH fallback: GPU engine utilisation per adapter LUID, summed over the engines (no power, no temperature). */
|
||||
static int pdhSample(void) {
|
||||
PDH_HQUERY q = NULL;
|
||||
PDH_HCOUNTER c = NULL;
|
||||
DWORD size = 0, count = 0, i;
|
||||
PDH_FMT_COUNTERVALUE_ITEM_A* items;
|
||||
int ordinal = 0;
|
||||
if (PdhOpenQueryA(NULL, 0, &q) != ERROR_SUCCESS) { printf("info perfcounter: PdhOpenQuery failed\n"); return 0; }
|
||||
if (PdhAddEnglishCounterA(q, "\\GPU Engine(*)\\Utilization Percentage", 0, &c) != ERROR_SUCCESS) { printf("info perfcounter: no GPU Engine counters\n"); PdhCloseQuery(q); return 0; }
|
||||
PdhCollectQueryData(q); Sleep(1000); PdhCollectQueryData(q);
|
||||
PdhGetFormattedCounterArrayA(c, PDH_FMT_DOUBLE, &size, &count, NULL);
|
||||
items = (PDH_FMT_COUNTERVALUE_ITEM_A*)malloc(size ? size : 1);
|
||||
if (PdhGetFormattedCounterArrayA(c, PDH_FMT_DOUBLE, &size, &count, items) == ERROR_SUCCESS) {
|
||||
/* instance names: pid_1234_luid_0x00000000_0x0000D4E3_phys_0_eng_0_engtype_3D; sum per luid */
|
||||
char luids[16][40]; double sums[16]; int n = 0, k;
|
||||
for (i = 0; i < count; ++i) {
|
||||
const char* p = strstr(items[i].szName, "luid_");
|
||||
char luid[40];
|
||||
if (!p) continue;
|
||||
snprintf(luid, sizeof(luid), "%.39s", p); { char* e = strstr(luid, "_phys"); if (e) *e = 0; }
|
||||
for (k = 0; k < n; ++k) if (strcmp(luids[k], luid) == 0) break;
|
||||
if (k == n && n < 16) { strcpy(luids[n], luid); sums[n] = 0; ++n; }
|
||||
if (k < 16) sums[k] += items[i].FmtValue.doubleValue;
|
||||
}
|
||||
for (k = 0; k < n; ++k) {
|
||||
Sample s; sampleInit(&s); s.source = "perfcounter";
|
||||
snprintf(s.bus, sizeof(s.bus), "%s", luids[k]);
|
||||
s.util = sums[k] > 100.0 ? 100.0 : sums[k];
|
||||
printSample(ordinal++, &s);
|
||||
}
|
||||
}
|
||||
free(items);
|
||||
PdhCloseQuery(q);
|
||||
return ordinal;
|
||||
}
|
||||
|
||||
int main(int argc, char** argv) {
|
||||
int every = 0, i, haveAdlx;
|
||||
BusEntry buses[32];
|
||||
int nBuses;
|
||||
for (i = 1; i < argc; ++i) if (strcmp(argv[i], "-l") == 0 && i + 1 < argc) every = atoi(argv[++i]);
|
||||
signal(SIGINT, onSignal); signal(SIGTERM, onSignal);
|
||||
setvbuf(stdout, NULL, _IOLBF, 0);
|
||||
nBuses = listBuses(buses, 32);
|
||||
for (i = 0; i < nBuses; ++i) printf("info display device \"%s\" bus %d\n", buses[i].name, buses[i].bus);
|
||||
haveAdlx = adlxOpen();
|
||||
do {
|
||||
double t0 = nowMs();
|
||||
int n = haveAdlx ? adlxSample(buses, nBuses) : pdhSample();
|
||||
printf("end %.1f ms %d card(s)\n", nowMs() - t0, n);
|
||||
fflush(stdout); /* a redirected stdout is fully buffered on the Windows CRT whatever setvbuf asks (PC 1 lost 60 s of samples at the kill) */
|
||||
if (every > 0) Sleep((DWORD)every * 1000);
|
||||
} while (every > 0 && !gStop);
|
||||
if (haveAdlx) { if (gPerf) gPerf->pVtbl->Release(gPerf); ADLXHelper_Terminate(); }
|
||||
return 0;
|
||||
}
|
||||
#else
|
||||
#include <dirent.h>
|
||||
#include <unistd.h>
|
||||
#include <time.h>
|
||||
static double nowMs(void) { struct timespec ts; clock_gettime(CLOCK_MONOTONIC, &ts); return ts.tv_sec * 1000.0 + ts.tv_nsec / 1e6; }
|
||||
static int readText(const char* path, char* out, size_t cap) { FILE* f = fopen(path, "r"); size_t n; if (!f) return 0; n = fread(out, 1, cap - 1, f); fclose(f); out[n] = 0; return 1; }
|
||||
static double readNumber(const char* path) { char b[64]; if (!readText(path, b, sizeof(b))) return -1.0; return atof(b); }
|
||||
/* pp_dpm_mclk: lines "0: 96Mhz", "3: 1258Mhz *"; the starred line is the current state */
|
||||
static double dpmCurrent(const char* text) {
|
||||
const char* p = text;
|
||||
while (p && *p) {
|
||||
const char* nl = strchr(p, '\n');
|
||||
size_t len = nl ? (size_t)(nl - p) : strlen(p);
|
||||
const char* star = memchr(p, '*', len);
|
||||
if (star) { const char* colon = memchr(p, ':', len); if (colon) return atof(colon + 1); }
|
||||
p = nl ? nl + 1 : NULL;
|
||||
}
|
||||
return -1.0;
|
||||
}
|
||||
static int sysfsSample(const char* root) {
|
||||
DIR* d = opendir(root);
|
||||
struct dirent* e;
|
||||
int ordinal = 0;
|
||||
if (!d) { printf("info sysfs: no %s\n", root); return 0; }
|
||||
while ((e = readdir(d)) != NULL) {
|
||||
char dev[512], path[640], text[4096], link[512];
|
||||
ssize_t ln;
|
||||
Sample s;
|
||||
DIR* hw; struct dirent* he;
|
||||
if (strncmp(e->d_name, "card", 4) != 0 || strchr(e->d_name + 4, '-')) continue;
|
||||
snprintf(dev, sizeof(dev), "%s/%s/device", root, e->d_name);
|
||||
snprintf(path, sizeof(path), "%s/vendor", dev);
|
||||
if (!readText(path, text, sizeof(text)) || strtol(text, NULL, 16) != 0x1002) continue;
|
||||
sampleInit(&s);
|
||||
s.source = "sysfs";
|
||||
ln = readlink(dev, link, sizeof(link) - 1);
|
||||
if (ln > 0) { link[ln] = 0; { const char* base = strrchr(link, '/'); snprintf(s.bus, sizeof(s.bus), "%.63s", base ? base + 1 : link); } }
|
||||
snprintf(path, sizeof(path), "%s/product_name", dev);
|
||||
if (readText(path, text, sizeof(text))) { text[strcspn(text, "\n")] = 0; snprintf(s.name, sizeof(s.name), "%s", text); }
|
||||
else { snprintf(path, sizeof(path), "%s/device", dev); if (readText(path, text, sizeof(text))) { text[strcspn(text, "\n")] = 0; snprintf(s.name, sizeof(s.name), "amdgpu %s", text); } }
|
||||
snprintf(path, sizeof(path), "%s/boot_vga", dev);
|
||||
strcpy(s.kind, "discrete");
|
||||
snprintf(path, sizeof(path), "%s/hwmon", dev);
|
||||
hw = opendir(path);
|
||||
if (hw) {
|
||||
while ((he = readdir(hw)) != NULL) {
|
||||
char hp[900];
|
||||
if (strncmp(he->d_name, "hwmon", 5) != 0) continue;
|
||||
snprintf(hp, sizeof(hp), "%s/%s/power1_average", path, he->d_name); s.watts = readNumber(hp); if (s.watts < 0) { snprintf(hp, sizeof(hp), "%s/%s/power1_input", path, he->d_name); s.watts = readNumber(hp); } if (s.watts >= 0) s.watts /= 1e6;
|
||||
snprintf(hp, sizeof(hp), "%s/%s/temp1_input", path, he->d_name); s.tempC = readNumber(hp); if (s.tempC >= 0) s.tempC /= 1000.0;
|
||||
snprintf(hp, sizeof(hp), "%s/%s/fan1_input", path, he->d_name); s.fanRpm = readNumber(hp);
|
||||
{ double pwm, pwmMax; snprintf(hp, sizeof(hp), "%s/%s/pwm1", path, he->d_name); pwm = readNumber(hp); snprintf(hp, sizeof(hp), "%s/%s/pwm1_max", path, he->d_name); pwmMax = readNumber(hp); if (pwm >= 0 && pwmMax > 0) s.fanPct = 100.0 * pwm / pwmMax; else if (pwm >= 0) s.fanPct = 100.0 * pwm / 255.0; }
|
||||
break;
|
||||
}
|
||||
closedir(hw);
|
||||
}
|
||||
snprintf(path, sizeof(path), "%s/pp_dpm_mclk", dev); if (readText(path, text, sizeof(text))) s.mclk = dpmCurrent(text);
|
||||
snprintf(path, sizeof(path), "%s/pp_dpm_sclk", dev); if (readText(path, text, sizeof(text))) s.gclk = dpmCurrent(text);
|
||||
snprintf(path, sizeof(path), "%s/gpu_busy_percent", dev); s.util = readNumber(path);
|
||||
printSample(ordinal++, &s);
|
||||
}
|
||||
closedir(d);
|
||||
return ordinal;
|
||||
}
|
||||
int main(int argc, char** argv) {
|
||||
int every = 0, i;
|
||||
const char* root = getenv("IGNEUM_DRM_ROOT") ? getenv("IGNEUM_DRM_ROOT") : "/sys/class/drm"; /* a fixture tree for tests */
|
||||
for (i = 1; i < argc; ++i) if (strcmp(argv[i], "-l") == 0 && i + 1 < argc) every = atoi(argv[++i]);
|
||||
signal(SIGINT, onSignal); signal(SIGTERM, onSignal);
|
||||
setvbuf(stdout, NULL, _IOLBF, 0);
|
||||
do {
|
||||
double t0 = nowMs();
|
||||
int n = sysfsSample(root);
|
||||
printf("end %.1f ms %d card(s)\n", nowMs() - t0, n);
|
||||
fflush(stdout); /* a redirected stdout is fully buffered on the Windows CRT whatever setvbuf asks (PC 1 lost 60 s of samples at the kill) */
|
||||
if (every > 0) sleep((unsigned)every);
|
||||
} while (every > 0 && !gStop);
|
||||
return 0;
|
||||
}
|
||||
#endif
|
||||
|
|
@ -93,9 +93,16 @@ fn run() -> Result<()> {
|
|||
// proof (the chain rule of design 5.3), the measurement of docs/plans/proving-v1.md step 2
|
||||
let list = arg("--chain").or_else(|| args.get(1).filter(|a| !a.starts_with("--")).cloned()).context("--chain <f1.json,f2.json,...> (consecutive fixtures)")?;
|
||||
let fixtures: Vec<String> = list.split(',').map(|s| s.trim().to_string()).filter(|s| !s.is_empty()).collect();
|
||||
return run_chain(&pinned, &fixtures, prover, out_path.as_deref());
|
||||
// --save-shards writes every shard's compressed proof next to the results (block-N-shard-i-compressed.bin),
|
||||
// so `--mode aggregate` can re-run the aggregation of the same proofs under other settings
|
||||
let save_shards = args.iter().any(|a| a == "--save-shards");
|
||||
// --prev <file>: the previous segment's aggregated proof; the chain continues from it (chain_len grows past
|
||||
// the segment length, the chain rule of spec 7.8) instead of starting fresh. The app's segment path (6 October
|
||||
// 2026) passes it when the node reports the previous segment paid and its proof in the pool.
|
||||
let prev = arg("--prev");
|
||||
return run_chain(&pinned, &fixtures, prover, out_path.as_deref(), save_shards, prev.as_deref());
|
||||
}
|
||||
let path = args.get(1).filter(|a| !a.starts_with("--")).context("usage: igneum-prove-host <fixture.json> [--mode native|execute|shard|compressed|block|all] [--shard N] [--budget <test pgas>] [--prover 0x..] [--out results.json]; --mode chain --chain <f1,f2,...> [--prover 0x..] [--out results.json]; --mode aggregate --proofs <a.bin,...> --parent 0x.. [--prev prev.bin] [--out results.json]; --mode verify --proof <file> --statement 0x..; --mode verify-segment --proof <file> --statement 0x..; --mode id")?;
|
||||
let path = args.get(1).filter(|a| !a.starts_with("--")).context("usage: igneum-prove-host <fixture.json> [--mode native|execute|shard|compressed|block|all] [--shard N] [--budget <test pgas>] [--prover 0x..] [--out results.json]; --mode chain --chain <f1,f2,...> [--prover 0x..] [--out results.json] [--save-shards] [--prev prev.bin]; --mode aggregate --proofs <a.bin,...> --parent 0x.. [--prev prev.bin] [--out results.json]; --mode verify --proof <file> --statement 0x..; --mode verify-segment --proof <file> --statement 0x..; --mode id")?;
|
||||
let shard_index: usize = arg("--shard").map(|s| s.parse()).transpose()?.unwrap_or(0);
|
||||
// `--budget <pgas>`: re-plan the fixture's block at a TEST budget (the S_p curve of 5 October 2026); the fixture's
|
||||
// own per-shard plan is then not compared (the chain and the sums still are), and `--out` records the cut
|
||||
|
|
@ -577,6 +584,8 @@ fn setup_sp1(pinned: &pinned::Pinned, results: &mut serde_json::Map<String, serd
|
|||
let sp1 = Sp1ProofSystem::from_env(pinned.shard_elf(), pinned.agg_elf())?;
|
||||
let setup_s = t.elapsed().as_secs_f64();
|
||||
println!("RESULT setup: {:.2} s, ProofSystem v{} shard program id {} aggregator id {} at {}", setup_s, Sp1ProofSystem::VERSION, sp1.program_id(), sp1.aggregator_id(), now());
|
||||
println!("RESULT sp1 knobs: {}", Sp1ProofSystem::env_knobs());
|
||||
results.insert("sp1_env_knobs".into(), Sp1ProofSystem::env_knobs().into());
|
||||
if sp1.program_id() != pinned.shard_id || sp1.aggregator_id() != pinned.agg_id {
|
||||
bail!("SP1's key setup derived shard program id {} and aggregator id {} from the embedded guests, the pinned manifest says {} and {}: this host would make proofs no other node accepts (re-pin with proving/igneum-prove/pin-guests.sh)", sp1.program_id(), sp1.aggregator_id(), pinned.shard_id, pinned.agg_id);
|
||||
}
|
||||
|
|
@ -616,7 +625,7 @@ fn out_dir_of(out_path: Option<&str>) -> std::path::PathBuf {
|
|||
/// `--mode chain`: every fixture in order, consecutive on the chain (number and parent hash), each block's shards
|
||||
/// proven compressed and aggregated with the previous block's aggregated proof (`AggInput.prev`, the chain rule),
|
||||
/// every proof verified. One RESULT line per shard, per block (with the running totals) and for the chain.
|
||||
fn run_chain(pinned: &pinned::Pinned, fixtures: &[String], prover: Address, out_path: Option<&str>) -> Result<()> {
|
||||
fn run_chain(pinned: &pinned::Pinned, fixtures: &[String], prover: Address, out_path: Option<&str>, save_shards: bool, prev_path: Option<&str>) -> Result<()> {
|
||||
if fixtures.is_empty() {
|
||||
bail!("--chain needs at least one fixture");
|
||||
}
|
||||
|
|
@ -654,16 +663,34 @@ fn run_chain(pinned: &pinned::Pinned, fixtures: &[String], prover: Address, out_
|
|||
println!("RESULT chain native block {}: {} shard(s), pgas {}, gas {}, pre {} post {}", f.block.env.number, shards.len(), outcome.pgas_used, outcome.gas_used, pre_root, outcome.state_root);
|
||||
built.push((f.block.env.number, f.block.env.hash, shards, pre_root));
|
||||
}
|
||||
// the previous segment's proof: the chain continues from it (its block must be the parent of the first fixture)
|
||||
let mut prev: Option<proof_system::Sp1SegmentProof> = match prev_path {
|
||||
None => None,
|
||||
Some(p) => {
|
||||
let bytes = std::fs::read(p).with_context(|| format!("read {p}"))?;
|
||||
let proof: sp1_sdk::SP1ProofWithPublicValues = bincode::deserialize(&bytes).with_context(|| format!("{p} is not a bincode SP1 proof"))?;
|
||||
let output = BlockOutput::from_bytes(proof.public_values.as_slice()).with_context(|| format!("{p}: public values are not a block statement"))?;
|
||||
if output.number + 1 != first || output.block_hash != loaded[0].1.block.env.parent_hash {
|
||||
bail!("--prev attests block {} ({}), the chain starts at block {first} with parent {}: the previous proof must be the parent block's", output.number, output.block_hash, loaded[0].1.block.env.parent_hash);
|
||||
}
|
||||
println!("RESULT chain prev: block {} chain_len {} (the chain continues from it)", output.number, output.chain_len);
|
||||
Some(proof_system::Sp1SegmentProof { proof, output })
|
||||
}
|
||||
};
|
||||
let base_len = prev.as_ref().map(|p| p.output.chain_len).unwrap_or(0);
|
||||
let sp1 = setup_sp1(pinned, &mut results)?;
|
||||
let chain_t = Instant::now();
|
||||
let mut prev: Option<proof_system::Sp1SegmentProof> = None;
|
||||
let mut blocks_json = Vec::with_capacity(built.len());
|
||||
let (mut shard_total, mut agg_total, mut shards_total) = (0.0f64, 0.0f64, 0usize);
|
||||
let out_dir = out_dir_of(out_path);
|
||||
let mut shard_files: Vec<String> = Vec::new();
|
||||
for (number, _hash, shards, _) in &built {
|
||||
let block_t = Instant::now();
|
||||
let mut proofs: Vec<Sp1ShardProof> = Vec::with_capacity(shards.len());
|
||||
let mut shard_secs = Vec::new();
|
||||
// with --save-shards: what a shard's proof record carries (the statement, the proof's sha256, the file), so
|
||||
// the app's segment path signs and submits every shard of the chain from one run
|
||||
let mut shard_records: Vec<serde_json::Value> = Vec::new();
|
||||
for s in shards {
|
||||
let i = s.output.shard_index;
|
||||
stage(&format!("chain block {number} compressed shard {i}"));
|
||||
|
|
@ -676,12 +703,24 @@ fn run_chain(pinned: &pinned::Pinned, fixtures: &[String], prover: Address, out_
|
|||
}
|
||||
shard_secs.push(dt);
|
||||
shard_total += dt;
|
||||
if save_shards {
|
||||
let file = out_dir.join(format!("block-{number}-shard-{i}-compressed.bin"));
|
||||
let bytes = bincode::serialize(&p.proof)?;
|
||||
std::fs::write(&file, &bytes).with_context(|| format!("write {}", file.display()))?;
|
||||
let proof_hash: [u8; 32] = sha2::Sha256::digest(&bytes).into();
|
||||
shard_records.push(serde_json::json!({
|
||||
"number": number, "block_hash": s.input.env.hash.to_string(), "shard": i, "statement": alloy_primitives::keccak256(s.output.to_bytes()).to_string(),
|
||||
"proof_sha256": format!("0x{}", hex::encode(proof_hash)), "proof_bytes": bytes.len(), "proof_file": file.display().to_string(), "prove_seconds": dt,
|
||||
}));
|
||||
shard_files.push(file.display().to_string());
|
||||
}
|
||||
proofs.push(p);
|
||||
}
|
||||
shards_total += proofs.len();
|
||||
stage(&format!("chain block {number} aggregate {} shards{}", proofs.len(), if prev.is_some() { " with the previous block proof" } else { "" }));
|
||||
let seg = sp1.aggregate(prev.as_ref(), &proofs)?;
|
||||
let adt = sp1.last_timing("aggregate").unwrap_or_default().as_secs_f64();
|
||||
let sdt = sp1.last_timing("aggregate-stdin").unwrap_or_default().as_secs_f64();
|
||||
agg_total += adt;
|
||||
let claim = SegmentClaim::from_block(&seg.output);
|
||||
let ok = sp1.verify_segment(&seg, &claim);
|
||||
|
|
@ -690,9 +729,10 @@ fn run_chain(pinned: &pinned::Pinned, fixtures: &[String], prover: Address, out_
|
|||
let block_s = block_t.elapsed().as_secs_f64();
|
||||
let cumulative = chain_t.elapsed().as_secs_f64();
|
||||
println!(
|
||||
"RESULT chain block {number}: {} shards ({:.1} s of shard proofs), aggregate prove {adt:.1} s, proof {bytes} bytes, verify {vdt:.3} s, {}; chain_len {}, agg_vk {}; this block {block_s:.1} s, cumulative {cumulative:.1} s over {} block(s) at {}",
|
||||
"RESULT chain block {number}: {} shards ({:.1} s of shard proofs), aggregate prove {adt:.1} s (stdin {sdt:.3} s, {} deferred proofs), proof {bytes} bytes, verify {vdt:.3} s, {}; chain_len {}, agg_vk {}; this block {block_s:.1} s, cumulative {cumulative:.1} s over {} block(s) at {}",
|
||||
seg.output.shard_count,
|
||||
shard_secs.iter().sum::<f64>(),
|
||||
proofs.len() + usize::from(prev.is_some()),
|
||||
if ok { "VERIFIED (shard program id, aggregator id and claim checked)" } else { "VERIFY FAILED" },
|
||||
seg.output.chain_len,
|
||||
seg.output.agg_vk,
|
||||
|
|
@ -702,19 +742,21 @@ fn run_chain(pinned: &pinned::Pinned, fixtures: &[String], prover: Address, out_
|
|||
if !ok {
|
||||
bail!("the aggregated proof of block {number} did not verify");
|
||||
}
|
||||
let expected_len = blocks_json.len() as u64 + 1;
|
||||
let expected_len = base_len + blocks_json.len() as u64 + 1;
|
||||
if seg.output.chain_len != expected_len {
|
||||
bail!("block {number}: chain_len {} is not {expected_len}", seg.output.chain_len);
|
||||
}
|
||||
blocks_json.push(serde_json::json!({
|
||||
"number": number, "shards": seg.output.shard_count, "shard_prove_seconds": shard_secs, "aggregate_prove_seconds": adt,
|
||||
"number": number, "shards": seg.output.shard_count, "shard_prove_seconds": shard_secs, "aggregate_prove_seconds": adt, "aggregate_stdin_seconds": sdt,
|
||||
"aggregate_verify_seconds": vdt, "proof_bytes": bytes, "chain_len": seg.output.chain_len, "block_seconds": block_s, "cumulative_seconds": cumulative,
|
||||
"post_root": seg.output.post_root.to_string(), "statement": alloy_primitives::keccak256(seg.output.to_bytes()).to_string(),
|
||||
"shard_records": shard_records,
|
||||
}));
|
||||
prev = Some(seg);
|
||||
}
|
||||
let seg = prev.unwrap();
|
||||
let total = chain_t.elapsed().as_secs_f64();
|
||||
results.insert("base_chain_len".into(), base_len.into());
|
||||
let (statement, bytes) = segment_results(&seg, &out_dir, &mut results)?;
|
||||
println!(
|
||||
"RESULT chain: {} blocks {first}..={last}, {shards_total} shards, shard proofs {shard_total:.1} s, aggregation {agg_total:.1} s, end to end {total:.1} s; final proof {bytes} bytes attests chain_len {} (statement {statement}), pre {} post {} provers {} at {}",
|
||||
|
|
@ -732,6 +774,7 @@ fn run_chain(pinned: &pinned::Pinned, fixtures: &[String], prover: Address, out_
|
|||
results.insert("shard_prove_seconds_total".into(), shard_total.into());
|
||||
results.insert("aggregate_prove_seconds_total".into(), agg_total.into());
|
||||
results.insert("chain_seconds".into(), total.into());
|
||||
results.insert("shard_proof_files".into(), shard_files.into_iter().map(serde_json::Value::from).collect::<Vec<_>>().into());
|
||||
drop(sp1);
|
||||
finish(results, out_path.map(|s| s.to_string()))
|
||||
}
|
||||
|
|
@ -794,16 +837,18 @@ fn run_aggregate(pinned: &pinned::Pinned, proofs: &str, parent: &str, prev_path:
|
|||
for shards in &blocks {
|
||||
let number = shards[0].output.number;
|
||||
stage(&format!("aggregate block {number}, {} shards", shards.len()));
|
||||
let deferred = shards.len() + usize::from(prev.is_some());
|
||||
let seg = sp1.aggregate(prev.as_ref(), shards)?;
|
||||
let adt = sp1.last_timing("aggregate").unwrap_or_default().as_secs_f64();
|
||||
let sdt = sp1.last_timing("aggregate-stdin").unwrap_or_default().as_secs_f64();
|
||||
let claim = SegmentClaim::from_block(&seg.output);
|
||||
let ok = sp1.verify_segment(&seg, &claim);
|
||||
let vdt = sp1.last_timing("verify-block").unwrap_or_default().as_secs_f64();
|
||||
println!("RESULT aggregate block {number}: {} shards, prove {adt:.1} s, proof {} bytes, verify {vdt:.3} s, {}; chain_len {}, post {} at {}", seg.output.shard_count, bincode::serialize(&seg.proof)?.len(), if ok { "VERIFIED" } else { "VERIFY FAILED" }, seg.output.chain_len, seg.output.post_root, now());
|
||||
println!("RESULT aggregate block {number}: {} shards, prove {adt:.1} s (stdin {sdt:.3} s, {deferred} deferred proofs), proof {} bytes, verify {vdt:.3} s, {}; chain_len {}, post {} at {}", seg.output.shard_count, bincode::serialize(&seg.proof)?.len(), if ok { "VERIFIED" } else { "VERIFY FAILED" }, seg.output.chain_len, seg.output.post_root, now());
|
||||
if !ok {
|
||||
bail!("the aggregated proof of block {number} did not verify");
|
||||
}
|
||||
per_block.push(serde_json::json!({ "number": number, "shards": seg.output.shard_count, "aggregate_prove_seconds": adt, "aggregate_verify_seconds": vdt, "chain_len": seg.output.chain_len }));
|
||||
per_block.push(serde_json::json!({ "number": number, "shards": seg.output.shard_count, "aggregate_prove_seconds": adt, "aggregate_stdin_seconds": sdt, "deferred_proofs": deferred, "aggregate_verify_seconds": vdt, "chain_len": seg.output.chain_len }));
|
||||
prev = Some(seg);
|
||||
}
|
||||
let seg = prev.unwrap();
|
||||
|
|
|
|||
|
|
@ -259,6 +259,16 @@ impl Sp1ProofSystem {
|
|||
self.timings.lock().unwrap().push((what.to_string(), dt));
|
||||
}
|
||||
|
||||
/// The SP1 prover knobs set in this process's environment (the GPU server inherits them; the names from
|
||||
/// sp1-core-executor 6.8.1 `opts.rs` and sp1-prover 6.8.1 `worker/config.rs`), for the RESULT lines, so a
|
||||
/// measurement names the settings it ran under. "none" when the defaults apply.
|
||||
pub fn env_knobs() -> String {
|
||||
let fixed = ["SHARD_SIZE", "ELEMENT_THRESHOLD", "HEIGHT_THRESHOLD", "FULL_SIZE_SHARDS", "MINIMAL_TRACE_CHUNK_THRESHOLD", "TRACE_CHUNK_SLOTS", "MEMORY_LIMIT", "WITHOUT_VK_VERIFICATION", "RUST_LOG"];
|
||||
let mut out: Vec<String> = std::env::vars().filter(|(k, _)| k.starts_with("SP1_WORKER_") || fixed.contains(&k.as_str())).map(|(k, v)| format!("{k}={v}")).collect();
|
||||
out.sort();
|
||||
if out.is_empty() { "none".into() } else { out.join(" ") }
|
||||
}
|
||||
|
||||
pub fn last_timing(&self, what: &str) -> Option<Duration> {
|
||||
self.timings.lock().unwrap().iter().rev().find(|(k, _)| k == what).map(|(_, d)| *d)
|
||||
}
|
||||
|
|
@ -286,6 +296,9 @@ impl ProofSystem for Sp1ProofSystem {
|
|||
/// The aggregator guest over the shard proofs (and the previous segment's proof when given), by recursion.
|
||||
fn aggregate(&self, prev: Option<&Sp1SegmentProof>, shards: &[Sp1ShardProof]) -> Result<Sp1SegmentProof> {
|
||||
let first = shards.first().ok_or_else(|| anyhow!("no shards"))?;
|
||||
// 5 October 2026 (aggregation cost): the stdin build (the proof clones into the request) is timed apart
|
||||
// from the prove call, so the host's own share of an aggregation is visible next to the GPU's.
|
||||
let t_stdin = Instant::now();
|
||||
let mut stdin = SP1Stdin::new();
|
||||
let input = AggInput {
|
||||
shard_vk: self.shard_vk_hash(),
|
||||
|
|
@ -302,6 +315,7 @@ impl ProofSystem for Sp1ProofSystem {
|
|||
let SP1Proof::Compressed(proof) = p.proof.proof.clone() else { return Err(anyhow!("the previous block proof is not a compressed proof")) };
|
||||
stdin.write_proof(*proof, self.agg_vk.vk.clone());
|
||||
}
|
||||
self.record("aggregate-stdin", t_stdin.elapsed());
|
||||
let t = Instant::now();
|
||||
let proof = self.client.prove(&self.agg_pk, stdin).compressed().run()?;
|
||||
self.record("aggregate", t.elapsed());
|
||||
|
|
|
|||
|
|
@ -34,9 +34,13 @@ if [ "${SKIP_GATE:-0}" != "1" ]; then
|
|||
fi
|
||||
H="$ROOT/proving/igneum-prove/target/release/igneum-prove-host"
|
||||
for f in "$ROOT"/proving/fixtures/block-*.json; do
|
||||
# the exporter's side files (block-N.json.node-plan.json, 5 October 2026) are not fixtures
|
||||
case "$f" in *.node-plan.json) continue ;; esac
|
||||
if ! "$H" "$f" --mode native >>"$GATE_LOG" 2>&1; then echo "GATE FAILED: native run of $(basename "$f"); see $GATE_LOG"; exit 1; fi
|
||||
done
|
||||
if ! "$ROOT/tools/lock/with-lock.sh" measure "$H" "$ROOT/proving/fixtures/block-338-shard1.json" --mode execute --shard 0 >>"$GATE_LOG" 2>&1; then
|
||||
# the execute step reports a cycle count, not a time: the `run` lock (tools/lock/with-lock.sh: counts, not ms), so the
|
||||
# gate does not queue behind every build and measurement on the Mac (5 October 2026: 25 min behind a packbench run)
|
||||
if ! "$ROOT/tools/lock/with-lock.sh" run "$H" "$ROOT/proving/fixtures/block-338-shard1.json" --mode execute --shard 0 >>"$GATE_LOG" 2>&1; then
|
||||
echo "GATE FAILED: the guest did not execute the shard fixture (the exact failure the PC hit on 4 October); see $GATE_LOG"; exit 1
|
||||
fi
|
||||
"$H" --mode id | tee -a "$GATE_LOG" # the pinned program ids this package carries (the PC's build embeds the same elf/ files)
|
||||
|
|
|
|||
|
|
@ -9,10 +9,13 @@
|
|||
// GET chain igneum.network/api/live trimmed + the Hetzner results item
|
||||
// GET log?limit=&since= work-log items (kinds log, build, note), newest first
|
||||
// GET results bench entries (synced from docs/bench-log.md) + the FUD ledger counts
|
||||
// GET tuning?days=30&min=5 Ember Tune: the fleet priors per (card model, driver major, program class) from the
|
||||
// TUNE records in miner_logs (relay/lib/ember.mjs), with the sample counts and MH/W
|
||||
// POST post {kind,title,body,who,key?,meta?} one item; with key it upserts
|
||||
// POST sync {items:[...]} bulk upsert by key
|
||||
import { neon, authed, readJson, str, iso } from '../lib/relay.mjs';
|
||||
import { kv, kvNum, lastMatch, FAULT, parseLabel, parseMinerTail, parseHeader, parseAppTail, STALE_S, markStale } from '../lib/parse.mjs';
|
||||
import { parseRecords, aggregate } from '../lib/ember.mjs';
|
||||
|
||||
const json = (res, status, obj) => { res.status(status).setHeader('Content-Type', 'application/json; charset=utf-8'); res.end(JSON.stringify(obj)); };
|
||||
const CACHE_MS = 10_000;
|
||||
|
|
@ -213,6 +216,13 @@ async function chain(sql) {
|
|||
hetzner: het.length ? itemOut(het[0]) : null,
|
||||
};
|
||||
}
|
||||
/// Ember Tune: every TUNE record of the window, folded into priors (the same aggregation the publisher uses).
|
||||
async function tuning(sql, days, min) {
|
||||
const rows = await sql(`SELECT lines FROM miner_logs WHERE received_at > now() - ($1 || ' days')::interval AND lines LIKE '%TUNE {%' ORDER BY received_at DESC LIMIT 2000`, [String(days)]);
|
||||
const records = rows.flatMap(r => parseRecords(r.lines));
|
||||
const { priors, table } = aggregate(records, { minSamples: min });
|
||||
return { days, min_samples: min, records: records.length, priors, table };
|
||||
}
|
||||
async function results(sql) {
|
||||
const [bench, ledger] = await Promise.all([
|
||||
sql(`SELECT * FROM console_items WHERE kind = 'bench' ORDER BY (meta->>'date') DESC NULLS LAST, (meta->>'pos')::int DESC LIMIT 60`),
|
||||
|
|
@ -253,6 +263,11 @@ export default async function handler(req, res) {
|
|||
if (fn === 'builds') return json(res, 200, { ok: true, now: new Date().toISOString(), ...(await cached('builds', () => builds(sql))) });
|
||||
if (fn === 'chain') return json(res, 200, { ok: true, ...(await cached('chain', () => chain(sql))) });
|
||||
if (fn === 'results') return json(res, 200, { ok: true, ...(await cached('results', () => results(sql))) });
|
||||
if (fn === 'tuning') {
|
||||
const days = Math.min(365, Math.max(1, Number(q.days) || 30));
|
||||
const min = Math.min(100, Math.max(1, Number(q.min) || 5));
|
||||
return json(res, 200, { ok: true, now: new Date().toISOString(), ...(await cached(`tuning-${days}-${min}`, () => tuning(sql, days, min))) });
|
||||
}
|
||||
if (fn === 'log') {
|
||||
const limit = Math.min(300, Math.max(1, Number(q.limit) || 100));
|
||||
const params = []; let where = `kind IN ('log','build','note')`;
|
||||
|
|
|
|||
131
relay/lib/ember.mjs
Normal file
131
relay/lib/ember.mjs
Normal file
|
|
@ -0,0 +1,131 @@
|
|||
// Ember Tune, the fleet side (docs/plans/ember-tune.md): the TUNE records every app uploads with its log are folded
|
||||
// into one prior per (card model, driver major, program class): the median chosen point, its spread and the sample
|
||||
// count. The publisher writes the priors into the signed manifest's `tuning` section beside the kernel-variant
|
||||
// cards (tools/tuning.mjs --write), the console shows them (api/console.mjs fn=tuning, tools/console.mjs tuning),
|
||||
// and the public bench table lists them per model (site/miner-priors.json). No dependencies; the tests in
|
||||
// relay/test/ember.test.mjs drive these functions with a fixture of captured records.
|
||||
//
|
||||
// A record (app/igneum-app/src/ember.rs record_json): {ts, machine (a hash of the install id), app, os, card, vendor,
|
||||
// driver, driver_major, class, key, plan: full|confirm|baseline, steps: [{clock_mhz, power_pct, limit_w, watts, mhs,
|
||||
// eff, gclk, mclk, tmax, faults, mark}], chosen: {...}, before: {...}|null, eff, mhs, watts}. Nothing identifies the
|
||||
// owner: no address, no hostname, no raw machine id.
|
||||
|
||||
/// The TUNE records inside uploaded log text, de-duplicated on (machine, card, ts) because the log is re-sent every
|
||||
/// minute. Baseline records (measure only) are kept apart: they say what a card does untuned, never what to set.
|
||||
export function parseRecords(text) {
|
||||
const out = [];
|
||||
for (const line of String(text || '').split('\n')) {
|
||||
const i = line.indexOf('TUNE {');
|
||||
if (i < 0) continue;
|
||||
let rec;
|
||||
try { rec = JSON.parse(line.slice(i + 5)); } catch { continue; }
|
||||
if (!rec || !rec.card || !rec.key || !rec.plan) continue;
|
||||
out.push(rec);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
export function dedupe(records) {
|
||||
const seen = new Set();
|
||||
const out = [];
|
||||
for (const r of records) {
|
||||
const k = `${r.machine}|${r.card}|${r.ts}`;
|
||||
if (seen.has(k)) continue;
|
||||
seen.add(k);
|
||||
out.push(r);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
export const median = xs => { const s = xs.filter(x => Number.isFinite(x)).sort((a, b) => a - b); return s.length ? (s.length % 2 ? s[(s.length - 1) / 2] : (s[s.length / 2 - 1] + s[s.length / 2]) / 2) : 0; };
|
||||
|
||||
/// The median absolute deviation as a percent of the median (0 for one sample or a zero median).
|
||||
export function spreadPct(xs) {
|
||||
const m = median(xs);
|
||||
if (!m || xs.length < 2) return 0;
|
||||
return Number((median(xs.map(x => Math.abs(x - m))) / m * 100).toFixed(2));
|
||||
}
|
||||
|
||||
const usable = r => r && r.chosen && r.chosen.mark === 'ok' && r.chosen.eff > 0 && (r.plan === 'full' || r.plan === 'confirm');
|
||||
|
||||
/// Folds records into priors: one per key, from the full and confirm records with a usable chosen point. The point
|
||||
/// is the median clock cap and the median power percent (each rounded to the step the apps use: 10 MHz, 1%), the
|
||||
/// efficiency, rate and draw are medians, the spread is the MAD of the efficiency in percent, `samples` counts the
|
||||
/// records and `machines` the distinct install hashes. An outlier (one bad card, one hot room) moves the median by
|
||||
/// at most one rank, never by its size. Baseline records are summarised beside the prior as `baseline` (median
|
||||
/// MH/W untuned) so the console can show the gain.
|
||||
export function aggregate(records, { minSamples = 1 } = {}) {
|
||||
const byKey = new Map();
|
||||
for (const r of dedupe(records)) {
|
||||
const g = byKey.get(r.key) || { key: r.key, card: r.card, vendor: r.vendor || '', driver_major: r.driver_major || '', class: r.class || 'v2', tuned: [], baseline: [], machines: new Set() };
|
||||
byKey.set(r.key, g);
|
||||
g.machines.add(r.machine);
|
||||
if (usable(r)) g.tuned.push(r);
|
||||
else if (r.plan === 'baseline' && r.chosen && r.chosen.eff > 0) g.baseline.push(r);
|
||||
}
|
||||
const priors = {};
|
||||
const table = [];
|
||||
for (const g of byKey.values()) {
|
||||
const t = g.tuned;
|
||||
const row = {
|
||||
key: g.key, card: g.card, vendor: g.vendor, driver_major: g.driver_major, class: g.class,
|
||||
samples: t.length, machines: g.machines.size,
|
||||
baseline_samples: g.baseline.length,
|
||||
baseline_eff: g.baseline.length ? Number(median(g.baseline.map(r => r.chosen.eff)).toFixed(4)) : null,
|
||||
baseline_mhs: g.baseline.length ? Number(median(g.baseline.map(r => r.chosen.mhs)).toFixed(2)) : null,
|
||||
baseline_watts: g.baseline.length ? Number(median(g.baseline.map(r => r.chosen.watts)).toFixed(1)) : null,
|
||||
};
|
||||
if (t.length) {
|
||||
const effs = t.map(r => r.chosen.eff);
|
||||
const prior = {
|
||||
clock_mhz: Math.round(median(t.map(r => r.chosen.clock_mhz)) / 10) * 10,
|
||||
power_pct: Math.round(median(t.map(r => r.chosen.power_pct))),
|
||||
eff: Number(median(effs).toFixed(4)),
|
||||
mhs: Number(median(t.map(r => r.chosen.mhs)).toFixed(2)),
|
||||
watts: Number(median(t.map(r => r.chosen.watts)).toFixed(1)),
|
||||
spread_pct: spreadPct(effs),
|
||||
samples: t.length,
|
||||
machines: g.machines.size,
|
||||
card: g.card,
|
||||
vendor: g.vendor,
|
||||
driver_major: g.driver_major,
|
||||
class: g.class,
|
||||
updated: new Date(Math.max(...t.map(r => Number(r.ts) || 0)) * 1000).toISOString().replace(/\.\d{3}Z$/, 'Z'),
|
||||
};
|
||||
// the untuned reference: the full plan's first step (the power ladder's 100%), else the baseline records
|
||||
const befores = t.map(r => r.before && r.before.eff > 0 ? r.before.eff : null).filter(x => x !== null);
|
||||
if (befores.length) prior.before_eff = Number(median(befores).toFixed(4));
|
||||
else if (row.baseline_eff) prior.before_eff = row.baseline_eff;
|
||||
if (prior.before_eff) prior.gain_pct = Number(((prior.eff / prior.before_eff - 1) * 100).toFixed(1));
|
||||
Object.assign(row, prior);
|
||||
if (t.length >= minSamples) priors[g.key] = prior;
|
||||
}
|
||||
table.push(row);
|
||||
}
|
||||
table.sort((a, b) => (b.samples - a.samples) || (a.key < b.key ? -1 : 1));
|
||||
return { priors, table };
|
||||
}
|
||||
|
||||
/// The manifest's tuning section with the priors folded in: the kernel-variant `cards` object is kept as is,
|
||||
/// `priors` replaces the previous priors (a key that lost its samples drops out), `ember` carries the settings.
|
||||
export function mergeTuning(existing, priors, ember = {}) {
|
||||
const base = existing && typeof existing === 'object' ? existing : {};
|
||||
const cards = base.cards && typeof base.cards === 'object' && !Array.isArray(base.cards) ? base.cards : {};
|
||||
const settings = { enabled: true, min_samples: 5, rate_tolerance_pct: 1, ...(base.ember && typeof base.ember === 'object' ? base.ember : {}), ...ember };
|
||||
return { ...base, updated: new Date().toISOString().replace(/\.\d{3}Z$/, 'Z'), cards, ember: settings, priors: priors || {} };
|
||||
}
|
||||
|
||||
/// A prior as a card starts from it (app/igneum-app/src/ember.rs prior_of): None under the sample floor.
|
||||
export function priorFor(tuning, key, minSamples) {
|
||||
const p = tuning && tuning.priors && tuning.priors[key];
|
||||
const floor = Number.isFinite(minSamples) ? minSamples : (tuning && tuning.ember && tuning.ember.min_samples) || 5;
|
||||
if (!p || !(p.samples >= floor)) return null;
|
||||
return { clock_mhz: p.clock_mhz || 0, power_pct: Math.min(100, Math.max(50, p.power_pct || 100)), eff: p.eff, samples: p.samples };
|
||||
}
|
||||
|
||||
/// One text line per prior for the console and the CLI.
|
||||
export function priorLine(p) {
|
||||
const point = p.clock_mhz ? `${p.clock_mhz} MHz at ${p.power_pct}%` : `${p.power_pct}% (clock unlocked)`;
|
||||
const gain = p.gain_pct != null ? ` (${p.gain_pct >= 0 ? '+' : ''}${p.gain_pct}% over untuned ${p.before_eff} MH/W)` : '';
|
||||
return `${p.card.replace(/_/g, ' ')} | driver ${p.driver_major} | ${p.class}: ${point}, ${p.eff} MH/W${gain}, ${p.mhs} MH/s at ${p.watts} W, spread ${p.spread_pct}%, ${p.samples} sample(s) from ${p.machines} machine(s)`;
|
||||
}
|
||||
201
relay/playbooks/ember-tune-pc1.ps1
Normal file
201
relay/playbooks/ember-tune-pc1.ps1
Normal file
|
|
@ -0,0 +1,201 @@
|
|||
# Igneum run job: Ember Tune end to end on PC 1 (machine ae432dc7), unattended. 5 October 2026.
|
||||
# Published as a `run` job with --stop-miners (docs/plans/ember-tune.md): the installed app stops its miners and
|
||||
# holds them; this script takes the engine that carries src/ember.rs (the one the fetch job put in
|
||||
# <app dir>\jobs\ember-kit-1\igneum-app-ember.exe, else the installed one; the AMD helper with the --tune and --set
|
||||
# commands from the telemetry agent's fetch job, <app dir>\jobs\amd-kit-1\kit\igneum-gpu-telemetry.exe), copies the install folder to a scratch
|
||||
# folder beside it, swaps the engine in, and starts that SECOND engine with `--sweep` in a scratch data folder (the
|
||||
# real settings.json, machine-id and wallet.json copied in; remote jobs, auto-update and proving switched off
|
||||
# there). That engine finds the installed app's node on 127.0.0.1:26610, mines on every card with the live program,
|
||||
# tunes them one after the other (the full two-knob plan where the card can be controlled, the baseline measurement
|
||||
# where it cannot: NVIDIA without administrator rights, Apple), prints every TUNE line on stdout, uploads its log
|
||||
# (the TUNE {json} record reaches the intake) and quits. Every TUNE line is re-emitted as a RESULT line, so
|
||||
# `node tools/jobs.mjs <job id>` shows the table. Before and after, nvidia-smi's limits and clocks and the AMD
|
||||
# helper's `--tune` lines are printed, so the restore can be read. The installed app's miners restart when the job
|
||||
# ends. Not elevated: nothing asks for administrator rights (the project lead asleep, 5 October 2026); the NVIDIA card is
|
||||
# therefore measure only tonight unless the engine finds itself elevated.
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$resultTag = 'TUNE'
|
||||
$budgetMinutes = 35
|
||||
if (-not ($budgetMinutes -is [int]) -or $budgetMinutes -lt 5) { $budgetMinutes = 35 } # a budget under 5 minutes is a bug, not a budget (C35)
|
||||
$started = Get-Date
|
||||
$deadline = $started.AddMinutes($budgetMinutes)
|
||||
function Say([string] $m) { Write-Host ("[" + (Get-Date -Format 'HH:mm:ss') + "] " + $m) }
|
||||
|
||||
# the installed engine: the per-user install (0.3.3+), else Program Files
|
||||
$installDir = $null
|
||||
foreach ($d in @((Join-Path $env:LOCALAPPDATA 'Programs\Igneum Miner'), (Join-Path $env:ProgramFiles 'Igneum Miner'))) {
|
||||
if (Test-Path (Join-Path $d 'igneum-app.exe')) { $installDir = $d; break }
|
||||
}
|
||||
if (-not $installDir) { Write-Output 'RESULT TUNE error=no_engine reason=igneum-app.exe_not_found'; exit 2 }
|
||||
$appData = $env:IGNEUM_APP_DATA
|
||||
if (-not $appData) { $appData = Join-Path $env:LOCALAPPDATA 'igneum' }
|
||||
$appDir = $env:IGNEUM_APP_DIR
|
||||
if (-not $appDir) { $appDir = Join-Path $appData 'app' }
|
||||
|
||||
# the scratch install: the whole folder (workers, node, helper, DLLs) with the Ember engine swapped in
|
||||
$root = Join-Path $env:LOCALAPPDATA 'igneum-tune'
|
||||
$bin = Join-Path $root 'bin'
|
||||
$sApp = Join-Path $root 'app'
|
||||
$sLogs = Join-Path $root 'logs'
|
||||
New-Item -ItemType Directory -Force -Path $root, $sApp, $sLogs | Out-Null
|
||||
if (Test-Path $bin) { Remove-Item -LiteralPath $bin -Recurse -Force -ErrorAction SilentlyContinue }
|
||||
Copy-Item -LiteralPath $installDir -Destination $bin -Recurse -Force
|
||||
$ember = $null
|
||||
foreach ($cand in @((Join-Path $appDir 'jobs\ember-kit-2\igneum-app-ember.exe'), (Join-Path $appDir 'jobs\ember-kit-1\igneum-app-ember.exe'))) { if (Test-Path $cand) { $ember = $cand; break } }
|
||||
if ($ember) {
|
||||
Copy-Item -LiteralPath $ember -Destination (Join-Path $bin 'igneum-app.exe') -Force
|
||||
Say ("engine: the Ember build from " + $ember)
|
||||
} else {
|
||||
Say 'engine: the installed one (no jobs\ember-kit-1\igneum-app-ember.exe); an older engine ignores the tune and reports no_rows'
|
||||
}
|
||||
$helper = $null
|
||||
$found = Get-ChildItem -Path (Join-Path $appDir 'jobs') -Recurse -Filter 'igneum-gpu-telemetry.exe' -ErrorAction SilentlyContinue | Where-Object { $_.FullName -match 'amd-kit' } | Sort-Object LastWriteTime -Descending | Select-Object -First 1
|
||||
if ($found) { $helper = $found.FullName }
|
||||
if ($helper) {
|
||||
Copy-Item -LiteralPath $helper -Destination (Join-Path $bin 'igneum-gpu-telemetry.exe') -Force
|
||||
Say ("helper: the Ember build of igneum-gpu-telemetry from " + $helper + " sha256=" + (Get-FileHash -LiteralPath $helper -Algorithm SHA256).Hash.ToLower())
|
||||
} else { Say 'helper: the installed igneum-gpu-telemetry (no jobs\amd-kit-1\kit\igneum-gpu-telemetry.exe); without --tune the AMD card measures only' }
|
||||
$exe = Join-Path $bin 'igneum-app.exe'
|
||||
$ver = (& $exe --version 2>&1 | Out-String).Trim()
|
||||
Say ("engine: " + $exe + " (" + $ver + ")")
|
||||
Write-Output ("RESULT TUNE engine " + $ver + " sha256=" + (Get-FileHash -LiteralPath $exe -Algorithm SHA256).Hash.ToLower())
|
||||
if ($ver -notmatch 'igneum-app (\d+)\.(\d+)\.(\d+)') { Write-Output 'RESULT TUNE error=version_unknown'; exit 2 }
|
||||
|
||||
foreach ($f in @('settings.json', 'machine-id', 'wallet.json', 'tuning.json')) {
|
||||
$src = Join-Path $appDir $f
|
||||
if (Test-Path $src) { Copy-Item -LiteralPath $src -Destination (Join-Path $sApp $f) -Force }
|
||||
}
|
||||
# the second engine must not poll jobs (it would see this one), update itself, or prove; the tune is on
|
||||
$sj = Join-Path $sApp 'settings.json'
|
||||
if (Test-Path $sj) {
|
||||
try {
|
||||
$j = Get-Content -LiteralPath $sj -Raw | ConvertFrom-Json
|
||||
$j.remote_jobs = $false; $j.auto_update = $false; $j.prove = $false; $j.paused = $false; $j.setup_done = $true; $j.sweep = $true
|
||||
# the project lead, 6 October 2026, 07:20Z: never raise an administrator prompt. The installed app's Power control is READ and
|
||||
# reported, but the copy runs with it OFF so the second engine can never start the elevated helper; the 5090 is
|
||||
# measured as it runs either way (the two-knob tune is the installed engine's job once it carries Ember Tune)
|
||||
$installedPowerControl = $false
|
||||
try { $installedPowerControl = [bool]$j.power_control } catch { }
|
||||
Write-Output ('RESULT TUNE installed_power_control=' + $installedPowerControl.ToString().ToLower() + ' (the copy runs with it off: no prompt)')
|
||||
if ($j.PSObject.Properties.Name -contains 'power_control') { $j.power_control = $false } else { $j | Add-Member -NotePropertyName power_control -NotePropertyValue $false }
|
||||
# every card is due: the stored results are cleared in the COPY only
|
||||
if ($j.cards) { foreach ($p in $j.cards.PSObject.Properties) { $p.Value.sweep_at = 0; $p.Value.pinned = $false } }
|
||||
# PowerShell 5.1's Set-Content -Encoding utf8 writes a BOM, which the engine's JSON parser refuses: the copy then
|
||||
# read as defaults (no payout address, no cards) and the miners never started (runs 1 and 2, 5 and 6 October 2026)
|
||||
[IO.File]::WriteAllText($sj, ($j | ConvertTo-Json -Depth 8), (New-Object System.Text.UTF8Encoding $false))
|
||||
} catch { Say ("settings.json: " + $_.Exception.Message) }
|
||||
$back = $null
|
||||
try { $back = Get-Content -LiteralPath $sj -Raw | ConvertFrom-Json } catch { }
|
||||
$addr = ''; if ($back) { $addr = [string]$back.address }
|
||||
$bom = (Get-Content -LiteralPath $sj -Encoding Byte -TotalCount 3 -ErrorAction SilentlyContinue) -join ','
|
||||
Write-Output ('RESULT TUNE scratch settings: address ' + $(if ($addr) { $addr.Substring(0, [Math]::Min(10, $addr.Length)) + '...' } else { 'EMPTY' }) + ', cards ' + $(if ($back -and $back.cards) { @($back.cards.PSObject.Properties).Count } else { 0 }) + ', first bytes ' + $bom)
|
||||
if (-not $addr -and -not (Test-Path (Join-Path $sApp 'wallet.json'))) { Write-Output 'RESULT TUNE error=no_address reason=the_copied_settings_carry_no_payout_address_and_no_wallet.json'; exit 2 }
|
||||
if ($bom -eq '239,187,191') { Write-Output 'RESULT TUNE error=bom reason=settings.json_starts_with_a_BOM'; exit 2 }
|
||||
} else { Write-Output 'RESULT TUNE error=no_settings reason=the_installed_app_has_no_settings.json'; exit 2 }
|
||||
Remove-Item -LiteralPath (Join-Path $sApp 'app.url') -Force -ErrorAction SilentlyContinue
|
||||
|
||||
# the state before, for the report
|
||||
$smi = Join-Path $env:ProgramFiles 'NVIDIA Corporation\NVSMI\nvidia-smi.exe'
|
||||
if (-not (Test-Path $smi)) { $smi = Join-Path $env:SystemRoot 'System32\nvidia-smi.exe' }
|
||||
$tele = Join-Path $bin 'igneum-gpu-telemetry.exe'
|
||||
function Snapshot([string] $tag) {
|
||||
if (Test-Path $smi) {
|
||||
$q = (& $smi --query-gpu=index,name,driver_version,power.draw,power.limit,power.default_limit,power.min_limit,power.max_limit,clocks.gr,clocks.max.gr,clocks.mem --format=csv,noheader 2>&1 | Out-String).Trim()
|
||||
Write-Output ("RESULT TUNE " + $tag + " nvidia " + ($q -replace "`r?`n", ' | '))
|
||||
}
|
||||
if (Test-Path $tele) {
|
||||
$t = (& $tele --tune 2>&1 | Out-String).Trim()
|
||||
Write-Output ("RESULT TUNE " + $tag + " amd " + ($t -replace "`r?`n", ' | '))
|
||||
} else { Write-Output ("RESULT TUNE " + $tag + " amd no_helper") }
|
||||
}
|
||||
Snapshot 'before'
|
||||
|
||||
# the tune engine: status every 10 s (6 rate samples per 60 s hold)
|
||||
$env:IGNEUM_APP_DATA = $root
|
||||
$env:IGNEUM_APP_LOGS = $sLogs
|
||||
$env:IGNEUM_APP_STATUS_SECS = '10'
|
||||
$env:IGNEUM_APP_NO_OTA = '1' # C35: a second engine never runs the updater (the installer would quit the installed app)
|
||||
# C35 (5 October 2026): the engine's output goes to a FILE, never a pipe. A pipe's write end is inherited by every
|
||||
# process the engine starts (its miners and workers), so after an abort the installed app's jobs runner waits for an
|
||||
# EOF that never comes and hangs in its own quit; and the engine's whole tree is killed at the end (nothing orphaned).
|
||||
$outFile = Join-Path $root 'engine-stdout.log'
|
||||
$errFile = Join-Path $root 'engine-stderr.log'
|
||||
Remove-Item -LiteralPath $outFile, $errFile -Force -ErrorAction SilentlyContinue
|
||||
$p = Start-Process -FilePath $exe -ArgumentList '--sweep' -WorkingDirectory (Split-Path $exe) -WindowStyle Hidden -PassThru -RedirectStandardOutput $outFile -RedirectStandardError $errFile
|
||||
Say ("engine started, pid " + $p.Id + ", data " + $root + ", stdout " + $outFile)
|
||||
function EndTree([int] $procId, [string] $why) {
|
||||
$before = @(Get-Process -Name 'igneum-app', 'igneum-miner', 'igneum-worker-cuda', 'igneum-worker-opencl', 'igneum-worker-metal' -ErrorAction SilentlyContinue).Count
|
||||
& taskkill /T /F /PID $procId 2>&1 | Out-Null
|
||||
Start-Sleep -Seconds 2
|
||||
$after = @(Get-Process -Name 'igneum-app', 'igneum-miner', 'igneum-worker-cuda', 'igneum-worker-opencl', 'igneum-worker-metal' -ErrorAction SilentlyContinue).Count
|
||||
Write-Output ('RESULT ' + $resultTag + ' tree ended (' + $why + '): igneum processes ' + $before + ' -> ' + $after + ' (the installed app''s own miners are stopped and held by the job)')
|
||||
}
|
||||
$seen = 0
|
||||
$rows = 0
|
||||
# the watchdog (coordinator, 6 October 2026): a tune engine that mines nothing for 120 s after its first status line
|
||||
# (every card "waiting" or 0.00 MH/s: no payout address, no worker, no node) fails the job at once with the engine's
|
||||
# last log line in the RESULT, its tree ended, mining restored by the job runner; a job that cannot mine never burns
|
||||
# its budget silently again
|
||||
$firstStatusAt = $null
|
||||
$lastMining = $null
|
||||
$lastEngineLine = ''
|
||||
function EngineTail() { $t = Get-ChildItem -Path $sLogs -Filter 'app-*.log' -ErrorAction SilentlyContinue | Sort-Object LastWriteTime -Descending | Select-Object -First 1; if ($t) { $l = Get-Content -LiteralPath $t.FullName -Tail 1 -ErrorAction SilentlyContinue; if ($l) { return [string]$l } }; return '' }
|
||||
while (-not $p.HasExited) {
|
||||
Start-Sleep -Seconds 5
|
||||
$tailLine = EngineTail
|
||||
if ($tailLine) { $lastEngineLine = $tailLine }
|
||||
if ($lastEngineLine -match ' status: ') {
|
||||
if (-not $firstStatusAt) { $firstStatusAt = Get-Date }
|
||||
if ($lastEngineLine -match ', mining \|' -or ($lastEngineLine -match '(\d+\.\d+) MH/s' -and [double]$Matches[1] -gt 0)) { $lastMining = Get-Date }
|
||||
}
|
||||
if ($firstStatusAt -and -not $lastMining -and ((Get-Date) - $firstStatusAt).TotalSeconds -gt 120) {
|
||||
Write-Output ('RESULT TUNE error=not_mining reason=no_card_mined_within_120_s_of_the_first_status_line last_log_line=' + ($lastEngineLine -replace '\s+', '_'))
|
||||
EndTree $p.Id 'watchdog: not mining'
|
||||
Write-Output 'RESULT TUNE error=no_rows'
|
||||
exit 3
|
||||
}
|
||||
if ($lastMining -and ((Get-Date) - $lastMining).TotalSeconds -gt 300) {
|
||||
Write-Output ('RESULT TUNE error=stopped_mining reason=every_card_idle_for_300_s last_log_line=' + ($lastEngineLine -replace '\s+', '_'))
|
||||
EndTree $p.Id 'watchdog: stopped mining'
|
||||
Write-Output 'RESULT TUNE error=no_rows'
|
||||
exit 3
|
||||
}
|
||||
$all = @(); if (Test-Path -LiteralPath $outFile) { $all = @(Get-Content -LiteralPath $outFile -ErrorAction SilentlyContinue) }
|
||||
while ($seen -lt $all.Count) {
|
||||
$l = [string]$all[$seen]; $seen++
|
||||
if ($l -match '^TUNE ') { Write-Output ('RESULT ' + $l); if ($l -match '^TUNE card=') { $rows++ } }
|
||||
elseif ($l -match '^SWEEP ') { Write-Output ('RESULT ' + $l) }
|
||||
elseif ($l -match '^(URL|STATE) ') { }
|
||||
else { Say $l }
|
||||
}
|
||||
if ((Get-Date) -gt $deadline) {
|
||||
Say ("budget of " + $budgetMinutes + " min spent; asking the tune engine to quit")
|
||||
# C35 (5 October 2026): the only quit this script may send goes to the TUNE engine's own URL file in the scratch
|
||||
# root, never to a file under the installed app's folder; the RESULT line names the file it used
|
||||
$u = Join-Path $sApp 'app.url'
|
||||
$installedUrl = Join-Path $appDir 'app.url'
|
||||
if ((Resolve-Path -LiteralPath $u -ErrorAction SilentlyContinue).Path -eq (Resolve-Path -LiteralPath $installedUrl -ErrorAction SilentlyContinue).Path -or $u -like '*\igneum\app\*') {
|
||||
Write-Output ('RESULT TUNE quit refused: ' + $u + ' is the installed app''s URL file')
|
||||
} elseif (Test-Path -LiteralPath $u) {
|
||||
Write-Output ('RESULT TUNE quit asked of the tune engine through ' + $u + ' (pid ' + $p.Id + ')')
|
||||
try { Invoke-WebRequest -Uri ((Get-Content -LiteralPath $u -Raw).Trim() + 'api/quit') -Method POST -Body '{}' -ContentType 'application/json' -UseBasicParsing -TimeoutSec 5 | Out-Null } catch { }
|
||||
} else { Write-Output ('RESULT TUNE quit not sent: no URL file at ' + $u + '; killing pid ' + $p.Id) }
|
||||
Start-Sleep -Seconds 20
|
||||
if (-not $p.HasExited) { EndTree $p.Id 'budget' }
|
||||
Write-Output 'RESULT TUNE error=budget_exceeded'
|
||||
}
|
||||
}
|
||||
$all = @(); if (Test-Path -LiteralPath $outFile) { $all = @(Get-Content -LiteralPath $outFile -ErrorAction SilentlyContinue) }
|
||||
while ($seen -lt $all.Count) { $l = [string]$all[$seen]; $seen++; if ($l -match '^TUNE ') { Write-Output ('RESULT ' + $l); if ($l -match '^TUNE card=') { $rows++ } } }
|
||||
Say ("tune engine exited " + $p.ExitCode + " after " + [int]((Get-Date) - $started).TotalSeconds + " s, " + $rows + " table rows")
|
||||
EndTree $p.Id 'end of run'
|
||||
# the project lead, 6 October 2026, 07:25Z: both cards stay on their best MH/W points (the tune pins them); no factory reset here.
|
||||
Snapshot 'after'
|
||||
# the tune engine's own log: the TUNE lines and what happened around them
|
||||
$log = Get-ChildItem -Path $sLogs -Filter 'app-*.log' -ErrorAction SilentlyContinue | Sort-Object LastWriteTime -Descending | Select-Object -First 1
|
||||
if ($log) {
|
||||
Say ("engine log " + $log.FullName + ":")
|
||||
Get-Content -LiteralPath $log.FullName | Where-Object { $_ -match 'TUNE|tune|power cap|GPUs:|worker ready|STATUS|exited|upload' } | Select-Object -Last 100 | ForEach-Object { Say (' ' + $_) }
|
||||
}
|
||||
if ($rows -eq 0) { Write-Output 'RESULT TUNE error=no_rows'; exit 1 }
|
||||
exit 0
|
||||
|
|
@ -8,6 +8,7 @@
|
|||
# miners restart when the job ends. Elevated, so nvidia-smi -pl needs no prompt (the engine detects that: mode=direct).
|
||||
# UNTESTED on a PC as of 4 Oct 2026 (parse-checked only).
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$resultTag = 'SWEEP'
|
||||
$budgetMinutes = 40
|
||||
$started = Get-Date
|
||||
$deadline = $started.AddMinutes($budgetMinutes)
|
||||
|
|
@ -42,7 +43,7 @@ if (Test-Path $sj) {
|
|||
try {
|
||||
$j = Get-Content -LiteralPath $sj -Raw | ConvertFrom-Json
|
||||
$j.remote_jobs = $false; $j.auto_update = $false; $j.prove = $false; $j.paused = $false; $j.setup_done = $true
|
||||
$j | ConvertTo-Json -Depth 8 | Set-Content -LiteralPath $sj -Encoding utf8
|
||||
[IO.File]::WriteAllText($sj, ($j | ConvertTo-Json -Depth 8), (New-Object System.Text.UTF8Encoding $false)) # no BOM: the engine's JSON parser refuses one (C35, runs 1 and 2)
|
||||
} catch { Say ("settings.json: " + $_.Exception.Message) }
|
||||
} else { Write-Output 'RESULT SWEEP error=no_settings reason=the_installed_app_has_no_settings.json'; exit 2 }
|
||||
Remove-Item -LiteralPath (Join-Path $sApp 'app.url') -Force -ErrorAction SilentlyContinue
|
||||
|
|
@ -59,29 +60,29 @@ if (Test-Path $smi) {
|
|||
$env:IGNEUM_APP_DATA = $root
|
||||
$env:IGNEUM_APP_LOGS = $sLogs
|
||||
$env:IGNEUM_APP_STATUS_SECS = '10'
|
||||
$psi = New-Object System.Diagnostics.ProcessStartInfo
|
||||
$psi.FileName = $exe
|
||||
$psi.Arguments = '--sweep'
|
||||
$psi.WorkingDirectory = Split-Path $exe
|
||||
$psi.UseShellExecute = $false
|
||||
$psi.RedirectStandardOutput = $true
|
||||
$psi.RedirectStandardError = $true
|
||||
$psi.CreateNoWindow = $true
|
||||
$p = New-Object System.Diagnostics.Process
|
||||
$p.StartInfo = $psi
|
||||
$lines = New-Object System.Collections.ArrayList
|
||||
$h = { if ($EventArgs.Data) { [void]$Event.MessageData.Add($EventArgs.Data) } }
|
||||
Register-ObjectEvent -InputObject $p -EventName OutputDataReceived -Action $h -MessageData $lines | Out-Null
|
||||
Register-ObjectEvent -InputObject $p -EventName ErrorDataReceived -Action $h -MessageData $lines | Out-Null
|
||||
[void]$p.Start()
|
||||
$p.BeginOutputReadLine(); $p.BeginErrorReadLine()
|
||||
Say ("sweep engine started, pid " + $p.Id + ", data " + $root)
|
||||
$env:IGNEUM_APP_NO_OTA = '1' # C35: a second engine never runs the updater (the installer would quit the installed app)
|
||||
# C35 (5 October 2026): the engine's output goes to a FILE, never a pipe. A pipe's write end is inherited by every
|
||||
# process the engine starts (its miners and workers), so after an abort the installed app's jobs runner waits for an
|
||||
# EOF that never comes and hangs in its own quit; and the engine's whole tree is killed at the end (nothing orphaned).
|
||||
$outFile = Join-Path $root 'engine-stdout.log'
|
||||
$errFile = Join-Path $root 'engine-stderr.log'
|
||||
Remove-Item -LiteralPath $outFile, $errFile -Force -ErrorAction SilentlyContinue
|
||||
$p = Start-Process -FilePath $exe -ArgumentList '--sweep' -WorkingDirectory (Split-Path $exe) -WindowStyle Hidden -PassThru -RedirectStandardOutput $outFile -RedirectStandardError $errFile
|
||||
Say ("engine started, pid " + $p.Id + ", data " + $root + ", stdout " + $outFile)
|
||||
function EndTree([int] $procId, [string] $why) {
|
||||
$before = @(Get-Process -Name 'igneum-app', 'igneum-miner', 'igneum-worker-cuda', 'igneum-worker-opencl', 'igneum-worker-metal' -ErrorAction SilentlyContinue).Count
|
||||
& taskkill /T /F /PID $procId 2>&1 | Out-Null
|
||||
Start-Sleep -Seconds 2
|
||||
$after = @(Get-Process -Name 'igneum-app', 'igneum-miner', 'igneum-worker-cuda', 'igneum-worker-opencl', 'igneum-worker-metal' -ErrorAction SilentlyContinue).Count
|
||||
Write-Output ('RESULT ' + $resultTag + ' tree ended (' + $why + '): igneum processes ' + $before + ' -> ' + $after + ' (the installed app''s own miners are stopped and held by the job)')
|
||||
}
|
||||
$seen = 0
|
||||
$rows = 0
|
||||
while (-not $p.HasExited) {
|
||||
Start-Sleep -Seconds 5
|
||||
while ($seen -lt $lines.Count) {
|
||||
$l = [string]$lines[$seen]; $seen++
|
||||
$all = @(); if (Test-Path -LiteralPath $outFile) { $all = @(Get-Content -LiteralPath $outFile -ErrorAction SilentlyContinue) }
|
||||
while ($seen -lt $all.Count) {
|
||||
$l = [string]$all[$seen]; $seen++
|
||||
if ($l -match '^SWEEP ') { Write-Output ('RESULT ' + $l); if ($l -match '^SWEEP card=') { $rows++ } }
|
||||
elseif ($l -match '^(URL|STATE) ') { }
|
||||
else { Say $l }
|
||||
|
|
@ -91,12 +92,14 @@ while (-not $p.HasExited) {
|
|||
$u = Join-Path $sApp 'app.url'
|
||||
if (Test-Path $u) { try { Invoke-WebRequest -Uri ((Get-Content -LiteralPath $u -Raw).Trim() + 'api/quit') -Method POST -Body '{}' -ContentType 'application/json' -UseBasicParsing -TimeoutSec 5 | Out-Null } catch { } }
|
||||
Start-Sleep -Seconds 20
|
||||
if (-not $p.HasExited) { $p.Kill() }
|
||||
if (-not $p.HasExited) { EndTree $p.Id 'budget' }
|
||||
Write-Output 'RESULT SWEEP error=budget_exceeded'
|
||||
}
|
||||
}
|
||||
while ($seen -lt $lines.Count) { $l = [string]$lines[$seen]; $seen++; if ($l -match '^SWEEP ') { Write-Output ('RESULT ' + $l); if ($l -match '^SWEEP card=') { $rows++ } } }
|
||||
$all = @(); if (Test-Path -LiteralPath $outFile) { $all = @(Get-Content -LiteralPath $outFile -ErrorAction SilentlyContinue) }
|
||||
while ($seen -lt $all.Count) { $l = [string]$all[$seen]; $seen++; if ($l -match '^SWEEP ') { Write-Output ('RESULT ' + $l); if ($l -match '^SWEEP card=') { $rows++ } } }
|
||||
Say ("sweep engine exited " + $p.ExitCode + " after " + [int]((Get-Date) - $started).TotalSeconds + " s, " + $rows + " table rows")
|
||||
EndTree $p.Id 'end of run'
|
||||
if (Test-Path $smi) {
|
||||
$q = (& $smi --query-gpu=index,power.draw,power.limit --format=csv,noheader 2>&1 | Out-String).Trim()
|
||||
Write-Output ("RESULT SWEEP after " + ($q -replace "`r?`n", ' | '))
|
||||
|
|
|
|||
110
relay/test/ember.test.mjs
Normal file
110
relay/test/ember.test.mjs
Normal file
|
|
@ -0,0 +1,110 @@
|
|||
// node --test relay/test/ember.test.mjs (no dependencies; CI runs it in the site job)
|
||||
// The fleet aggregation of Ember Tune records (relay/lib/ember.mjs) on a fixture of records in the shape
|
||||
// app/igneum-app/src/ember.rs record_json writes: known-good (five samples converge on one point), known-bad (an
|
||||
// outlier does not move the median), the de-duplication of re-sent logs, the manifest merge and the prior lookup.
|
||||
import { test } from 'node:test';
|
||||
import assert from 'node:assert/strict';
|
||||
import { parseRecords, dedupe, aggregate, mergeTuning, priorFor, priorLine, median, spreadPct } from '../lib/ember.mjs';
|
||||
|
||||
const step = (clock_mhz, power_pct, watts, mhs, mark = 'ok') => ({ clock_mhz, power_pct, limit_w: 575 * power_pct / 100, watts, mhs, eff: Number((mhs / watts).toFixed(4)), gclk: clock_mhz || 2800, mclk: 10500, tmax: 68, faults: 0, mark });
|
||||
const rec = (machine, ts, chosen, over = {}) => ({
|
||||
ts, machine, app: '0.3.10', os: 'windows', card: 'NVIDIA_GeForce_RTX_5090', vendor: 'nvidia', driver: '581.57', driver_major: '581', class: 'l128w16',
|
||||
key: 'NVIDIA_GeForce_RTX_5090|581|l128w16', plan: 'full', steps: [step(0, 100, 290, 124.0), chosen], chosen, before: step(0, 100, 290, 124.0), eff: chosen.eff, mhs: chosen.mhs, watts: chosen.watts, ...over,
|
||||
});
|
||||
// five machines, each landing near 2,470 MHz at 100%: 0.55 to 0.57 MH/W
|
||||
const good = [
|
||||
rec('a1', 1000, step(2472, 100, 220, 123.5)),
|
||||
rec('b2', 1001, step(2472, 100, 222, 123.1)),
|
||||
rec('c3', 1002, step(2781, 100, 236, 123.8)),
|
||||
rec('d4', 1003, step(2472, 100, 218, 123.6)),
|
||||
rec('e5', 1004, step(2163, 100, 212, 122.9)),
|
||||
];
|
||||
const outlier = rec('f6', 1005, step(1854, 50, 130, 118.0)); // 0.908 MH/W: a card with a broken draw reading
|
||||
const baseline = rec('g7', 1006, step(0, 80, 290, 122.3), { plan: 'baseline', steps: [step(0, 80, 290, 122.3)], before: null });
|
||||
const amd = (machine, ts, chosen) => rec(machine, ts, chosen, { card: 'AMD_Radeon_RX_9070_XT', vendor: 'amd', driver: '32.0.15801.1', driver_major: '32', key: 'AMD_Radeon_RX_9070_XT|32|l128w16', plan: 'confirm', before: null });
|
||||
|
||||
test('five samples converge on the median point and an outlier does not move it', () => {
|
||||
const { priors, table } = aggregate(good, { minSamples: 5 });
|
||||
const p = priors['NVIDIA_GeForce_RTX_5090|581|l128w16'];
|
||||
assert.ok(p, 'a prior at the sample floor');
|
||||
assert.equal(p.clock_mhz, 2470, 'the median clock cap, rounded to 10 MHz');
|
||||
assert.equal(p.power_pct, 100);
|
||||
assert.equal(p.samples, 5);
|
||||
assert.equal(p.machines, 5);
|
||||
assert.ok(p.eff > 0.55 && p.eff < 0.57, `eff ${p.eff}`);
|
||||
assert.ok(p.spread_pct >= 0 && p.spread_pct < 3, `spread ${p.spread_pct}`);
|
||||
assert.equal(p.before_eff, Number((124 / 290).toFixed(4)));
|
||||
assert.ok(p.gain_pct > 25, `gain ${p.gain_pct}% over the untuned 100% point`);
|
||||
assert.equal(p.updated, '1970-01-01T00:16:44Z');
|
||||
// the outlier: 0.908 MH/W at 1,854 MHz joins; the median moves by one rank at most
|
||||
const with6 = aggregate(good.concat(outlier), { minSamples: 5 }).priors['NVIDIA_GeForce_RTX_5090|581|l128w16'];
|
||||
assert.equal(with6.samples, 6);
|
||||
assert.equal(with6.clock_mhz, 2470);
|
||||
assert.equal(with6.power_pct, 100);
|
||||
assert.ok(with6.eff < 0.58, `the outlier's 0.908 MH/W did not drag the median: ${with6.eff}`);
|
||||
assert.ok(with6.spread_pct < 5, `spread ${with6.spread_pct}`);
|
||||
assert.equal(table[0].key, 'NVIDIA_GeForce_RTX_5090|581|l128w16');
|
||||
});
|
||||
|
||||
test('under the floor there is no prior, and baseline records never make one', () => {
|
||||
const { priors, table } = aggregate(good.slice(0, 4), { minSamples: 5 });
|
||||
assert.deepEqual(priors, {});
|
||||
assert.equal(table[0].samples, 4, 'the table still shows the count');
|
||||
const b = aggregate([baseline, baseline], { minSamples: 1 });
|
||||
assert.deepEqual(b.priors, {}, 'measure-only records say what a card does, never what to set');
|
||||
assert.equal(b.table[0].baseline_samples, 1, 'the duplicate upload counted once');
|
||||
assert.equal(b.table[0].baseline_eff, Number((122.3 / 290).toFixed(4)));
|
||||
// a marked chosen step (a faulted or hot winner cannot exist, but a record with one is ignored)
|
||||
const bad = rec('h8', 1007, step(2000, 100, 200, 120, 'faulted'));
|
||||
assert.deepEqual(aggregate([bad], { minSamples: 1 }).priors, {});
|
||||
});
|
||||
|
||||
test('records are parsed out of log text and de-duplicated on machine, card and time', () => {
|
||||
const line = `1791230000 TUNE ${JSON.stringify(good[0])}`;
|
||||
const text = ['1791229999 status: x', line, line, `1791230001 TUNE ${JSON.stringify(good[1])}`, '1791230002 TUNE {not json'].join('\n');
|
||||
const rs = parseRecords(text);
|
||||
assert.equal(rs.length, 3);
|
||||
assert.equal(dedupe(rs).length, 2);
|
||||
assert.equal(parseRecords('').length, 0);
|
||||
});
|
||||
|
||||
test('the manifest merge keeps the kernel-variant cards and carries the settings', () => {
|
||||
const existing = { updated: '2026-10-04T21:00:00Z', window_days: 7, cards: { NVIDIA_GeForce_RTX_5090: { variant: 'u2-ldg', race: true, candidates: ['u2-ldg', 'ldg', 'base'] } }, priors: { 'old|1|v2': { samples: 9 } } };
|
||||
const { priors } = aggregate(good, { minSamples: 5 });
|
||||
const t = mergeTuning(existing, priors, { rate_tolerance_pct: 1 });
|
||||
assert.equal(t.cards.NVIDIA_GeForce_RTX_5090.variant, 'u2-ldg', 'lever 2 untouched');
|
||||
assert.equal(t.window_days, 7);
|
||||
assert.deepEqual(t.ember, { enabled: true, min_samples: 5, rate_tolerance_pct: 1 });
|
||||
assert.ok(!t.priors['old|1|v2'], 'a key without samples in the window drops out');
|
||||
assert.ok(t.priors['NVIDIA_GeForce_RTX_5090|581|l128w16']);
|
||||
// the kill switch rides the same section
|
||||
assert.equal(mergeTuning(existing, {}, { enabled: false }).ember.enabled, false);
|
||||
assert.deepEqual(mergeTuning(null, {}).cards, {});
|
||||
// the round trip: canonical JSON (what publish-manifest.sh signs) parses back to the same prior
|
||||
const back = JSON.parse(JSON.stringify(t));
|
||||
assert.deepEqual(priorFor(back, 'NVIDIA_GeForce_RTX_5090|581|l128w16'), { clock_mhz: 2470, power_pct: 100, eff: priors['NVIDIA_GeForce_RTX_5090|581|l128w16'].eff, samples: 5 });
|
||||
assert.equal(priorFor(back, 'NVIDIA_GeForce_RTX_5090|581|l128w16', 6), null, 'six wanted, five there');
|
||||
assert.equal(priorFor(back, 'nothing|0|v2'), null);
|
||||
assert.equal(priorFor(null, 'x'), null);
|
||||
});
|
||||
|
||||
test('AMD confirm records aggregate by their own key, and the line reads', () => {
|
||||
const rs = [amd('p1', 2000, step(2600, 90, 177, 17.7)), amd('p2', 2001, step(2600, 90, 180, 17.6)), amd('p3', 2002, step(2500, 90, 170, 17.4))];
|
||||
const { priors, table } = aggregate(rs, { minSamples: 3 });
|
||||
const p = priors['AMD_Radeon_RX_9070_XT|32|l128w16'];
|
||||
assert.equal(p.clock_mhz, 2600);
|
||||
assert.equal(p.power_pct, 90);
|
||||
assert.equal(p.vendor, 'amd');
|
||||
assert.equal(p.gain_pct, undefined, 'confirm records carry no before step and no baseline was uploaded');
|
||||
assert.match(priorLine(p), /^AMD Radeon RX 9070 XT \| driver 32 \| l128w16: 2600 MHz at 90%, 0\.\d+ MH\/W, 17\.6 MH\/s at 177 W, spread \d+(\.\d+)?%, 3 sample\(s\) from 3 machine\(s\)$/);
|
||||
assert.equal(table.length, 1);
|
||||
});
|
||||
|
||||
test('median and spread', () => {
|
||||
assert.equal(median([3, 1, 2]), 2);
|
||||
assert.equal(median([4, 1, 2, 3]), 2.5);
|
||||
assert.equal(median([]), 0);
|
||||
assert.equal(spreadPct([1, 1, 1]), 0);
|
||||
assert.equal(spreadPct([10]), 0);
|
||||
assert.equal(spreadPct([9, 10, 11]), 10);
|
||||
});
|
||||
|
|
@ -340,6 +340,21 @@ for (const [file, active] of PAGES) {
|
|||
];
|
||||
const table = '<div class="tbl"><table><thead><tr>' + ['Card', 'Generator', 'Best MH/s', 'MH per watt', 'Miner', 'Date', 'Source', 'Who measured it'].map(h => `<th>${h}</th>`).join('') + '</tr></thead><tbody>' +
|
||||
rows.map(r => '<tr>' + cell(r).map(c => `<td>${esc(String(c))}</td>`).join('') + '</tr>').join('') + '</tbody></table></div>';
|
||||
// Ember Tune's fleet priors (site/miner-priors.json, tools/tuning.mjs --priors --site): one row per card model,
|
||||
// driver major and program class; a row under the sample floor shows its count and no point
|
||||
const pj = JSON.parse(readFileSync(join(here, 'miner-priors.json'), 'utf8'));
|
||||
const prows = (pj.rows || []).slice().sort((a, b) => (b.samples - a.samples) || (a.card < b.card ? -1 : 1));
|
||||
const pcell = r => [
|
||||
r.card, r.driver_major, r.class, r.samples + (r.machines ? ' from ' + r.machines + ' machine' + (r.machines === 1 ? '' : 's') : ''),
|
||||
r.prior ? (r.clock_mhz ? fmt(r.clock_mhz) + ' MHz at ' + r.power_pct + '%' : r.power_pct + '%, clock unlocked') : 'under the floor (' + pj.min_samples + ' needed)',
|
||||
r.mh_per_w == null ? 'not yet' : Number(r.mh_per_w).toFixed(3) + (r.spread_pct != null ? ' (spread ' + r.spread_pct + '%)' : ''),
|
||||
r.mh_s == null ? '' : fmt(r.mh_s) + ' MH/s at ' + fmt(r.watts) + ' W',
|
||||
r.untuned_mh_per_w == null ? 'not measured' : Number(r.untuned_mh_per_w).toFixed(3) + (r.gain_pct != null ? ' (' + (r.gain_pct >= 0 ? '+' : '') + r.gain_pct + '%)' : ''),
|
||||
];
|
||||
const ptable = prows.length
|
||||
? '<div class="tbl"><table><thead><tr>' + ['Card', 'Driver', 'Program class', 'Samples', 'Tuned point', 'MH per watt', 'Rate and draw', 'Untuned MH per watt (gain)'].map(h => `<th>${h}</th>`).join('') + '</tr></thead><tbody>' +
|
||||
prows.map(r => '<tr>' + pcell(r).map(c => `<td>${esc(String(c))}</td>`).join('') + '</tr>').join('') + '</tbody></table></div>'
|
||||
: '<p>No tune reports yet. The first rows appear once five machines with the same card model have reported.</p>';
|
||||
const body = scrubBench([
|
||||
'<h2 id="table">The table</h2>',
|
||||
'<p>One row per card, generator version and miner version. The rate is the best one measured. Integrated GPUs are not listed. Prototype rows are bench numbers from before the devnet and say so in the miner column.</p>',
|
||||
|
|
@ -349,8 +364,12 @@ for (const [file, active] of PAGES) {
|
|||
'<p>MH per watt needs the card\'s power draw during the run. The app reads it on NVIDIA cards through the driver. Rows get the figure when a run records it.</p>',
|
||||
'<p>There is no other Igneum miner to compare with yet, so this table compares cards, not miners. The app that produces these rows: <a href="/miner">the miner page</a>.</p>',
|
||||
`<p>Rows: ${rows.length}. Source file: <code>site/miner-bench.json</code> in the repository.</p>`,
|
||||
'<h2 id="priors">Fleet tuning priors</h2>',
|
||||
'<p>Ember Tune runs on every card the app mines with: the power limit and the core clock are stepped on the live program and the card keeps the point with the best MH per watt within 1% of its top rate. Every finished tune is reported back without anything that identifies the owner, and the fleet\'s median point per card model, driver major and program class comes back down inside the signed update manifest as the starting point for the next card of that model. A model needs ' + pj.min_samples + ' reports before its prior is used.</p>',
|
||||
ptable,
|
||||
`<p>Rows: ${prows.length}${pj.generated ? ', generated ' + pj.generated : ''}. Source file: <code>site/miner-priors.json</code> in the repository, written from the fleet records by <code>tools/tuning.mjs --priors --site</code>.</p>`,
|
||||
].join('\n'));
|
||||
const toc = [{ lvl: 2, t: 'The table', id: 'table' }, { lvl: 2, t: 'How a row gets here', id: 'how' }];
|
||||
const toc = [{ lvl: 2, t: 'The table', id: 'table' }, { lvl: 2, t: 'How a row gets here', id: 'how' }, { lvl: 2, t: 'Fleet tuning priors', id: 'priors' }];
|
||||
writeFileSync(join(here, 'miners.html'), page('Igneum GPU bench table', 'Measured Igneum hash rates per GPU: card, generator version, best MH/s, MH per watt where measured, miner version, date and the log entry each number came from.', body, toc,
|
||||
'Measured hash rates per card on the Igneum lottery hash, with the generator version, the miner version, the date and the log entry behind each number.',
|
||||
{ path: '/miners', heading: 'GPU bench table', active: 'miner' }));
|
||||
|
|
|
|||
6
site/miner-priors.json
Normal file
6
site/miner-priors.json
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
{
|
||||
"_about": "Rows of the fleet priors table at /miners (site/build.mjs), written by tools/tuning.mjs --priors --site from the TUNE records every Igneum Miner uploads. One row per card model, driver major and program class: the median tuned point, MH per watt, the spread and the sample count. No machine names, no addresses.",
|
||||
"generated": null,
|
||||
"min_samples": 5,
|
||||
"rows": []
|
||||
}
|
||||
|
|
@ -353,7 +353,7 @@ pre b{color:var(--molten);font-weight:500}
|
|||
<div class="feat"><b>Variant racing every hour</b><p>At every hourly prepare the worker compiles the program in several shapes (unroll, load path, register budget, threads per group), times each for 2 seconds and keeps the fastest for the hour. Base keeps its place unless beaten.</p><a class="src" href="/bench#4-october-2026-miner-performance-variant-racing-metal-worker-on-the-m5-max-the-rtx-5090-job-is-ready-not-run">log · 4 Oct 2026</a></div>
|
||||
<div class="feat"><b>Measured on Apple silicon</b><p>On the M5 Max, 256-thread groups ran 17.3% and 21.2% faster than base on two programs, under load from other work. The NVIDIA race is built and has not yet run on a GPU.</p><a class="src" href="/bench#4-october-2026-miner-performance-variant-racing-metal-worker-on-the-m5-max-the-rtx-5090-job-is-ready-not-run">log · 4 Oct 2026, ratios under contention</a></div>
|
||||
<div class="feat"><b>Fleet tuning manifest</b><p>Every race is one record in the app log. The fleet's best variant per card model goes back out inside the signed update manifest, so a card starts from the known best and keeps racing.</p></div>
|
||||
<div class="feat"><b>Efficiency mode</b><p>Hash per watt, the number miners compare. The app steps an NVIDIA card's power cap from 100% to 50%, holds each step for 60 seconds on the live kernel and leaves the cap at the best MH per watt. Built and unit-tested; not yet run on a card.</p></div>
|
||||
<div class="feat"><b>Ember Tune</b><p>Hash per watt, the number miners compare. Out of the box the app steps every card's power limit and core clock on the live kernel, 60 seconds a step, and keeps the point with the best MH per watt within 1% of the card's top rate. The memory clock is never touched; a step with a rejected hash, a hot GPU or a dragged memory clock is reverted and marked. Every result feeds a fleet prior per card model that the next card of that model starts from. Built and unit-tested; the first measured tune is owed.</p></div>
|
||||
<div class="feat"><b>Latency work</b><p>A block built on a stale tip earns less. The miner is moving from polling to a template subscription and shorter jobs, so a new tip reaches the card in milliseconds. In progress, no number published yet.</p></div>
|
||||
<div class="feat"><b>Remote signed jobs, opt in</b><p>Our own fleet only. A switch, "Allow remote jobs from Igneum (signed)", with the key's fingerprint beside it. Off aborts the running job and stops polling. Jobs run at most once each and only on the machines they name.</p></div>
|
||||
</div>
|
||||
|
|
@ -403,8 +403,8 @@ pre b{color:var(--molten);font-weight:500}
|
|||
<thead><tr><th>Lever</th><th>The idea</th><th>Measured state</th></tr></thead>
|
||||
<tbody>
|
||||
<tr><td><b>1 · Race the compiler every hour</b></td><td>For each new program, five to ten kernel variants are compiled, benched for two seconds each, and the winner is kept for the hour.</td><td>Measured on the Mac's Metal worker: the winning variant +17.3% on the genesis seed and +21.2% on the hourly seed over the base compile, under load from other work. The RTX 5090 race is prepared and not yet run. <a href="/bench#4-october-2026-miner-performance-variant-racing-metal-worker-on-the-m5-max-the-rtx-5090-job-is-ready-not-run">Log, 4 Oct 2026</a></td></tr>
|
||||
<tr><td><b>2 · Auto-tune that learns from the fleet</b></td><td>Apps report the race per card and program class. The best settings come back down through the signed update manifest as defaults, so the miner gets faster for everyone as the fleet grows.</td><td>Shipped. The fleet is still small, so no fleet table yet.</td></tr>
|
||||
<tr><td><b>3 · Hash per watt, not hash</b></td><td>A sweep per card finds the power point with the best MH per watt and holds it. Miners pay for electricity; that is the number they compare.</td><td>Shipped for NVIDIA cards through the power cap. Not yet run on a card; the first sweep on a 5090 is the owed measurement. Clocks are the next lever.</td></tr>
|
||||
<tr><td><b>2 · Auto-tune that learns from the fleet</b></td><td>Apps report the race per card and program class, and every finished Ember Tune. The best settings come back down through the signed update manifest as defaults, so the miner gets faster and more efficient for everyone as the fleet grows.</td><td>Shipped. The fleet is still small, so no fleet table yet; the priors table on <a href="/miners#priors">/miners</a> fills as machines report.</td></tr>
|
||||
<tr><td><b>3 · Hash per watt, not hash</b></td><td>Ember Tune: two knobs per card (power limit, core clock), the memory clock held, the point with the best MH per watt within 1% of the top rate kept and pinned. Miners pay for electricity; that is the number they compare.</td><td>Built for NVIDIA (through the driver, with Power control on) and AMD (through the app's own helper, no administrator rights), measure only on Apple silicon. Unit-tested on every rule; the first tune measured on a card is the owed number.</td></tr>
|
||||
<tr><td><b>4 · Template latency</b></td><td>Solo against the local node, a new template within 50 ms of a new tip, because a late block on a BlockDAG goes red and earns nothing.</td><td>Approximate, from the 0.3.6 release plan and not yet in the engineering log: switched p50 46 to 52 ms on a three-node CPU run, 5 Oct 2026. Ships in 0.3.6.</td></tr>
|
||||
<tr><td><b>5 · Never lose a second</b></td><td>Zero-loss hourly program swaps, fault guards, automatic restart, a CPU re-check of every found hash, per-worker health.</td><td>Shipped. Swap 0.01 ms on Metal and 0.00 ms on CUDA with 0 rejected blocks; recoveries in the <a href="#reliability">table above</a>. <a href="/bench#4-october-2026-first-hourly-program-swap-on-the-live-devnet-compile-ahead-no-pause-two-cards">Log, 4 Oct 2026</a></td></tr>
|
||||
<tr><td><b>6 · Prove it in public</b></td><td>The bench table per card, fed from the job channel. We claim fastest only when the table says so.</td><td>Live at <a href="/miners">/miners</a>, one row per card, generator version and miner version, each with its log entry.</td></tr>
|
||||
|
|
|
|||
|
|
@ -172,12 +172,12 @@ th{font-family:var(--f-mono);font-size:12px;letter-spacing:.12em;text-transform:
|
|||
<!-- nav:end -->
|
||||
<main id="main" class="wrap">
|
||||
<div class="head">
|
||||
<div class="eyebrow">2 entries, newest at the bottom</div>
|
||||
<div class="eyebrow">3 entries, newest at the bottom</div>
|
||||
<h1>GPU bench table</h1>
|
||||
<p class="note">Measured hash rates per card on the Igneum lottery hash, with the generator version, the miner version, the date and the log entry behind each number.</p>
|
||||
</div>
|
||||
<div class="layout">
|
||||
<nav class="toc" aria-label="Contents"><div class="lbl">Contents</div><ol><li><a href="#table">The table</a></li><li><a href="#how">How a row gets here</a></li></ol></nav>
|
||||
<nav class="toc" aria-label="Contents"><div class="lbl">Contents</div><ol><li><a href="#table">The table</a></li><li><a href="#how">How a row gets here</a></li><li><a href="#priors">Fleet tuning priors</a></li></ol></nav>
|
||||
<article><h2 id="table">The table</h2>
|
||||
<p>One row per card, generator version and miner version. The rate is the best one measured. Integrated GPUs are not listed. Prototype rows are bench numbers from before the devnet and say so in the miner column.</p>
|
||||
<div class="tbl"><table><thead><tr><th>Card</th><th>Generator</th><th>Best MH/s</th><th>MH per watt</th><th>Miner</th><th>Date</th><th>Source</th><th>Who measured it</th></tr></thead><tbody><tr><td>Apple M5 Max (40 GPU cores, Metal)</td><td>v1</td><td>45.2</td><td>not measured</td><td>proto-metal bench (prototype, not mining)</td><td>2026-10-03</td><td>bench log: 3 October 2026, RTX 5090 first run (the Apple row of the same table)</td><td>measured by the team. genesis program, 1 GiB dataset</td></tr><tr><td>Apple M5 Max (40 GPU cores, Metal)</td><td>v2</td><td>26.7</td><td>not measured</td><td>igneum-miner devnet v4, Metal worker with prepare</td><td>2026-10-04</td><td>bench log: 4 October 2026, first hourly program swap on the live devnet: compile-ahead, no pause, two cards</td><td>measured by the team. live devnet v4, unbroken through the hour boundary</td></tr><tr><td>Apple silicon laptop (model not reported)</td><td>v2</td><td>24.3</td><td>not measured</td><td>Igneum Miner 0.3.1 (DMG)</td><td>2026-10-04</td><td>bench log: 4 October 2026, first outside machine on the devnet: an Apple silicon laptop through the Igneum Miner app</td><td>reported by the fleet. 21.0 MH/s average over 7 minutes, 24.3 MH/s at the moment of the report, 33 accepted blocks</td></tr><tr><td>NVIDIA RTX 5090 (32 GB)</td><td>v1</td><td>229</td><td>not measured</td><td>proto-cuda bench (prototype, not mining)</td><td>2026-10-03</td><td>bench log: 3 October 2026, RTX 5090, memory-hard dataset (pack igneum-genesis-mh)</td><td>measured by the team. genesis program, 104 loads per hash, 1 GiB dataset</td></tr><tr><td>NVIDIA RTX 5090 (32 GB)</td><td>v1</td><td>185.3</td><td>not measured</td><td>proto-cuda bench (prototype, not mining)</td><td>2026-10-03</td><td>bench log: 3 October 2026, RTX 5090 first run, dataset sweep and second program</td><td>measured by the team. hourly program, 128 loads per hash, 1 GiB dataset</td></tr><tr><td>NVIDIA RTX 5090 (32 GB)</td><td>v2</td><td>124.2</td><td>not measured</td><td>Igneum Miner 0.3.0 package, prebuilt NVRTC worker</td><td>2026-10-04</td><td>bench log: 4 October 2026, the gfx1036 worker fault and what the Apple M5 Max could and could not reproduce</td><td>measured by the team. live devnet v4, 128 loads per hash, CPU re-check clean, 0 rejected</td></tr></tbody></table></div>
|
||||
|
|
@ -185,7 +185,11 @@ th{font-family:var(--f-mono);font-size:12px;letter-spacing:.12em;text-transform:
|
|||
<p>Every row names the engineering log entry or the job it came from. "Measured by the team" means our own hardware and our own log. "Reported by the fleet" means a machine we do not own, read from the status lines its miner uploads.</p>
|
||||
<p>MH per watt needs the card's power draw during the run. The app reads it on NVIDIA cards through the driver. Rows get the figure when a run records it.</p>
|
||||
<p>There is no other Igneum miner to compare with yet, so this table compares cards, not miners. The app that produces these rows: <a href="/miner">the miner page</a>.</p>
|
||||
<p>Rows: 6. Source file: <code>site/miner-bench.json</code> in the repository.</p></article>
|
||||
<p>Rows: 6. Source file: <code>site/miner-bench.json</code> in the repository.</p>
|
||||
<h2 id="priors">Fleet tuning priors</h2>
|
||||
<p>Ember Tune runs on every card the app mines with: the power limit and the core clock are stepped on the live program and the card keeps the point with the best MH per watt within 1% of its top rate. Every finished tune is reported back without anything that identifies the owner, and the fleet's median point per card model, driver major and program class comes back down inside the signed update manifest as the starting point for the next card of that model. A model needs 5 reports before its prior is used.</p>
|
||||
<p>No tune reports yet. The first rows appear once five machines with the same card model have reported.</p>
|
||||
<p>Rows: 0. Source file: <code>site/miner-priors.json</code> in the repository, written from the fleet records by <code>tools/tuning.mjs --priors --site</code>.</p></article>
|
||||
</div>
|
||||
<p class="gen">Generated from the repository at build time. Times are UTC. Machine names are model names.</p>
|
||||
</main>
|
||||
|
|
|
|||
31
tools/ci/second-engine-check.sh
Executable file
31
tools/ci/second-engine-check.sh
Executable file
|
|
@ -0,0 +1,31 @@
|
|||
#!/usr/bin/env bash
|
||||
# The second-engine class (C35, 5 October 2026, PC 1 22:31 UTC): a playbook started a second Igneum engine beside the
|
||||
# installed app through a redirected PIPE (PowerShell's Process.Start with RedirectStandardOutput). A pipe's write end is
|
||||
# inherited by every process the engine starts (its miners and workers); when the job was aborted, the installed app's
|
||||
# jobs runner waited for an EOF the orphaned grandchildren never sent and its quit hung for 24 minutes, and the second
|
||||
# engine's miners mined on against the relaunched app. Rule for every playbook that starts an engine: (1) the engine's
|
||||
# output goes to a FILE (Start-Process -RedirectStandardOutput <file>), never a pipe into the script; (2) the engine's
|
||||
# whole process tree is ended at the end and on the budget (taskkill /T /F), so nothing is orphaned; the installed
|
||||
# app's miners come back only after that (the job runner restarts them when the script ends). This check fails CI when
|
||||
# a playbook starts an engine without both, or without IGNEUM_APP_NO_OTA = '1' (the third line, same night: the second
|
||||
# engine's updater found itself under min_supported_version and ran the installer, which quit the installed app).
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")/../.."
|
||||
fail=0
|
||||
while IFS= read -r f; do
|
||||
grep -qE "igneum-app(\.exe)?['\"]? *(--sweep|-ArgumentList '--sweep'|--no-open)|ArgumentList '--sweep'|\.Arguments = '--sweep'" "$f" || continue
|
||||
if grep -qE 'RedirectStandardOutput *= *\$true|UseShellExecute *= *\$false|Register-ObjectEvent|BeginOutputReadLine' "$f"; then
|
||||
echo "second-engine: $f starts an engine through a pipe (RedirectStandardOutput/BeginOutputReadLine); use Start-Process -RedirectStandardOutput <file>"; fail=1
|
||||
fi
|
||||
if ! grep -qE 'taskkill /T /F' "$f"; then
|
||||
echo "second-engine: $f starts an engine without ending its process tree (taskkill /T /F) at the end"; fail=1
|
||||
fi
|
||||
if grep -qE 'settings\.json|\.json' "$f" && grep -vE '^\s*#' "$f" | grep -qE 'Set-Content[^\n]*-Encoding +utf8'; then
|
||||
echo "second-engine: $f writes JSON with Set-Content -Encoding utf8 (a BOM the engine refuses: the copy read as defaults, no payout address, nothing mined); use [IO.File]::WriteAllText with UTF8Encoding(\$false)"; fail=1
|
||||
fi
|
||||
if ! grep -qE "IGNEUM_APP_NO_OTA *= *'1'" "$f"; then
|
||||
echo "second-engine: $f starts an engine without IGNEUM_APP_NO_OTA = '1' (its updater would run the installer, which quits the installed app: PC 1, 5 October 2026, 22:31 UTC)"; fail=1
|
||||
fi
|
||||
done < <(git ls-files 'relay/playbooks/**' 'tools/windows/**' 'packaging/**' | grep -E '\.ps1$')
|
||||
[ "$fail" = 0 ] && echo "second-engine: every playbook that starts an engine logs to a file and ends its tree"
|
||||
exit $fail
|
||||
106
tools/ci/windows-spawn-check.mjs
Normal file
106
tools/ci/windows-spawn-check.mjs
Normal file
|
|
@ -0,0 +1,106 @@
|
|||
#!/usr/bin/env node
|
||||
// The console-window class (5 October 2026, PC 1): a child the app starts on Windows without CREATE_NO_WINDOW, or an
|
||||
// elevated child started without SW_HIDE, gets a console of its own, and on Windows 11 with Windows Terminal as the
|
||||
// default terminal that console is a visible Terminal window on the user's desk. Rule: every process the app starts
|
||||
// on Windows runs with a hidden console. This check fails CI when
|
||||
// - a `Command::new(` in app/igneum-app/src is not quieted within 20 lines: crate::platform::quiet, run_timeout,
|
||||
// run_capture, run_streamed, spawn_detached, elevated_command, or creation_flags(0x0800_0000) (CREATE_NO_WINDOW);
|
||||
// programs that only exist off Windows (nohup, osascript, pkexec, hdiutil, ...) are allowed, and a
|
||||
// `// console: <reason>` comment on the line or the line above allows a builder the caller quiets;
|
||||
// the check looks 2 lines back as well, for `run_timeout(\n Command::new(...)`;
|
||||
// - a `creation_flags(` carries anything but 0x0800_0000 (DETACHED_PROCESS made powershell exit at start-up,
|
||||
// 0.3.0 to 0.3.4, docs/bugs.md);
|
||||
// - a PowerShell `Start-Process` written by the Rust code lacks `-WindowStyle Hidden` or `-NoNewWindow` (a GUI
|
||||
// program, which never gets a console, takes a `# console: <reason>` comment on the line or the line above);
|
||||
// - app/windows/host.cpp calls CreateProcessW without CREATE_NO_WINDOW or sets up a ShellExecuteExW without
|
||||
// nShow = SW_HIDE (ShellExecuteW "open" of a URL is the browser, allowed).
|
||||
// node tools/ci/windows-spawn-check.mjs the tree
|
||||
// node tools/ci/windows-spawn-check.mjs --self-test the rules on known-good and known-bad samples
|
||||
import { readFileSync, readdirSync, statSync } from 'node:fs';
|
||||
import { join, resolve, dirname, relative } from 'node:path';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
|
||||
const ROOT = resolve(dirname(fileURLToPath(import.meta.url)), '..', '..');
|
||||
const QUIET = /\bquiet\(|\brun_timeout\(|\brun_capture\(|\brun_streamed\(|\bspawn_detached\(|\belevated_command\(|creation_flags\(0x0800_0000\)/;
|
||||
const OFF_WINDOWS = /Command::new\((?:crate::platform::)?tool\("(?:nohup|osascript|pkexec|hdiutil|ditto|open|xattr|system_profiler|sysctl|sntp|scutil|caffeinate)"\)|Command::new\("(?:pkexec|xdg-open|\/bin\/bash|\/usr\/bin\/[a-z]+)"\)|Command::new\(staged\.join\("Contents\/MacOS/;
|
||||
const WINDOW = 20;
|
||||
|
||||
export function checkRust(text, file) {
|
||||
const lines = text.split('\n');
|
||||
const out = [];
|
||||
for (let i = 0; i < lines.length; i++) {
|
||||
const l = lines[i];
|
||||
if (/Command::new\(/.test(l) && !/^\s*\/\//.test(l)) {
|
||||
const allowed = OFF_WINDOWS.test(l) || /\/\/ console:/.test(l) || (i > 0 && /\/\/ console:/.test(lines[i - 1]));
|
||||
if (!allowed) {
|
||||
const span = lines.slice(Math.max(0, i - 2), i + WINDOW).join('\n'); // 2 lines back: run_timeout(\n Command::new(...)
|
||||
if (!QUIET.test(span)) out.push(`${file}:${i + 1}: Command::new without a hidden console within ${WINDOW} lines (quiet, run_timeout, run_capture, run_streamed, spawn_detached, elevated_command or creation_flags(0x0800_0000)); add one, or a '// console: <reason>' comment`);
|
||||
}
|
||||
}
|
||||
const cf = /creation_flags\(([^)]*)\)/.exec(l);
|
||||
if (cf && cf[1].trim() !== '0x0800_0000') out.push(`${file}:${i + 1}: creation_flags(${cf[1]}) is not CREATE_NO_WINDOW alone (0x0800_0000)`);
|
||||
if (/Start-Process\b/.test(l) && !/^\s*\/\//.test(l) && !/-WindowStyle Hidden|-NoNewWindow/.test(l) && !/(\/\/|#) console:/.test(l) && !(i > 0 && /(\/\/|#) console:/.test(lines[i - 1]))) out.push(`${file}:${i + 1}: Start-Process without -WindowStyle Hidden or -NoNewWindow (a GUI program takes a '# console: <reason>' comment)`);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
export function checkHost(text, file) {
|
||||
const lines = text.split('\n');
|
||||
const out = [];
|
||||
for (let i = 0; i < lines.length; i++) {
|
||||
const l = lines[i];
|
||||
if (/CreateProcessW?\s*\(/.test(l) && !/CREATE_NO_WINDOW/.test(lines.slice(i, i + 3).join('\n'))) out.push(`${file}:${i + 1}: CreateProcess without CREATE_NO_WINDOW`);
|
||||
if (/ShellExecuteExW?\s*\(/.test(l) && !/nShow\s*=\s*SW_HIDE/.test(lines.slice(Math.max(0, i - 12), i + 1).join('\n'))) out.push(`${file}:${i + 1}: ShellExecuteEx without nShow = SW_HIDE in the 12 lines before it`);
|
||||
if (/ShellExecuteW?\s*\(/.test(l) && !/ShellExecuteExW?/.test(l) && !/L"open"/.test(l)) out.push(`${file}:${i + 1}: ShellExecute that is not the browser "open" of a URL`);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
function walk(dir, ext, acc = []) {
|
||||
for (const e of readdirSync(dir)) {
|
||||
const p = join(dir, e);
|
||||
if (statSync(p).isDirectory()) { if (e !== 'target') walk(p, ext, acc); } else if (p.endsWith(ext)) acc.push(p);
|
||||
}
|
||||
return acc;
|
||||
}
|
||||
|
||||
function selfTest() {
|
||||
const good = `fn a() {\n let mut c = Command::new(crate::platform::tool("powershell"));\n c.args(["-NoProfile"]);\n crate::platform::quiet(&mut c);\n c.spawn();\n}\n`;
|
||||
const bad = `fn a() {\n let mut c = Command::new(crate::platform::tool("powershell"));\n c.args(["-NoProfile"]);\n c.spawn();\n}\n`;
|
||||
const badFlag = `c.creation_flags(0x0000_0008);\n`;
|
||||
const badPs = `let ps = format!("Start-Process -FilePath '{}' -Wait", exe);\n`;
|
||||
const okPs = `let ps = format!("Start-Process -FilePath '{}' -Wait -WindowStyle Hidden", exe);\n`;
|
||||
const offWin = `let out = Command::new(tool("osascript")).args(["-e", "x"]).output();\n`;
|
||||
const allowed = `// console: the caller quiets it\nlet mut c = Command::new(wsl_exe);\n`;
|
||||
const hostGood = `sei.nShow = SW_HIDE;\nif (!ShellExecuteExW(&sei)) {}\nCreateProcessW(exe, buf, nullptr, nullptr, TRUE, CREATE_NO_WINDOW, nullptr, dir, &si, &pi);\nShellExecuteW(nullptr, L"open", url, nullptr, nullptr, SW_SHOWNORMAL);\n`;
|
||||
const hostBad = `sei.nShow = SW_SHOW;\nif (!ShellExecuteExW(&sei)) {}\nCreateProcessW(exe, buf, nullptr, nullptr, TRUE, 0, nullptr, dir, &si, &pi);\n`;
|
||||
const cases = [
|
||||
['quieted Command', checkRust(good, 't.rs').length === 0],
|
||||
['bare Command fails', checkRust(bad, 't.rs').length === 1],
|
||||
['DETACHED_PROCESS fails', checkRust(badFlag, 't.rs').length === 1],
|
||||
['Start-Process without Hidden fails', checkRust(badPs, 't.rs').length === 1],
|
||||
['Start-Process with Hidden passes', checkRust(okPs, 't.rs').length === 0],
|
||||
['off-Windows program passes', checkRust(offWin, 't.rs').length === 0],
|
||||
['console: comment passes', checkRust(allowed, 't.rs').length === 0],
|
||||
['host.cpp good passes', checkHost(hostGood, 'h.cpp').length === 0],
|
||||
['host.cpp bad fails twice', checkHost(hostBad, 'h.cpp').length === 2],
|
||||
];
|
||||
let fail = 0;
|
||||
for (const [name, ok] of cases) { console.log(`${ok ? 'ok ' : 'FAIL'} ${name}`); if (!ok) fail++; }
|
||||
return fail;
|
||||
}
|
||||
|
||||
if (process.argv[1] && resolve(process.argv[1]) === fileURLToPath(import.meta.url)) {
|
||||
if (process.argv.includes('--self-test')) {
|
||||
const f = selfTest();
|
||||
console.log(f ? `windows-spawn: self-test FAILED (${f})` : 'windows-spawn: self-test ok');
|
||||
process.exit(f ? 1 : 0);
|
||||
}
|
||||
const problems = [];
|
||||
for (const f of walk(join(ROOT, 'app', 'igneum-app', 'src'), '.rs')) problems.push(...checkRust(readFileSync(f, 'utf8'), relative(ROOT, f)));
|
||||
const host = join(ROOT, 'app', 'windows', 'host.cpp');
|
||||
try { problems.push(...checkHost(readFileSync(host, 'utf8'), relative(ROOT, host))); } catch {}
|
||||
for (const p of problems) console.log(p);
|
||||
console.log(problems.length ? `windows-spawn: ${problems.length} spawn(s) without a hidden console` : 'windows-spawn: every Windows spawn in app/igneum-app/src and app/windows/host.cpp runs with a hidden console');
|
||||
process.exit(problems.length ? 1 : 0);
|
||||
}
|
||||
|
|
@ -5,6 +5,7 @@
|
|||
// node tools/console.mjs machines the machine cards as text
|
||||
// node tools/console.mjs chain the chain numbers
|
||||
// node tools/console.mjs jobs | builds | results the other tabs as text
|
||||
// node tools/console.mjs tuning [--days 30] [--min 5] Ember Tune's fleet priors per card model (samples, MH/W)
|
||||
// node tools/console.mjs sync-bench push docs/bench-log.md entries and the FUD ledger counts
|
||||
// node tools/console.mjs sync-dl push the downloads folder listing (names, sizes, times)
|
||||
// node tools/console.mjs sync-hetzner push the newest infra/cloud-devnet/results/ summary
|
||||
|
|
@ -124,6 +125,12 @@ try {
|
|||
}
|
||||
else if (cmd === 'jobs') { const j = await api('jobs'); if (!j.jobs.length) console.log(j.note || 'no jobs'); for (const jb of j.jobs) console.log(`${jb.id} ${jb.title} | ${jb.runs.map(r => `${r.machine} ${r.status}${r.exit_code != null ? ' exit ' + r.exit_code : ''}${r.summary ? ': ' + r.summary.slice(0, 80) : ''}`).join(' | ') || 'queued everywhere'}`); }
|
||||
else if (cmd === 'builds') { const j = await api('builds'); console.log(`manifest ${j.manifest ? j.manifest.version + ' (' + j.manifest.channel + ') ' + Object.keys(j.manifest.platforms || {}).join('+') + ': ' + j.manifest.notes : 'none'}`); if (j.ci) console.log(`ci ${j.ci.installer} fetched ${j.ci.fetched_at} ${j.ci.run}`); for (const b of j.builds) console.log(`${when(b.ts)} ${b.who.padEnd(8)} ${b.title}${b.body ? ': ' + b.body.split('\n')[0].slice(0, 100) : ''}`); }
|
||||
else if (cmd === 'tuning') {
|
||||
const j = await api('tuning', { q: { days: flags.days || 30, min: flags.min || 5 } });
|
||||
console.log(`${j.records} tune record(s) in ${j.days} day(s); a prior needs ${j.min_samples} sample(s)`);
|
||||
if (!j.table.length) console.log('no tune records yet (the apps log one per finished tune, 0.3.10 and later)');
|
||||
for (const t of j.table) console.log(`${t.card.replace(/_/g, ' ')} | driver ${t.driver_major} | ${t.class}: ${t.samples} sample(s) from ${t.machines} machine(s)${t.samples ? `, ${t.clock_mhz ? t.clock_mhz + ' MHz at ' : ''}${t.power_pct}%${t.clock_mhz ? '' : ' (clock unlocked)'}, ${t.eff} MH/W (${t.mhs} MH/s at ${t.watts} W, spread ${t.spread_pct}%)` : ''}${t.baseline_eff ? `; untuned ${t.baseline_eff} MH/W from ${t.baseline_samples} baseline(s)` : ''}${t.gain_pct != null ? `; gain ${t.gain_pct >= 0 ? '+' : ''}${t.gain_pct}%` : ''}; prior ${t.key in j.priors ? 'yes' : 'no'}`);
|
||||
}
|
||||
else if (cmd === 'results') { const j = await api('results'); if (j.ledger) console.log(`${j.ledger.title}: ${j.ledger.body}`); for (const b of j.bench) console.log(`${b.meta.date || '-'} ${b.title}`); }
|
||||
else if (cmd === 'sync-bench' || cmd === 'sync-dl' || cmd === 'sync-hetzner' || cmd === 'sync') {
|
||||
const items = [];
|
||||
|
|
|
|||
50
tools/proving-v1/agg-cost-table.mjs
Normal file
50
tools/proving-v1/agg-cost-table.mjs
Normal file
|
|
@ -0,0 +1,50 @@
|
|||
#!/usr/bin/env node
|
||||
// The aggregation-cost curve from one or more pc2-agg-cost.ps1 jobs (5 October 2026 night): per phase, the shard and
|
||||
// aggregation seconds per block (the chain and aggregate-only RESULT lines), the GPU utilisation and memory peak of
|
||||
// the phase, the own miner's rate (MH/s wall) and its batch-log2. Reads the job uploads from the log intake like
|
||||
// tools/jobs.mjs (DATABASE_URL in ~/.config/igneum/env). Prints a markdown table.
|
||||
// node tools/proving-v1/agg-cost-table.mjs agg-cost-pc2-1 agg-cost-pc2-2 ...
|
||||
import { readFileSync } from 'node:fs';
|
||||
import { homedir } from 'node:os';
|
||||
const ids = process.argv.slice(2);
|
||||
if (!ids.length) { console.error('usage: agg-cost-table.mjs <job id> [...]'); process.exit(2); }
|
||||
const m = /^DATABASE_URL=(.*)$/m.exec(readFileSync(`${homedir()}/.config/igneum/env`, 'utf8'));
|
||||
const url = m[1].trim().replace(/^['"]|['"]$/g, '');
|
||||
const host = new URL(url).hostname.replace('-pooler', '');
|
||||
const sql = async (query, params = []) => {
|
||||
const r = await fetch(`https://${host}/sql`, { method: 'POST', headers: { 'Neon-Connection-String': url, 'Content-Type': 'application/json' }, body: JSON.stringify({ query, params }) });
|
||||
const j = await r.json();
|
||||
if (!r.ok) throw new Error(j.message || JSON.stringify(j));
|
||||
return j.rows;
|
||||
};
|
||||
const f1 = x => (Math.round(x * 10) / 10).toFixed(1);
|
||||
const rows = [];
|
||||
for (const id of ids) {
|
||||
const ups = await sql('SELECT lines FROM miner_logs WHERE run_id LIKE $1 ORDER BY received_at DESC LIMIT 1', [`job-${id}-%`]);
|
||||
if (!ups.length) { console.error(`${id}: no upload`); continue; }
|
||||
const lines = ups[0].lines.split('\n');
|
||||
const phases = new Map();
|
||||
const ph = l => { if (!phases.has(l)) phases.set(l, { job: id, label: l, shard: [], agg: [], deferred: [], util: '', mem: '', rate: '', batch: '', note: '' }); return phases.get(l); };
|
||||
for (const raw of lines) {
|
||||
const l = raw.replace(/^\d+(\.\d+)? job \S+: /, '');
|
||||
let x;
|
||||
if ((x = /^([A-Z0-9]+): RESULT chain block \d+ shard \d+: compressed prove ([\d.]+) s/.exec(l))) ph(x[1]).shard.push(Number(x[2]));
|
||||
else if ((x = /^([A-Z0-9]+): RESULT (?:chain|aggregate) block \d+: .*?(?:aggregate )?prove ([\d.]+) s \(stdin [\d.]+ s, (\d+) deferred/.exec(l))) { ph(x[1]).agg.push(Number(x[2])); ph(x[1]).deferred.push(Number(x[3])); }
|
||||
else if ((x = /^RESULT phase_gpu ([A-Z0-9]+) samples=\d+ memory_used_max_mib=(\d+) util_mean_pct=([\d.]+)/.exec(l))) { ph(x[1]).mem = x[2]; ph(x[1]).util = x[3]; }
|
||||
else if ((x = /^RESULT ([A-Z0-9]+) miner rate (n=\d+ mean=[\d.]+)/.exec(l))) ph(x[1]).rate = x[2].replace('n=', 'n ').replace(' mean=', ', mean ');
|
||||
else if ((x = /^RESULT ([A-Z0-9]+) own miner started .* batch_log2=(\d+)/.exec(l))) ph(x[1]).batch = x[2];
|
||||
else if ((x = /^RESULT ([A-Z0-9]+) own miner (not hashing|EXITED|: no app miner)/.exec(l))) ph(x[1]).note = 'own miner ' + x[2];
|
||||
else if ((x = /^RESULT phase ([A-Z0-9]+) end .* exit (\d+) wall ([\d.]+) s/.exec(l))) { const p = ph(x[1]); p.exit = x[2]; p.wall = x[3]; }
|
||||
else if ((x = /^RESULT phase ([A-Z0-9]+) skipped/.exec(l))) ph(x[1]).note = 'skipped';
|
||||
else if ((x = /^RESULT (H (?:choice|knobs)): (.*)$/.exec(l))) console.error(`${id}: ${x[1]}: ${x[2]}`);
|
||||
}
|
||||
for (const p of phases.values()) rows.push(p);
|
||||
}
|
||||
const mean = a => a.length ? a.reduce((s, v) => s + v, 0) / a.length : NaN;
|
||||
console.log('| Job | Phase | batch-log2 | Shard s (each) | Aggregation s (each, deferred proofs) | Block s (shard + aggregation, mean) | GPU util % | GPU peak MiB | Miner MH/s | Note |');
|
||||
console.log('|---|---|---|---|---|---|---|---|---|---|');
|
||||
for (const p of rows) {
|
||||
const chained = p.agg.filter((_, i) => p.deferred[i] >= 2);
|
||||
const block = p.shard.length && p.agg.length ? f1(mean(p.shard) + mean(chained.length ? chained : p.agg)) : '';
|
||||
console.log(`| ${p.job} | ${p.label} | ${p.batch || (p.label.startsWith('E') && /^E\d+$/.test(p.label) ? p.label.slice(1) : '')} | ${p.shard.map(f1).join(', ')} | ${p.agg.map((a, i) => `${f1(a)} (${p.deferred[i]})`).join(', ')} | ${block} | ${p.util} | ${p.mem} | ${p.rate} | ${[p.note, p.exit && p.exit !== '0' ? `exit ${p.exit}` : ''].filter(Boolean).join('; ')} |`);
|
||||
}
|
||||
35
tools/proving-v1/miner-rate.mjs
Normal file
35
tools/proving-v1/miner-rate.mjs
Normal file
|
|
@ -0,0 +1,35 @@
|
|||
#!/usr/bin/env node
|
||||
// The hash rate of one miner over a UTC window, from its STATUS lines in the log intake (the app uploads the miner's
|
||||
// log every minute; each STATUS line starts with a unix timestamp and carries `now=<MH/s> MH/s wall`, the last
|
||||
// interval's rate). For the aggregation-cost phases that keep the app's own miner running (docs/bench-log.md,
|
||||
// 5 October 2026 night), where the job cannot read the app's state from PowerShell 5.1.
|
||||
// node tools/proving-v1/miner-rate.mjs <label> <start ISO> <end ISO> e.g. miner-nvidia-1ccfe586-1 2026-10-05T21:00:00Z 2026-10-05T21:05:00Z
|
||||
import { readFileSync } from 'node:fs';
|
||||
import { homedir } from 'node:os';
|
||||
const [label, start, end] = process.argv.slice(2);
|
||||
if (!label || !start || !end) { console.error('usage: miner-rate.mjs <label> <start ISO> <end ISO>'); process.exit(2); }
|
||||
const m = /^DATABASE_URL=(.*)$/m.exec(readFileSync(`${homedir()}/.config/igneum/env`, 'utf8'));
|
||||
const url = m[1].trim().replace(/^['"]|['"]$/g, '');
|
||||
const host = new URL(url).hostname.replace('-pooler', '');
|
||||
const sql = async (query, params = []) => {
|
||||
const r = await fetch(`https://${host}/sql`, { method: 'POST', headers: { 'Neon-Connection-String': url, 'Content-Type': 'application/json' }, body: JSON.stringify({ query, params }) });
|
||||
const j = await r.json();
|
||||
if (!r.ok) throw new Error(j.message || JSON.stringify(j));
|
||||
return j.rows;
|
||||
};
|
||||
const t0 = Date.parse(start) / 1000, t1 = Date.parse(end) / 1000;
|
||||
// every upload received up to 15 min after the window (the log is re-uploaded whole, so the newest covering upload holds every line)
|
||||
const rows = await sql(`SELECT received_at, lines FROM miner_logs WHERE label = $1 AND received_at > to_timestamp($2) AND received_at < to_timestamp($3) ORDER BY received_at DESC LIMIT 20`, [label, t0, t1 + 900]);
|
||||
const seen = new Map();
|
||||
for (const r of rows) {
|
||||
for (const l of r.lines.split('\n')) {
|
||||
const mm = /^(\d+(?:\.\d+)?) STATUS .* now=([\d.]+) MH\/s wall/.exec(l);
|
||||
if (!mm) continue;
|
||||
const t = Number(mm[1]);
|
||||
if (t >= t0 && t <= t1) seen.set(mm[1], Number(mm[2]));
|
||||
}
|
||||
}
|
||||
const h = [...seen.values()].sort((a, b) => a - b);
|
||||
if (!h.length) { console.log(`${label} ${start}..${end}: no STATUS lines in the window (${rows.length} uploads read)`); process.exit(1); }
|
||||
const mean = h.reduce((a, b) => a + b, 0) / h.length;
|
||||
console.log(`${label} ${start}..${end}: n=${h.length} mean=${mean.toFixed(2)} p50=${h[Math.floor((h.length - 1) / 2)].toFixed(2)} min=${h[0].toFixed(2)} max=${h[h.length - 1].toFixed(2)} MH/s wall (now= of the STATUS lines, ${rows.length} uploads read)`);
|
||||
|
|
@ -37,6 +37,10 @@ const args = process.argv.slice(2);
|
|||
const flag = (name, dflt) => { const i = args.indexOf(name); return i >= 0 && args[i + 1] !== undefined ? +args[i + 1] : dflt; };
|
||||
const SECS = flag('--secs', 1200);
|
||||
const V0 = 60, V1 = flag('--v1', 120), SEG = flag('--segment', 4), UNPROVEN = flag('--unproven', 60), SHARE_BPS = 1000;
|
||||
// --fresh-rule <daa>: the fresh-record rule switch (6 October 2026); default never (the 0.3.11 rule), so case 3 keeps
|
||||
// its known-failed line; with --fresh-rule 0 a fresh record after a PENDING segment is accepted (the new rule) and
|
||||
// case 2's refusal after a PROVEN one still stands
|
||||
const FRESH_RULE = process.argv.includes('--fresh-rule') ? Number(process.argv[process.argv.indexOf('--fresh-rule') + 1]) : null;
|
||||
const started = [];
|
||||
const t0 = Date.now();
|
||||
const since = () => ((Date.now() - t0) / 1000).toFixed(1);
|
||||
|
|
@ -56,6 +60,7 @@ let overrideText = readFileSync(FILE, 'utf8')
|
|||
.replace(/"proving_v1_segment_blocks":\s*\d+/, `"proving_v1_segment_blocks": ${SEG}`)
|
||||
.replace(/"proving_v1_unproven_daa":\s*\d+/, `"proving_v1_unproven_daa": ${UNPROVEN}`)
|
||||
.replace(/"proving_v1_aggregator_share_bps":\s*\d+/, `"proving_v1_aggregator_share_bps": ${SHARE_BPS}`)
|
||||
.replace(/"proving_v1_fresh_rule_daa":\s*\d+/, FRESH_RULE === null ? '"proving_v1_fresh_rule_daa": 18446744073709551615' : `"proving_v1_fresh_rule_daa": ${FRESH_RULE}`)
|
||||
.replace(/"skip_proof_of_work":\s*(true|false)/, '"skip_proof_of_work": true');
|
||||
for (const re of [/"skip_proof_of_work": true/, new RegExp(`"proving_v1_activation_daa": ${V1}`), new RegExp(`"proving_v1_segment_blocks": ${SEG}`), new RegExp(`"proving_v1_unproven_daa": ${UNPROVEN}`)]) if (!re.test(overrideText)) throw new Error(`override edit failed: ${re}`);
|
||||
writeFileSync(override, overrideText);
|
||||
|
|
@ -211,7 +216,13 @@ try {
|
|||
const lastHash3 = stmt3early.blocks[stmt3early.blocks.length - 1].hash;
|
||||
const fresh3 = signSegment(seg3[0], seg3[1], lastHash3, stmt3early.publicValuesFresh, 'seg3-fresh');
|
||||
const subEarly = await n0.eth('igneum_submitSegmentRecord', [{ record: fresh3.record, proof: fresh3.proof }]);
|
||||
check('known-failed: segment 3 cannot start a fresh chain while segment 2 is pending', subEarly.accepted === false && /pending until DAA/.test(subEarly.reason), subEarly);
|
||||
if (FRESH_RULE === null) {
|
||||
check('known-failed: segment 3 cannot start a fresh chain while segment 2 is pending', subEarly.accepted === false && /pending until DAA/.test(subEarly.reason), subEarly);
|
||||
} else {
|
||||
check('fresh-record rule: segment 3 starts a fresh chain while segment 2 is pending', subEarly.accepted === true, subEarly);
|
||||
const stmt3now = await n0.eth('igneum_getSegmentStatement', ['0x' + seg3[0].toString(16)]);
|
||||
check('fresh-record rule: the statement reports the fresh record admissible', stmt3now.freshAdmissible === true, { freshAdmissible: stmt3now.freshAdmissible });
|
||||
}
|
||||
await waitDaa(n0, deadline2 + 2, 'past segment 2 deadline');
|
||||
const stmt2late = await n0.eth('igneum_getSegmentStatement', ['0x' + seg2[0].toString(16)]);
|
||||
check('known-failed: segment 2 is unproven after its deadline', stmt2late.status.status === 'unproven', stmt2late.status);
|
||||
|
|
@ -220,7 +231,13 @@ try {
|
|||
const subLate = await n0.eth('igneum_submitSegmentRecord', [{ record: late2.record, proof: late2.proof }]);
|
||||
check('known-failed: a late record for segment 2 pays nothing (refused as unproven)', subLate.accepted === false && /unproven/.test(subLate.reason), subLate);
|
||||
const subRestart = await n1.eth('igneum_submitSegmentRecord', [{ record: fresh3.record, proof: fresh3.proof }]);
|
||||
check('segment 3 restarts the chain with a fresh-chain record after the unproven segment', subRestart.accepted === true, subRestart);
|
||||
if (FRESH_RULE === null) {
|
||||
check('segment 3 restarts the chain with a fresh-chain record after the unproven segment', subRestart.accepted === true, subRestart);
|
||||
} else {
|
||||
// under the fresh-record rule the record was accepted while segment 2 was still pending (above) and is carried
|
||||
// by now: the second offer is a duplicate or a paid segment, never a chain-rule refusal
|
||||
check('fresh-record rule: the second offer of segment 3\'s record is a duplicate, not a chain-rule refusal', subRestart.accepted === true || !/does not chain/.test(subRestart.reason || ''), subRestart);
|
||||
}
|
||||
const paid3 = await waitSegmentPaid(n0, seg3[0]);
|
||||
check('segment 3 paid with chain_len 4', hexn(paid3.paid.chainLen) === SEG, paid3.paid);
|
||||
st = await n0.eth('igneum_getProvingStatus');
|
||||
|
|
|
|||
57
tools/proving-v1/pc2-agg-cost-restore.ps1
Normal file
57
tools/proving-v1/pc2-agg-cost-restore.ps1
Normal file
|
|
@ -0,0 +1,57 @@
|
|||
# Aggregation cost, the restore job (5 October 2026, night): what pc2-agg-cost.ps1's finally block does, for the case
|
||||
# where the app restarted under the job (the 0.3.10 rollout; a running job dies with the app and its finally block
|
||||
# never runs): stops any miner or CUDA worker the job started itself (a process whose parent is not igneum-app.exe),
|
||||
# resumes the app's miners, switches the prover back on, resets the GPU compute-policy timeslice. Idempotent.
|
||||
$ErrorActionPreference = 'Continue'
|
||||
function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') }
|
||||
$urlFile = if ($env:IGNEUM_APP_DIR) { Join-Path $env:IGNEUM_APP_DIR 'app.url' } else { Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
$base = (Get-Content $urlFile -Raw).Trim().TrimEnd('/')
|
||||
# the state as text parsed by regex (app 0.3.10's response comes back empty to PowerShell 5.1's JSON reader)
|
||||
function State {
|
||||
try {
|
||||
$raw = (Invoke-WebRequest -UseBasicParsing -Uri "$base/api/state" -TimeoutSec 20).Content
|
||||
$cards = @()
|
||||
# one card = the text from its "index" key to the next card's (the fields come in any order inside it; the first
|
||||
# version took the first "identities" AFTER "vendor", which belongs to the NEXT card: 23:16:53Z, the 5090 set to the
|
||||
# iGPU's 2 identities)
|
||||
$idx = @([regex]::Matches($raw, '\{"index":\d+,"key":"') | ForEach-Object { $_.Index })
|
||||
for ($i = 0; $i -lt $idx.Count; $i++) {
|
||||
$end = if ($i + 1 -lt $idx.Count) { $idx[$i + 1] } else { $raw.Length }
|
||||
$seg = $raw.Substring($idx[$i], $end - $idx[$i])
|
||||
$f = { param($re, $def) $x = [regex]::Match($seg, $re); if ($x.Success) { $x.Groups[1].Value } else { $def } }
|
||||
$v = (& $f '"vendor":"([^"]*)"' '')
|
||||
if (-not $v) { continue }
|
||||
$cards += [pscustomobject]@{ key = (& $f '"key":"([^"]*)"' ''); name = (& $f '"name":"([^"]*)"' ''); vendor = $v; enabled = ((& $f '"enabled":(true|false)' 'true') -eq 'true'); identities = [int](& $f '"identities":(\d+)' '1'); state = (& $f '"state":"([^"]*)"' ''); hash_now = [double](& $f '"hash_now":([\d.eE+-]+)' '0') }
|
||||
}
|
||||
$g = { param($re, $def) $x = [regex]::Match($raw, $re); if ($x.Success) { $x.Groups[1].Value } else { $def } }
|
||||
return [pscustomobject]@{ version = (& $g '"version":"([^"]*)"' ''); settings = [pscustomobject]@{ prove = ((& $g '"settings":\{[^}]*?"prove":(true|false)' 'false') -eq 'true') }; mining = [pscustomobject]@{ paused = ((& $g '"mining":\{[^}]*?"paused":(true|false)' 'false') -eq 'true'); cards = $cards }; proving = [pscustomobject]@{ status = (& $g '"proving":\{[^}]*?"status":"([^"]*)"' '') } }
|
||||
} catch { $null }
|
||||
}
|
||||
function Post($path, $body) { try { (Invoke-RestMethod -Method Post -Uri "$base$path" -ContentType 'application/json' -Body ($body | ConvertTo-Json -Compress -Depth 5) -TimeoutSec 15) | ConvertTo-Json -Compress } catch { "error: $_" } }
|
||||
$st = State
|
||||
"RESULT before $(Stamp) app $($st.version) paused=$($st.mining.paused) prove=$($st.settings.prove) proving=$($st.proving.status)"
|
||||
$apps = @(Get-Process -ErrorAction SilentlyContinue | Where-Object { $_.ProcessName -like 'igneum-app*' } | ForEach-Object { $_.Id })
|
||||
"RESULT app_pids $(Stamp) $($apps -join ',')"
|
||||
$killed = 0
|
||||
foreach ($p in (Get-CimInstance Win32_Process | Where-Object { $_.Name -like 'igneum-miner*' -or $_.Name -like 'igneum-worker*' })) {
|
||||
$parent = Get-CimInstance Win32_Process -Filter "ProcessId = $($p.ParentProcessId)" -ErrorAction SilentlyContinue
|
||||
$stray = (-not $parent) -or (($parent.Name -notlike 'igneum-app*') -and ($parent.Name -notlike 'igneum-miner*'))
|
||||
if ($stray) { "RESULT stray $(Stamp) pid $($p.ProcessId) $($p.Name) parent=$($p.ParentProcessId) ($(if ($parent) { $parent.Name } else { 'gone' })): stopped"; Stop-Process -Id $p.ProcessId -Force -ErrorAction SilentlyContinue; $killed++ }
|
||||
}
|
||||
"RESULT strays_stopped $(Stamp) $killed"
|
||||
& wsl.exe -d Ubuntu-24.04 -u root -- bash -c 'pkill -x sp1-gpu-server; rm -f /tmp/sp1-cuda-*.sock; echo "RESULT socket_cleanup $(date -u +%Y-%m-%dT%H:%M:%SZ) servers=$(pgrep -x sp1-gpu-server | wc -l) sockets=$(ls /tmp/sp1-cuda-*.sock 2>/dev/null | wc -l)"' 2>&1 | ForEach-Object { ($_ -replace "`0", '') }
|
||||
function Card($st) { if ($null -eq $st) { return $null }; $st.mining.cards | Where-Object { $_.vendor -eq 'nvidia' } | Select-Object -First 1 }
|
||||
$c = Card (State)
|
||||
if ($c) {
|
||||
$stc = State
|
||||
"RESULT cards_before $(Stamp) $(($stc.mining.cards | ForEach-Object { "$($_.name):enabled=$($_.enabled):identities=$($_.identities):$($_.state)" }) -join ',')"
|
||||
$list = @(); foreach ($k in $stc.mining.cards) { $list += @{ key = $k.key; enabled = $(if ($k.key -eq $c.key) { $true } else { [bool]$k.enabled }); identities = [int]$k.identities } }
|
||||
"RESULT card_on $(Stamp) $(Post '/api/cards' @{ cards = $list })"
|
||||
}
|
||||
"RESULT timeslice $(Stamp) $((& nvidia-smi compute-policy --set-timeslice=0 2>&1) -join ' ')"
|
||||
"RESULT resume $(Stamp) $(Post '/api/resume' @{})"
|
||||
"RESULT start $(Stamp) $(Post '/api/start' @{})"
|
||||
"RESULT prove_on $(Stamp) $(Post '/api/prove' @{on=$true})"
|
||||
Start-Sleep -Seconds 20
|
||||
$st = State
|
||||
"RESULT after $(Stamp) app $($st.version) paused=$($st.mining.paused) prove=$($st.settings.prove) proving=$($st.proving.status) cards=$(($st.mining.cards | ForEach-Object { "$($_.name):$($_.state):$([math]::Round($_.hash_now,1))" }) -join ',')"
|
||||
433
tools/proving-v1/pc2-agg-cost.ps1
Normal file
433
tools/proving-v1/pc2-agg-cost.ps1
Normal file
|
|
@ -0,0 +1,433 @@
|
|||
# Aggregation cost (5 October 2026, night; the project lead: "fix everything else in the numbers tonight"): what a per-block
|
||||
# aggregation on PC 2's RTX 5090 spends and what each lever gives. A signed `run` job (shell powershell, not elevated).
|
||||
# The live prover is switched OFF for the run (its sp1-gpu-server would otherwise be shared, socket /tmp/sp1-cuda-0.sock)
|
||||
# and ON again at the end; the app's miners are paused for the idle and own-miner phases and resumed at the end
|
||||
# (try/finally). The live /opt/igneum host is untouched: this build lands in /opt/igneum-aggcost.
|
||||
# Phases (the set at the top; every number is a RESULT line prefixed with the phase):
|
||||
# A0 app miner mining, chain of 1 with RUST_LOG=info: the SP1 GPU server's own log lines (the profile)
|
||||
# A app miner mining, chain of 4 (--save-shards), default knobs: tonight's baseline, shard proofs kept
|
||||
# B0 app miner mining, aggregate-only over A's shard proofs, default knobs
|
||||
# B app miner mining, aggregate-only, SP1_WORKER_VERIFY_INTERMEDIATES=false
|
||||
# C miners paused: chain of 4, default knobs (the idle baseline tonight)
|
||||
# C1 miners paused: aggregate-only over A's shard proofs
|
||||
# C2 miners paused: block-344-shards4 (four full shards, one block, --save-shards): a 4-deferred aggregation idle
|
||||
# D own miner, the app's command line, batch-log2 22 (the worker's default): chain of 4; MH/s from its STATUS lines
|
||||
# E own miner at batch-log2 20, 18, 16: chain of 4 each; MH/s each
|
||||
# E0 own miner 22: aggregate-only over C2's four shard proofs (a 4-deferred aggregation under mining)
|
||||
# F own miner 22 with nvidia-smi compute-policy --set-timeslice=1 (SHORT), chain of 4; the policy restored after
|
||||
# G0 miners paused: two host processes at once (chain of 2 each, disjoint blocks) on the one card
|
||||
# G own miner 22: two host processes at once (chain of 2 each, disjoint blocks) on the one GPU server
|
||||
# H own miner at $BestBatch with the best knobs: chain of 4 (the combination the plan adopts)
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$PhaseSet = @('A0','A','B0','B','C','C1','C2','D','E')
|
||||
$ShardSourceDir = '' # job 2: the WSL directory holding job 1's block-N-shard-i-compressed.bin and block-344 proofs ('' = this job's own)
|
||||
$ShardSourceFirst = 0 # job 2: job 1's first block number (its four shard proofs are consecutive from it)
|
||||
$ShardSourceParent = '' # job 2: the parent chain block hash of that first block
|
||||
$BestBatch = 20
|
||||
$BestKnobs = ''
|
||||
function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') }
|
||||
$urlFile = if ($env:IGNEUM_APP_DIR) { Join-Path $env:IGNEUM_APP_DIR 'app.url' } else { Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
if (-not (Test-Path $urlFile)) { $urlFile = Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
$base = (Get-Content $urlFile -Raw).Trim().TrimEnd('/')
|
||||
$app = if ($env:IGNEUM_APP_DIR) { $env:IGNEUM_APP_DIR } else { Join-Path $env:LOCALAPPDATA 'igneum\app' }
|
||||
# The app's state as plain text parsed by regex (job 3, 22:41Z, app 0.3.10: Invoke-RestMethod's JSON conversion came back
|
||||
# empty to PowerShell 5.1, so the card switch found no card and the own miner ran beside the app's). The card objects are
|
||||
# serialised in struct order (src/state.rs CardState: key ... vendor ... enabled, state, hash_now), so the fields after a
|
||||
# "vendor" belong to that card and the last "key" before it is its key.
|
||||
$script:stateErr = ''
|
||||
$script:nvidiaSeg = ''
|
||||
function State {
|
||||
try {
|
||||
$raw = (Invoke-WebRequest -UseBasicParsing -Uri "$base/api/state" -TimeoutSec 20).Content
|
||||
if (-not $raw) { $script:stateErr = 'empty body'; return $null }
|
||||
$cards = @()
|
||||
# one card = the text from its "index" key to the next card's (the fields come in any order inside it; the first
|
||||
# version took the first "identities" AFTER "vendor", which belongs to the NEXT card: 23:16:53Z, the 5090 set to the
|
||||
# iGPU's 2 identities)
|
||||
$idx = @([regex]::Matches($raw, '\{"index":\d+,"key":"') | ForEach-Object { $_.Index })
|
||||
for ($i = 0; $i -lt $idx.Count; $i++) {
|
||||
$end = if ($i + 1 -lt $idx.Count) { $idx[$i + 1] } else { $raw.Length }
|
||||
$seg = $raw.Substring($idx[$i], $end - $idx[$i])
|
||||
$f = { param($re, $def) $x = [regex]::Match($seg, $re); if ($x.Success) { $x.Groups[1].Value } else { $def } }
|
||||
$v = (& $f '"vendor":"([^"]*)"' '')
|
||||
if (-not $v) { continue }
|
||||
if ($v -eq 'nvidia' -and -not $script:nvidiaSeg) { $script:nvidiaSeg = $seg.Substring(0, [Math]::Min(700, $seg.Length)) }
|
||||
$cards += [pscustomobject]@{ key = (& $f '"key":"([^"]*)"' ''); name = (& $f '"name":"([^"]*)"' ''); vendor = $v; enabled = ((& $f '"enabled":(true|false)' 'true') -eq 'true'); identities = [int](& $f '"identities":(\d+)' '1'); state = (& $f '"state":"([^"]*)"' ''); hash_now = [double](& $f '"hash_now":([\d.eE+-]+)' '0') }
|
||||
}
|
||||
$g = { param($re, $def) $x = [regex]::Match($raw, $re); if ($x.Success) { $x.Groups[1].Value } else { $def } }
|
||||
$script:stateErr = ''
|
||||
return [pscustomobject]@{
|
||||
version = (& $g '"version":"([^"]*)"' ''); machine_id = (& $g '"machine_id":"([^"]*)"' '')
|
||||
settings = [pscustomobject]@{ prove = ((& $g '"settings":\{[^}]*?"prove":(true|false)' 'false') -eq 'true') }
|
||||
mining = [pscustomobject]@{ paused = ((& $g '"mining":\{[^}]*?"paused":(true|false)' 'false') -eq 'true'); cards = $cards }
|
||||
proving = [pscustomobject]@{ status = (& $g '"proving":\{[^}]*?"status":"([^"]*)"' '') }
|
||||
raw_len = $raw.Length
|
||||
}
|
||||
} catch { $script:stateErr = "$_"; $null }
|
||||
}
|
||||
function Post($path, $body) { try { (Invoke-RestMethod -Method Post -Uri "$base$path" -ContentType 'application/json' -Body ($body | ConvertTo-Json -Compress -Depth 5) -TimeoutSec 15) | ConvertTo-Json -Compress } catch { "error: $_" } }
|
||||
function Rpc($port, $method, $params) {
|
||||
try { (Invoke-RestMethod -Method Post -Uri "http://127.0.0.1:$port" -ContentType 'application/json' -Body (@{jsonrpc='2.0'; id=1; method=$method; params=$params} | ConvertTo-Json -Compress -Depth 6) -TimeoutSec 180).result } catch { $null }
|
||||
}
|
||||
function Card($st) { if ($null -eq $st) { return $null }; $st.mining.cards | Where-Object { $_.vendor -eq 'nvidia' } | Select-Object -First 1 }
|
||||
function Smi { try { (& nvidia-smi --query-gpu=index,name,memory.used,utilization.gpu,power.draw --format=csv,noheader,nounits 2>$null) -join ' | ' } catch { 'nvidia-smi failed' } }
|
||||
# The card keys come from the app's own settings.json (`cards`: key -> {enabled, identities, ...}), never from /api/state:
|
||||
# that reply is "{}" whenever the state fails to serialise (job 4, 00:18Z: raw_len=2; the app's state_json falls back
|
||||
# to an empty object), which no parser can read. The state is still read for the hash rate when it answers.
|
||||
function CardPrefs {
|
||||
$f = Join-Path $app 'settings.json'
|
||||
if (-not (Test-Path $f)) { return @{} }
|
||||
$raw = Get-Content $f -Raw
|
||||
$prefs = @{}
|
||||
$cm = [regex]::Match($raw, '"cards":\s*\{')
|
||||
if (-not $cm.Success) { return $prefs }
|
||||
$i = $cm.Index + $cm.Length; $depth = 1; $start = $i
|
||||
while ($i -lt $raw.Length -and $depth -gt 0) { $ch = $raw[$i]; if ($ch -eq '{') { $depth++ } elseif ($ch -eq '}') { $depth-- }; $i++ }
|
||||
$body = $raw.Substring($start, $i - $start - 1)
|
||||
foreach ($m in [regex]::Matches($body, '"([^"]+)":\s*\{([^}]*)\}')) {
|
||||
$k = $m.Groups[1].Value; $v = $m.Groups[2].Value
|
||||
$en = [regex]::Match($v, '"enabled":\s*(true|false)'); $id = [regex]::Match($v, '"identities":\s*(\d+)')
|
||||
$prefs[$k] = @{ enabled = $(if ($en.Success) { $en.Groups[1].Value -eq 'true' } else { $true }); identities = $(if ($id.Success) { [int]$id.Groups[1].Value } else { 1 }) }
|
||||
}
|
||||
return $prefs
|
||||
}
|
||||
function NvidiaKey { $p = CardPrefs; ($p.Keys | Where-Object { $_ -like 'nvidia:*' } | Select-Object -First 1) }
|
||||
function CardSwitch($on, $identities) {
|
||||
$p = CardPrefs; $nk = NvidiaKey
|
||||
if (-not $nk) { return "no nvidia card key in settings.json (keys: $($p.Keys -join ','))" }
|
||||
$list = @(); foreach ($k in $p.Keys) { $list += @{ key = $k; enabled = $(if ($k -eq $nk) { [bool]$on } else { [bool]$p[$k].enabled }); identities = $(if ($k -eq $nk -and $identities) { [int]$identities } else { [int]$p[$k].identities }) } }
|
||||
"prefs_before=$(($p.Keys | ForEach-Object { "$($_):enabled=$($p[$_].enabled):identities=$($p[$_].identities)" }) -join ',') " + (Post '/api/cards' @{ cards = $list })
|
||||
}
|
||||
function CudaWorkers { (Get-Process -ErrorAction SilentlyContinue | Where-Object { $_.ProcessName -like 'igneum-worker-cuda*' } | Measure-Object).Count }
|
||||
$job = $env:IGNEUM_JOB_DIR
|
||||
if (-not $job) { $job = Join-Path $env:TEMP 'igneum-agg-cost' }
|
||||
New-Item -ItemType Directory -Force -Path $job | Out-Null
|
||||
$jobStart = Get-Date
|
||||
$st0 = State
|
||||
# the 5090 back to its 8 identities (the restore job of 23:16:53Z set it to 2 through the parser fault above)
|
||||
$RestoreIdentities = 8
|
||||
$p0 = CardPrefs; $nk0 = NvidiaKey
|
||||
"RESULT card_prefs $(Stamp) settings.json cards: $(($p0.Keys | ForEach-Object { "$($_):enabled=$($p0[$_].enabled):identities=$($p0[$_].identities)" }) -join ',') nvidia_key='$nk0'"
|
||||
if (-not $nk0) { "RESULT refused $(Stamp) no nvidia card key in settings.json: nothing is paused and nothing runs"; exit 1 }
|
||||
if ($p0[$nk0].identities -ne $RestoreIdentities) {
|
||||
"RESULT identities_set $(Stamp) $nk0 $($p0[$nk0].identities) -> ${RestoreIdentities}: $(CardSwitch ([bool]$p0[$nk0].enabled) $RestoreIdentities)"
|
||||
Start-Sleep -Seconds 5; $p1 = CardPrefs
|
||||
"RESULT identities_after $(Stamp) settings.json says identities=$($p1[$nk0].identities) enabled=$($p1[$nk0].enabled)"
|
||||
}
|
||||
"RESULT state_read $(Stamp) ok=$($null -ne $st0) raw_len=$($st0.raw_len) cards=$(($st0.mining.cards | Measure-Object).Count) err=$script:stateErr"
|
||||
"RESULT nvidia_card_json $(Stamp) $script:nvidiaSeg"
|
||||
if ($st0 -and ($st0.mining.cards | Measure-Object).Count -eq 0) { "RESULT state_head $(Stamp) no card parsed; the body starts: $((Invoke-WebRequest -UseBasicParsing -Uri "$base/api/state" -TimeoutSec 20).Content.Substring(0, 1500))" }
|
||||
"RESULT start $(Stamp) app $($st0.version) machine $($st0.machine_id) prove_setting=$($st0.settings.prove) proving=$($st0.proving.status) paused=$($st0.mining.paused) phases=$($PhaseSet -join ',') shard_source='$ShardSourceDir' $ShardSourceFirst $ShardSourceParent"
|
||||
"RESULT gpus $(Stamp) $(Smi)"
|
||||
$evmPort = $null
|
||||
foreach ($p in 26790, 26800, 26810) { if (Rpc $p 'igneum_getProvingStatus' @()) { $evmPort = $p; break } }
|
||||
# the NVIDIA worker must be hashing (the app's own miner) before the mining phases mean anything
|
||||
$waited = 0
|
||||
while ($waited -lt 180) {
|
||||
$c = Card (State)
|
||||
if ($c -and $c.hash_now -gt 10) { break }
|
||||
if ($waited % 60 -eq 0) { "WAIT $(Stamp) nvidia worker state=$(if ($c) { $c.state } else { 'na' }) hash_now=$(if ($c) { $c.hash_now } else { 'na' })" }
|
||||
Start-Sleep -Seconds 15; $waited += 15
|
||||
}
|
||||
$c = Card (State)
|
||||
"RESULT miner $(Stamp) nvidia worker $(if ($c -and $c.hash_now -gt 10) { 'hashing' } else { 'NOT hashing: the app-miner phases are void' }) hash_now=$(if ($c) { $c.hash_now } else { 'na' }) after $waited s"
|
||||
# the app's miner command line, for the own-miner phases (the same binary, args, cwd and tuning file)
|
||||
$mp = Get-CimInstance Win32_Process | Where-Object { $_.Name -like 'igneum-miner*' -and $_.CommandLine -like '* mine *' -and $_.CommandLine -like '*igneum-worker-cuda*' } | Select-Object -First 1
|
||||
$minerExe = $null; $minerArgs = $null
|
||||
if ($mp) {
|
||||
$minerExe = $mp.ExecutablePath
|
||||
$cl = $mp.CommandLine
|
||||
if ($cl -match '^\s*"[^"]*"\s*(.*)$') { $minerArgs = $Matches[1] } elseif ($cl -match '^\s*\S+\s+(.*)$') { $minerArgs = $Matches[1] }
|
||||
$minerArgs = $minerArgs -replace '--status-secs \d+', '--status-secs 10'
|
||||
}
|
||||
if (-not $mp) {
|
||||
# the 5090 miner is not running (job 2, 21:34Z: "off" after job 1's resume): the same command line as the app builds for
|
||||
# it (engine.rs miner_args), synthesised from the iGPU miner's line when that one runs, else from the app's install
|
||||
$amd = Get-CimInstance Win32_Process | Where-Object { $_.Name -like 'igneum-miner*' -and $_.CommandLine -like '* mine *' } | Select-Object -First 1
|
||||
$prog = Join-Path $env:LOCALAPPDATA 'Programs\Igneum Miner'
|
||||
$cuda = Join-Path $prog 'igneum-worker-cuda.exe'
|
||||
if ($amd -and $amd.CommandLine -match '--evm-address (0x[0-9a-fA-F]{40})') {
|
||||
$evm = $Matches[1]; $minerExe = $amd.ExecutablePath
|
||||
$rpc = if ($amd.CommandLine -match ' mine (\S+) ') { $Matches[1] } else { 'grpc://127.0.0.1:26610' }
|
||||
$minerArgs = "mine $rpc 1 100000000 nvidia-$($st0.machine_id.Substring(0,8))-1 --worker `"$cuda`" --status-secs 10 --exit-on-seed-change --evm-address $evm --identities 8 --payout-label nvidia-$($st0.machine_id.Substring(0,8))-1 --prepare-packs packs\prepare --worker-args `"--device 0 --pack packs\devnet`""
|
||||
"RESULT miner_cmdline_synthesised $(Stamp) from the iGPU miner's line; exe=$minerExe cuda_worker_exists=$(Test-Path $cuda)"
|
||||
} else { "RESULT miner_cmdline_synthesised $(Stamp) no running miner to copy from: the own-miner phases are void" }
|
||||
}
|
||||
$tuning = Join-Path $app 'tuning.json'
|
||||
if (Test-Path $tuning) { $env:IGNEUM_TUNING_FILE = $tuning }
|
||||
"RESULT miner_cmdline $(Stamp) exe=$minerExe tuning_file=$(Test-Path $tuning) args=$minerArgs"
|
||||
# strays from an earlier run that died with an app restart (job 3, 23:03Z): a miner or worker whose parent is not the
|
||||
# app (nor a miner) is ours and is stopped before anything is measured
|
||||
$killed = 0
|
||||
foreach ($p in (Get-CimInstance Win32_Process | Where-Object { $_.Name -like 'igneum-miner*' -or $_.Name -like 'igneum-worker*' })) {
|
||||
$parent = Get-CimInstance Win32_Process -Filter "ProcessId = $($p.ParentProcessId)" -ErrorAction SilentlyContinue
|
||||
if ((-not $parent) -or (($parent.Name -notlike 'igneum-app*') -and ($parent.Name -notlike 'igneum-miner*'))) { "RESULT stray $(Stamp) pid $($p.ProcessId) $($p.Name) parent=$($p.ParentProcessId): stopped"; Stop-Process -Id $p.ProcessId -Force -ErrorAction SilentlyContinue; $killed++ }
|
||||
}
|
||||
"RESULT strays_stopped $(Stamp) $killed"
|
||||
# the live prover off for the whole run: its sp1-gpu-server would be shared with ours (same socket, its environment)
|
||||
"RESULT prove_off $(Stamp) $(Post '/api/prove' @{on=$false})"
|
||||
# WSL side: the package, the export, the fixtures, the phase runner
|
||||
$pkg = Join-Path $env:LOCALAPPDATA 'igneum\prove\igneum-prove-wsl2-aggcost\igneum-prove-wsl2'
|
||||
if (-not (Test-Path $pkg)) { $pkg = Join-Path $env:LOCALAPPDATA 'igneum\prove\igneum-prove-wsl2-aggcost' }
|
||||
"RESULT package $(Stamp) $pkg exists=$(Test-Path (Join-Path $pkg 'package'))"
|
||||
function WslPath($p) { $w = (& wsl.exe -d Ubuntu-24.04 -u root -- wslpath -a ($p -replace '\\', '/') 2>$null); if ($w) { ($w -replace "`0", '').Trim() } else { '/mnt/c' + ($p.Substring(2) -replace '\\', '/') } }
|
||||
$pkgW = WslPath $pkg; $jobW = WslPath $job
|
||||
$srcW = if ($ShardSourceDir) { $ShardSourceDir.TrimEnd('/') } else { $jobW }
|
||||
$tipHex = Rpc $evmPort 'eth_blockNumber' @()
|
||||
$tip = [Convert]::ToInt64($tipHex, 16)
|
||||
$first = $tip - 30; $last = $first + 3
|
||||
"RESULT chain $(Stamp) node_evm_port=$evmPort tip=$tip blocks $first..$last"
|
||||
$t = Get-Date
|
||||
$body = (@{jsonrpc='2.0'; id=1; method='igneum_exportSegments'; params=@('0x0', ('0x{0:x}' -f $last))} | ConvertTo-Json -Compress)
|
||||
$seqFile = Join-Path $job 'seq.json'; $bodyFile = Join-Path $job 'export-request.json'
|
||||
[IO.File]::WriteAllText($bodyFile, $body, (New-Object System.Text.UTF8Encoding $false))
|
||||
& curl.exe -s -S -m 600 -X POST "http://127.0.0.1:$evmPort" -H 'Content-Type: application/json' --data-binary "@$bodyFile" -o $seqFile 2>&1 | ForEach-Object { "curl: $_" }
|
||||
if (-not (Test-Path $seqFile) -or (Get-Item $seqFile).Length -lt 1000) { "RESULT export FAILED: no reply file"; exit 1 }
|
||||
"RESULT export $(Stamp) $((Get-Item $seqFile).Length) bytes in $([math]::Round(((Get-Date) - $t).TotalSeconds,1)) s"
|
||||
$vars = @"
|
||||
export PKG='$pkgW'
|
||||
export JOB='$jobW'
|
||||
export SRC='$srcW'
|
||||
export FIRST=$first
|
||||
export LAST=$last
|
||||
"@
|
||||
[IO.File]::WriteAllText((Join-Path $job 'vars.sh'), ($vars -replace "`r`n", "`n"), (New-Object System.Text.UTF8Encoding $false))
|
||||
$prep = @'
|
||||
set -uo pipefail
|
||||
. "$(dirname "$0")/vars.sh"
|
||||
export PATH="$HOME/.cargo/bin:$HOME/.sp1/bin:$PATH"
|
||||
CUDA_DIR="$(ls -d /usr/local/cuda-12.* 2>/dev/null | sort -V | tail -1 || true)"; [ -n "$CUDA_DIR" ] && export PATH="$CUDA_DIR/bin:$PATH" && export LD_LIBRARY_PATH="$CUDA_DIR/lib64:/usr/lib/wsl/lib:${LD_LIBRARY_PATH:-}"
|
||||
stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; }
|
||||
DEST="$HOME/igneum-prove-aggcost"; LIVE_TARGET="$HOME/igneum-prove/proving/igneum-prove/target"
|
||||
mkdir -p "$DEST"
|
||||
rsync -a --delete --exclude target "$PKG/package/" "$DEST/"
|
||||
find "$DEST" -name target -prune -o -type f -exec touch {} + 2>/dev/null
|
||||
grep -o '"program_id": "0x[0-9a-f]*"' "$DEST/proving/igneum-prove/elf/manifest.json" | sed 's/^/RESULT manifest /'
|
||||
cd "$DEST/proving/igneum-prove"
|
||||
t0=$(date +%s)
|
||||
if ! CARGO_TARGET_DIR="$LIVE_TARGET" cargo build --release -p igneum-prove-export -p igneum-prove-host --features igneum-prove-host/cuda 2>&1 | tail -3; then echo "RESULT build FAILED"; exit 1; fi
|
||||
echo "RESULT build $(stamp) exit 0 in $(( $(date +%s) - t0 )) s"
|
||||
mkdir -p /opt/igneum-aggcost && cp "$LIVE_TARGET/release/igneum-prove-host" "$LIVE_TARGET/release/igneum-prove-export" /opt/igneum-aggcost/
|
||||
H=/opt/igneum-aggcost/igneum-prove-host; X=/opt/igneum-aggcost/igneum-prove-export
|
||||
echo "RESULT installed $(sha256sum $H | cut -c1-16) host, $(sha256sum $X | cut -c1-16) export; live /opt/igneum untouched: $(sha256sum /opt/igneum/igneum-prove-host | cut -c1-16)"
|
||||
$H --mode id | sed 's/^/RESULT aggcost-host /'
|
||||
python3 -c "import json,sys; d=json.load(open('$JOB/seq.json')); json.dump(d['result'], open('$JOB/export.json','w'))"
|
||||
LIST=""
|
||||
for n in $(seq $FIRST $LAST); do
|
||||
if ! $X "$JOB/export.json" $n "$JOB/block-$n.json" --source "PC 2 live devnet export, aggregation cost run" 2>&1 | tail -1 | sed "s/^/export $n: /"; then echo "RESULT cut $n FAILED"; exit 1; fi
|
||||
$H "$JOB/block-$n.json" --mode native 2>&1 | grep -E "^RESULT (native|plan)" | sed "s/^/block $n /"
|
||||
LIST="$LIST${LIST:+,}$JOB/block-$n.json"
|
||||
done
|
||||
echo "$LIST" > "$JOB/fixtures.txt"
|
||||
cp "$DEST/proving/fixtures/block-344-shards4.json" "$JOB/"
|
||||
# the parent hash of every fixture, for --mode aggregate
|
||||
python3 - <<PY
|
||||
import json
|
||||
for n in range($FIRST, $LAST + 1):
|
||||
f = json.load(open('$JOB/block-%d.json' % n))
|
||||
print('RESULT parent', n, f['block']['env']['parent_hash'] if 'parent_hash' in f['block']['env'] else f['block']['env'].get('parentHash'))
|
||||
open('$JOB/parent-%d.txt' % n, 'w').write(f['block']['env'].get('parent_hash') or f['block']['env'].get('parentHash'))
|
||||
f = json.load(open('$JOB/block-344-shards4.json'))
|
||||
open('$JOB/parent-344.txt', 'w').write(f['block']['env'].get('parent_hash') or f['block']['env'].get('parentHash'))
|
||||
PY
|
||||
echo "RESULT fixtures $(stamp) $LIST"
|
||||
# the live prover's GPU server must be gone (its environment and socket would be ours otherwise); and this job runs as
|
||||
# root, so the socket it leaves would be root-owned and the app's prover (another WSL user) could not connect (the
|
||||
# 21:25Z fault of job 1): the server is killed and the socket removed here and again at the end (the rule of
|
||||
# tools/ci/prover-socket-check.sh)
|
||||
for i in $(seq 1 36); do pgrep -x sp1-gpu-server >/dev/null || break; sleep 5; done
|
||||
pkill -f sp1-gpu-server 2>/dev/null; rm -f /tmp/sp1-cuda-*.sock # the root-socket class: the CI check (prover-socket-check.sh) wants pkill -f
|
||||
echo "RESULT gpu_server_before $(stamp) running=$(pgrep -x sp1-gpu-server | wc -l) socket=$(ls /tmp/sp1-cuda-0.sock 2>/dev/null || echo none)"
|
||||
'@
|
||||
$phase = @'
|
||||
# bash phase.sh <label> <cmd file>: runs the host command in the file (one line, env assignments first) with a 1-s
|
||||
# nvidia-smi sampler underneath; every RESULT/STAGE line comes out prefixed "<label>:"; a gpu line closes the phase.
|
||||
set -uo pipefail
|
||||
. "$(dirname "$0")/vars.sh"
|
||||
export PATH="$HOME/.cargo/bin:$HOME/.sp1/bin:$PATH"
|
||||
CUDA_DIR="$(ls -d /usr/local/cuda-12.* 2>/dev/null | sort -V | tail -1 || true)"; [ -n "$CUDA_DIR" ] && export PATH="$CUDA_DIR/bin:$PATH" && export LD_LIBRARY_PATH="$CUDA_DIR/lib64:/usr/lib/wsl/lib:${LD_LIBRARY_PATH:-}"
|
||||
stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; }
|
||||
L="$1"; CMD="$2"
|
||||
H=/opt/igneum-aggcost/igneum-prove-host
|
||||
export H
|
||||
nvidia-smi --query-gpu=timestamp,index,memory.used,utilization.gpu,power.draw --format=csv,noheader,nounits -l 1 > "$JOB/smi-$L.csv" 2>/dev/null &
|
||||
SMI=$!
|
||||
t0=$(date +%s.%N); START=$(stamp)
|
||||
echo "RESULT phase $L start $START cmd: $(cat "$CMD" | cut -c1-400)"
|
||||
bash "$CMD" > "$JOB/host-$L.log" 2> "$JOB/host-$L.err"
|
||||
RC=$?
|
||||
WALL=$(python3 -c "import time; print(round(time.time() - $t0, 1))")
|
||||
kill $SMI 2>/dev/null; sleep 1
|
||||
grep -E "^(RESULT|STAGE|igneum-prove-host sources|segment proof written)" "$JOB/host-$L.log" | sed "s/^/$L: /"
|
||||
awk -F', *' '{ if ($3+0 > max) max=$3+0; u+=$4; n++ } END { if (n) printf "RESULT phase_gpu LABEL samples=%d memory_used_max_mib=%d util_mean_pct=%.1f\n", n, max, u/n }' "$JOB/smi-$L.csv" | sed "s/LABEL/$L/"
|
||||
echo "RESULT phase $L end $(stamp) exit $RC wall $WALL s (started $START); stderr $(wc -l < "$JOB/host-$L.err") lines"
|
||||
if [ $RC -ne 0 ]; then tail -5 "$JOB/host-$L.err" | sed "s/^/$L stderr: /"; fi
|
||||
'@
|
||||
foreach ($pair in @(@('prep.sh', $prep), @('phase.sh', $phase))) {
|
||||
[IO.File]::WriteAllText((Join-Path $job $pair[0]), ($pair[1] -replace "`r`n", "`n"), (New-Object System.Text.UTF8Encoding $false))
|
||||
}
|
||||
function Wsl($file) { & wsl.exe -d Ubuntu-24.04 -u root -- bash (WslPath $file) 2>&1 | ForEach-Object { ($_ -replace "`0", '') } }
|
||||
function Cmd($label, $line) {
|
||||
$f = Join-Path $job "cmd-$label.sh"
|
||||
[IO.File]::WriteAllText($f, ($line + "`n"), (New-Object System.Text.UTF8Encoding $false))
|
||||
return (WslPath $f)
|
||||
}
|
||||
function RunPhase($label, $line) {
|
||||
$el = ((Get-Date) - $jobStart).TotalMinutes
|
||||
if ($el -gt 16.5) { "RESULT phase $label skipped: $([math]::Round($el,1)) min into the job (the 20-min window)"; return }
|
||||
$cmdW = Cmd $label $line
|
||||
& wsl.exe -d Ubuntu-24.04 -u root -- bash "$jobW/phase.sh" $label $cmdW 2>&1 | ForEach-Object { ($_ -replace "`0", '') }
|
||||
}
|
||||
function Fixtures { (Get-Content (Join-Path $job 'fixtures.txt') -Raw).Trim() }
|
||||
function Parent($n) { (Get-Content (Join-Path $job "parent-$n.txt") -Raw).Trim() }
|
||||
# proof groups for --mode aggregate: the four blocks' shard 0 proofs from the source job (one shard a block tonight)
|
||||
function SrcFirst { if ($ShardSourceFirst -gt 0) { $ShardSourceFirst } else { $first } }
|
||||
function SrcParent { if ($ShardSourceParent) { $ShardSourceParent } else { Parent $first } }
|
||||
function AggGroups($src) { $f = SrcFirst; (($f..($f + 3)) | ForEach-Object { "$src/block-$_-shard-0-compressed.bin" }) -join ';' }
|
||||
# --- own miner ---
|
||||
$script:miner = $null
|
||||
# (job 1, 21:10Z: `if (StartMiner ...)` was always true because the function's RESULT strings are part of its output; the
|
||||
# outcome is the script-scope flag $script:minerOk now)
|
||||
function StartMiner($label, $batchLog2) {
|
||||
$script:minerOk = $false
|
||||
if (-not $minerExe) { "RESULT $label own miner: no app miner command line was captured; phase void"; return }
|
||||
$alive = (Get-Process -ErrorAction SilentlyContinue | Where-Object { $_.ProcessName -like 'igneum-worker-cuda*' } | Measure-Object).Count
|
||||
if ($alive -gt 0) { "RESULT $label own miner: the app's CUDA worker is still running ($alive); the phase would mine twice on the card, so it is void (job 3, 22:48Z)"; return }
|
||||
$a = $minerArgs
|
||||
$extra = "--race off"
|
||||
if ($batchLog2 -and $batchLog2 -ne 22) { $extra = "$extra --batch-log2 $batchLog2" }
|
||||
if ($a -match '--worker-args "([^"]*)"') { $a = $a -replace '--worker-args "([^"]*)"', ('--worker-args "$1 ' + $extra + '"') }
|
||||
elseif ($a -match '--worker-args (\S+)') { $a = $a -replace '--worker-args (\S+)', ('--worker-args "$1 ' + $extra + '"') }
|
||||
else { $a = $a + ' --worker-args "' + $extra + '"' }
|
||||
$out = Join-Path $job "miner-$label.out"; $err = Join-Path $job "miner-$label.err"
|
||||
$script:miner = Start-Process -FilePath $minerExe -ArgumentList $a -WorkingDirectory $app -RedirectStandardOutput $out -RedirectStandardError $err -NoNewWindow -PassThru
|
||||
"RESULT $label own miner started $(Stamp) pid $($script:miner.Id) batch_log2=$batchLog2 args=$a"
|
||||
$w = 0
|
||||
while ($w -lt 150) {
|
||||
Start-Sleep -Seconds 10; $w += 10
|
||||
$s = Get-Content $out -ErrorAction SilentlyContinue | Where-Object { $_ -match 'STATUS' } | Select-Object -Last 1
|
||||
if ($s -and $s -match 'hash=([\d.]+) MH/s' -and [double]$Matches[1] -gt 10) { "RESULT $label own miner hashing after $w s: $s"; $script:minerOk = $true; return }
|
||||
if ($script:miner.HasExited) { "RESULT $label own miner EXITED (code $($script:miner.ExitCode)); err: $((Get-Content $err -ErrorAction SilentlyContinue | Select-Object -Last 3) -join ' / ')"; return }
|
||||
}
|
||||
"RESULT $label own miner not hashing after $w s; last lines: $((Get-Content $out -ErrorAction SilentlyContinue | Select-Object -Last 3) -join ' / ')"
|
||||
}
|
||||
function StopMiner($label) {
|
||||
if (-not $script:miner) { return }
|
||||
$pid0 = $script:miner.Id
|
||||
Get-CimInstance Win32_Process | Where-Object { $_.ParentProcessId -eq $pid0 } | ForEach-Object { Stop-Process -Id $_.ProcessId -Force -ErrorAction SilentlyContinue }
|
||||
Stop-Process -Id $pid0 -Force -ErrorAction SilentlyContinue
|
||||
Start-Sleep -Seconds 3
|
||||
$left = (Get-Process -ErrorAction SilentlyContinue | Where-Object { $_.ProcessName -like 'igneum-worker*' -or $_.ProcessName -like 'igneum-miner*' } | Measure-Object).Count
|
||||
"RESULT $label own miner stopped $(Stamp); miner/worker processes left: $left"
|
||||
$script:miner = $null
|
||||
}
|
||||
function MinerRate($label, $skip) {
|
||||
$out = Join-Path $job "miner-$label.out"
|
||||
$h = @(); foreach ($l in (Get-Content $out -ErrorAction SilentlyContinue | Where-Object { $_ -match 'STATUS' })) { if ($l -match 'now=([\d.]+) MH/s wall') { $h += [double]$Matches[1] } }
|
||||
if ($h.Count -gt $skip) { $h = $h[$skip..($h.Count - 1)] }
|
||||
if ($h.Count -eq 0) { return "n=0" }
|
||||
$m = $h | Measure-Object -Average -Minimum -Maximum
|
||||
"n=$($h.Count) mean=$([math]::Round($m.Average,2)) min=$([math]::Round($m.Minimum,2)) max=$([math]::Round($m.Maximum,2)) MH/s wall (the now= field of the STATUS lines every 10 s, the first $skip skipped)"
|
||||
}
|
||||
$paused = $false
|
||||
try {
|
||||
Wsl (Join-Path $job 'prep.sh')
|
||||
if (-not (Test-Path (Join-Path $job 'fixtures.txt'))) { "RESULT prep FAILED: no fixtures"; exit 1 }
|
||||
$list = Fixtures
|
||||
$first2 = ($list -split ',')[0..1] -join ','; $last2 = ($list -split ',')[2..3] -join ','
|
||||
if ($PhaseSet -contains 'A0') {
|
||||
RunPhase 'A0' "SP1_PROVER=cuda RUST_LOG=info `$H --mode chain --chain $(($list -split ',')[0]) --prover 0xCAfc6e74000000000000000000000000000000c2 --out `$JOB/results-A0.json"
|
||||
$errf = Join-Path $job 'host-A0.err'
|
||||
"RESULT A0 profile: $((Get-Content $errf -ErrorAction SilentlyContinue | Measure-Object).Count) stderr lines; the first 160 with a level or a duration follow"
|
||||
Get-Content $errf -ErrorAction SilentlyContinue | Where-Object { $_ -match 'INFO|WARN|DEBUG|ERROR|time\.|elapsed|took|ms\b|\bs\b' } | Select-Object -First 160 | ForEach-Object { "PROFILE $_" }
|
||||
}
|
||||
if ($PhaseSet -contains 'A') { RunPhase 'A' "SP1_PROVER=cuda RUST_LOG=off `$H --mode chain --chain $list --prover 0xCAfc6e74000000000000000000000000000000c2 --save-shards --out `$JOB/results-A.json" }
|
||||
if ($PhaseSet -contains 'B0') { RunPhase 'B0' "SP1_PROVER=cuda RUST_LOG=off `$H --mode aggregate --proofs '$(AggGroups $srcW)' --parent $(SrcParent) --out `$JOB/results-B0.json" }
|
||||
if ($PhaseSet -contains 'B') { RunPhase 'B' "SP1_WORKER_VERIFY_INTERMEDIATES=false SP1_PROVER=cuda RUST_LOG=off `$H --mode aggregate --proofs '$(AggGroups $srcW)' --parent $(SrcParent) --out `$JOB/results-B.json" }
|
||||
$needIdle = @('C','C1','C2','G0','D','E','E0','F','G','H') | Where-Object { $PhaseSet -contains $_ }
|
||||
if ($needIdle) {
|
||||
# the 5090 switched off through /api/cards (and on again in the finally block): /api/pause then /api/resume left
|
||||
# the worker off on 0.3.9 (21:25Z, an app defect in the resume path, fixed for 0.3.11)
|
||||
"RESULT card_off $(Stamp) $(CardSwitch $false $null)"; $paused = $true
|
||||
$w = 0; $left = 1
|
||||
while ($w -lt 120 -and $left -gt 0) { Start-Sleep -Seconds 5; $w += 5; $left = CudaWorkers }
|
||||
if ($left -gt 0) { "RESULT card_off_failed $(Stamp) the app's CUDA worker is still running after $w s ($left); /api/pause as the fallback: $(Post '/api/pause' @{})"; $w2 = 0; while ($w2 -lt 60 -and (CudaWorkers) -gt 0) { Start-Sleep -Seconds 5; $w2 += 5 }; $left = CudaWorkers }
|
||||
"RESULT paused $(Stamp) after $w s; CUDA workers left: $left; $(Smi)"
|
||||
Start-Sleep -Seconds 5
|
||||
}
|
||||
if ($PhaseSet -contains 'C') { RunPhase 'C' "SP1_PROVER=cuda RUST_LOG=off `$H --mode chain --chain $list --prover 0xCAfc6e74000000000000000000000000000000c2 --out `$JOB/results-C.json" }
|
||||
if ($PhaseSet -contains 'C1') { RunPhase 'C1' "SP1_PROVER=cuda RUST_LOG=off `$H --mode aggregate --proofs '$(AggGroups $srcW)' --parent $(SrcParent) --out `$JOB/results-C1.json" }
|
||||
if ($PhaseSet -contains 'C2') { RunPhase 'C2' "SP1_PROVER=cuda RUST_LOG=off `$H --mode chain --chain `$JOB/block-344-shards4.json --prover 0xCAfc6e74000000000000000000000000000000c2 --save-shards --out `$JOB/results-C2.json" }
|
||||
if ($PhaseSet -contains 'G0') {
|
||||
$c1 = Cmd 'G01' "SP1_PROVER=cuda RUST_LOG=off `$H --mode chain --chain $first2 --prover 0xCAfc6e74000000000000000000000000000000c2 --out `$JOB/results-G01.json"
|
||||
$c2 = Cmd 'G02' "SP1_PROVER=cuda RUST_LOG=off `$H --mode chain --chain $last2 --prover 0xCAfc6e74000000000000000000000000000000c2 --out `$JOB/results-G02.json"
|
||||
$p1 = Start-Process -FilePath 'wsl.exe' -ArgumentList "-d Ubuntu-24.04 -u root -- bash $jobW/phase.sh G01 $c1" -RedirectStandardOutput (Join-Path $job 'G01.out') -NoNewWindow -PassThru
|
||||
Start-Sleep -Seconds 2
|
||||
$p2 = Start-Process -FilePath 'wsl.exe' -ArgumentList "-d Ubuntu-24.04 -u root -- bash $jobW/phase.sh G02 $c2" -RedirectStandardOutput (Join-Path $job 'G02.out') -NoNewWindow -PassThru
|
||||
$p1.WaitForExit(); $p2.WaitForExit()
|
||||
foreach ($f in 'G01.out', 'G02.out') { ((Get-Content (Join-Path $job $f) -Raw -ErrorAction SilentlyContinue) -replace "`0", '') -split "`n" | ForEach-Object { $_.TrimEnd("`r") } | Where-Object { $_ } }
|
||||
}
|
||||
if ($PhaseSet -contains 'D') {
|
||||
StartMiner 'D' 22; if ($script:minerOk) { RunPhase 'D' "SP1_PROVER=cuda RUST_LOG=off `$H --mode chain --chain $list --prover 0xCAfc6e74000000000000000000000000000000c2 --out `$JOB/results-D.json"; "RESULT D miner rate $(MinerRate 'D' 2)" }
|
||||
StopMiner 'D'
|
||||
}
|
||||
if ($PhaseSet -contains 'E') {
|
||||
foreach ($b in 20, 18, 16) {
|
||||
$lab = "E$b"
|
||||
StartMiner $lab $b; if ($script:minerOk) { RunPhase $lab "SP1_PROVER=cuda RUST_LOG=off `$H --mode chain --chain $list --prover 0xCAfc6e74000000000000000000000000000000c2 --out `$JOB/results-$lab.json"; "RESULT $lab miner rate $(MinerRate $lab 2)" }
|
||||
StopMiner $lab
|
||||
}
|
||||
}
|
||||
if ($PhaseSet -contains 'E0') {
|
||||
$g = (0..3 | ForEach-Object { "$srcW/block-344-shard-$_-compressed.bin" }) -join ','
|
||||
StartMiner 'E0' 22; if ($script:minerOk) { RunPhase 'E0' "SP1_PROVER=cuda RUST_LOG=off `$H --mode aggregate --proofs '$g' --parent $(Parent 344) --out `$JOB/results-E0.json"; "RESULT E0 miner rate $(MinerRate 'E0' 2)" }
|
||||
StopMiner 'E0'
|
||||
}
|
||||
if ($PhaseSet -contains 'F') {
|
||||
$pol = (& nvidia-smi compute-policy --set-timeslice=1 2>&1) -join ' '
|
||||
"RESULT F timeslice set: $pol"
|
||||
if ($pol -notmatch 'rror|nsufficient|not supported|Unknown') {
|
||||
StartMiner 'F' 22; if ($script:minerOk) { RunPhase 'F' "SP1_PROVER=cuda RUST_LOG=off `$H --mode chain --chain $list --prover 0xCAfc6e74000000000000000000000000000000c2 --out `$JOB/results-F.json"; "RESULT F miner rate $(MinerRate 'F' 2)" }
|
||||
StopMiner 'F'
|
||||
}
|
||||
"RESULT F timeslice restored: $((& nvidia-smi compute-policy --set-timeslice=0 2>&1) -join ' ')"
|
||||
}
|
||||
if ($PhaseSet -contains 'G') {
|
||||
StartMiner 'G' 22; if ($script:minerOk) {
|
||||
$c1 = Cmd 'G1' "SP1_PROVER=cuda RUST_LOG=off `$H --mode chain --chain $first2 --prover 0xCAfc6e74000000000000000000000000000000c2 --out `$JOB/results-G1.json"
|
||||
$c2 = Cmd 'G2' "SP1_PROVER=cuda RUST_LOG=off `$H --mode chain --chain $last2 --prover 0xCAfc6e74000000000000000000000000000000c2 --out `$JOB/results-G2.json"
|
||||
$p1 = Start-Process -FilePath 'wsl.exe' -ArgumentList "-d Ubuntu-24.04 -u root -- bash $jobW/phase.sh G1 $c1" -RedirectStandardOutput (Join-Path $job 'G1.out') -NoNewWindow -PassThru
|
||||
Start-Sleep -Seconds 2
|
||||
$p2 = Start-Process -FilePath 'wsl.exe' -ArgumentList "-d Ubuntu-24.04 -u root -- bash $jobW/phase.sh G2 $c2" -RedirectStandardOutput (Join-Path $job 'G2.out') -NoNewWindow -PassThru
|
||||
$p1.WaitForExit(); $p2.WaitForExit()
|
||||
foreach ($f in 'G1.out', 'G2.out') { ((Get-Content (Join-Path $job $f) -Raw -ErrorAction SilentlyContinue) -replace "`0", '') -split "`n" | ForEach-Object { $_.TrimEnd("`r") } | Where-Object { $_ } }
|
||||
"RESULT G miner rate $(MinerRate 'G' 2)"
|
||||
}
|
||||
StopMiner 'G'
|
||||
}
|
||||
if ($PhaseSet -contains 'H') {
|
||||
# the combination from this job's own phases: the batch-log2 with the shortest chain among D and E (the miner's
|
||||
# rate alongside), the knob when phase B beat B0 by 5% or more; skipped when the job is past 15 min (the window)
|
||||
try {
|
||||
$cands = @(); foreach ($lab in 'D', 'E20', 'E18', 'E16') { $f = Join-Path $job "results-$lab.json"; if (Test-Path $f) { $r = (Get-Content $f -Raw | ConvertFrom-Json); $cands += [pscustomobject]@{ lab = $lab; batch = $(if ($lab -eq 'D') { 22 } else { [int]$lab.Substring(1) }); secs = [double]$r.chain_seconds } } }
|
||||
if ($cands.Count -gt 0) { $bestC = $cands | Sort-Object secs | Select-Object -First 1; $BestBatch = $bestC.batch; "RESULT H choice: $(($cands | ForEach-Object { "$($_.lab)=$([math]::Round($_.secs,1))s" }) -join ' ') -> batch_log2 $BestBatch" }
|
||||
$b0 = Join-Path $job 'results-B0.json'; $b1 = Join-Path $job 'results-B.json'
|
||||
if ((Test-Path $b0) -and (Test-Path $b1)) { $s0 = [double](Get-Content $b0 -Raw | ConvertFrom-Json).aggregate_seconds; $s1 = [double](Get-Content $b1 -Raw | ConvertFrom-Json).aggregate_seconds; if ($s1 -lt 0.95 * $s0) { $BestKnobs = 'SP1_WORKER_VERIFY_INTERMEDIATES=false' }; "RESULT H knobs: B0 $([math]::Round($s0,1)) s, B $([math]::Round($s1,1)) s -> '$BestKnobs'" }
|
||||
} catch { "RESULT H choice error: $_" }
|
||||
$elapsedMin = ((Get-Date) - $jobStart).TotalMinutes
|
||||
if ($elapsedMin -gt 15) { "RESULT H skipped: $([math]::Round($elapsedMin,1)) min into the job (the 20-min window)" }
|
||||
else { StartMiner 'H' $BestBatch }
|
||||
if ($elapsedMin -le 15 -and $script:minerOk) { RunPhase 'H' "$BestKnobs SP1_PROVER=cuda RUST_LOG=off `$H --mode chain --chain $list --prover 0xCAfc6e74000000000000000000000000000000c2 --out `$JOB/results-H.json"; "RESULT H miner rate $(MinerRate 'H' 2)" }
|
||||
StopMiner 'H'
|
||||
}
|
||||
} finally {
|
||||
StopMiner 'final'
|
||||
& wsl.exe -d Ubuntu-24.04 -u root -- bash -c 'pkill -x sp1-gpu-server; rm -f /tmp/sp1-cuda-*.sock; echo "RESULT socket_cleanup $(date -u +%Y-%m-%dT%H:%M:%SZ) servers=$(pgrep -x sp1-gpu-server | wc -l) sockets=$(ls /tmp/sp1-cuda-*.sock 2>/dev/null | wc -l)"' 2>&1 | ForEach-Object { ($_ -replace "`0", '') }
|
||||
if ($paused) {
|
||||
"RESULT card_on $(Stamp) $(CardSwitch $true $null)"; "RESULT resume $(Stamp) $(Post '/api/resume' @{})"
|
||||
$w3 = 0; while ($w3 -lt 90 -and (CudaWorkers) -eq 0) { Start-Sleep -Seconds 5; $w3 += 5 }
|
||||
if ((CudaWorkers) -eq 0) { "RESULT miner_back_failed $(Stamp) no CUDA worker after $w3 s; /api/start: $(Post '/api/start' @{})" } else { "RESULT miner_back $(Stamp) CUDA worker running after $w3 s" }
|
||||
}
|
||||
"RESULT prove_on $(Stamp) $(Post '/api/prove' @{on=$true})"
|
||||
"RESULT end $(Stamp) $(Smi)"
|
||||
}
|
||||
342
tools/proving-v1/pc2-segments.ps1
Normal file
342
tools/proving-v1/pc2-segments.ps1
Normal file
|
|
@ -0,0 +1,342 @@
|
|||
# Proving v1, the segment-aligned prover on PC 2 (6 October 2026, the project lead: "proving v1 is active but has proven zero
|
||||
# segments, and cannot with one prover; find a way to solve this"). A signed `run` job (shell powershell, not
|
||||
# elevated; the app keeps mining; the app's own prover is switched OFF for the run through /api/prove and ON again
|
||||
# at the end; the live /opt/igneum host is untouched: this build lands in /opt/igneum-segal).
|
||||
#
|
||||
# What it does, for RUN_MINUTES, exactly what the app's segment path (app/igneum-app/src/prover.rs, pick_segment and
|
||||
# prove_segment) does:
|
||||
# 1. the node's v1 status and the work list (igneum_getAssignedShards, lookback 600) grouped into whole untouched
|
||||
# segments (every block present, every shard open, unpaid, not in our pool); the newest one inside its deadline by
|
||||
# the margin (240 DAA, or 1.5x the last segment's time) is claimed; the app ranks by FNV of (first, key) for the
|
||||
# multi-prover spread, which with one prover changes nothing but the order
|
||||
# 2. the segment statement: executed, pending; fresh when the previous segment is not paid and no verified record of
|
||||
# it waits in the pool, else chained to the previous proof (--prev) when the pool holds it
|
||||
# 3. one export (igneum_exportSegments 0..last), one fixture per block, one host run (--mode chain --save-shards
|
||||
# [--prev]) with a 1-s nvidia-smi sampler underneath
|
||||
# 4. every shard record signed (igneum-miner.exe sign-record) and submitted (igneum_submitProofRecord), then the
|
||||
# segment record (sign-segment-record, igneum_submitSegmentRecord)
|
||||
# 5. the paid state of every submitted segment polled each pass (igneum_getSegmentRecords: carried, paid)
|
||||
# Every number is a RESULT line. The miner's hash rate comes from the app log's "status: ... MH/s" lines (every 30 s),
|
||||
# the 30 minutes before the job against the run. Card keys, when needed, come from settings.json, never /api/state
|
||||
# (that reply is "{}" on 0.3.11 once paid_wei passes u64::MAX; fixed in 6714a45).
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$RunMinutes = 30
|
||||
function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') }
|
||||
$app = if ($env:IGNEUM_APP_DIR) { $env:IGNEUM_APP_DIR } else { Join-Path $env:LOCALAPPDATA 'igneum\app' }
|
||||
$urlFile = Join-Path $app 'app.url'
|
||||
$base = if (Test-Path $urlFile) { (Get-Content $urlFile -Raw).Trim().TrimEnd('/') } else { '' }
|
||||
function Post($path, $obj) { try { (Invoke-RestMethod -Method Post -Uri "$base$path" -ContentType 'application/json' -Body ($obj | ConvertTo-Json -Compress) -TimeoutSec 10) | ConvertTo-Json -Compress } catch { "error: $_" } }
|
||||
function Rpc($method, $params) {
|
||||
try { (Invoke-RestMethod -Method Post -Uri "http://127.0.0.1:$evmPort" -ContentType 'application/json' -Body (@{jsonrpc='2.0'; id=1; method=$method; params=$params} | ConvertTo-Json -Compress -Depth 8) -TimeoutSec 120).result } catch { $script:rpcErr = "$_"; $null }
|
||||
}
|
||||
function RpcRaw($method, $params) {
|
||||
try { Invoke-RestMethod -Method Post -Uri "http://127.0.0.1:$evmPort" -ContentType 'application/json' -Body (@{jsonrpc='2.0'; id=1; method=$method; params=$params} | ConvertTo-Json -Compress -Depth 8) -TimeoutSec 120 } catch { $script:rpcErr = "$_"; $null }
|
||||
}
|
||||
function Hex($v) { if ($null -eq $v) { return 0 }; if ($v -is [string] -and $v.StartsWith('0x')) { return [Convert]::ToInt64($v.Substring(2), 16) }; return [int64]$v }
|
||||
$job = $env:IGNEUM_JOB_DIR
|
||||
if (-not $job) { $job = Join-Path $env:TEMP 'igneum-segments' }
|
||||
New-Item -ItemType Directory -Force -Path $job | Out-Null
|
||||
$jobStart = Get-Date
|
||||
"RESULT start $(Stamp) app_dir=$app base_len=$($base.Length) job=$job run_minutes=$RunMinutes"
|
||||
$evmPort = $null
|
||||
foreach ($p in 26790, 26800, 26810) { $evmPort = $p; if (Rpc 'igneum_getProvingStatus' @()) { break }; $evmPort = $null }
|
||||
if (-not $evmPort) { "RESULT refused $(Stamp) no node EVM RPC on 26790/26800/26810: $script:rpcErr"; exit 1 }
|
||||
# the app's miner (the signer) and its label, the payout address
|
||||
$mp = Get-CimInstance Win32_Process | Where-Object { $_.Name -like 'igneum-miner*' -and $_.CommandLine -like '* mine *' } | Select-Object -First 1
|
||||
$minerExe = if ($mp) { $mp.ExecutablePath } else { $null }
|
||||
if (-not $minerExe -or -not (Test-Path $minerExe)) {
|
||||
$cand = Get-ChildItem -Path (Split-Path $app -Parent) -Recurse -Filter 'igneum-miner.exe' -ErrorAction SilentlyContinue | Select-Object -First 1
|
||||
if ($cand) { $minerExe = $cand.FullName }
|
||||
}
|
||||
if (-not $minerExe) { "RESULT refused $(Stamp) no igneum-miner.exe found (the signer)"; exit 1 }
|
||||
$settingsRaw = Get-Content (Join-Path $app 'settings.json') -Raw
|
||||
$payout = [regex]::Match($settingsRaw, '"address":\s*"(0x[0-9a-fA-F]{40})"').Groups[1].Value
|
||||
if ($payout.Length -ne 42) { "RESULT refused $(Stamp) no payout address in settings.json"; exit 1 }
|
||||
$identities = 1; $im = [regex]::Match($settingsRaw, '"nvidia:[^"]*":\s*\{[^}]*"identities":\s*(\d+)'); if ($im.Success) { $identities = [int]$im.Groups[1].Value }
|
||||
$machineId = ''; $mm = [regex]::Match($settingsRaw, '"machine_id":\s*"([0-9a-f]+)"'); if ($mm.Success) { $machineId = $mm.Groups[1].Value }
|
||||
if (-not $machineId) { $mf = Join-Path $app 'machine-id'; if (Test-Path $mf) { $machineId = (Get-Content $mf -Raw).Trim() } }
|
||||
if (-not $machineId -and $job -match '([0-9a-f]{8})') { $machineId = $Matches[1] }
|
||||
$id8 = if ($machineId.Length -ge 8) { $machineId.Substring(0, 8) } else { '1ccfe586' }
|
||||
$label = if ($identities -gt 1) { "win-$id8-1-1" } else { "win-$id8-1" }
|
||||
$chain = 'igneum-devnet' # the app's chain_name with no devnet suffix (IGNEUM_APP_DEVNET_SUFFIX unset on the fleet)
|
||||
$keyHash = (& $minerExe key-hash $label 2>$null | Select-Object -Last 1).Trim()
|
||||
if ($keyHash -and -not $keyHash.StartsWith('0x')) { $keyHash = "0x$keyHash" }
|
||||
"RESULT signer $(Stamp) miner=$minerExe label=$label key=$keyHash chain=$chain payout=$payout identities=$identities"
|
||||
if (-not $keyHash -or $keyHash.Length -lt 64) { "RESULT refused $(Stamp) key-hash gave nothing"; exit 1 }
|
||||
# the live prover off for the run: its sp1-gpu-server would be shared with ours (same socket, its environment)
|
||||
"RESULT prove_off $(Stamp) $(Post '/api/prove' @{on=$false})"
|
||||
Start-Sleep -Seconds 5
|
||||
# --- WSL side: the package built into /opt/igneum-segal (root, the warm target of the live build), the segment runner
|
||||
$pkg = Join-Path $env:LOCALAPPDATA 'igneum\prove\igneum-prove-wsl2-segal\igneum-prove-wsl2'
|
||||
if (-not (Test-Path $pkg)) { $pkg = Join-Path $env:LOCALAPPDATA 'igneum\prove\igneum-prove-wsl2-segal' }
|
||||
"RESULT package $(Stamp) $pkg exists=$(Test-Path (Join-Path $pkg 'package'))"
|
||||
function WslPath($p) { $w = (& wsl.exe -d Ubuntu-24.04 -u root -- wslpath -a ($p -replace '\\', '/') 2>$null); if ($w) { ($w -replace "`0", '').Trim() } else { '/mnt/c' + ($p.Substring(2) -replace '\\', '/') } }
|
||||
$pkgW = WslPath $pkg; $jobW = WslPath $job
|
||||
$prep = @'
|
||||
set -uo pipefail
|
||||
export PATH="$HOME/.cargo/bin:$HOME/.sp1/bin:$PATH"
|
||||
CUDA_DIR="$(ls -d /usr/local/cuda-12.* 2>/dev/null | sort -V | tail -1 || true)"; [ -n "$CUDA_DIR" ] && export PATH="$CUDA_DIR/bin:$PATH" && export LD_LIBRARY_PATH="$CUDA_DIR/lib64:/usr/lib/wsl/lib:${LD_LIBRARY_PATH:-}"
|
||||
stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; }
|
||||
PKG="$1"
|
||||
DEST="$HOME/igneum-prove-segal"; LIVE_TARGET="$HOME/igneum-prove/proving/igneum-prove/target"
|
||||
mkdir -p "$DEST"
|
||||
rsync -a --delete --exclude target "$PKG/package/" "$DEST/"
|
||||
find "$DEST" -name target -prune -o -type f -exec touch {} + 2>/dev/null
|
||||
grep -o '"program_id": "0x[0-9a-f]*"' "$DEST/proving/igneum-prove/elf/manifest.json" | sed 's/^/RESULT manifest /'
|
||||
cd "$DEST/proving/igneum-prove"
|
||||
t0=$(date +%s)
|
||||
if ! CARGO_TARGET_DIR="$LIVE_TARGET" cargo build --release -p igneum-prove-export -p igneum-prove-host --features igneum-prove-host/cuda 2>&1 | tail -3; then echo "RESULT build FAILED"; exit 1; fi
|
||||
echo "RESULT build $(stamp) exit 0 in $(( $(date +%s) - t0 )) s"
|
||||
mkdir -p /opt/igneum-segal && cp "$LIVE_TARGET/release/igneum-prove-host" "$LIVE_TARGET/release/igneum-prove-export" /opt/igneum-segal/
|
||||
H=/opt/igneum-segal/igneum-prove-host; X=/opt/igneum-segal/igneum-prove-export
|
||||
echo "RESULT installed $(sha256sum $H | cut -c1-16) host, $(sha256sum $X | cut -c1-16) export; live /opt/igneum untouched: $(sha256sum /opt/igneum/igneum-prove-host | cut -c1-16)"
|
||||
$H --mode id | sed 's/^/RESULT segal-host /'
|
||||
echo "RESULT gpu_server_before $(stamp) running=$(pgrep -x sp1-gpu-server | wc -l) socket=$(ls -la /tmp/sp1-cuda-0.sock 2>/dev/null || echo none)"
|
||||
# the root-socket rule (tools/ci/prover-socket-check.sh): this run is root; the app's prover is off, so its server goes too
|
||||
pkill -f sp1-gpu-server; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
echo "RESULT socket_start $(stamp) servers=$(pgrep -x sp1-gpu-server | wc -l) sockets=$(ls /tmp/sp1-cuda-*.sock 2>/dev/null | wc -l)"
|
||||
'@
|
||||
# bash seg.sh <job dir> <first> <last> <payout> [prev file]: unwrap the export, cut the blocks, run the chain with the sampler
|
||||
$seg = @'
|
||||
set -uo pipefail
|
||||
export PATH="$HOME/.cargo/bin:$HOME/.sp1/bin:$PATH"
|
||||
CUDA_DIR="$(ls -d /usr/local/cuda-12.* 2>/dev/null | sort -V | tail -1 || true)"; [ -n "$CUDA_DIR" ] && export PATH="$CUDA_DIR/bin:$PATH" && export LD_LIBRARY_PATH="$CUDA_DIR/lib64:/usr/lib/wsl/lib:${LD_LIBRARY_PATH:-}"
|
||||
stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; }
|
||||
JOB="$1"; FIRST="$2"; LAST="$3"; PAYOUT="$4"; PREV="${5:-}"
|
||||
H=/opt/igneum-segal/igneum-prove-host; X=/opt/igneum-segal/igneum-prove-export
|
||||
D="$JOB/seg-$FIRST"; mkdir -p "$D"
|
||||
t0=$(date +%s.%N)
|
||||
python3 -c "import json; d=json.load(open('$D/seq.json')); json.dump(d['result'], open('$D/export.json','w'))" || { echo "RESULT seg $FIRST unwrap FAILED"; exit 1; }
|
||||
LIST=""
|
||||
for n in $(seq $FIRST $LAST); do
|
||||
if ! $X "$D/export.json" $n "$D/block-$n.json" --source "PC 2 live devnet, segment-aligned prover" >"$D/export-$n.log" 2>&1; then echo "RESULT seg $FIRST cut $n FAILED: $(tail -1 "$D/export-$n.log")"; exit 1; fi
|
||||
LIST="$LIST${LIST:+,}$D/block-$n.json"
|
||||
done
|
||||
CUT=$(python3 -c "import time; print(round(time.time() - $t0, 1))")
|
||||
echo "RESULT seg $FIRST cut $(stamp) $LAST blocks in $CUT s"
|
||||
nvidia-smi --query-gpu=timestamp,index,memory.used,utilization.gpu,power.draw --format=csv,noheader,nounits -l 1 > "$D/smi.csv" 2>/dev/null &
|
||||
SMI=$!
|
||||
t1=$(date +%s.%N)
|
||||
PREVARG=""; [ -n "$PREV" ] && PREVARG="--prev $PREV"
|
||||
SP1_PROVER=cuda RUST_LOG=off $H --mode chain --chain "$LIST" --prover "$PAYOUT" --save-shards $PREVARG --out "$D/chain-results.json" >"$D/chain.log" 2>"$D/chain.err"
|
||||
RC=$?
|
||||
WALL=$(python3 -c "import time; print(round(time.time() - $t1, 1))")
|
||||
kill $SMI 2>/dev/null; sleep 1
|
||||
grep -E "^RESULT (chain block|chain:|chain prev|setup)" "$D/chain.log" | sed "s/^/seg $FIRST: /"
|
||||
awk -F', *' '{ if ($3+0 > max) max=$3+0; u+=$4; n++ } END { if (n) printf "RESULT seg FIRST gpu samples=%d memory_used_max_mib=%d util_mean_pct=%.1f\n", n, max, u/n }' "$D/smi.csv" | sed "s/FIRST/$FIRST/"
|
||||
echo "RESULT seg $FIRST chain $(stamp) exit $RC wall $WALL s"
|
||||
if [ $RC -ne 0 ]; then tail -4 "$D/chain.err" | sed "s/^/seg $FIRST stderr: /"; exit 1; fi
|
||||
# the submit bodies: one JSON-RPC request per shard record and one for the segment, the proof bytes as hex
|
||||
python3 - "$D" <<'PY'
|
||||
import json, sys, os
|
||||
d = sys.argv[1]
|
||||
r = json.load(open(os.path.join(d, 'chain-results.json')))
|
||||
recs = []
|
||||
for b in r['blocks']:
|
||||
for s in b.get('shard_records', []):
|
||||
recs.append(s)
|
||||
json.dump(recs, open(os.path.join(d, 'shard-records.json'), 'w'))
|
||||
print('RESULT seg %s records %d shard records, segment chain_len %s, proof %s bytes' % (r['first'], len(recs), r['segment_chain_len'], r['segment_proof_bytes']))
|
||||
PY
|
||||
'@
|
||||
# bash body.sh <proof file> <record hex> <method> <out file>: the JSON-RPC body with the proof as hex
|
||||
$body = @'
|
||||
set -u
|
||||
python3 - "$1" "$2" "$3" "$4" <<'PY'
|
||||
import json, sys, binascii
|
||||
proof = '0x' + binascii.hexlify(open(sys.argv[1], 'rb').read()).decode()
|
||||
json.dump({'jsonrpc': '2.0', 'id': 1, 'method': sys.argv[3], 'params': [{'record': sys.argv[2], 'proof': proof}]}, open(sys.argv[4], 'w'))
|
||||
PY
|
||||
'@
|
||||
# bash unhex.sh <hex file> <out file>: a hex string (0x...) to bytes
|
||||
$unhex = @'
|
||||
set -u
|
||||
python3 -c "import sys,binascii; h=open(sys.argv[1]).read().strip(); h=h[2:] if h.startswith('0x') else h; open(sys.argv[2],'wb').write(binascii.unhexlify(h))" "$1" "$2"
|
||||
'@
|
||||
foreach ($pair in @(@('prep.sh', $prep), @('seg.sh', $seg), @('body.sh', $body), @('unhex.sh', $unhex))) {
|
||||
[IO.File]::WriteAllText((Join-Path $job $pair[0]), ($pair[1] -replace "`r`n", "`n"), (New-Object System.Text.UTF8Encoding $false))
|
||||
}
|
||||
function Wsl { param([string[]]$a) & wsl.exe -d Ubuntu-24.04 -u root -- bash @a 2>&1 | ForEach-Object { ($_ -replace "`0", '') } }
|
||||
Wsl @("$jobW/prep.sh", $pkgW)
|
||||
if (-not (Test-Path (Join-Path $job 'prep.sh'))) { "RESULT refused $(Stamp) prep.sh missing"; exit 1 }
|
||||
# --- the app log's miner rate before the run (baseline): "status: ... <x> MH/s" every 30 s
|
||||
$logDir = Join-Path $app 'logs'
|
||||
function RateLines($since, $until) {
|
||||
$out = @()
|
||||
$files = @(Get-ChildItem -Path $logDir, (Join-Path (Split-Path $app -Parent) 'logs'), $app -Filter '*.log' -ErrorAction SilentlyContinue | Where-Object { $_.LastWriteTime -gt (Get-Date).AddHours(-3) } | Sort-Object LastWriteTime -Descending | Select-Object -First 4)
|
||||
if ($files.Count -eq 0) { "RESULT rate_files none under $logDir, $(Join-Path (Split-Path $app -Parent) 'logs'), $app" }
|
||||
foreach ($f in $files) {
|
||||
foreach ($l in (Get-Content $f.FullName -ErrorAction SilentlyContinue | Where-Object { $_ -match '^(\d{10}) status: .* ([\d.]+) MH/s' })) {
|
||||
$t = [int64]$Matches[1]; if ($t -ge $since -and $t -le $until) { $out += [double]$Matches[2] }
|
||||
}
|
||||
}
|
||||
return $out
|
||||
}
|
||||
function Epoch($d) { [int64](($d.ToUniversalTime()) - (Get-Date '1970-01-01')).TotalSeconds }
|
||||
$t0 = Epoch $jobStart
|
||||
$before = RateLines ($t0 - 1800) $t0
|
||||
if ($before.Count) { $m = $before | Measure-Object -Average -Minimum -Maximum; "RESULT rate_before $(Stamp) n=$($before.Count) mean=$([math]::Round($m.Average,2)) min=$($m.Minimum) max=$($m.Maximum) MH/s (the app log's status lines, the 30 min before the job)" } else { "RESULT rate_before $(Stamp) no status lines found in $logDir" }
|
||||
# --- the loop
|
||||
$submittedSegments = @{} # first -> @{last; aggWei; at}
|
||||
$attempted = @{}
|
||||
$lastSegSecs = 0
|
||||
$n = 1; $start = 0; $unproven = 600
|
||||
$passes = 0; $claimed = 0; $submitted = 0; $shardsAccepted = 0; $shardsRefused = 0
|
||||
$runStart = Get-Date
|
||||
function Candidates {
|
||||
$st = Rpc 'igneum_getProvingStatus' @()
|
||||
if (-not $st -or -not $st.v1 -or -not $st.v1.active -or -not $st.v1.start) { return @() }
|
||||
$script:start = Hex $st.v1.start; $script:n = [math]::Max(1, (Hex $st.v1.segmentBlocks)); $script:unproven = Hex $st.v1.unprovenDaa
|
||||
$tipDaa = Hex $st.tipDaa
|
||||
$work = Rpc 'igneum_getAssignedShards' @(@($keyHash), 600)
|
||||
if (-not $work) { return @() }
|
||||
$by = @{}
|
||||
foreach ($w in $work) { [int64]$num = Hex $w.number; if ($num -lt $script:start) { continue }; if (-not $by.ContainsKey($num)) { $by[$num] = @() }; $by[$num] += $w }
|
||||
"RESULT worklist $(Stamp) entries=$(@($work).Count) blocks=$($by.Count) start=$($script:start) n=$($script:n) tipDaa=$tipDaa"
|
||||
$need = [math]::Max(240, [math]::Ceiling($lastSegSecs * 1.5))
|
||||
$segs = @()
|
||||
$seen = @{}
|
||||
foreach ($num in ($by.Keys | Sort-Object)) {
|
||||
# [int64] throughout: [math]::Floor gives a double, and a double never matches an int64 hashtable key (job b,
|
||||
# 07:20Z to 07:50Z: every segment "not whole", 88 passes, nothing claimed)
|
||||
[int64]$k = [math]::Floor(($num - $script:start) / $script:n); [int64]$first = $script:start + $k * $script:n; [int64]$last = $first + $script:n - 1
|
||||
if ($seen.ContainsKey($first)) { continue }; $seen[$first] = $true
|
||||
if ($attempted.ContainsKey($first)) { continue }
|
||||
$whole = $true; $shards = @(); $lastDaa = 0; $wei = 0
|
||||
for ([int64]$b = $first; $b -le $last; $b++) {
|
||||
if (-not $by.ContainsKey($b)) { $whole = $false; break }
|
||||
$es = $by[$b] | Sort-Object { $_.shard } -Unique
|
||||
foreach ($e in $es) {
|
||||
$inPool = ($e.pool -and (@($e.pool)).Count -gt 0)
|
||||
if (-not $e.open -or $null -ne $e.paid -or $inPool) { $whole = $false; break }
|
||||
$shards += @{ number = $b; hash = $e.hash; shard = [int]$e.shard }
|
||||
if ($b -eq $last) { $lastDaa = Hex $e.daaScore }
|
||||
}
|
||||
if (-not $whole) { break }
|
||||
}
|
||||
if ($whole -and $shards.Count -gt 0) {
|
||||
$deadline = $lastDaa + $script:unproven
|
||||
if ($deadline -ge ($tipDaa + 1 + $need)) { $segs += @{ first = $first; last = $last; lastDaa = $lastDaa; deadline = $deadline; shards = $shards; margin = [int64]($deadline - $tipDaa - 1) } }
|
||||
}
|
||||
}
|
||||
# newest first (the most time before the deadline); the app ranks by FNV(first, key) for the spread
|
||||
return @($segs | Sort-Object { $_.first } -Descending)
|
||||
}
|
||||
while (((Get-Date) - $runStart).TotalMinutes -lt $RunMinutes) {
|
||||
$passes++
|
||||
# paid segments
|
||||
foreach ($f in @($submittedSegments.Keys)) {
|
||||
$rec = Rpc 'igneum_getSegmentRecords' @(('0x{0:x}' -f $f))
|
||||
if ($rec) {
|
||||
$carried = (@($rec.carried)).Count
|
||||
if ($rec.paid) { "RESULT paid $(Stamp) segment $f..$($submittedSegments[$f].last) paid wei=$($rec.paid.wei) ($([math]::Round((Hex $rec.paid.wei) / 1e18, 4)) IGN) carrier=$(Hex $rec.paid.carrierNumber) payout=$($rec.paid.payout) after $([math]::Round(((Get-Date) - $submittedSegments[$f].at).TotalSeconds)) s"; $submittedSegments.Remove($f) }
|
||||
elseif ($carried -gt 0) { "RESULT carried $(Stamp) segment $f carried=$carried valid=$(@($rec.carried | ForEach-Object { $_.valid }) -join ',') rejected=$(@($rec.carried | ForEach-Object { $_.rejected }) -join ';')" }
|
||||
}
|
||||
}
|
||||
$cands = Candidates
|
||||
if (-not $cands -or $cands.Count -eq 0) { "RESULT pass $passes $(Stamp) no whole segment inside the margin (need=$([math]::Max(240, [math]::Ceiling($lastSegSecs * 1.5))) DAA); waiting 20 s"; Start-Sleep -Seconds 20; continue }
|
||||
$picked = $null; $prevFile = $null; $expected = ''
|
||||
$tries = 0
|
||||
foreach ($c in $cands) {
|
||||
if ($tries -ge 3) { break }; $tries++
|
||||
$stmt = Rpc 'igneum_getSegmentStatement' @(('0x{0:x}' -f $c.first))
|
||||
if (-not $stmt -or -not $stmt.executed -or $stmt.status.status -ne 'pending') { $attempted[$c.first] = $true; "RESULT skip $(Stamp) segment $($c.first): executed=$($stmt.executed) status=$($stmt.status.status)"; continue }
|
||||
if ($null -eq $stmt.previous) {
|
||||
if ($c.first -ge ($start + $n)) {
|
||||
$pr = Rpc 'igneum_getSegmentRecords' @(('0x{0:x}' -f ($c.first - $n)))
|
||||
$waiting = $false
|
||||
if ($pr -and $pr.pool) { foreach ($e in @($pr.pool)) { if ($e.verified -eq $true -and $null -eq $e.includedIn) { $waiting = $true } } }
|
||||
if ($waiting -or ($pr -and $pr.paid)) { "RESULT skip $(Stamp) segment $($c.first): the previous segment has a record waiting or paid; a fresh chain would be refused"; continue }
|
||||
}
|
||||
$picked = $c; $expected = $stmt.publicValuesFresh; break
|
||||
}
|
||||
if ($stmt.previous.proofInPool -ne $true) { "RESULT skip $(Stamp) segment $($c.first): previous paid, its proof not in this pool"; continue }
|
||||
$got = Rpc 'igneum_getSegmentProofBytes' @($stmt.previous.first, $stmt.previous.keyHash)
|
||||
if (-not $got -or -not $got.proof) { continue }
|
||||
$hexFile = Join-Path $job "prev-$($c.first).hex"; $binFile = Join-Path $job "prev-$($c.first).bin"
|
||||
[IO.File]::WriteAllText($hexFile, $got.proof)
|
||||
Wsl @("$jobW/unhex.sh", (WslPath $hexFile), (WslPath $binFile))
|
||||
if (-not (Test-Path $binFile)) { continue }
|
||||
$picked = $c; $prevFile = WslPath $binFile; $expected = $stmt.publicValuesContinuing; break
|
||||
}
|
||||
if (-not $picked) { "RESULT pass $passes $(Stamp) $($cands.Count) candidates, none usable this pass; waiting 20 s"; Start-Sleep -Seconds 20; continue }
|
||||
$first = $picked.first; $last = $picked.last
|
||||
$attempted[$first] = $true; $claimed++
|
||||
$tipNow = Hex (Rpc 'eth_blockNumber' @())
|
||||
"RESULT claim $(Stamp) segment $first..$last ($($picked.shards.Count) shards, $(if ($prevFile) { 'continuing' } else { 'fresh' })), margin=$($picked.margin) DAA, tip=$tipNow, candidates=$($cands.Count)"
|
||||
$segStart = Get-Date
|
||||
$d = Join-Path $job "seg-$first"; New-Item -ItemType Directory -Force -Path $d | Out-Null
|
||||
# 3. the export (curl.exe streams the reply to a file; Invoke-WebRequest's Content is a string there)
|
||||
$t = Get-Date
|
||||
$bodyTxt = (@{jsonrpc='2.0'; id=1; method='igneum_exportSegments'; params=@('0x0', ('0x{0:x}' -f $last))} | ConvertTo-Json -Compress)
|
||||
$bodyFile = Join-Path $d 'export-request.json'; $seqFile = Join-Path $d 'seq.json'
|
||||
[IO.File]::WriteAllText($bodyFile, $bodyTxt, (New-Object System.Text.UTF8Encoding $false))
|
||||
& curl.exe -s -S -m 600 -X POST "http://127.0.0.1:$evmPort" -H 'Content-Type: application/json' --data-binary "@$bodyFile" -o $seqFile 2>&1 | ForEach-Object { "curl: $_" }
|
||||
if (-not (Test-Path $seqFile) -or (Get-Item $seqFile).Length -lt 1000) { "RESULT seg $first export FAILED"; continue }
|
||||
"RESULT seg $first export $(Stamp) $((Get-Item $seqFile).Length) bytes in $([math]::Round(((Get-Date) - $t).TotalSeconds,1)) s"
|
||||
$segArgs = @("$jobW/seg.sh", $jobW, "$first", "$last", $payout); if ($prevFile) { $segArgs += $prevFile }
|
||||
Wsl $segArgs
|
||||
Remove-Item $seqFile -ErrorAction SilentlyContinue
|
||||
$resFile = Join-Path $d 'chain-results.json'
|
||||
if (-not (Test-Path $resFile)) { "RESULT seg $first FAILED: no chain results"; continue }
|
||||
$res = Get-Content $resFile -Raw | ConvertFrom-Json
|
||||
$recs = Get-Content (Join-Path $d 'shard-records.json') -Raw | ConvertFrom-Json
|
||||
# 4. the shard records
|
||||
$okShards = 0
|
||||
foreach ($r in $recs) {
|
||||
$sg = (& $minerExe sign-record $label $chain $r.block_hash "$($r.number)" "$($r.shard)" $payout $r.statement $r.proof_sha256 2>&1 | Select-Object -Last 1)
|
||||
$signed = $null; try { $signed = $sg | ConvertFrom-Json } catch {}
|
||||
if (-not $signed -or -not $signed.record) { "RESULT seg $first shard $($r.number)/$($r.shard) sign FAILED: $sg"; $shardsRefused++; continue }
|
||||
$proofWin = $r.proof_file -replace '^/mnt/c/', 'C:/' -replace '/', '\'
|
||||
$bf = Join-Path $d "body-$($r.number)-$($r.shard).json"
|
||||
Wsl @("$jobW/body.sh", $r.proof_file, $signed.record, 'igneum_submitProofRecord', (WslPath $bf))
|
||||
$reply = (& curl.exe -s -S -m 120 -X POST "http://127.0.0.1:$evmPort" -H 'Content-Type: application/json' --data-binary "@$bf" 2>&1)
|
||||
$rr = $null; try { $rr = ($reply | ConvertFrom-Json).result } catch {}
|
||||
if ($rr -and $rr.accepted) { $okShards++; $shardsAccepted++ } else { $shardsRefused++; "RESULT seg $first shard $($r.number)/$($r.shard) refused: $(if ($rr) { $rr.reason } else { "$reply".Substring(0, [math]::Min(300, "$reply".Length)) })" }
|
||||
Remove-Item $bf -ErrorAction SilentlyContinue
|
||||
}
|
||||
"RESULT seg $first shards $(Stamp) accepted $okShards of $($recs.Count) (prove $([math]::Round($res.shard_prove_seconds_total,1)) s, aggregation $([math]::Round($res.aggregate_prove_seconds_total,1)) s, chain $([math]::Round($res.chain_seconds,1)) s)"
|
||||
if ($okShards -ne $recs.Count) { "RESULT seg $first FAILED: not every shard record accepted; the segment record is not submitted"; continue }
|
||||
# 5. the segment record: the statement against the node's (every field but provers, bytes 236..268 of the 340)
|
||||
$pv = $res.segment_public_values
|
||||
function Strip($h) { $h = $h -replace '^0x', ''; if ($h.Length -eq 680) { $h.Substring(0, 472) + $h.Substring(536) } else { $h } }
|
||||
if ((Strip $pv) -ne (Strip $expected)) { "RESULT seg $first FAILED: statement differs from the node's native one; ours $($pv.Substring(0,66)) node $($expected.Substring(0, [math]::Min(66, $expected.Length)))"; continue }
|
||||
$lastHash = ($picked.shards | Where-Object { $_.number -eq $last } | Select-Object -First 1).hash
|
||||
$sg = (& $minerExe sign-segment-record $label $chain "$first" "$last" $lastHash $payout $pv $res.segment_proof_sha256 2>&1 | Select-Object -Last 1)
|
||||
$signed = $null; try { $signed = $sg | ConvertFrom-Json } catch {}
|
||||
if (-not $signed -or -not $signed.record) { "RESULT seg $first segment sign FAILED: $sg"; continue }
|
||||
$bf = Join-Path $d 'body-segment.json'
|
||||
Wsl @("$jobW/body.sh", $res.segment_proof_file, $signed.record, 'igneum_submitSegmentRecord', (WslPath $bf))
|
||||
$reply = (& curl.exe -s -S -m 120 -X POST "http://127.0.0.1:$evmPort" -H 'Content-Type: application/json' --data-binary "@$bf" 2>&1)
|
||||
$rr = $null; try { $rr = ($reply | ConvertFrom-Json).result } catch {}
|
||||
Remove-Item $bf -ErrorAction SilentlyContinue
|
||||
$segSecs = [math]::Round(((Get-Date) - $segStart).TotalSeconds, 1)
|
||||
if ($rr -and $rr.accepted) {
|
||||
$submitted++; $lastSegSecs = $segSecs
|
||||
$stmt2 = Rpc 'igneum_getSegmentStatement' @(('0x{0:x}' -f $first))
|
||||
$aggWei = if ($stmt2) { Hex $stmt2.aggregatorWei } else { 0 }
|
||||
$submittedSegments[$first] = @{ last = $last; aggWei = $aggWei; at = (Get-Date) }
|
||||
"RESULT seg $first submitted $(Stamp) segment $first..$last record accepted (new=$($rr.new), chain_len $($res.segment_chain_len)), aggregator share $([math]::Round($aggWei / 1e18, 4)) IGN, end to end $segSecs s (export+cut+chain+sign+submit)"
|
||||
} else {
|
||||
"RESULT seg $first segment refused $(Stamp): $(if ($rr) { $rr.reason } else { "$reply".Substring(0, [math]::Min(300, "$reply".Length)) }); end to end $segSecs s"
|
||||
}
|
||||
foreach ($b in $first..$last) { Remove-Item (Join-Path $d "block-$b.json") -ErrorAction SilentlyContinue }
|
||||
Remove-Item (Join-Path $d 'export.json') -ErrorAction SilentlyContinue
|
||||
}
|
||||
# --- the end: paid state once more, the rate during the run, the sockets, the prover back on
|
||||
Start-Sleep -Seconds 30
|
||||
$paidTotal = 0; $paidCount = 0
|
||||
foreach ($f in @($submittedSegments.Keys)) {
|
||||
$rec = Rpc 'igneum_getSegmentRecords' @(('0x{0:x}' -f $f))
|
||||
if ($rec -and $rec.paid) { $paidCount++; $paidTotal += (Hex $rec.paid.wei); "RESULT paid $(Stamp) segment $f..$($submittedSegments[$f].last) paid wei=$($rec.paid.wei) carrier=$(Hex $rec.paid.carrierNumber)"; $submittedSegments.Remove($f) }
|
||||
else { "RESULT unpaid $(Stamp) segment $f..$($submittedSegments[$f].last) carried=$(if ($rec) { (@($rec.carried)).Count } else { '?' }) pool=$(if ($rec) { (@($rec.pool)).Count } else { '?' }) $(if ($rec -and $rec.pool) { ($rec.pool | ForEach-Object { "verified=$($_.verified) included=$($_.includedIn) note=$($_.note)" }) -join '; ' })" }
|
||||
}
|
||||
$st = Rpc 'igneum_getProvingStatus' @()
|
||||
if ($st) { "RESULT node_v1 $(Stamp) paidSegments=$($st.v1.paidSegments) paidSegmentWei=$($st.v1.paidSegmentWei) window pending=$($st.v1.segmentsInWindow.pending) proven=$($st.v1.segmentsInWindow.proven) unproven=$($st.v1.segmentsInWindow.unproven) pool entries=$($st.v1.pool.entries) verified=$($st.v1.pool.verified) failed=$($st.v1.pool.failed)" }
|
||||
$t1 = Epoch (Get-Date)
|
||||
$during = RateLines (Epoch $runStart) $t1
|
||||
if ($during.Count) { $m = $during | Measure-Object -Average -Minimum -Maximum; "RESULT rate_during $(Stamp) n=$($during.Count) mean=$([math]::Round($m.Average,2)) min=$($m.Minimum) max=$($m.Maximum) MH/s (the app log's status lines over the run)" } else { "RESULT rate_during $(Stamp) no status lines found" }
|
||||
"RESULT summary $(Stamp) passes=$passes claimed=$claimed submitted=$submitted paid=$paidCount paid_wei=$paidTotal shards_accepted=$shardsAccepted shards_refused=$shardsRefused last_segment_s=$lastSegSecs run_min=$([math]::Round(((Get-Date) - $runStart).TotalMinutes,1))"
|
||||
& wsl.exe -d Ubuntu-24.04 -u root -- bash -c 'pkill -f sp1-gpu-server; sleep 1; rm -f /tmp/sp1-cuda-*.sock; echo "RESULT socket_cleanup $(date -u +%Y-%m-%dT%H:%M:%SZ) servers=$(pgrep -x sp1-gpu-server | wc -l) sockets=$(ls /tmp/sp1-cuda-*.sock 2>/dev/null | wc -l)"' 2>&1 | ForEach-Object { ($_ -replace "`0", '') }
|
||||
"RESULT prove_on $(Stamp) $(Post '/api/prove' @{on=$true})"
|
||||
"RESULT end $(Stamp)"
|
||||
|
|
@ -14,10 +14,17 @@
|
|||
// variant without a race; the record then carries only the self-test), the default keeps racing (the fleet keeps
|
||||
// learning while the card starts from the known best).
|
||||
// node tools/tuning.mjs --records [--days 7] [--card <model>] the raw records, newest first
|
||||
// node tools/tuning.mjs --priors [--days 30] [--min-samples 5] Ember Tune (docs/plans/ember-tune.md): the fleet
|
||||
// priors per (card model, driver major, program class) from the TUNE records (relay/lib/ember.mjs aggregate);
|
||||
// --write adds them to the tuning file under "priors" with the "ember" settings (kill switch --tuning-off,
|
||||
// --rate-tolerance N), beside the kernel-variant cards; --site writes site/miner-priors.json for /miners.
|
||||
//
|
||||
// Reads DATABASE_URL from ~/.config/igneum/env. No dependencies: Neon HTTP SQL over fetch.
|
||||
import { readFileSync, writeFileSync } from 'node:fs';
|
||||
import { readFileSync, writeFileSync, existsSync } from 'node:fs';
|
||||
import { homedir } from 'node:os';
|
||||
import { join, dirname, resolve } from 'node:path';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
import { parseRecords, aggregate, mergeTuning, priorLine } from '../relay/lib/ember.mjs';
|
||||
|
||||
process.stdout.on('error', e => { if (e.code === 'EPIPE') process.exit(0); throw e; });
|
||||
|
||||
|
|
@ -46,6 +53,43 @@ const minSamples = Number(opt('--min-samples', 3));
|
|||
const by = opt('--by', 'mhs');
|
||||
const outFile = opt('--write', '');
|
||||
const onlyCard = opt('--card', '');
|
||||
const ROOT = resolve(dirname(fileURLToPath(import.meta.url)), '..');
|
||||
|
||||
// ---- Ember Tune priors (--priors): the TUNE records of the window, folded per key --------------------------------
|
||||
if (flag('--priors')) {
|
||||
const pdays = Number(opt('--days', 30));
|
||||
const pmin = Number(opt('--min-samples', 5));
|
||||
const prow = await sql(
|
||||
`SELECT machine, run_id, lines FROM miner_logs WHERE received_at > now() - ($1 || ' days')::interval AND lines LIKE '%TUNE {%' ORDER BY received_at DESC`,
|
||||
[String(pdays)]);
|
||||
const recs = prow.flatMap(r => parseRecords(r.lines)).filter(r => !onlyCard || r.card === onlyCard);
|
||||
const { priors, table } = aggregate(recs, { minSamples: pmin });
|
||||
if (!recs.length) console.log(`No TUNE records in the last ${pdays} day(s). The apps log one per finished tune (0.3.10 and later).`);
|
||||
else {
|
||||
console.log(`${recs.length} tune record(s) in the last ${pdays} day(s); a prior needs ${pmin} sample(s)`);
|
||||
console.table(table.map(t => ({ key: t.key, samples: t.samples, machines: t.machines, 'clock MHz': t.clock_mhz ?? '-', 'power %': t.power_pct ?? '-', 'MH/W': t.eff ?? '-', 'MH/s': t.mhs ?? '-', W: t.watts ?? '-', 'spread %': t.spread_pct ?? '-', 'untuned MH/W': t.before_eff ?? t.baseline_eff ?? '-', 'gain %': t.gain_pct ?? '-', prior: t.key in priors ? 'yes' : 'no' })));
|
||||
for (const p of Object.values(priors)) console.log(priorLine(p));
|
||||
}
|
||||
if (outFile) {
|
||||
const existing = existsSync(outFile) ? JSON.parse(readFileSync(outFile, 'utf8')) : {};
|
||||
const ember = {};
|
||||
if (flag('--tuning-off')) ember.enabled = false;
|
||||
if (flag('--tuning-on')) ember.enabled = true;
|
||||
if (opt('--rate-tolerance', '')) ember.rate_tolerance_pct = Number(opt('--rate-tolerance', '1'));
|
||||
ember.min_samples = pmin;
|
||||
const merged = mergeTuning(existing, priors, ember);
|
||||
writeFileSync(outFile, JSON.stringify(merged, null, 2) + '\n');
|
||||
console.log(`written ${outFile}: ${Object.keys(merged.cards).length} kernel-variant card(s) kept, ${Object.keys(priors).length} prior(s), ember ${JSON.stringify(merged.ember)}; publish with: packaging/ota/publish-manifest.sh --version <current> --tuning ${outFile} [--deploy]`);
|
||||
}
|
||||
if (flag('--site')) {
|
||||
const site = join(ROOT, 'site', 'miner-priors.json');
|
||||
const rows = table.map(t => ({ card: t.card.replace(/_/g, ' '), vendor: t.vendor, driver_major: t.driver_major, class: t.class, clock_mhz: t.clock_mhz ?? null, power_pct: t.power_pct ?? null, mh_per_w: t.eff ?? null, mh_s: t.mhs ?? null, watts: t.watts ?? null, spread_pct: t.spread_pct ?? null, samples: t.samples, machines: t.machines, untuned_mh_per_w: t.before_eff ?? t.baseline_eff ?? null, gain_pct: t.gain_pct ?? null, prior: t.key in priors, updated: t.updated || null }));
|
||||
const about = existsSync(site) ? JSON.parse(readFileSync(site, 'utf8'))._about : undefined;
|
||||
writeFileSync(site, JSON.stringify({ _about: about || 'Rows of the fleet priors table at /miners (site/build.mjs), written by tools/tuning.mjs --priors --site from the TUNE records every Igneum Miner uploads. One row per card model, driver major and program class: the median tuned point, MH per watt, the spread and the sample count. No machine names, no addresses.', generated: new Date().toISOString().replace(/\.\d{3}Z$/, 'Z'), min_samples: pmin, rows }, null, 2) + '\n');
|
||||
console.log(`written ${site} (${rows.length} row(s))`);
|
||||
}
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
// Every app-log upload of the window; the TUNING lines out of them. The same race is uploaded many times (the log
|
||||
// is re-sent every minute), so records are de-duplicated on (machine, card, epoch).
|
||||
|
|
|
|||
101
tools/windows/console-watch-bg.ps1
Normal file
101
tools/windows/console-watch-bg.ps1
Normal file
|
|
@ -0,0 +1,101 @@
|
|||
# Background console-window watcher for a Windows PC running the Igneum Miner app: the second half of
|
||||
# tools/windows/console-watch.ps1. That one proved (PC 1, 5 October 2026, job run-20261005-182528) that no child a
|
||||
# job script starts from the app's headless console opens a window; this one finds what does. A signed `run` job
|
||||
# starts a detached PowerShell (Start-Process -WindowStyle Hidden, the shape the first watcher showed opens nothing)
|
||||
# and returns at once; the detached process samples for WATCH_MINUTES and writes <job dir>\console-windows.log:
|
||||
# <utc> window <process> pid <p> [<class>] <title> a new visible console or terminal window
|
||||
# <utc> proc <name> pid <p> cmd <command line> <- <parent chain, name pid and command line each>
|
||||
# a new cmd, powershell, wsl, conhost, OpenConsole or WindowsTerminal
|
||||
# <utc> gone <name> pid <p> one of those ended (the window's life)
|
||||
# Read it back with a collect job: packaging/ota/publish-jobs.sh add --kind collect --target ae432dc7 \
|
||||
# --glob "app/jobs/<this job id>/console-windows.log" --deploy
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$minutes = 25
|
||||
$dir = $env:IGNEUM_JOB_DIR
|
||||
$log = Join-Path $dir 'console-windows.log'
|
||||
$bg = Join-Path $dir 'bg.ps1'
|
||||
# the detached body: params first (PowerShell wants them at the top), then the sampler
|
||||
$body = @'
|
||||
param([string] $Log, [int] $Minutes)
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$src = @"
|
||||
using System;
|
||||
using System.Collections.Generic;
|
||||
using System.Runtime.InteropServices;
|
||||
using System.Text;
|
||||
public static class IgWin2 {
|
||||
public delegate bool EnumProc(IntPtr h, IntPtr l);
|
||||
[DllImport("user32.dll")] public static extern bool EnumWindows(EnumProc p, IntPtr l);
|
||||
[DllImport("user32.dll")] public static extern bool IsWindowVisible(IntPtr h);
|
||||
[DllImport("user32.dll")] public static extern uint GetWindowThreadProcessId(IntPtr h, out uint pid);
|
||||
[DllImport("user32.dll", CharSet = CharSet.Unicode)] public static extern int GetWindowText(IntPtr h, StringBuilder s, int n);
|
||||
[DllImport("user32.dll", CharSet = CharSet.Unicode)] public static extern int GetClassName(IntPtr h, StringBuilder s, int n);
|
||||
public static List<string> Visible() {
|
||||
var list = new List<string>();
|
||||
EnumWindows((h, l) => {
|
||||
if (!IsWindowVisible(h)) return true;
|
||||
uint pid; GetWindowThreadProcessId(h, out pid);
|
||||
var t = new StringBuilder(512); GetWindowText(h, t, 512);
|
||||
var c = new StringBuilder(256); GetClassName(h, c, 256);
|
||||
list.Add(((long)h).ToString() + "|" + pid + "|" + c + "|" + t);
|
||||
return true;
|
||||
}, IntPtr.Zero);
|
||||
return list;
|
||||
}
|
||||
}
|
||||
"@
|
||||
Add-Type -TypeDefinition $src
|
||||
function Now() { (Get-Date).ToUniversalTime().ToString('HH:mm:ss.fff') }
|
||||
function Put([string] $l) { Add-Content -Path $Log -Value $l -Encoding UTF8 }
|
||||
function Chain([int] $procId) {
|
||||
$out = @(); $seen = @{}; $p = $procId
|
||||
for ($i = 0; $i -lt 6 -and $p -gt 0 -and -not $seen.ContainsKey($p); $i++) {
|
||||
$seen[$p] = 1
|
||||
$ci = Get-CimInstance Win32_Process -Filter "ProcessId=$p" -ErrorAction SilentlyContinue
|
||||
if (-not $ci) { $out += ("pid " + $p + " gone"); break }
|
||||
$cl = [string]$ci.CommandLine; if ($cl.Length -gt 160) { $cl = $cl.Substring(0, 160) + '...' }
|
||||
$out += ($ci.Name + " pid " + $p + " [" + $cl + "]")
|
||||
$p = $ci.ParentProcessId
|
||||
}
|
||||
return ($out -join ' <- ')
|
||||
}
|
||||
$watch = 'cmd', 'powershell', 'pwsh', 'wsl', 'wslhost', 'conhost', 'OpenConsole', 'WindowsTerminal'
|
||||
$classes = 'ConsoleWindowClass', 'CASCADIA_HOSTING_WINDOW_CLASS', 'PseudoConsoleWindow'
|
||||
$knownWin = @{}; $knownProc = @{}
|
||||
$first = $true
|
||||
$end = (Get-Date).AddMinutes($Minutes)
|
||||
Put ((Now) + " start: watching for " + $Minutes + " min, pid " + $PID)
|
||||
while ((Get-Date) -lt $end) {
|
||||
try {
|
||||
foreach ($w in [IgWin2]::Visible()) {
|
||||
$f = $w.Split('|', 4)
|
||||
if ($knownWin.ContainsKey($f[0])) { continue }
|
||||
$knownWin[$f[0]] = 1
|
||||
if ($first) { continue }
|
||||
$pn = try { (Get-Process -Id ([int]$f[1]) -ErrorAction Stop).ProcessName } catch { 'gone' }
|
||||
if ($classes -contains $f[2] -or $watch -contains $pn) { Put ((Now) + " window " + $pn + " pid " + $f[1] + " [" + $f[2] + "] " + $f[3]) }
|
||||
}
|
||||
$live = @{}
|
||||
foreach ($p in @(Get-Process -Name $watch -ErrorAction SilentlyContinue)) {
|
||||
$live[$p.Id] = 1
|
||||
if ($knownProc.ContainsKey($p.Id)) { continue }
|
||||
$knownProc[$p.Id] = $p.ProcessName
|
||||
if ($first) { continue }
|
||||
Put ((Now) + " proc " + $p.ProcessName + " pid " + $p.Id + " " + (Chain $p.Id))
|
||||
}
|
||||
foreach ($k in @($knownProc.Keys)) { if (-not $live.ContainsKey($k)) { if (-not $first) { Put ((Now) + " gone " + $knownProc[$k] + " pid " + $k) }; $knownProc.Remove($k) } }
|
||||
if ($first) { Put ((Now) + " baseline: " + $knownWin.Count + " visible windows, " + $knownProc.Count + " watched processes: " + (($knownProc.GetEnumerator() | ForEach-Object { $_.Value + ' ' + $_.Key }) -join ', ')) }
|
||||
$first = $false
|
||||
} catch { Put ((Now) + " error " + $_) }
|
||||
Start-Sleep -Milliseconds 200
|
||||
}
|
||||
Put ((Now) + " end")
|
||||
'@
|
||||
[IO.File]::WriteAllText($bg, $body, (New-Object System.Text.UTF8Encoding($true)))
|
||||
if (Test-Path $log) { Remove-Item $log -Force }
|
||||
$p = Start-Process -FilePath powershell.exe -ArgumentList @('-NoProfile', '-ExecutionPolicy', 'Bypass', '-File', $bg, '-Log', $log, '-Minutes', $minutes) -WindowStyle Hidden -PassThru
|
||||
Start-Sleep -Seconds 3
|
||||
$alive = try { -not (Get-Process -Id $p.Id -ErrorAction Stop).HasExited } catch { $false }
|
||||
Write-Output ("RESULT watcher: pid " + $p.Id + " alive " + $alive + " for " + $minutes + " min, log " + $log)
|
||||
if (Test-Path $log) { Get-Content $log | ForEach-Object { Write-Output ("RESULT first: " + $_) } }
|
||||
exit $(if ($alive) { 0 } else { 1 })
|
||||
84
tools/windows/console-watch-elevated.ps1
Normal file
84
tools/windows/console-watch-elevated.ps1
Normal file
|
|
@ -0,0 +1,84 @@
|
|||
# Console-window watcher for the ELEVATED job path (app/igneum-app/src/jobrun.rs run_script with elevated=true: the
|
||||
# app's headless powershell runs `Start-Process powershell.exe -Verb RunAs -Wait -WindowStyle Hidden`, the AppInfo
|
||||
# service creates this process after the UAC prompt). The engine's power cap, the sweep helper and the clock sync take
|
||||
# the same road with cmd.exe (platform.rs run_elevated, sync_clock; app/windows/host.cpp runElevated). This script
|
||||
# runs INSIDE the elevated process and reports whether its own console has a window, which host serves it, and
|
||||
# whether a Windows Terminal window appeared for it. One UAC prompt on the PC; a few seconds.
|
||||
# packaging/ota/publish-jobs.sh add --kind run --target ae432dc7 --elevated --timeout-minutes 3 \
|
||||
# --script tools/windows/console-watch-elevated.ps1 --title "PC 1: elevated console watcher" --deploy
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$src = @'
|
||||
using System;
|
||||
using System.Collections.Generic;
|
||||
using System.Runtime.InteropServices;
|
||||
using System.Text;
|
||||
public static class IgWin3 {
|
||||
public delegate bool EnumProc(IntPtr h, IntPtr l);
|
||||
[DllImport("user32.dll")] public static extern bool EnumWindows(EnumProc p, IntPtr l);
|
||||
[DllImport("user32.dll")] public static extern bool IsWindowVisible(IntPtr h);
|
||||
[DllImport("user32.dll")] public static extern uint GetWindowThreadProcessId(IntPtr h, out uint pid);
|
||||
[DllImport("user32.dll", CharSet = CharSet.Unicode)] public static extern int GetWindowText(IntPtr h, StringBuilder s, int n);
|
||||
[DllImport("user32.dll", CharSet = CharSet.Unicode)] public static extern int GetClassName(IntPtr h, StringBuilder s, int n);
|
||||
[DllImport("kernel32.dll")] public static extern IntPtr GetConsoleWindow();
|
||||
public static List<string> Visible() {
|
||||
var list = new List<string>();
|
||||
EnumWindows((h, l) => {
|
||||
if (!IsWindowVisible(h)) return true;
|
||||
uint pid; GetWindowThreadProcessId(h, out pid);
|
||||
var t = new StringBuilder(512); GetWindowText(h, t, 512);
|
||||
var c = new StringBuilder(256); GetClassName(h, c, 256);
|
||||
list.Add(((long)h).ToString() + "|" + pid + "|" + c + "|" + t);
|
||||
return true;
|
||||
}, IntPtr.Zero);
|
||||
return list;
|
||||
}
|
||||
}
|
||||
'@
|
||||
Add-Type -TypeDefinition $src
|
||||
function Say([string] $m) { Write-Output $m }
|
||||
function ProcName([int] $procId) { try { (Get-Process -Id $procId -ErrorAction Stop).ProcessName } catch { 'gone' } }
|
||||
function Chain([int] $procId) {
|
||||
$out = @(); $seen = @{}; $p = $procId
|
||||
for ($i = 0; $i -lt 6 -and $p -gt 0 -and -not $seen.ContainsKey($p); $i++) {
|
||||
$seen[$p] = 1
|
||||
$ci = Get-CimInstance Win32_Process -Filter "ProcessId=$p" -ErrorAction SilentlyContinue
|
||||
if (-not $ci) { $out += ("pid " + $p + " gone"); break }
|
||||
$cl = [string]$ci.CommandLine; if ($cl.Length -gt 140) { $cl = $cl.Substring(0, 140) + '...' }
|
||||
$out += ($ci.Name + " pid " + $p + " [" + $cl + "]")
|
||||
$p = $ci.ParentProcessId
|
||||
}
|
||||
return ($out -join ' <- ')
|
||||
}
|
||||
$me = [System.Diagnostics.Process]::GetCurrentProcess()
|
||||
$id = [Security.Principal.WindowsIdentity]::GetCurrent()
|
||||
$admin = (New-Object Security.Principal.WindowsPrincipal($id)).IsInRole([Security.Principal.WindowsBuiltInRole]::Administrator)
|
||||
Say ("RESULT elevated: " + $admin + " user " + $id.Name + " session " + $me.SessionId + " chain " + (Chain $me.Id))
|
||||
$hwnd = [IgWin3]::GetConsoleWindow()
|
||||
$cls = ''
|
||||
if ($hwnd -ne [IntPtr]::Zero) { $sb = New-Object System.Text.StringBuilder 256; [void][IgWin3]::GetClassName($hwnd, $sb, 256); $cls = $sb.ToString() }
|
||||
$vis = if ($hwnd -ne [IntPtr]::Zero) { [IgWin3]::IsWindowVisible($hwnd) } else { 'no window' }
|
||||
Say ("RESULT self-console: hwnd " + $hwnd + " class [" + $cls + "] visible " + $vis)
|
||||
# the hosts that serve this process: a conhost with this pid as parent (classic), or an OpenConsole + WindowsTerminal
|
||||
# pair started by svchost in the last seconds (the default-terminal handoff)
|
||||
$since = (Get-Date).AddSeconds(-20)
|
||||
foreach ($h in @(Get-CimInstance Win32_Process -Filter "Name='conhost.exe' OR Name='OpenConsole.exe' OR Name='WindowsTerminal.exe'" -ErrorAction SilentlyContinue)) {
|
||||
$created = try { [Management.ManagementDateTimeConverter]::ToDateTime($h.CreationDate) } catch { $null }
|
||||
if ($h.ParentProcessId -eq $me.Id -or ($created -and $created -gt $since)) {
|
||||
Say ("RESULT host: " + $h.Name + " pid " + $h.ProcessId + " parent " + (ProcName $h.ParentProcessId) + " started " + $(if ($created) { $created.ToUniversalTime().ToString('HH:mm:ss') } else { '?' }) + " cmd " + $h.CommandLine)
|
||||
}
|
||||
}
|
||||
$wins = @([IgWin3]::Visible() | Where-Object { $f = $_.Split('|', 4); $f[2] -eq 'CASCADIA_HOSTING_WINDOW_CLASS' -or $f[2] -eq 'ConsoleWindowClass' -or $f[2] -eq 'PseudoConsoleWindow' })
|
||||
Say ("RESULT console-windows-now: " + $wins.Count)
|
||||
foreach ($w in $wins) { $f = $w.Split('|', 4); Say ("RESULT window: " + (ProcName ([int]$f[1])) + " pid " + $f[1] + " [" + $f[2] + "] " + $f[3]) }
|
||||
# a child the elevated script starts the way the sweep helper and the power cap do (cmd, inherited console), watched
|
||||
$before = @{}; foreach ($w in [IgWin3]::Visible()) { $before[$w.Split('|', 4)[0]] = 1 }
|
||||
$p = Start-Process -FilePath cmd.exe -ArgumentList '/c ping -n 3 127.0.0.1 >nul' -NoNewWindow -PassThru
|
||||
$seen = @{}
|
||||
for ($i = 0; $i -lt 30; $i++) {
|
||||
foreach ($w in [IgWin3]::Visible()) { $f = $w.Split('|', 4); if (-not $before.ContainsKey($f[0]) -and -not $seen.ContainsKey($f[0])) { $seen[$f[0]] = (ProcName ([int]$f[1])) + " pid " + $f[1] + " [" + $f[2] + "] " + $f[3] } }
|
||||
Start-Sleep -Milliseconds 100
|
||||
}
|
||||
try { $p.WaitForExit(10000) | Out-Null } catch {}
|
||||
Say ("RESULT probe cmd-inherit-elevated: " + $seen.Count + " new window(s)")
|
||||
foreach ($s in $seen.Values) { Say ("RESULT window: " + $s + " (probe cmd-inherit-elevated)") }
|
||||
exit 0
|
||||
181
tools/windows/console-watch.ps1
Normal file
181
tools/windows/console-watch.ps1
Normal file
|
|
@ -0,0 +1,181 @@
|
|||
# Console-window watcher for a Windows PC running the Igneum Miner app. A signed `run` job (app/igneum-app/src/jobrun.rs,
|
||||
# shell powershell, not elevated): while a sampler thread enumerates the visible top-level windows (user32 EnumWindows,
|
||||
# IsWindowVisible, GetWindowThreadProcessId, GetClassName, GetWindowText) and the console host processes (conhost,
|
||||
# OpenConsole, WindowsTerminal, with their command lines and parents) every 30 ms, the main thread starts each
|
||||
# candidate child the way a job script or the app does, and every window or host that appears during a probe is
|
||||
# reported against it:
|
||||
# RESULT terminal: ... the default-terminal delegation (HKCU\Console\%%Startup) and the host process counts
|
||||
# RESULT self: ... the job's own console (hidden or not) and the conhost that serves it
|
||||
# RESULT probe <n>: ... exit code, duration, how many windows and hosts appeared
|
||||
# RESULT window: <process> [<class>] <title> (probe <n>)
|
||||
# RESULT host: <name> pid <p> parent <process> cmd <command line> (probe <n>)
|
||||
# Trust test (CLAUDE.md: a watcher is trusted only after a known-finished and a known-failed case): probe
|
||||
# start-process-new-console MUST report a window (cmd in a new console); start-process-hidden is the same with
|
||||
# -WindowStyle Hidden. 5 October 2026: written for PC 1 (ae432dc7, Windows 11 Pro 26200), where the project lead saw
|
||||
# "Windows Command Processor" windows whenever a remote job ran.
|
||||
# packaging/ota/publish-jobs.sh add --kind run --target ae432dc7 --script tools/windows/console-watch.ps1 \
|
||||
# --timeout-minutes 5 --title "PC 1: console window watcher" --deploy
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$src = @'
|
||||
using System;
|
||||
using System.Collections.Generic;
|
||||
using System.Runtime.InteropServices;
|
||||
using System.Text;
|
||||
public static class IgWin {
|
||||
public delegate bool EnumProc(IntPtr h, IntPtr l);
|
||||
[DllImport("user32.dll")] public static extern bool EnumWindows(EnumProc p, IntPtr l);
|
||||
[DllImport("user32.dll")] public static extern bool IsWindowVisible(IntPtr h);
|
||||
[DllImport("user32.dll")] public static extern uint GetWindowThreadProcessId(IntPtr h, out uint pid);
|
||||
[DllImport("user32.dll", CharSet = CharSet.Unicode)] public static extern int GetWindowText(IntPtr h, StringBuilder s, int n);
|
||||
[DllImport("user32.dll", CharSet = CharSet.Unicode)] public static extern int GetClassName(IntPtr h, StringBuilder s, int n);
|
||||
[DllImport("kernel32.dll")] public static extern IntPtr GetConsoleWindow();
|
||||
public static List<string> Visible() {
|
||||
var list = new List<string>();
|
||||
EnumWindows((h, l) => {
|
||||
if (!IsWindowVisible(h)) return true;
|
||||
uint pid; GetWindowThreadProcessId(h, out pid);
|
||||
var t = new StringBuilder(512); GetWindowText(h, t, 512);
|
||||
var c = new StringBuilder(256); GetClassName(h, c, 256);
|
||||
list.Add(((long)h).ToString() + "|" + pid + "|" + c + "|" + t);
|
||||
return true;
|
||||
}, IntPtr.Zero);
|
||||
return list;
|
||||
}
|
||||
}
|
||||
'@
|
||||
try { Add-Type -TypeDefinition $src -ErrorAction Stop } catch { if (-not ([System.Management.Automation.PSTypeName]'IgWin').Type) { Write-Output ("RESULT error: Add-Type failed: " + $_); exit 2 } }
|
||||
|
||||
function Say([string] $m) { Write-Output $m }
|
||||
function ProcName([int] $procId) { try { (Get-Process -Id $procId -ErrorAction Stop).ProcessName } catch { 'gone' } }
|
||||
|
||||
# ---- the default terminal and the hosts present before anything starts ------------------------------------------------
|
||||
$CONHOST_ID = '{B23D10C0-E52E-411E-9D5B-C09FDF709C7D}'
|
||||
$DECIDE_ID = '{00000000-0000-0000-0000-000000000000}'
|
||||
$k = Get-ItemProperty -Path 'HKCU:\Console\%%Startup' -ErrorAction SilentlyContinue
|
||||
$dc = if ($k) { [string]$k.DelegationConsole } else { '(absent)' }
|
||||
$dt = if ($k) { [string]$k.DelegationTerminal } else { '(absent)' }
|
||||
$meaning = if ($dc -eq $CONHOST_ID) { 'Windows Console Host (conhost)' } elseif ($dc -eq $DECIDE_ID -or $dc -eq '(absent)') { 'Let Windows decide (Windows Terminal on Windows 11 22H2 and later when it is installed)' } else { 'a terminal package, Windows Terminal or its preview' }
|
||||
$wtPkg = (Get-AppxPackage -Name 'Microsoft.WindowsTerminal*' -ErrorAction SilentlyContinue | ForEach-Object { $_.Name + ' ' + $_.Version }) -join ', '
|
||||
Say ("RESULT terminal: DelegationConsole=" + $dc + " DelegationTerminal=" + $dt + " -> " + $meaning + "; Windows Terminal package: " + $(if ($wtPkg) { $wtPkg } else { 'none' }))
|
||||
$os = Get-CimInstance Win32_OperatingSystem
|
||||
Say ("RESULT os: " + $os.Caption + " build " + $os.BuildNumber + " user " + $env:USERNAME + " session " + [System.Diagnostics.Process]::GetCurrentProcess().SessionId)
|
||||
$counts = @{}
|
||||
foreach ($n in 'conhost', 'OpenConsole', 'WindowsTerminal', 'cmd', 'powershell', 'wsl', 'wslhost') { $counts[$n] = @(Get-Process -Name $n -ErrorAction SilentlyContinue).Count }
|
||||
Say ("RESULT hosts-before: conhost " + $counts['conhost'] + " OpenConsole " + $counts['OpenConsole'] + " WindowsTerminal " + $counts['WindowsTerminal'] + " cmd " + $counts['cmd'] + " powershell " + $counts['powershell'] + " wsl " + $counts['wsl'] + " wslhost " + $counts['wslhost'])
|
||||
|
||||
# the job's own console and the chain above it
|
||||
$me = [System.Diagnostics.Process]::GetCurrentProcess()
|
||||
$meCim = Get-CimInstance Win32_Process -Filter "ProcessId=$($me.Id)"
|
||||
$parent = if ($meCim) { ProcName $meCim.ParentProcessId } else { '?' }
|
||||
$hwnd = [IgWin]::GetConsoleWindow()
|
||||
$selfVis = if ($hwnd -ne [IntPtr]::Zero) { [IgWin]::IsWindowVisible($hwnd) } else { 'no window' }
|
||||
$ownHost = Get-CimInstance Win32_Process -Filter "Name='conhost.exe' OR Name='OpenConsole.exe'" | Where-Object { $_.ParentProcessId -eq $me.Id -or $_.ParentProcessId -eq $meCim.ParentProcessId }
|
||||
$ownLine = if ($ownHost) { ($ownHost | ForEach-Object { $_.Name + ' pid ' + $_.ProcessId + ' parent ' + (ProcName $_.ParentProcessId) + ' cmd ' + $_.CommandLine }) -join ' ; ' } else { 'none with this script or its parent as parent' }
|
||||
Say ("RESULT self: powershell pid " + $me.Id + " parent " + $parent + " (pid " + $meCim.ParentProcessId + "); console hwnd " + $hwnd + " visible " + $selfVis + "; host " + $ownLine)
|
||||
foreach ($w in [IgWin]::Visible()) {
|
||||
$f = $w.Split('|', 4)
|
||||
if ($f[2] -eq 'ConsoleWindowClass' -or $f[2] -eq 'CASCADIA_HOSTING_WINDOW_CLASS') { Say ("RESULT window-before: " + (ProcName ([int]$f[1])) + " [" + $f[2] + "] " + $f[3]) }
|
||||
}
|
||||
|
||||
# ---- the sampler thread: every new visible window and every new console host, with the time it was first seen -------
|
||||
$sync = [hashtable]::Synchronized(@{ win = [hashtable]::Synchronized(@{}); hosts = [hashtable]::Synchronized(@{}); stop = $false; ticks = 0; err = '' })
|
||||
$rs = [runspacefactory]::CreateRunspace()
|
||||
$rs.Open()
|
||||
$rs.SessionStateProxy.SetVariable('sync', $sync)
|
||||
$rs.SessionStateProxy.SetVariable('src', $src)
|
||||
$sampler = [powershell]::Create()
|
||||
$sampler.Runspace = $rs
|
||||
[void]$sampler.AddScript({
|
||||
try {
|
||||
if (-not ([System.Management.Automation.PSTypeName]'IgWin').Type) { Add-Type -TypeDefinition $src }
|
||||
$first = $true
|
||||
while (-not $sync.stop) {
|
||||
$now = Get-Date
|
||||
foreach ($w in [IgWin]::Visible()) {
|
||||
$f = $w.Split('|', 4)
|
||||
if (-not $sync.win.ContainsKey($f[0])) {
|
||||
$pn = try { (Get-Process -Id ([int]$f[1]) -ErrorAction Stop).ProcessName } catch { 'gone' }
|
||||
$sync.win[$f[0]] = @{ t = $now; procId = [int]$f[1]; proc = $pn; cls = $f[2]; title = $f[3]; base = $first }
|
||||
}
|
||||
}
|
||||
foreach ($p in @(Get-Process -Name conhost, OpenConsole, WindowsTerminal -ErrorAction SilentlyContinue)) {
|
||||
if (-not $sync.hosts.ContainsKey($p.Id)) {
|
||||
$ci = Get-CimInstance Win32_Process -Filter "ProcessId=$($p.Id)" -ErrorAction SilentlyContinue
|
||||
$ppid = if ($ci) { $ci.ParentProcessId } else { 0 }
|
||||
$ppn = try { (Get-Process -Id $ppid -ErrorAction Stop).ProcessName } catch { 'gone' }
|
||||
$sync.hosts[$p.Id] = @{ t = $now; name = $p.ProcessName; cmd = $(if ($ci) { [string]$ci.CommandLine } else { '?' }); ppid = $ppid; pproc = $ppn; base = $first }
|
||||
}
|
||||
}
|
||||
$first = $false
|
||||
$sync.ticks++
|
||||
Start-Sleep -Milliseconds 30
|
||||
}
|
||||
} catch { $sync.err = [string]$_ }
|
||||
})
|
||||
$handle = $sampler.BeginInvoke()
|
||||
$t = 0
|
||||
while ($sync.ticks -lt 2 -and $t -lt 100) { Start-Sleep -Milliseconds 50; $t++ }
|
||||
if ($sync.ticks -lt 2) { Say ("RESULT error: the sampler did not start: " + $sync.err); exit 2 }
|
||||
Say ("sampler running: " + $sync.win.Count + " visible windows and " + $sync.hosts.Count + " console hosts at the start")
|
||||
|
||||
# ---- probes: each one as a job script or the app would start it ----------------------------------------------------
|
||||
$report = New-Object System.Collections.ArrayList
|
||||
function Probe([string] $name, [scriptblock] $body) {
|
||||
Start-Sleep -Milliseconds 400
|
||||
$t0 = Get-Date
|
||||
$global:LASTEXITCODE = 0
|
||||
$err = ''
|
||||
try { & $body 2>&1 | Out-Null } catch { $err = [string]$_ }
|
||||
$rc = $LASTEXITCODE
|
||||
Start-Sleep -Milliseconds 600
|
||||
$t1 = Get-Date
|
||||
$ms = [int]($t1 - $t0).TotalMilliseconds - 600
|
||||
$wins = @($sync.win.GetEnumerator() | Where-Object { -not $_.Value.base -and $_.Value.t -ge $t0 -and $_.Value.t -le $t1 -and -not $_.Value.reported })
|
||||
$hosts = @($sync.hosts.GetEnumerator() | Where-Object { -not $_.Value.base -and $_.Value.t -ge $t0 -and $_.Value.t -le $t1 -and -not $_.Value.reported })
|
||||
Say ("RESULT probe " + $name + ": exit " + $rc + " in " + $ms + " ms, " + $wins.Count + " window(s), " + $hosts.Count + " host(s)" + $(if ($err) { "; error " + $err } else { '' }))
|
||||
foreach ($w in $wins) { $w.Value.reported = $true; Say ("RESULT window: " + $w.Value.proc + " [" + $w.Value.cls + "] " + $w.Value.title + " (probe " + $name + ")") }
|
||||
foreach ($h in $hosts) { $h.Value.reported = $true; Say ("RESULT host: " + $h.Value.name + " pid " + $h.Key + " parent " + $h.Value.pproc + " cmd " + $h.Value.cmd + " (probe " + $name + ")") }
|
||||
}
|
||||
$distro = 'Ubuntu-24.04'
|
||||
$haveDistro = $false
|
||||
try { $haveDistro = ((& wsl.exe -l -q 2>$null) -replace "`0", '' | Where-Object { $_.Trim() -eq $distro }).Count -gt 0 } catch {}
|
||||
Say ("wsl distro " + $distro + ": " + $(if ($haveDistro) { 'present' } else { 'absent, the distro probes are skipped' }))
|
||||
|
||||
# a. the plain children a job script starts (CreateProcess, the console inherited)
|
||||
Probe 'powershell-inherit' { & powershell.exe -NoProfile -ExecutionPolicy Bypass -Command 'Start-Sleep -Milliseconds 1200' }
|
||||
Probe 'cmd-c-inherit' { & cmd.exe /c 'ping -n 3 127.0.0.1 >nul' }
|
||||
Probe 'query-session' { & query.exe session }
|
||||
Probe 'curl-version' { & curl.exe --version }
|
||||
Probe 'nvidia-smi-L' { & nvidia-smi.exe -L }
|
||||
Probe 'powershell-windowstyle-hidden-inherit' { & powershell.exe -NoProfile -WindowStyle Hidden -Command 'Start-Sleep -Milliseconds 1200' }
|
||||
Probe 'wsl-status' { & wsl.exe --status }
|
||||
if ($haveDistro) {
|
||||
Probe 'wsl-distro-sleep' { & wsl.exe -d Ubuntu-24.04 -u root -- sleep 1 }
|
||||
Probe 'wsl-interop-cmd' { & wsl.exe -d Ubuntu-24.04 -u root -- cmd.exe /c 'ping -n 3 127.0.0.1' }
|
||||
Probe 'wsl-interop-powershell' { & wsl.exe -d Ubuntu-24.04 -u root -- powershell.exe -NoProfile -Command 'Start-Sleep -Milliseconds 1200' }
|
||||
}
|
||||
# b. the trust test: a new console (ShellExecute) must show; the same hidden
|
||||
Probe 'start-process-new-console' { Start-Process -FilePath cmd.exe -ArgumentList '/c ping -n 3 127.0.0.1' -Wait }
|
||||
Probe 'start-process-hidden' { Start-Process -FilePath cmd.exe -ArgumentList '/c ping -n 3 127.0.0.1' -WindowStyle Hidden -Wait }
|
||||
# c. the elevated job's shape without the elevation: powershell in a new hidden console through ShellExecute
|
||||
Probe 'start-process-powershell-hidden' { Start-Process -FilePath powershell.exe -ArgumentList '-NoProfile -ExecutionPolicy Bypass -Command Start-Sleep -Milliseconds 1200' -WindowStyle Hidden -Wait }
|
||||
# d. candidate fixes: a headless conhost of our own around the child; the children of a headless session
|
||||
Probe 'conhost-headless-cmd' { & conhost.exe --headless cmd.exe /c 'ping -n 3 127.0.0.1 >nul' }
|
||||
if ($haveDistro) {
|
||||
Probe 'conhost-headless-wsl-interop' { & conhost.exe --headless wsl.exe -d Ubuntu-24.04 -u root -- cmd.exe /c 'ping -n 3 127.0.0.1' }
|
||||
}
|
||||
Probe 'conhost-headless-start-process-hidden' { & conhost.exe --headless powershell.exe -NoProfile -Command "Start-Process -FilePath cmd.exe -ArgumentList '/c ping -n 3 127.0.0.1' -WindowStyle Hidden -Wait" }
|
||||
|
||||
# ---- the end: anything the probes did not claim ---------------------------------------------------------------------
|
||||
Start-Sleep -Milliseconds 800
|
||||
$sync.stop = $true
|
||||
try { [void]$sampler.EndInvoke($handle) } catch {}
|
||||
$sampler.Dispose(); $rs.Close()
|
||||
$stray = @($sync.win.GetEnumerator() | Where-Object { -not $_.Value.base -and -not $_.Value.reported })
|
||||
foreach ($w in $stray) { Say ("RESULT window: " + $w.Value.proc + " [" + $w.Value.cls + "] " + $w.Value.title + " (between probes)") }
|
||||
$strayH = @($sync.hosts.GetEnumerator() | Where-Object { -not $_.Value.base -and -not $_.Value.reported })
|
||||
foreach ($h in $strayH) { Say ("RESULT host: " + $h.Value.name + " pid " + $h.Key + " parent " + $h.Value.pproc + " cmd " + $h.Value.cmd + " (between probes)") }
|
||||
$counts = @{}
|
||||
foreach ($n in 'conhost', 'OpenConsole', 'WindowsTerminal') { $counts[$n] = @(Get-Process -Name $n -ErrorAction SilentlyContinue).Count }
|
||||
Say ("RESULT hosts-after: conhost " + $counts['conhost'] + " OpenConsole " + $counts['OpenConsole'] + " WindowsTerminal " + $counts['WindowsTerminal'] + "; sampler ticks " + $sync.ticks + $(if ($sync.err) { "; sampler error " + $sync.err } else { '' }))
|
||||
exit 0
|
||||
28
tools/windows/power-prompts-off.ps1
Normal file
28
tools/windows/power-prompts-off.ps1
Normal file
|
|
@ -0,0 +1,28 @@
|
|||
# Stops the administrator prompts a 0.3.9 Igneum Miner app raises on its own on a PC (the project lead, 5 October 2026: "if we
|
||||
# don't have to ask then don't ask"), through the settings API that app has, until the build with the Power control
|
||||
# setting ships: the weekly/first-hour efficiency sweep off (api/sweep/enable), a running sweep stopped
|
||||
# (api/sweep/stop). The power cap has no off switch in 0.3.9 (it asks at every app start and on a slider change, and
|
||||
# nowhere else); this script reports the cards' cap state so the next prompt's source is known. A signed `run` job,
|
||||
# not elevated, a few seconds:
|
||||
# packaging/ota/publish-jobs.sh add --kind run --target ae432dc7 --timeout-minutes 3 \
|
||||
# --script tools/windows/power-prompts-off.ps1 --title "PC 1: sweep off (no administrator prompts)" --deploy
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$base = (Get-Content (Join-Path $env:LOCALAPPDATA 'igneum\app\app.url') -Raw).Trim().TrimEnd('/')
|
||||
function Snapshot([string] $tag) {
|
||||
$s = Invoke-RestMethod -Uri "$base/api/state" -TimeoutSec 20
|
||||
$set = $s.settings
|
||||
"RESULT $tag settings: sweep " + $set.sweep + " | power_control " + $(if ($null -ne $set.power_control) { $set.power_control } else { '(not in this build)' }) + " | prove " + $set.prove + " | remote_jobs " + $set.remote_jobs + " | version " + $s.version
|
||||
foreach ($c in $s.mining.cards) {
|
||||
if ($c.vendor -ne 'nvidia') { continue }
|
||||
"RESULT $tag card: " + $c.name + " | enabled " + $c.enabled + " | power_pct " + $c.power_pct + " | limit " + $c.power_limit_w + " W of " + $c.power_default_w + " W default | applied " + $c.power_applied + " | pinned " + $c.pinned + " | sweep_state " + $c.sweep_state + " | note: " + $c.power_note + " | " + $c.sweep_note
|
||||
}
|
||||
}
|
||||
Snapshot 'before'
|
||||
$r = Invoke-RestMethod -Uri "$base/api/sweep/stop" -Method Post -ContentType 'application/json' -Body '{}' -TimeoutSec 30
|
||||
"RESULT api/sweep/stop: " + ($r | ConvertTo-Json -Compress)
|
||||
$r = Invoke-RestMethod -Uri "$base/api/sweep/enable" -Method Post -ContentType 'application/json' -Body '{"on":false}' -TimeoutSec 30
|
||||
"RESULT api/sweep/enable off: " + ($r | ConvertTo-Json -Compress)
|
||||
Start-Sleep -Seconds 3
|
||||
Snapshot 'after'
|
||||
"RESULT note: the 0.3.9 power cap asks only at an app start or a slider change; no setting turns it off before the Power control build"
|
||||
exit 0
|
||||
Loading…
Reference in a new issue